diff --git a/.claude/skills/gitnexus-cli/SKILL.md b/.claude/skills/gitnexus-cli/SKILL.md index 1b1f733b6..170703439 100644 --- a/.claude/skills/gitnexus-cli/SKILL.md +++ b/.claude/skills/gitnexus-cli/SKILL.md @@ -27,9 +27,12 @@ Run from the project root. This parses all source files, builds the knowledge gr | `--embeddings` | Enable embedding generation for semantic search (off by default) | | `--drop-embeddings` | Drop existing embeddings on rebuild. By default, an `analyze` without `--embeddings` preserves them. | | `--pdg` | Build the program-dependence layers used by `explain` and `pdg_query` (taint, CDG, and REACHING_DEF). | +| `--spring-actuator ` | Import opt-in Spring Boot Actuator mappings, beans, conditions, configprops, and env snapshots. Forces a full rebuild; unsupported with `--watch`. | **When to run:** First time in a project, after major code changes, or when `gitnexus://repo/{name}/context` reports the index is stale. In Claude Code, a PostToolUse hook detects staleness after `git commit` and `git merge` and notifies the agent to run `analyze` — the hook does not run analyze itself, to avoid blocking the agent for up to 120s and risking KuzuDB corruption on timeout. +For Spring runtime enrichment, pass a JSON bundle, one endpoint JSON file, or a directory containing endpoint files. Route evidence is authoritative only when `runtimeConfirmed === true`; `runtimeSource` records provenance and may also accompany `handler-conflict`. Env/configprops values are never persisted. + Use `node .gitnexus/run.cjs analyze --watch` for a long-lived local Git repository. It performs an initial analysis, queues scanner-admitted file changes, and retries intact failed batches with bounded backoff. Watch refreshes update only the graph: they skip AGENTS.md / CLAUDE.md injection and standard skill installation, so run a one-shot `analyze` when those generated files need updating. Watch rejects one-shot or context-output flags including `--force`, embedding flags, `--skills`, `--default-branch`, `--skip-agents-md`, `--skip-skills`, `--no-stats`, `--self-commit`, `--index-only`, and `--skip-git`. It never pulls remotes. Running MCP and `serve` processes periodically check for a published replacement and reopen it without a restart. MCP checks are throttled to once every five seconds, so a tool call before the next check can briefly use the previous index. ### status — Check index freshness diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index 91d44ac45..be7b4d813 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -481,6 +481,20 @@ jobs: node --import tsx bench/python-scope/import-target-fingerprint.mjs --check working-directory: gitnexus + - name: Java wildcard-static route constant guards (#3110) + if: ${{ !cancelled() }} + # Build-free: named-import control vs wildcard materialization; + # fingerprints bindings and guards scaling + absolute wall time. + run: node --import tsx bench/java-wildcard-route-constants/measure.mjs --check + working-directory: gitnexus + + - name: Kotlin package-star route constant guards (#3110) + if: ${{ !cancelled() }} + # Build-free: explicit-import control vs package-star folding; + # fingerprints route facts and guards scaling + widening overhead. + run: node --import tsx bench/kotlin-star-route-constants/measure.mjs --check + working-directory: gitnexus + - name: Cross-language scope-capture fingerprint + scaling guards # Runs even after an earlier guard fails (#2895). Every step here was # fail-fast, so the FIRST failing --check aborted the job and every guard @@ -509,6 +523,31 @@ jobs: run: node --import tsx bench/callable-value-flow/measure.mjs --check working-directory: gitnexus + - name: Java Lombok accessor synthesis guards (#2885) + if: ${{ !cancelled() }} + # Build-free: no-Lombok vs Lombok-heavy corpora; fingerprint over + # synthetic Method ids; scaling + widening overhead budgets. + run: node --import tsx bench/java-lombok-synthesis/measure.mjs --check + working-directory: gitnexus + + - name: Kotlin JVM accessor synthesis guards (#2885) + if: ${{ !cancelled() }} + # Build-free: no-property vs data-class corpora; fingerprint over + # synthetic Method ids; scaling + widening overhead budgets. + run: node --import tsx bench/kotlin-jvm-accessors/measure.mjs --check + working-directory: gitnexus + + - name: Kotlin Spring config-consumer capture guards (#2412) + if: ${{ !cancelled() }} + # Build-free: explicit-import control vs wildcard-import feature path; + # fingerprints @Value / @ConfigurationProperties facts and guards scaling + # + widening overhead. The parity check is the regression gate: each file + # declares a sibling nested type named `Value`, which must not suppress + # the imported Spring annotation (file-wide shadowing dropped 2 of every + # 3 facts on this corpus). + run: node --import tsx bench/spring-config-bindings/measure.mjs --check + working-directory: gitnexus + - name: Re-export closure scaling guards (#2864) # Build-free: asserts buildReexportClosures stays linear in chain depth # and within an absolute ceiling on a wide package corpus. #2864 changed @@ -711,20 +750,22 @@ jobs: - name: Cross-language pipeline benchmarks (GITNEXUS_BENCH, serial) if: ${{ !cancelled() }} - # cpp-adl-benchmark.test.ts is not a `*-pipeline-benchmark.test.ts` but - # belongs here for the same reason: it is skipIf-gated on GITNEXUS_BENCH, - # so it had never run in CI and the PR #1990 ADL emit-scaling guard it - # holds was dead. ~45s of test time. + # cpp-adl-benchmark.test.ts and csharp-razor-view-components-benchmark.test.ts + # are not `*-pipeline-benchmark.test.ts` files but belong here for the + # same reason: they are skipIf-gated on GITNEXUS_BENCH, so the scaling + # guards they hold never run in the main coverage job. env: GITNEXUS_BENCH: '1' run: >- npx vitest run --no-file-parallelism test/integration/cobol-pipeline-benchmark.test.ts test/integration/csharp-pipeline-benchmark.test.ts + test/integration/csharp-razor-view-components-benchmark.test.ts test/integration/cpp-adl-benchmark.test.ts test/integration/data-route-table-benchmark.test.ts test/integration/instance-ownership-pipeline-benchmark.test.ts test/integration/spring-bean-resource-benchmark.test.ts + test/integration/spring-dynamic-lookup-benchmark.test.ts test/integration/rust-pipeline-benchmark.test.ts test/integration/php-pipeline-benchmark.test.ts test/integration/ruby-pipeline-benchmark.test.ts diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index 41440d557..a768b69e7 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -71,6 +71,12 @@ jobs: # deliberately contain use-before-init / unused-variable shapes). - '**/test/fixtures/**' - '**/test/**/fixtures/**' + # GET /api/grep intentionally builds RegExp from the query string + # (literal=1 escapes). ReDoS is handled by worker terminate() — + # see SECURITY.md. Inline codeql[] comments do not clear the + # GitHub PR CodeQL gate, so this file is excluded to avoid + # re-filing js/regex-injection on every push of the same line. + - 'gitnexus/src/server/grep-params.ts' - name: Perform CodeQL Analysis uses: github/codeql-action/analyze@ff2f1c621b7f889edc0d3c761ac2e6a3f8cdb0dd # v4.37.7 diff --git a/.gitleaksignore b/.gitleaksignore index c713b712a..243cd4410 100644 --- a/.gitleaksignore +++ b/.gitleaksignore @@ -1,2 +1,4 @@ # Deleted README placeholder from PR #2458; no credential was present. c9fdab17f25ebaf332fba6e6ba55ee328f20fe66:README.md:curl-auth-header:348 +# Synthetic Kotlin Actuator fixture value from PR #3107; no credential was present. +3951079300a18b14e79f5b5f5dd778ae19ced6e3:gitnexus/test/integration/spring-actuator-kotlin-runtime-pipeline.test.ts:generic-api-key:8 diff --git a/AGENTS.md b/AGENTS.md index 05be52af2..251f410c1 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -121,8 +121,7 @@ This project is indexed by GitNexus as **GitNexus** (248612 symbols, 565510 rela - **MUST analyze graph changes before committing.** Use `detect_changes({scope: "all"})` (MCP) or `node .gitnexus/run.cjs detect-changes --scope all --repo .` (CLI fallback). `partial: true` or `truncated: true` is not a clean check — a zero means unseen, not unaffected; re-run it. For regression review: `detect_changes({scope: "compare", base_ref: "main"})` or `node .gitnexus/run.cjs detect-changes --scope compare --base-ref "main" --repo .`. - MUST warn on HIGH/CRITICAL `risk` pre-edit; never use `riskSharedAxes` to waive a HIGH/CRITICAL `risk` warning. Compare File/symbol: MCP File omits axes; Graph-RAG expands File. - **MUST treat `risk: UNKNOWN` as unresolved, not as low.** An empty caller set is not evidence the symbol is unused — it can also mean the callers are not resolvable by the index (plain-object property access, dynamic dispatch, cross-language calls). `impact` pairs `UNKNOWN` with a `riskNote` saying so. Confirm with a text search before treating the symbol as safe to change or delete; do not proceed on the strength of a zero. -- Explore with `query({search_query: "concept"})` for process-grouped flows. -- Use `context({name: "symbolName"})` for callers, callees, and flows. +- **MUST use `query({search_query: "concept"})` for concepts/flows, `context({name: "symbolName"})` for a named symbol, or `impact` for blast radius, on read-only callers, dependencies, imports, or execution flow.** Graph first; text search only for empty/`UNKNOWN`/literals. - For security review, `explain({target: "fileOrSymbol"})` lists taint findings (source→sink flows; needs `analyze --pdg`). - For control/data dependence, `pdg_query({mode: "controls", target: "fileOrSymbol"})` answers "under what condition does X run?" (CDG, incl. guard clauses) and `pdg_query({mode: "flows", target, variable})` traces "where does variable Y flow?" (REACHING_DEF). `--pdg` layer. diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 6fe547b70..b1076d1d5 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -108,7 +108,7 @@ scan → structure → [springConfig, markdown, cobol] → parse → [routes, to | `pruneLocalSymbols` | `prune-local-symbols.ts` | `scopeResolution` | Drops inert block-local `Const`/`Variable`/`Static` nodes (only a `File→DEFINES` edge) post-resolution | | `mro` | `mro.ts` | `crossFile`, `scopeResolution`, `pruneLocalSymbols`, `structure` | METHOD_OVERRIDES + METHOD_IMPLEMENTS edges | | `springAopInheritance` | `spring-aop.ts` | `springAop`, `mro` | Propagates declarative behavior through class/interface inheritance decisions | -| `di` | `di.ts` | `mro` | INJECTS edges from consumer Classes or factory Methods to provider Classes/declaration CodeElements (framework-neutral DI resolution; per-language matchers registered in `di-extractors/`) | +| `di` | `di.ts` | `mro` | INJECTS edges from consumer Classes, factory Methods, or AST-captured programmatic lookup callables to provider Classes/declaration CodeElements (framework-neutral DI resolution; per-language matchers registered in `di-extractors/`) | | `communities` | `communities.ts` | `mro`, `pruneLocalSymbols`, `structure` | Community nodes + MEMBER_OF edges (Leiden algorithm) | | `processes` | `processes.ts` | `communities`, `routes`, `tools`, `pruneLocalSymbols`, `structure` | Process nodes + STEP_IN_PROCESS edges | diff --git a/CLAUDE.md b/CLAUDE.md index 10681d819..069163232 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -72,8 +72,7 @@ This project is indexed by GitNexus as **GitNexus** (248612 symbols, 565510 rela - **MUST analyze graph changes before committing.** Use `detect_changes({scope: "all"})` (MCP) or `node .gitnexus/run.cjs detect-changes --scope all --repo .` (CLI fallback). `partial: true` or `truncated: true` is not a clean check — a zero means unseen, not unaffected; re-run it. For regression review: `detect_changes({scope: "compare", base_ref: "main"})` or `node .gitnexus/run.cjs detect-changes --scope compare --base-ref "main" --repo .`. - MUST warn on HIGH/CRITICAL `risk` pre-edit; never use `riskSharedAxes` to waive a HIGH/CRITICAL `risk` warning. Compare File/symbol: MCP File omits axes; Graph-RAG expands File. - **MUST treat `risk: UNKNOWN` as unresolved, not as low.** An empty caller set is not evidence the symbol is unused — it can also mean the callers are not resolvable by the index (plain-object property access, dynamic dispatch, cross-language calls). `impact` pairs `UNKNOWN` with a `riskNote` saying so. Confirm with a text search before treating the symbol as safe to change or delete; do not proceed on the strength of a zero. -- Explore with `query({search_query: "concept"})` for process-grouped flows. -- Use `context({name: "symbolName"})` for callers, callees, and flows. +- **MUST use `query({search_query: "concept"})` for concepts/flows, `context({name: "symbolName"})` for a named symbol, or `impact` for blast radius, on read-only callers, dependencies, imports, or execution flow.** Graph first; text search only for empty/`UNKNOWN`/literals. - For security review, `explain({target: "fileOrSymbol"})` lists taint findings (source→sink flows; needs `analyze --pdg`). - For control/data dependence, `pdg_query({mode: "controls", target: "fileOrSymbol"})` answers "under what condition does X run?" (CDG, incl. guard clauses) and `pdg_query({mode: "flows", target, variable})` traces "where does variable Y flow?" (REACHING_DEF). `--pdg` layer. diff --git a/README.md b/README.md index 405d85201..f011447fd 100644 --- a/README.md +++ b/README.md @@ -1,4 +1,4 @@ -# GitNexus (Akon Labs) +# GitNexus (Akon Labs) **⚠️ Important Notice:** GitNexus has NO official cryptocurrency, token, or coin. Any token/coin using the GitNexus name on Pump.fun or any other platform is **not affiliated with, endorsed by, or created by** this project or its maintainers. Do not purchase any cryptocurrency claiming association with GitNexus. @@ -449,10 +449,13 @@ gitnexus analyze --verbose # Log skipped files when parsers are unavailabl gitnexus analyze --worker-timeout 60 # Increase worker idle timeout for slow parses gitnexus analyze --workers # Parse worker pool size (>=1; default: cores-1, capped at 16, # auto-sized to the repo). 0 is rejected — there is no sequential mode. +gitnexus analyze --spring-actuator ./actuator # Enrich with local Spring Boot Actuator JSON snapshots gitnexus analyze --wal-checkpoint-threshold 67108864 # LadybugDB WAL auto-checkpoint threshold in bytes # (default 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB) ``` +`--spring-actuator` is explicitly opt-in and accepts either a JSON bundle keyed by `mappings`, `beans`, `conditions`, `configprops`, and/or `env`, or a directory containing endpoint-named JSON files. It confirms matching static nodes and adds conservative runtime-only routes, beans, and property keys. The configured input is excluded from source scanning; only normalized repository-relative exclusions are retained for future scans, never absolute paths. Env/configprops values, origins, condition messages, and source names are never persisted or printed. Because snapshots are external runtime state, an enabled run always rebuilds; the first later run without the option rebuilds once to remove runtime evidence. The same path can be set as `springActuator` in `.gitnexusrc`. + If `analyze` reports a worker parse timeout on a large or unusual repository, it keeps running and falls back safely. To give slow worker jobs more time, use `--worker-timeout 60` or set `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS=60000`. For very large files, `GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES` controls the worker job byte budget. **Embeddings node limit** — `gitnexus analyze --embeddings` generates semantic search vectors with a default 50,000-node safety cap to protect memory on large repositories: @@ -500,6 +503,7 @@ Commit a `.gitnexusrc` JSON file at the repo root to preconfigure recurring `ana "skipContextFiles": true, // alias of skipAgentsMd: keep your own AGENTS.md/CLAUDE.md "skipSkills": true, // don't install standard skill files under .claude/skills/ and .agents/skills/ "embeddings": true, // generate embeddings by default + "springActuator": "./actuator", // optional local runtime snapshot directory or bundle "workerTimeout": 60, } ``` @@ -514,7 +518,7 @@ Notes: - The default branch is resolved as: `--default-branch` > `.gitnexusrc` `defaultBranch`/`branch` > auto-detected `origin/HEAD` > `main`. - `skipContextFiles` / `skipAiContext` are aliases for `skipAgentsMd` — they skip the `AGENTS.md` / `CLAUDE.md` block only. They do **not** imply `skipSkills`. `indexOnly` is the stronger option that skips all file injection. -- Supported keys: `defaultBranch` (`branch`), `skipAgentsMd` (`skipContextFiles`, `skipAiContext`), `skipSkills`, `indexOnly`, `stats`/`noStats`, `embeddings`, `dropEmbeddings`, `name`, `allowDuplicateName`, `maxFileSize`, `workerTimeout`, `walCheckpointThreshold`, `workers`, `embeddingThreads`, `embeddingBatchSize`, `embeddingSubBatchSize`, `embeddingDevice`. +- Supported keys: `defaultBranch` (`branch`), `skipAgentsMd` (`skipContextFiles`, `skipAiContext`), `skipSkills`, `indexOnly`, `stats`/`noStats`, `embeddings`, `dropEmbeddings`, `name`, `allowDuplicateName`, `maxFileSize`, `workerTimeout`, `walCheckpointThreshold`, `workers`, `springActuator`, `embeddingThreads`, `embeddingBatchSize`, `embeddingSubBatchSize`, `embeddingDevice`. - The file is JSON only. Unknown keys and invalid values fail fast with an actionable error before analysis starts. @@ -524,36 +528,39 @@ Notes: Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max-file-size`, `--verbose`). Use the env-var form when you'd otherwise repeat the same flag every run, or when invoking GitNexus from a long-running host (MCP server, eval-server, CI shell) that already manages its own environment. CLI flags take precedence over env vars; env vars take precedence over built-in defaults. -| Variable | Default | Effect | Tune when… | -| ----------------------------------------------- | ------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `GITNEXUS_WORKER_POOL_SIZE` | `cores - 1`, capped at 16 | Parse worker pool size (must be ≥ 1). Equivalent to `--workers `. The worker pool is the sole parse path — there is no sequential parser, so `0` is rejected with an actionable error (the pool self-heals via quarantine + respawn). | Constrained containers (cgroup CPU limits) or CI runners with explicit quotas. To narrow down a worker crash set `1` for a single-worker pool — not `0`. | -| `GITNEXUS_PARSE_CHUNK_CONCURRENCY` | `2` | Number of chunks whose file contents may be read into memory in parallel while the pool dispatches the current chunk. Worker dispatch itself stays serial. | Repos large enough to chunk (multi-MB total source) where disk I/O is a measurable fraction of analyze wall-clock. | -| `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. | -| `GITNEXUS_AUTH_TOKEN` | unset | Bearer token required when `eval-server` binds beyond loopback. May also be read from `.env.local` or `.env`; shell values take precedence. | Exposing the evaluation HTTP tools to a container, VM, or LAN. | -| `GITNEXUS_PROFILE_DEFERRED` | unset | When `1`, emits `[deferred-profile]` timing/progress logs for the post-chunk deferred resolution band (imports → heritage → buildHeritageMap → legacy call resolution). Implied by `GITNEXUS_VERBOSE`. | Diagnosing analyze stalls in "Resolving calls (all chunks)" on large Java/Kotlin repos (issue #1741) without the full verbose ingestion noise. | -| `GITNEXUS_PROFILE_DEFERRED_SLOW_MS` | `3000` (verbose) / `5000` | Per-file threshold in ms above which `processCallsFromExtracted` emits a `slow file …` log line. Parsed via `Number()`: accepts integers (`5000`), scientific notation (`2.5e3`), decimals (`.5`), and hex (`0x10`). Non-finite or non-positive values fall back to the default. | Hunting a few outlier files dominating the deferred call-resolution stage; lower to surface more, raise to focus only on the worst. | -| `PROF_LBUG_LOAD` | unset | When `1`, emits one `[lbug-load prof]` summary line per `loadGraphToLbug` call breaking the graph-DB persistence wall into stages (`csv-emit` / `copy-nodes` / `copy-rels` / `fallback` / `total`) plus node & edge counts. Zero-cost when unset. | Attributing large-repo analyze wall time across CSV generation vs. LadybugDB `COPY` (issue #2203) — the analyze "emit" timing is the scope-resolution bucket, not this DB-write path. | -| `GITNEXUS_MAX_FILE_SIZE` | `512` (KB) | Walker skip threshold in KB. Hard cap is `32768` (tree-sitter buffer ceiling). Equivalent to `--max-file-size `. | Indexing repos with intentionally-large source files (generated parsers, vendored bundles) that should still be parsed. | -| `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS` | `30000` | Worker idle timeout in milliseconds before retry/fallback. Equivalent to `--worker-timeout ` × 1000. | Slow-parsing files (large minified JS, deeply-nested TS types) that legitimately need more than 30s. | -| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget in milliseconds for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. | Slow or heavily loaded hosts where a full pool cold-starting concurrently needs more than 5s, and analyze aborts with "did not report ready within 5000ms". | -| `GITNEXUS_FTS_STEMMER` | `porter` | Stemmer used when rebuilding BM25/FTS indexes. Use `none` for CJK-heavy repositories, or a language stemmer such as `german`, `french`, or `spanish` for matching repository comments. Re-run `gitnexus analyze --repair-fts` after changing it. | Keyword search quality is poor for non-English comments or identifiers under English stemming. | -| `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold in bytes. Equivalent to `--wal-checkpoint-threshold `. `-1` keeps LadybugDB's stock threshold (~16 MiB). Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | You need a larger or smaller WAL auto-checkpoint threshold for your analyze workload. | -| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling in bytes for every GitNexus database (analyze, MCP server, serve, group bridges). `0` restores LadybugDB's native unbounded default of 80% of system RAM; invalid values warn and fall back to the default (#2557). During `analyze` the pool is right-sized to the graph, scaled on non-4 KiB-page hosts by the page-size granule ratio up to min(2 GiB × pageSize/4 KiB, 80% RAM) (#2631); this env var overrides all of that as an absolute value. | A long-lived `gitnexus mcp` or a big incremental `analyze` uses too much memory, or a huge repo's working set genuinely needs a pool larger than 2 GiB. | -| `GITNEXUS_LBUG_MAX_DB_SIZE` | `17179869184` (16 GiB) | Maximum size in bytes of a single LadybugDB database file — an mmap/disk-address-space ceiling, not a memory limit (it does not constrain the buffer pool). Invalid values silently fall back to the default. | Indexing a genuinely huge monorepo whose on-disk graph index approaches 16 GiB. | -| `GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES` | `8388608` (8 MB) | Per-job byte budget the pool will send to a worker in one `postMessage`. | Very large individual files; mostly diagnostic — bumping past 8 MB risks structured-clone memory pressure. | -| `GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT` | `3` | Max replacement spawns per worker slot before the slot is dropped from the active rotation. Bounds respawn loops on a chronically-crashing slot. | Hosts where a flaky worker should retry more (raise) or fail-fast (lower) before the slot is dropped. | -| `GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS` | `5 × subBatchTimeoutMs` | Total retry wall-time budget per job before quarantining. Combined with `timeoutBackoffFactor`, prevents exponentially-growing retries from stalling for hours. | Slow files that legitimately need long total retry windows; lower to fail-fast on stalls. | -| `GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD` | `max(3, poolSize)` | Per-slot consecutive deaths before the pool's circuit breaker trips. After tripping, every subsequent dispatch rejects until a fresh pool is created. | Hosts where a SIGSEGV-prone native grammar should trip the breaker sooner; CI runners that should fail loudly. | -| `GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS` | `30000` | Max wait at pool shutdown for a retired worker still inside native code. The worker is terminated at its next JS-safe point instead of mid-native-call (which aborts the whole process with `Napi::Error`, #2432); on expiry it is left running, unref'd, and terminated when it surfaces. | Shutdown latency matters more than draining a wedged worker (lower), or a legitimately-slow native grammar needs longer to surface (raise). | -| `GITNEXUS_CPP_CAPTURE_BUDGET_MS` | `20000` | Per-file wall-clock budget for C++ capture extraction. On breach the file keeps the captures accumulated so far and logs a warning — the worker returns to JS instead of stalling in native-heavy loops (#2432). `0` expires immediately. | Pathological generated C++ that still exceeds the budget after the indexed lookups; raise for completeness, lower to fail-fast. | -| `GITNEXUS_CHUNK_BYTE_BUDGET` | `2097152` (2 MB) | Per-bucket byte budget for parse-cache packing. Files are grouped by `(language, hash(path) mod 128)`; packs inside a bucket are cut at this limit. Smaller = finer-grained invalidation and more dispatch. Default is always 2 MiB and no longer scales with worker count. | Tuning incremental-analyze cache invalidation on monorepos without changing `--workers`. | -| `GITNEXUS_NO_GITIGNORE` | unset | When set, skips `.gitignore` parsing. `.gitnexusignore` is still honored. | Indexing a repo whose `.gitignore` excludes files you actually want indexed (e.g., generated code committed for cross-repo lookup). | -| `GITNEXUS_SKIP_OPTIONAL_GRAMMARS` | unset | When `=1` strictly, skips the vendored grammar materialize for `tree-sitter-dart`, `tree-sitter-proto`, `tree-sitter-swift`, and `tree-sitter-kotlin` at install time (and the Dart/Proto source builds). Those four won't be parsed; the install still succeeds. | Installing on a host without a C++ toolchain or where the vendored prebuilds don't match; willing to skip Dart/Proto/Swift/Kotlin parsing. | -| `GITNEXUS_MCP_READ_ONLY` | unset | Set to `1` to expose only proven single-repository read tools and resources; `0` disables the policy and any other value fails startup. | The MCP server runs in an environment where graph mutation, raw Cypher, and cross-repository group routing must be unavailable. | -| `GITNEXUS_MCP_ALLOWED_REPOS` | unset | Comma-separated allowlist of canonical indexed repository names or absolute paths. Invalid, ambiguous, or blank entries fail startup. | One MCP process must expose only a bounded subset of the repositories in the global registry. | -| `GITNEXUS_MCP_DEFAULT_REPO` | unset | Canonical indexed repository name or absolute path used when a tool or resource omits its repository. Must belong to the allowlist when one is set. | Several repositories are available but unqualified MCP calls should resolve deterministically. | -| `GITNEXUS_MCP_DEFAULT_MAX_TOKENS` | unset | Default positive-integer response budget for MCP `query`, `context`, and `impact`, estimated at four UTF-8 bytes per token. Explicit `maxTokens` wins. | Long MCP responses consume too much model context and callers cannot reliably add a per-request budget. | -| `GITNEXUS_PUBLIC_ORIGIN` | unset | The single browser origin `serve` is reached through, added to the CORS allowlist and to the write-route origin guard. A wildcard bind (`0.0.0.0`) has no host identity, so without this the server's own UI is refused. **Setting it currently refuses to start:** `serve` has no authentication, requests carrying no `Origin` header already reach `POST /api/analyze` and `DELETE /api/repo`, and this is the setting that would admit browser writes on top of that. Matching rules for when the gate lifts: the hostname must match exactly, and so must the scheme. A value with no scheme (`app.example.com`) means `https`, since a bare host comes from platform service discovery and those terminate TLS; spell out `http://app.example.com` for plain HTTP. An explicit port must match; with no port, any port on that hostname is accepted. Anything that is not one reachable host (a list, `*`, a bare port number, a `:0` port, a trailing dot) warns at startup and allows nothing. | `gitnexus serve` runs behind a reverse proxy or on a wildcard bind, and the UI's index/delete requests return `origin_not_allowed`. | +| Variable | Default | Effect | Tune when… | +| ----------------------------------------------- | ---------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `GITNEXUS_WORKER_POOL_SIZE` | `cores - 1`, capped at 16 | Parse worker pool size (must be ≥ 1). Equivalent to `--workers `. The worker pool is the sole parse path — there is no sequential parser, so `0` is rejected with an actionable error (the pool self-heals via quarantine + respawn). | Constrained containers (cgroup CPU limits) or CI runners with explicit quotas. To narrow down a worker crash set `1` for a single-worker pool — not `0`. | +| `GITNEXUS_PARSE_CHUNK_CONCURRENCY` | `2` | Number of chunks whose file contents may be read into memory in parallel while the pool dispatches the current chunk. Worker dispatch itself stays serial. | Repos large enough to chunk (multi-MB total source) where disk I/O is a measurable fraction of analyze wall-clock. | +| `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. | +| `GITNEXUS_ANALYZER_IDENTITY_IN_PROCESS_GUARDS` | unset | When truthy (`1`/`true`/`yes`), forces in-process cache-guard validation once a batch has ≥128 requests. In-process mode also auto-selects when `packageRoot`/`buildRoot` fail `W_OK` with `EACCES`/`EROFS`. Otherwise those large batches use a Node subprocess probe. Batches under 128 always stay in-process. | Trusted or read-only installs where two identity subprocess spawns per analyze dominate wall time; leave unset to keep the default isolation path on writable trees. | +| `GITNEXUS_RESOLVE_DEF_GRAPH_ID_MEMO` | on (unset) | Memoizes `resolveDefGraphId` per `nodeLookup` instance (WeakMap). Enabled by default. Set to `0`/`false`/`off`/`no` to disable and recompute on every call (debug / bisect memo bugs). | Suspecting stale graph-id resolution after a lookup rebuild, or comparing memo vs uncached cost on a large index. | +| `GITNEXUS_AUTH_TOKEN` | unset | Bearer token required when `eval-server` binds beyond loopback. May also be read from `.env.local` or `.env`; shell values take precedence. | Exposing the evaluation HTTP tools to a container, VM, or LAN. | +| `GITNEXUS_MCP_AUTH_TOKEN` | unset | Bearer token for the dedicated `gitnexus mcp --http` server, for a **directly reachable** `gitnexus serve` `/api/mcp` route, and for the `docker-server` / web proxy in front of one. A non-loopback dedicated MCP bind requires it; `serve` enables protocol-layer MCP auth when it is set. Behind a proxy, set the **same** value on both services: the proxy spends the edge `GITNEXUS_SERVE_AUTH_TOKEN`, then replaces `Authorization` with this token on `/api/mcp` only. | Dedicated MCP, a `serve` the client can reach directly, or a proxied deploy (Render Blueprint) where the backend runs protocol-layer MCP auth — configure it on the proxy too. | +| `GITNEXUS_PROFILE_DEFERRED` | unset | When `1`, emits `[deferred-profile]` timing/progress logs for the post-chunk deferred resolution band (imports → heritage → buildHeritageMap → legacy call resolution). Implied by `GITNEXUS_VERBOSE`. | Diagnosing analyze stalls in "Resolving calls (all chunks)" on large Java/Kotlin repos (issue #1741) without the full verbose ingestion noise. | +| `GITNEXUS_PROFILE_DEFERRED_SLOW_MS` | `3000` (verbose) / `5000` | Per-file threshold in ms above which `processCallsFromExtracted` emits a `slow file …` log line. Parsed via `Number()`: accepts integers (`5000`), scientific notation (`2.5e3`), decimals (`.5`), and hex (`0x10`). Non-finite or non-positive values fall back to the default. | Hunting a few outlier files dominating the deferred call-resolution stage; lower to surface more, raise to focus only on the worst. | +| `PROF_LBUG_LOAD` | unset | When `1`, emits one `[lbug-load prof]` summary line per `loadGraphToLbug` call breaking the graph-DB persistence wall into stages (`csv-emit` / `copy-nodes` / `copy-rels` / `fallback` / `total`) plus node & edge counts. Zero-cost when unset. | Attributing large-repo analyze wall time across CSV generation vs. LadybugDB `COPY` (issue #2203) — the analyze "emit" timing is the scope-resolution bucket, not this DB-write path. | +| `GITNEXUS_MAX_FILE_SIZE` | `512` (KB) | Walker skip threshold in KB. Hard cap is `32768` (tree-sitter buffer ceiling). Equivalent to `--max-file-size `. | Indexing repos with intentionally-large source files (generated parsers, vendored bundles) that should still be parsed. | +| `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS` | `30000` | Worker idle timeout in milliseconds before retry/fallback. Equivalent to `--worker-timeout ` × 1000. | Slow-parsing files (large minified JS, deeply-nested TS types) that legitimately need more than 30s. | +| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget in milliseconds for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. | Slow or heavily loaded hosts where a full pool cold-starting concurrently needs more than 5s, and analyze aborts with "did not report ready within 5000ms". | +| `GITNEXUS_FTS_STEMMER` | `porter` | Stemmer used when rebuilding BM25/FTS indexes. Use `none` for CJK-heavy repositories, or a language stemmer such as `german`, `french`, or `spanish` for matching repository comments. Re-run `gitnexus analyze --repair-fts` after changing it. | Keyword search quality is poor for non-English comments or identifiers under English stemming. | +| `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold in bytes. Equivalent to `--wal-checkpoint-threshold `. `-1` keeps LadybugDB's stock threshold (~16 MiB). Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | You need a larger or smaller WAL auto-checkpoint threshold for your analyze workload. | +| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling in bytes for every GitNexus database (analyze, MCP server, serve, group bridges). `0` restores LadybugDB's native unbounded default of 80% of system RAM; invalid values warn and fall back to the default (#2557). During `analyze` the pool is right-sized to the graph, scaled on non-4 KiB-page hosts by the page-size granule ratio up to min(2 GiB × pageSize/4 KiB, 80% RAM) (#2631); this env var overrides all of that as an absolute value. | A long-lived `gitnexus mcp` or a big incremental `analyze` uses too much memory, or a huge repo's working set genuinely needs a pool larger than 2 GiB. | +| `GITNEXUS_LBUG_MAX_DB_SIZE` | `17179869184` (16 GiB) | Maximum size in bytes of a single LadybugDB database file — an mmap/disk-address-space ceiling, not a memory limit (it does not constrain the buffer pool). Invalid values silently fall back to the default. | Indexing a genuinely huge monorepo whose on-disk graph index approaches 16 GiB. | +| `GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES` | `8388608` (8 MB) | Per-job byte budget the pool will send to a worker in one `postMessage`. | Very large individual files; mostly diagnostic — bumping past 8 MB risks structured-clone memory pressure. | +| `GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT` | `3` | Max replacement spawns per worker slot before the slot is dropped from the active rotation. Bounds respawn loops on a chronically-crashing slot. | Hosts where a flaky worker should retry more (raise) or fail-fast (lower) before the slot is dropped. | +| `GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS` | `5 × subBatchTimeoutMs` | Total retry wall-time budget per job before quarantining. Combined with `timeoutBackoffFactor`, prevents exponentially-growing retries from stalling for hours. | Slow files that legitimately need long total retry windows; lower to fail-fast on stalls. | +| `GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD` | `max(3, poolSize)` | Per-slot consecutive deaths before the pool's circuit breaker trips. After tripping, every subsequent dispatch rejects until a fresh pool is created. | Hosts where a SIGSEGV-prone native grammar should trip the breaker sooner; CI runners that should fail loudly. | +| `GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS` | `30000` | Max wait at pool shutdown for a retired worker still inside native code. The worker is terminated at its next JS-safe point instead of mid-native-call (which aborts the whole process with `Napi::Error`, #2432); on expiry it is left running, unref'd, and terminated when it surfaces. | Shutdown latency matters more than draining a wedged worker (lower), or a legitimately-slow native grammar needs longer to surface (raise). | +| `GITNEXUS_CPP_CAPTURE_BUDGET_MS` | `20000` | Per-file wall-clock budget for C++ capture extraction. On breach the file keeps the captures accumulated so far and logs a warning — the worker returns to JS instead of stalling in native-heavy loops (#2432). `0` expires immediately. | Pathological generated C++ that still exceeds the budget after the indexed lookups; raise for completeness, lower to fail-fast. | +| `GITNEXUS_CHUNK_BYTE_BUDGET` | `2097152` (2 MB) | Per-bucket byte budget for parse-cache packing. Files are grouped by `(language, hash(path) mod 128)`; packs inside a bucket are cut at this limit. Smaller = finer-grained invalidation and more dispatch. Default is always 2 MiB and no longer scales with worker count. | Tuning incremental-analyze cache invalidation on monorepos without changing `--workers`. | +| `GITNEXUS_NO_GITIGNORE` | unset | When set, skips `.gitignore` parsing. `.gitnexusignore` is still honored. | Indexing a repo whose `.gitignore` excludes files you actually want indexed (e.g., generated code committed for cross-repo lookup). | +| `GITNEXUS_SKIP_OPTIONAL_GRAMMARS` | unset | When `=1` strictly, skips the vendored grammar materialize for `tree-sitter-dart`, `tree-sitter-proto`, `tree-sitter-swift`, and `tree-sitter-kotlin` at install time (and the Dart/Proto source builds). Those four won't be parsed; the install still succeeds. | Installing on a host without a C++ toolchain or where the vendored prebuilds don't match; willing to skip Dart/Proto/Swift/Kotlin parsing. | +| `GITNEXUS_MCP_READ_ONLY` | unset | Set to `1` to expose only proven single-repository read tools and resources; `0` disables the policy and any other value fails startup. | The MCP server runs in an environment where graph mutation, raw Cypher, and cross-repository group routing must be unavailable. | +| `GITNEXUS_MCP_ALLOWED_REPOS` | unset | Comma-separated allowlist of canonical indexed repository names or absolute paths. Invalid, ambiguous, or blank entries fail startup. | One MCP process must expose only a bounded subset of the repositories in the global registry. | +| `GITNEXUS_MCP_DEFAULT_REPO` | unset | Canonical indexed repository name or absolute path used when a tool or resource omits its repository. Must belong to the allowlist when one is set. | Several repositories are available but unqualified MCP calls should resolve deterministically. | +| `GITNEXUS_MCP_DEFAULT_MAX_TOKENS` | unset | Default positive-integer response budget for MCP `query`, `context`, and `impact`, estimated at four UTF-8 bytes per token. Explicit `maxTokens` wins. | Long MCP responses consume too much model context and callers cannot reliably add a per-request budget. | +| `GITNEXUS_PUBLIC_ORIGIN` | unset | The single browser origin `serve` is reached through, added to the CORS allowlist and to the write-route origin guard. A wildcard bind (`0.0.0.0`) has no host identity, so without this the server's own UI is refused. **Setting it currently refuses to start:** `serve` has no authentication, requests carrying no `Origin` header already reach `POST /api/analyze` and `DELETE /api/repo`, and this is the setting that would admit browser writes on top of that. Matching rules for when the gate lifts: the hostname must match exactly, and so must the scheme. A value with no scheme (`app.example.com`) means `https`, since a bare host comes from platform service discovery and those terminate TLS; spell out `http://app.example.com` for plain HTTP. An explicit port must match; with no port, any port on that hostname is accepted. Anything that is not one reachable host (a list, `*`, a bare port number, a `:0` port, a trailing dot) warns at startup and allows nothing. | `gitnexus serve` runs behind a reverse proxy or on a wildcard bind, and the UI's index/delete requests return `origin_not_allowed`. | | `GITNEXUS_TRUST_PROXY` | `loopback, linklocal, uniquelocal` | Express `trust proxy` value — which upstream hops may set `X-Forwarded-*`, and so what the per-IP rate limiter reads as the client IP. Set it to the exact number of proxies you control. Every hop past that is one more entry of the chain the caller gets to write. `false`/`no`/`off` (and a `0` hop count) trust no hop; a proxy list Express can compile (`loopback`, `10.0.0.0/8, 127.0.0.1`) names them instead. `true`/`yes`/`on` is **rejected**: it reads the client-controlled leftmost `X-Forwarded-For` entry, so a spoofed chain earns a fresh rate-limit key per request, and express-rate-limit rejects it too (`ERR_ERL_PERMISSIVE_TRUST_PROXY`). Counts above `16` are rejected as well, as a sanity ceiling rather than a safety boundary. Any invalid value warns and falls back to the default. Bind non-loopback with this unset and `serve` warns: a load balancer outside the private ranges is untrusted, so every request keys to the balancer and the per-IP limit becomes one shared limit. | `serve` sits behind a load balancer outside the private ranges (AWS ALB, Cloudflare, CGNAT), where every request otherwise collapses to the proxy hop and rate limiting goes global. | diff --git a/SECURITY.md b/SECURITY.md index d1fbcd051..37d216911 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -59,11 +59,16 @@ The `render.yaml` Blueprint (see the README's **Deploy to Render**) puts `gitnex - **The generated `GITNEXUS_SERVE_AUTH_TOKEN` is the only access control.** The proxy rejects any `/api/*` request without it with a `401` before forwarding. Rotate it by editing the environment variable on the `gitnexus-web` service and redeploying. - **The CSRF guard is inert on this path.** The proxy strips `Origin` before forwarding, so the server's write-origin guard does nothing for proxied traffic — it passes `Origin`-less requests through by design. The token is not a second layer behind the guard. - **Anyone holding the token can read every indexed repo's source.** These routes carry no origin guard, and the first three carry no rate limiter either: `GET /api/repos`, `GET /api/graph`, `POST /api/query`, `GET /api/file`, `GET /api/grep`. Whoever has the token can also index and delete repositories. -- **`POST /api/mcp` rides the same path.** `serve` mounts the MCP handler via `mountMCPEndpoints`, and `createStreamableHttpHandler` is called with no `authToken` — a **pre-existing** gap in `serve` itself, not something this deploy introduces. On Render it is closed only by the edge token and the private network. A `serve` bound directly to a public interface has no such cover. +- **`POST /api/mcp` rides the same path.** When `GITNEXUS_MCP_AUTH_TOKEN` is set on the backend, `serve` protects `/api/mcp` with the same constant-time Bearer check as the dedicated HTTP MCP server, before parsing the request body. The Render Blueprint does not set a backend MCP token by default. To enable it behind the proxy, set the **same** `GITNEXUS_MCP_AUTH_TOKEN` on both the `gitnexus-web` proxy and the `gitnexus-server` backend: the proxy consumes the edge `GITNEXUS_SERVE_AUTH_TOKEN`, then replaces `Authorization` with the MCP token on `/api/mcp` (and its subpaths) only — the edge credential is never forwarded, and other `/api/*` routes stay stripped. Configuring it on the backend alone makes every proxied MCP request `401`. +- **A directly reachable `serve` still needs an explicit control.** If neither `GITNEXUS_MCP_AUTH_TOKEN` nor an authenticated edge/private-network boundary is present, `/api/mcp` is unauthenticated. Do not bind that topology to a LAN or public interface: MCP readers can access indexed source and graph context. - **Rate limits bound cost, not access.** They cap what a token holder can spend; they do not decide who gets in. Do not hand the URL out as a public demo. A token holder has read access to everything the deploy has indexed. +### `/api/grep` regex semantics and residual ReDoS exposure + +`GET /api/grep` executes caller-supplied patterns as real regular expressions (with an optional path-substring `fileFilter` and `caseSensitive` flag) to honor the web chat's grep tool contract; `literal=1` restores the older escaped-substring mode. Mitigations: a 200-character pattern cap, line-by-line matching, a max-200 result cap, and a 5-second wall-clock budget. Matching runs in a `worker_threads` worker so a catastrophic pattern (e.g. `(a+)+$`) can be killed with `terminate()` when the budget expires — the parent event loop (other routes + SSE) stays responsive. A timed-out scan returns partial results with `timedOut: true`; the web grep tool surfaces that flag so an agent does not treat a cut-off scan as exhaustive. CodeQL still flags constructing a `RegExp` from the query string; that is the advertised contract, not accidental injection. Hosted deploys continue to gate the route behind the edge token. + ## Automated Scans Running in CI This repository runs the following scans automatically. Findings appear under the repository's **Security → Code scanning** tab. diff --git a/docker-server.mjs b/docker-server.mjs index e3e9f68ee..3b036f7db 100644 --- a/docker-server.mjs +++ b/docker-server.mjs @@ -112,6 +112,14 @@ const upstreamOrigin = upstreamBase ? new URL(upstreamBase).origin : null; // (gitnexus/src/mcp/http-transport.ts). const authToken = process.env.GITNEXUS_SERVE_AUTH_TOKEN?.trim() || null; +// The protocol-layer credential the upstream `serve` expects on /api/mcp when it +// runs with MCP Bearer auth enabled. Set it to the SAME value on both services: +// the edge token is spent here and replaced with this one for MCP requests only +// (see proxyToUpstream). Unset — the default — means no injection, so a backend +// without MCP auth is unaffected. Blank-is-absent follows resolveAuthToken +// (gitnexus/src/mcp/http-transport.ts). Never logged. +const mcpAuthToken = process.env.GITNEXUS_MCP_AUTH_TOKEN?.trim() || null; + // Mirrors the non-loopback refusal in http-transport.ts (startMcpHttpServer), // relocated because the trust boundary is here: an unguarded `serve` behind a // private service is legitimate, an unguarded public proxy is not. @@ -341,11 +349,17 @@ async function proxyToUpstream(req, res) { // talks to this same-origin web service. delete headers.origin; delete headers.referer; - // The edge token is spent here. `serve` reads no Authorization header - // (gitnexus/src/server/mcp-http.ts mounts /api/mcp unguarded), so forwarding - // it would only copy a live credential into another service's logs. Pinned by - // test. + // The edge token is spent here and must never be forwarded: copying + // Authorization would put a live credential into another service's logs. So + // drop it unconditionally first, then — for the MCP route alone, and only + // when a backend token is configured — replace it with that separate + // protocol credential. Unset GITNEXUS_MCP_AUTH_TOKEN (the default) leaves + // every request stripped, as before. The scope is the normalized pathname, + // so a query string can't widen it and /api/mcpfoo doesn't qualify. delete headers.authorization; + const upstreamPath = upstream.pathname; + const isMcpRoute = upstreamPath === '/api/mcp' || upstreamPath.startsWith('/api/mcp/'); + if (isMcpRoute && mcpAuthToken) headers.authorization = `Bearer ${mcpAuthToken}`; headers.host = upstream.host; // Replace, never forward, the inbound chain (see clientAddressFor). const clientAddress = clientAddressFor(req); diff --git a/docker-server.test.mjs b/docker-server.test.mjs index 80e742f7e..6d2a9f6c2 100644 --- a/docker-server.test.mjs +++ b/docker-server.test.mjs @@ -271,6 +271,12 @@ it('does not inject config into static assets', async () => { const TEST_AUTH_TOKEN = 'proxy-test-token-0123456789abcdefghij'; const TEST_BEARER = `Bearer ${TEST_AUTH_TOKEN}`; +// The protocol token the upstream expects on /api/mcp. Deliberately unlike the +// edge token, so "injected the backend credential" and "forwarded the edge one" +// can never both satisfy an assertion. +const TEST_MCP_TOKEN = 'backend-mcp-token-0123456789abcdefghij'; +const TEST_MCP_BEARER = `Bearer ${TEST_MCP_TOKEN}`; + // rawRequest never sends credentials; apiRequest does. In a file whose subject // is who gets let through, no test should pass because a helper quietly // authenticated for it. @@ -376,6 +382,11 @@ async function withProxy( const proc = spawnServerWithEnv(dir, port, { GITNEXUS_UPSTREAM_URL: schemeless ? target : `http://${target}`, GITNEXUS_SERVE_AUTH_TOKEN: TEST_AUTH_TOKEN, + // An ambient GITNEXUS_MCP_AUTH_TOKEN in the developer's shell would make the + // proxy inject one on /api/mcp, so drop it: spawn omits undefined entries, + // which unsets the inherited value. A test that wants injection sets it via + // `env` below. + GITNEXUS_MCP_AUTH_TOKEN: undefined, ...env, }); proc.stderr.setEncoding('utf8'); @@ -969,8 +980,9 @@ it('forwards an /api/* request that carries the correct token', async () => { }); it('strips the Authorization header instead of forwarding the edge token', async () => { - // The token is spent at this hop. `serve` reads no Authorization header, so - // forwarding would only copy a live credential into another service's logs. + // The edge credential is spent and stripped at this hop. Forwarding it + // would copy a live credential into another service's logs. With no + // GITNEXUS_MCP_AUTH_TOKEN configured — the default — nothing replaces it. await withProxy({}, async (port, ctx) => { const res = await apiRequest(port, '/api/mcp', { method: 'POST', body: '{}' }); assert.equal(res.status, 200, 'the request itself must still be proxied'); @@ -978,6 +990,72 @@ it('strips the Authorization header instead of forwarding the edge token', async }); }); +// -- Upstream MCP token injection (GITNEXUS_MCP_AUTH_TOKEN) ----------------- +// +// A backend running protocol-layer MCP auth expects its own Bearer on +// /api/mcp, and the edge credential can't serve as one. Both services are +// configured with the same GITNEXUS_MCP_AUTH_TOKEN; this hop spends the edge +// token and substitutes the backend one, for that route only. + +// Stands in for a `serve` with MCP Bearer auth enabled: only the exact backend +// credential gets through, so a passing two-hop request proves what was sent. +const mcpBackend = (req, res) => { + if (req.headers.authorization !== TEST_MCP_BEARER) { + res.writeHead(401, { 'Content-Type': 'application/json; charset=utf-8' }); + res.end('{"error":"unauthorized"}'); + return; + } + res.writeHead(200, { 'Content-Type': 'application/json; charset=utf-8' }); + res.end('{"ok":true}'); +}; + +it('treats a blank GITNEXUS_MCP_AUTH_TOKEN as unset and still strips', async () => { + const env = { GITNEXUS_MCP_AUTH_TOKEN: ' ' }; + await withProxy({ env }, async (port, ctx) => { + const res = await apiRequest(port, '/api/mcp', { method: 'POST', body: '{}' }); + assert.equal(res.status, 200); + assert.equal(ctx.received.headers.authorization, undefined); + }); +}); + +it('replaces the edge credential with the upstream MCP token on /api/mcp', async () => { + const env = { GITNEXUS_MCP_AUTH_TOKEN: TEST_MCP_TOKEN }; + await withProxy({ upstream: mcpBackend, env }, async (port, ctx) => { + const res = await apiRequest(port, '/api/mcp', { method: 'POST', body: '{}' }); + assert.equal(res.status, 200, 'a backend that demands the MCP token must accept this hop'); + assert.equal(ctx.received.headers.authorization, TEST_MCP_BEARER); + assert.notEqual( + ctx.received.headers.authorization, + TEST_BEARER, + 'the edge credential must never be forwarded', + ); + }); +}); + +it('injects the upstream MCP token on /api/mcp subpaths and ignores the query string', async () => { + const env = { GITNEXUS_MCP_AUTH_TOKEN: TEST_MCP_TOKEN }; + await withProxy({ upstream: mcpBackend, env }, async (port, ctx) => { + for (const path of ['/api/mcp/messages', '/api/mcp?session=abc']) { + const res = await apiRequest(port, path, { method: 'POST', body: '{}' }); + assert.equal(res.status, 200, `${path} must reach the MCP backend authenticated`); + assert.equal(ctx.received.headers.authorization, TEST_MCP_BEARER, path); + } + }); +}); + +it('leaves non-MCP routes stripped when an upstream MCP token is configured', async () => { + // /api/mcpfoo shares a prefix with the MCP route but is not it, and a plain + // API route never carries a protocol credential. + const env = { GITNEXUS_MCP_AUTH_TOKEN: TEST_MCP_TOKEN }; + await withProxy({ env }, async (port, ctx) => { + for (const path of ['/api/mcpfoo', '/api/health']) { + const res = await apiRequest(port, path); + assert.equal(res.status, 200); + assert.equal(ctx.received.headers.authorization, undefined, path); + } + }); +}); + it('never gates static assets behind the token', async () => { // The UI has to load before it can prompt for a token. await withProxy({}, async (port, ctx) => { diff --git a/gitnexus-claude-plugin/skills/gitnexus-cli/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-cli/SKILL.md index 1b1f733b6..170703439 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-cli/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-cli/SKILL.md @@ -27,9 +27,12 @@ Run from the project root. This parses all source files, builds the knowledge gr | `--embeddings` | Enable embedding generation for semantic search (off by default) | | `--drop-embeddings` | Drop existing embeddings on rebuild. By default, an `analyze` without `--embeddings` preserves them. | | `--pdg` | Build the program-dependence layers used by `explain` and `pdg_query` (taint, CDG, and REACHING_DEF). | +| `--spring-actuator ` | Import opt-in Spring Boot Actuator mappings, beans, conditions, configprops, and env snapshots. Forces a full rebuild; unsupported with `--watch`. | **When to run:** First time in a project, after major code changes, or when `gitnexus://repo/{name}/context` reports the index is stale. In Claude Code, a PostToolUse hook detects staleness after `git commit` and `git merge` and notifies the agent to run `analyze` — the hook does not run analyze itself, to avoid blocking the agent for up to 120s and risking KuzuDB corruption on timeout. +For Spring runtime enrichment, pass a JSON bundle, one endpoint JSON file, or a directory containing endpoint files. Route evidence is authoritative only when `runtimeConfirmed === true`; `runtimeSource` records provenance and may also accompany `handler-conflict`. Env/configprops values are never persisted. + Use `node .gitnexus/run.cjs analyze --watch` for a long-lived local Git repository. It performs an initial analysis, queues scanner-admitted file changes, and retries intact failed batches with bounded backoff. Watch refreshes update only the graph: they skip AGENTS.md / CLAUDE.md injection and standard skill installation, so run a one-shot `analyze` when those generated files need updating. Watch rejects one-shot or context-output flags including `--force`, embedding flags, `--skills`, `--default-branch`, `--skip-agents-md`, `--skip-skills`, `--no-stats`, `--self-commit`, `--index-only`, and `--skip-git`. It never pulls remotes. Running MCP and `serve` processes periodically check for a published replacement and reopen it without a restart. MCP checks are throttled to once every five seconds, so a tool call before the next check can briefly use the previous index. ### status — Check index freshness diff --git a/gitnexus-shared/src/graph/types.ts b/gitnexus-shared/src/graph/types.ts index d9d916e0d..8b46113df 100644 --- a/gitnexus-shared/src/graph/types.ts +++ b/gitnexus-shared/src/graph/types.ts @@ -45,6 +45,21 @@ export type NodeLabel = | 'Section' | 'Route' | 'Tool' + /** + * A message-broker destination — a Kafka topic, a Rabbit exchange/routing + * key, a JMS queue, a Spring Cloud Stream binding. The framework overlay for + * ASYNCHRONOUS entry/exit points, symmetric to `Route` for HTTP. + * + * Identity is `(broker, resolved ADDRESS)`, so a publisher and a consumer of + * the same address on the same broker land on one node and the connection is + * a single hop — while a Kafka topic and a Rabbit queue that share a name + * stay two nodes, the same way `GET /x` and `POST /x` are two Routes. A + * destination whose address could NOT be resolved is keyed by its source + * location instead and carries no `address` property at all. See + * `pipeline-phases/spring-destinations.ts` for why an unresolved spelling may + * not key a node, and `ingestion/destination-key.ts` for why the broker may. + */ + | 'Destination' // Taint/PDG substrate (issue #2080). Intra-procedural control-flow node. // Emitted by no phase yet — M1 (#2081) populates these behind an opt-in. | 'BasicBlock'; @@ -95,6 +110,30 @@ export type NodeProperties = { responseKeys?: string[]; errorKeys?: string[]; middleware?: string[]; + /** Route runtime evidence is authoritative only when this is exactly true. */ + runtimeConfirmed?: boolean; + /** Provenance of runtime evidence; presence alone does not imply confirmation. */ + runtimeSource?: string; + /** Runtime result such as runtime-confirmed or handler-conflict. */ + runtimeStatus?: string; + // Destination (async messaging overlay). See the `Destination` label above. + /** The RESOLVED broker address. Together with `broker` it is the key a + * cross-repository pass joins on. Present only when the address resolved: + * absent is the load-bearing state, because an absent property cannot match + * another absent property. */ + address?: string; + /** Broker family the syntax attests to (`kafka`, `rabbit`, `jms`, …). Part + * of the node's identity alongside `address`, not a label on it. */ + broker?: string; + /** How the address was arrived at (`literal`, `constant`) when it resolved, + * or the named reason it did not. */ + resolution?: string; + /** Configuration key named by an unresolvable `${…}` placeholder. The key + * only — configuration VALUES are deliberately absent from this graph. */ + configKey?: string; + /** The `${key:default}` default text. Not an address: configuration can + * override it and the graph cannot see whether it did. */ + configDefault?: string; // BasicBlock (taint/PDG substrate, issue #2080) — reuses filePath/startLine/endLine. text?: string; /** BasicBlock: space-joined leaf callee names invoked in the block — the @@ -122,6 +161,19 @@ export type RelationshipType = | 'MEMBER_OF' | 'STEP_IN_PROCESS' | 'HANDLES_ROUTE' + /** Outbound async messaging. Source = the callable that performs the publish + * (or its File); target = the `Destination` it publishes to. Emitted by + * `pipeline-phases/spring-destinations.ts` from Spring messaging-template + * calls (`kafkaTemplate.send(...)`, `rabbitTemplate.convertAndSend(...)`). + * One edge per address: a publish that names two destinations yields two + * edges, and `reason` records which argument each came from. */ + | 'PUBLISHES_TO' + /** Inbound async messaging — the mirror of `PUBLISHES_TO`. Source = the + * annotated handler callable (or its File); target = the `Destination` it + * subscribes to. Emitted from `@KafkaListener` / `@RabbitListener` / + * `@JmsListener` and their siblings. Together the two types make + * "who else reads what this service writes" a two-hop traversal. */ + | 'CONSUMES_FROM' | 'FETCHES' | 'HANDLES_TOOL' | 'ENTRY_POINT_OF' diff --git a/gitnexus-shared/src/lbug/schema-constants.ts b/gitnexus-shared/src/lbug/schema-constants.ts index 350aa273d..217c382a6 100644 --- a/gitnexus-shared/src/lbug/schema-constants.ts +++ b/gitnexus-shared/src/lbug/schema-constants.ts @@ -40,6 +40,8 @@ export const NODE_TABLES = [ 'Module', 'Route', 'Tool', + // Async messaging overlay — the broker-side counterpart of `Route`. + 'Destination', // Taint/PDG substrate (issue #2080) — inert until M1 (#2081) emits blocks. 'BasicBlock', ] as const; @@ -64,6 +66,8 @@ export const REL_TYPES = [ 'MEMBER_OF', 'STEP_IN_PROCESS', 'HANDLES_ROUTE', + 'PUBLISHES_TO', + 'CONSUMES_FROM', 'FETCHES', 'HANDLES_TOOL', 'ENTRY_POINT_OF', diff --git a/gitnexus-web/src/core/llm/tools.ts b/gitnexus-web/src/core/llm/tools.ts index 407018678..421725be3 100644 --- a/gitnexus-web/src/core/llm/tools.ts +++ b/gitnexus-web/src/core/llm/tools.ts @@ -14,7 +14,11 @@ import { tool } from '@langchain/core/tools'; import { z } from 'zod'; import { NODE_TABLES, REL_TYPES, scoreImpactRisk, unusedAxesForImpactWalk } from 'gitnexus-shared'; -import type { EnrichedSearchResult, GrepResult } from '../../services/backend-client'; +import type { + EnrichedSearchResult, + GrepOptions, + GrepResponse, +} from '../../services/backend-client'; /** * Tool names registered by createGraphRAGTools — kept in sync with each tool's `name` @@ -44,7 +48,7 @@ export interface GraphRAGBackend { query: string, opts?: { limit?: number; mode?: 'hybrid' | 'semantic' | 'bm25'; enrich?: boolean }, ) => Promise; - grep: (pattern: string, limit?: number) => Promise; + grep: (pattern: string, limit?: number, opts?: GrepOptions) => Promise; readFile: (filePath: string) => Promise; } @@ -375,20 +379,22 @@ MATCH (n:Function {id: emb.nodeId}) RETURN n`, } const limit = maxResults ?? 100; - const fullPattern = fileFilter - ? `(?=.*${fileFilter.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}).*${pattern}` - : pattern; - - const results = await backendGrep(fullPattern, limit); + const { results, timedOut } = await backendGrep(pattern, limit, { + fileFilter, + caseSensitive, + }); + const timeoutMsg = timedOut + ? '\n\n(Scan timed out after a few seconds — results may be incomplete)' + : ''; if (results.length === 0) { - return `No matches for "${pattern}"${fileFilter ? ` in files matching "${fileFilter}"` : ''}`; + return `No matches for "${pattern}"${fileFilter ? ` in files matching "${fileFilter}"` : ''}${timeoutMsg}`; } const formatted = results.map((r) => `${r.filePath}:${r.line}: ${r.text}`).join('\n'); const truncatedMsg = results.length >= limit ? `\n\n(Showing first ${limit} results)` : ''; - return `Found ${results.length} matches:\n\n${formatted}${truncatedMsg}`; + return `Found ${results.length} matches:\n\n${formatted}${truncatedMsg}${timeoutMsg}`; } catch (error) { return `Grep error: ${error instanceof Error ? error.message : String(error)}`; } @@ -396,16 +402,20 @@ MATCH (n:Function {id: emb.nodeId}) RETURN n`, { name: 'grep', description: - 'Search for exact text patterns across all files using regex. Use for finding specific strings, error messages, TODOs, variable names, etc.', + 'Search file contents with a regular expression (server executes it as a real regex — alternation like "sign|Sign" works). Matches are case-insensitive unless caseSensitive is set. fileFilter keeps only files whose path contains the substring. Each call caps at maxResults matches (default 100) and the server stops after a few seconds (the tool will say so if the scan was incomplete), so prefer precise patterns over catch-alls.', schema: z.object({ pattern: z .string() - .describe('Regex pattern to search for (e.g., "TODO", "console\\.log", "API_KEY")'), + .describe( + 'Regex pattern to search for (e.g., "TODO|FIXME", "console\\.log", "signOrder")', + ), fileFilter: z .string() .optional() .nullable() - .describe('Only search files containing this string (e.g., ".ts", "src/api")'), + .describe( + 'Only search files whose path contains this substring (e.g., ".ts", "src/api", "Controller.java")', + ), caseSensitive: z .boolean() .optional() @@ -1219,7 +1229,7 @@ MATCH (n:Function {id: emb.nodeId}) RETURN n`, const targetFileName = (targetFilePath || target).split('/').pop() || target; const baseName = targetFileName.replace(/\.[^/.]+$/, ''); try { - const hints = await backendGrep(`\\b${escapeRegex(baseName)}\\b`, 15); + const { results: hints } = await backendGrep(`\\b${escapeRegex(baseName)}\\b`, 15); const filtered = hints.filter((h) => h.filePath !== targetFilePath); if (filtered.length > 0) { diff --git a/gitnexus-web/src/hooks/useAppState.tsx b/gitnexus-web/src/hooks/useAppState.tsx index d698b87d5..deeda86bf 100644 --- a/gitnexus-web/src/hooks/useAppState.tsx +++ b/gitnexus-web/src/hooks/useAppState.tsx @@ -40,6 +40,7 @@ import { repoIdentity as repoIdentityOf, type BackendRepo, type ConnectResult, + type GrepOptions, type JobProgress, } from '../services/backend-client'; import { ERROR_RESET_DELAY_MS } from '../config/ui-constants'; @@ -671,7 +672,8 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => { const backend = { executeQuery, search: (query: string, opts?: any) => backendSearch(query, { ...opts, repo }), - grep: (pattern: string, limit?: number) => backendGrep(pattern, repo, limit), + grep: (pattern: string, limit?: number, opts?: GrepOptions) => + backendGrep(pattern, repo, limit, opts), readFile: (filePath: string) => backendReadFile(filePath, { repo }).then((r) => r.content), }; diff --git a/gitnexus-web/src/lib/constants.ts b/gitnexus-web/src/lib/constants.ts index 2f717cab3..5a85754cf 100644 --- a/gitnexus-web/src/lib/constants.ts +++ b/gitnexus-web/src/lib/constants.ts @@ -37,6 +37,7 @@ export const NODE_COLORS: Record = { Constructor: '#10b981', // Emerald - like Function Template: '#a78bfa', // Violet light - like Type Route: '#f43f5e', // Rose - like Process + Destination: '#fb7185', // Rose light - like Route, the broker-side counterpart Tool: '#a855f7', // Purple - like Project BasicBlock: '#475569', // Slate darker - control-flow node (muted, taint/PDG substrate) }; @@ -79,6 +80,7 @@ export const NODE_SIZES: Record = { Constructor: 4, // Like Function Template: 3, // Like Type Route: 5, // Like Enum + Destination: 5, // Like Route - the broker-side counterpart Tool: 5, // Like Enum BasicBlock: 2, // Tiny - control-flow node (taint/PDG substrate) }; diff --git a/gitnexus-web/src/services/backend-client.ts b/gitnexus-web/src/services/backend-client.ts index 52f8c65c9..706b6d90e 100644 --- a/gitnexus-web/src/services/backend-client.ts +++ b/gitnexus-web/src/services/backend-client.ts @@ -64,6 +64,12 @@ export interface GrepResult { text: string; } +/** Full `/api/grep` payload — `timedOut` is true when the 5s budget cut the scan short. */ +export interface GrepResponse { + results: GrepResult[]; + timedOut: boolean; +} + export interface JobProgress { phase: string; percent: number; @@ -869,23 +875,37 @@ export const search = async ( return (body.results ?? []) as EnrichedSearchResult[]; }; -/** Grep across file contents in the indexed repo. */ +/** Options for {@link grep} beyond pattern/repo/limit. */ +export interface GrepOptions { + /** Only search files whose path contains this substring (case-insensitive). */ + fileFilter?: string | null; + /** Case-sensitive matching (default: insensitive). */ + caseSensitive?: boolean; +} + +/** Grep across file contents in the indexed repo. Regex semantics server-side. */ export const grep = async ( pattern: string, repo?: string, limit?: number, -): Promise => { + opts?: GrepOptions, +): Promise => { const params = [ `pattern=${encodeURIComponent(pattern)}`, repoParam(repo), limit ? `limit=${limit}` : '', + opts?.fileFilter ? `fileFilter=${encodeURIComponent(opts.fileFilter)}` : '', + opts?.caseSensitive ? 'caseSensitive=1' : '', ] .filter(Boolean) .join('&'); const response = await fetchWithTimeout(`${_backendUrl}/api/grep?${params}`); await assertOk(response); - const body = await response.json(); - return (body.results ?? []) as GrepResult[]; + const body = (await response.json()) as Partial; + return { + results: body.results ?? [], + timedOut: body.timedOut === true, + }; }; /** Result from reading a file, optionally with line range. */ diff --git a/gitnexus-web/test/unit/agent-prompt.test.ts b/gitnexus-web/test/unit/agent-prompt.test.ts index c2cb1392f..8bf69a513 100644 --- a/gitnexus-web/test/unit/agent-prompt.test.ts +++ b/gitnexus-web/test/unit/agent-prompt.test.ts @@ -43,7 +43,7 @@ const FORBIDDEN_TOOL_NAMES = [ const stubBackend: GraphRAGBackend = { executeQuery: async () => [], search: async () => [], - grep: async () => [], + grep: async () => ({ results: [], timedOut: false }), readFile: async () => '', }; diff --git a/gitnexus-web/test/unit/backend-client-grep.test.ts b/gitnexus-web/test/unit/backend-client-grep.test.ts new file mode 100644 index 000000000..183068d5e --- /dev/null +++ b/gitnexus-web/test/unit/backend-client-grep.test.ts @@ -0,0 +1,75 @@ +/** + * `/api/grep` client: query params and `timedOut` must reach callers. + * Dropping `timedOut` made a 5s partial scan look like a complete miss. + */ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; +import { __resetBreakerRegistry__ } from 'gitnexus-shared/test-helpers'; +import { grep, setBackendUrl } from '../../src/services/backend-client'; + +const BASE = 'http://grep-client.test:4747'; + +const jsonOk = (body: unknown) => + new Response(JSON.stringify(body), { + status: 200, + headers: { 'Content-Type': 'application/json' }, + }); + +describe('backend-client grep', () => { + beforeEach(() => { + __resetBreakerRegistry__(); + setBackendUrl(BASE); + }); + + afterEach(() => { + vi.unstubAllGlobals(); + }); + + it('forwards fileFilter and caseSensitive and returns timedOut', async () => { + const fetchMock = vi.fn(async (input: RequestInfo | URL) => { + const url = String(input); + expect(url).toContain('/api/grep?'); + expect(url).toContain(`pattern=${encodeURIComponent('sign|Sign')}`); + expect(url).toContain(`fileFilter=${encodeURIComponent('src/api')}`); + expect(url).toContain('caseSensitive=1'); + expect(url).toContain('limit=12'); + return jsonOk({ + results: [{ filePath: 'src/api.ts', line: 3, text: 'signOrder()' }], + timedOut: true, + }); + }); + vi.stubGlobal('fetch', fetchMock); + + const body = await grep('sign|Sign', '/repo', 12, { + fileFilter: 'src/api', + caseSensitive: true, + }); + expect(body.results).toEqual([{ filePath: 'src/api.ts', line: 3, text: 'signOrder()' }]); + expect(body.timedOut).toBe(true); + }); + + it('reports timedOut false when the server completed the scan', async () => { + vi.stubGlobal( + 'fetch', + vi.fn(async () => { + return jsonOk({ results: [] }); + }), + ); + + const body = await grep('TODO'); + expect(body).toEqual({ results: [], timedOut: false }); + }); + + it('does not send fileFilter when it is null or empty', async () => { + const fetchMock = vi.fn(async (input: RequestInfo | URL) => { + const url = String(input); + expect(url).not.toContain('fileFilter='); + return jsonOk({ results: [] }); + }); + vi.stubGlobal('fetch', fetchMock); + + for (const fileFilter of ['', null] as const) { + await grep('x', undefined, undefined, { fileFilter }); + } + expect(fetchMock).toHaveBeenCalledTimes(2); + }); +}); diff --git a/gitnexus-web/test/unit/grep-tool.test.ts b/gitnexus-web/test/unit/grep-tool.test.ts new file mode 100644 index 000000000..8701ca665 --- /dev/null +++ b/gitnexus-web/test/unit/grep-tool.test.ts @@ -0,0 +1,36 @@ +import { describe, expect, it, vi } from 'vitest'; +import { createGraphRAGTools, type GraphRAGBackend } from '../../src/core/llm/tools'; + +const noOpBackend: GraphRAGBackend = { + executeQuery: async () => [], + search: async () => [], + grep: async () => ({ results: [], timedOut: false }), + readFile: async () => '', +}; + +function grepTool(backend: GraphRAGBackend) { + return createGraphRAGTools(backend).find((candidate) => candidate.name === 'grep')!; +} + +describe('grep tool timeout contract', () => { + it('says the scan was incomplete when the server sets timedOut with no hits', async () => { + const grep = vi.fn(async () => ({ results: [], timedOut: true })); + const output = await grepTool({ ...noOpBackend, grep }).invoke({ pattern: 'signOrder' }); + expect(output).toContain('No matches for "signOrder"'); + expect(output).toContain('results may be incomplete'); + }); + + it('still warns when a timed-out scan returned some hits below the limit', async () => { + const grep = vi.fn(async () => ({ + results: [{ filePath: 'a.ts', line: 1, text: 'signOrder()' }], + timedOut: true, + })); + const output = await grepTool({ ...noOpBackend, grep }).invoke({ + pattern: 'signOrder', + maxResults: 100, + }); + expect(output).toContain('Found 1 matches'); + expect(output).toContain('results may be incomplete'); + expect(output).not.toContain('Showing first'); + }); +}); diff --git a/gitnexus-web/test/unit/impact-tool.test.ts b/gitnexus-web/test/unit/impact-tool.test.ts index f04817ed1..704240e6e 100644 --- a/gitnexus-web/test/unit/impact-tool.test.ts +++ b/gitnexus-web/test/unit/impact-tool.test.ts @@ -4,7 +4,7 @@ import { createGraphRAGTools, type GraphRAGBackend } from '../../src/core/llm/to const noOpBackend: GraphRAGBackend = { executeQuery: async () => [], search: async () => [], - grep: async () => [], + grep: async () => ({ results: [], timedOut: false }), readFile: async () => '', }; diff --git a/gitnexus/README.md b/gitnexus/README.md index 20ea65ef5..12bd96d20 100644 --- a/gitnexus/README.md +++ b/gitnexus/README.md @@ -240,10 +240,11 @@ gitnexus analyze --force # Full rebuild: re-parse + graph rebuild + FTS gitnexus analyze --embeddings # Enable embedding generation (slower, better search) gitnexus embeddings install # Fetch the optional local embedding stack on demand (--cuda, --force) gitnexus analyze --skills # Generate repo-specific skill files from detected communities -gitnexus analyze --skip-agents-md # Preserve custom AGENTS.md/CLAUDE.md gitnexus section edits +gitnexus analyze --skip-agents-md # Preserve custom AGENTS.md/CLAUDE.md gitnexus section edits (does not skip standard skills; use --skip-skills; community --skills files are unaffected) gitnexus analyze --skip-skills # Skip installing standard .claude/skills/gitnexus-* skill files gitnexus analyze --skip-git # Index folders that are not Git repositories gitnexus analyze --workers # Parse worker pool size (>=1; default: cores-1, capped at 16) +gitnexus analyze --spring-actuator ./actuator # Enrich with local Spring Boot Actuator JSON snapshots gitnexus analyze --verbose # Log skipped files when parsers are unavailable gitnexus analyze --max-file-size 1024 # Skip files larger than N KB (default: 512, cap: 32768) gitnexus analyze --worker-timeout 60 # Increase worker idle timeout for slow parses @@ -323,6 +324,8 @@ and root fields. Dynamic decorator names, anonymous operations, and ambiguous or anchors are deliberately omitted. Add common infrastructure fields such as `/health` to `matching.exclude_links_paths` to keep those GraphQL contracts visible without cross-linking them. +`--spring-actuator` is explicitly opt-in. The path may be a JSON bundle keyed by `mappings`, `beans`, `conditions`, `configprops`, and/or `env`, or a directory containing endpoint-named JSON files. Runtime mappings and beans confirm matching static nodes; conditions and configuration property keys enrich existing evidence, with conservative runtime-only nodes added when no match exists. The configured input is excluded from source scanning; only normalized repository-relative exclusions are retained for future scans, never absolute paths. Env/configprops values, origins, condition messages, and source names are never persisted or printed. Enabled runs always rebuild because runtime snapshots are external to git freshness; omitting the option later rebuilds once to remove runtime evidence. Project config can set the same path with `springActuator` in `.gitnexusrc`. + > **`gitnexus uninstall`** reverses `gitnexus setup` — it removes the GitNexus MCP entries, hooks, and skill directories it added to each detected editor. Skill directories are identified **by bundled gitnexus skill name** (e.g. `gitnexus-cli/`), so if you customized files inside an installed skill directory, back them up first. It is a dry-run preview by default and prints the exact paths it would remove; pass `--force` to apply. Per-repo indexes (`gitnexus clean --all`) and the global npm package (`npm uninstall -g gitnexus`) are left for you to remove. ## Remote Embeddings @@ -431,7 +434,7 @@ Installed automatically by both `gitnexus analyze` (per-repo) and `gitnexus setu LadybugDB native binary ships as a prebuild against that floor, so on an older host it cannot load and reinstalling does not help — see [Linux: `GLIBC_2.34' not found`](#linux-glibc_234-not-found). -- **Windows, for full-text search:** the Microsoft Visual C++ 2015-2022 Redistributable (x64) *and* +- **Windows, for full-text search:** the Microsoft Visual C++ 2015-2022 Redistributable (x64) _and_ OpenSSL 3 (`libssl-3-x64.dll`, `libcrypto-3-x64.dll`) resolvable on `PATH` — see [Windows: full-text search unavailable](#windows-full-text-search-unavailable). @@ -708,17 +711,17 @@ For repositories with very large source files, `GITNEXUS_WORKER_SUB_BATCH_MAX_BY Four env vars expose the pool's resilience layers (respawn budget, cumulative-timeout cap, circuit breaker, startup handshake). Defaults are tuned for typical repos; bump them when an analyze legitimately needs more retries, or lower them to fail-fast on a known-bad shape. -| Variable | Default | Effect | -| ----------------------------------------------- | ----------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT` | `3` | Max replacement spawns per slot before the slot is dropped from the active rotation. | -| `GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS` | `5 × subBatchTimeoutMs` | Total retry wall-time budget per job before quarantining. Bounds exponentially-growing retry waits. | -| `GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD` | `max(3, poolSize)` | Per-slot consecutive deaths before the pool's circuit breaker trips. After tripping, dispatches require a fresh pool. | -| `GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS` | `30000` | Max wait at pool shutdown for a retired worker still inside native code — terminated at its next JS-safe point instead of mid-native-call, which would abort the process (`Napi::Error`, #2432). | -| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. Raise it on a slow or heavily loaded host where a full pool cold-starting concurrently needs more than 5s. | -| `GITNEXUS_MEMORY` | `off` | unset (autopilot on) | `off` declines GitNexus's memory autopilot: analyze will neither re-run itself with a RAM-aware heap cap nor abort the parse before V8 enters its ineffective-mark-compact death spiral. Use it when you want to drive memory manually; to simply pin a heap size, pass Node's own `--max-old-space-size`, which is already honoured as your decision. | -| `GITNEXUS_WORKER_HEAP_MB` | `clamp(512, RAM/2/poolSize, 4096)` | Per-worker V8 old-generation heap cap (#2649). Bounds pool RSS on large repos; a worker exceeding it dies with a real heap error handled by quarantine/respawn. | -| `GITNEXUS_SERVER_ANALYZE_HEAP_MB` | `min(8192, auto cap)` | Heap for the web/MCP server's forked analyze worker (#2649). Defaults to the historical 8192 MB bounded by the machine/container's RAM-aware auto cap; set an absolute MB value to override. | -| `GITNEXUS_CPP_CAPTURE_BUDGET_MS` | `20000` | Per-file wall-clock budget for C++ capture extraction; on breach the file keeps partial captures with a warning (#2432). `0` expires immediately. | +| Variable | Default | Effect | +| ----------------------------------------------- | ---------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| `GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT` | `3` | Max replacement spawns per slot before the slot is dropped from the active rotation. | +| `GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS` | `5 × subBatchTimeoutMs` | Total retry wall-time budget per job before quarantining. Bounds exponentially-growing retry waits. | +| `GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD` | `max(3, poolSize)` | Per-slot consecutive deaths before the pool's circuit breaker trips. After tripping, dispatches require a fresh pool. | +| `GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS` | `30000` | Max wait at pool shutdown for a retired worker still inside native code — terminated at its next JS-safe point instead of mid-native-call, which would abort the process (`Napi::Error`, #2432). | +| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. Raise it on a slow or heavily loaded host where a full pool cold-starting concurrently needs more than 5s. | +| `GITNEXUS_MEMORY` | `off` | unset (autopilot on) | `off` declines GitNexus's memory autopilot: analyze will neither re-run itself with a RAM-aware heap cap nor abort the parse before V8 enters its ineffective-mark-compact death spiral. Use it when you want to drive memory manually; to simply pin a heap size, pass Node's own `--max-old-space-size`, which is already honoured as your decision. | +| `GITNEXUS_WORKER_HEAP_MB` | `clamp(512, RAM/2/poolSize, 4096)` | Per-worker V8 old-generation heap cap (#2649). Bounds pool RSS on large repos; a worker exceeding it dies with a real heap error handled by quarantine/respawn. | +| `GITNEXUS_SERVER_ANALYZE_HEAP_MB` | `min(8192, auto cap)` | Heap for the web/MCP server's forked analyze worker (#2649). Defaults to the historical 8192 MB bounded by the machine/container's RAM-aware auto cap; set an absolute MB value to override. | +| `GITNEXUS_CPP_CAPTURE_BUDGET_MS` | `20000` | Per-file wall-clock budget for C++ capture extraction; on breach the file keeps partial captures with a warning (#2432). `0` expires immediately. | ### Graph cleanup tuning @@ -732,8 +735,8 @@ Programmatic callers can pass `keepLocalValueSymbols: true` in `PipelineOptions` ### Scope-resolution property-key dispatch cap -During scope resolution GitNexus synthesizes CALLS edges through *property-key -dispatch* — call sites like `hooks.emitScopeCaptures()` where a property key is +During scope resolution GitNexus synthesizes CALLS edges through _property-key +dispatch_ — call sites like `hooks.emitScopeCaptures()` where a property key is registered by multiple definitions across the codebase. To keep this fan-in bounded, each property key is capped at **32 registrations**: a key registered by more than 32 distinct functions is skipped entirely (no CALLS are synthesized @@ -741,8 +744,8 @@ through it), and the dropped key names are surfaced in the analyze log for operator visibility. The cap is calibrated at 2× this repo's own provider table (16 legitimate registrations, one per language provider). -| Variable | Default | Effect | -| --------------------------------------- | ------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Variable | Default | Effect | +| --------------------------------------- | ------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | | `GITNEXUS_MAX_PROPERTY_DISPATCH_FANOUT` | `32` | Per-property-key registration cap in the property-dispatch scope-resolution pass. Set to a positive integer to raise it for repositories whose provider/hook tables exceed the default and lose CALLS coverage on a legitimate key; non-integer or `< 1` values fall back to `32`. Lowering it tightens the overflow budget. | ```bash @@ -755,11 +758,11 @@ npx gitnexus analyze --force ### Scope-resolution dispatch-target cap -During scope resolution GitNexus resolves calls that flow through *callable -values* — function/method references bound to variables, passed as arguments, +During scope resolution GitNexus resolves calls that flow through _callable +values_ — function/method references bound to variables, passed as arguments, or stored in maps/tables. To keep that inclusion-based resolution finite, each callable site is capped at **32 dispatch targets**. When a site gathers more -candidates than the cap it is treated as **overflowed** and *all* of its call +candidates than the cap it is treated as **overflowed** and _all_ of its call edges are dropped — a cliff, not a tail, so a repository with a legitimately wide dispatch table (a single callable site resolving to 33+ targets) loses that site's whole call chain. In that case `analyze` logs @@ -769,8 +772,8 @@ candidate count, and the cap (32). Raise the cap for such repositories: -| Variable | Default | Effect | -| ------------------------------------- | ------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | +| Variable | Default | Effect | +| ------------------------------------- | ------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | | `GITNEXUS_MAX_CALLABLE_VALUE_TARGETS` | `32` | Per-callable-site dispatch-target cap in the callable-value-flow scope-resolution pass. Set to a positive integer to raise it for repositories whose wide dispatch tables overflow the default and lose a whole call chain; non-integer or `< 1` values fall back to `32`. Lowering it tightens the overflow budget. | ```bash diff --git a/gitnexus/bench/emit-persistence/baselines.json b/gitnexus/bench/emit-persistence/baselines.json index 5eeceb02b..5b3568256 100644 --- a/gitnexus/bench/emit-persistence/baselines.json +++ b/gitnexus/bench/emit-persistence/baselines.json @@ -1,8 +1,11 @@ { - "fingerprint": "c4d799c5336d616955b3530ba051b7dca300d1a0e412a66741cf2f27e04c533e", + "fingerprint": "72096279092d4f118de7e179333705c19c9aff2664f77d71a7da48cd9f73fb5a", "scaling_budget": 1.8, "max_ms_large": 1000, + "_rebaselined_destination_broker_conflict_column_removed": "The `Destination` node table lost its trailing `brokerConflict` STRING column (DESTINATION_SCHEMA in src/core/lbug/schema.ts, the COPY statement in lbug-adapter.ts, and the `destinationWriter` header plus its row cell in csv-generator.ts). The column existed only to say WHY a destination's address had been withdrawn when two brokers claimed it; the address is no longer withdrawn — a resolved `Destination` is now keyed by `(broker, address)` via `ingestion/destination-key.ts`, so two brokers on one name are two ordinary joinable nodes and there is nothing to diagnose. That makes this a header-only shrink, and it was verified as one rather than assumed: dumping every CSV this bench emits with csv-generator.ts at the merge base and again on this branch, then diffing per file by filename, byte length and sha256, shows the file SET identical at 36 CSVs on both sides, 35 of the 36 byte-IDENTICAL (same sha256, not merely same length), and the sole difference `destination.csv` shrinking 112 -> 97 bytes: `id,name,filePath,startLine,endLine,address,broker,resolution,configKey,configDefault,brokerConflict,description` -> the same list without `brokerConflict`. That file is header-only on both sides — the synthetic benchmark graph contains no Destination nodes — so no row moved, was re-routed to another pair file, or reordered, which is the class of change this fingerprint exists to catch. Prior 4b339233662b0eebb236738abad8f7c039b9930bc1222dcad7b814afa6332fbf -> 72096279092d4f118de7e179333705c19c9aff2664f77d71a7da48cd9f73fb5a, reproduced identically across two consecutive runs. Both timing gates passed while the guard was red (scaling_ratio 0.818 and 0.751 across those two runs against the 1.8 budget; elapsed_ms_large 65.64ms and 59.32ms against the 1000ms backstop), so no throughput claim is being rebaselined away.", + "_rebaselined_3132_destination_node_table": "A new `Destination` node table (async messaging overlay; see DESTINATION_SCHEMA in src/core/lbug/schema.ts) means `streamAllCSVsToDisk` writes one more FILE, not one more column — the first rebaseline here that changes the file SET rather than a header. That makes the usual evidence more important, not less, so it was gathered the same way: dump every CSV this bench emits on the merge base and on this branch, then diff per-file by filename, byte length and sha256. Result: 35 files -> 36, the sole addition is `destination.csv`, and ALL 35 pre-existing files are byte-IDENTICAL — not merely same-length, same sha256. So nothing was re-routed into the new file and nothing reordered, which is exactly what this fingerprint exists to catch. `destination.csv` is 112 bytes, header only: the synthetic benchmark graph has no Destination nodes, so no row exists to move. Prior 7b2ec01a110dcbc66868fba2c97714aaece8c3864c2eba00df68b5c027f034d2 -> 4b339233662b0eebb236738abad8f7c039b9930bc1222dcad7b814afa6332fbf, reproduced identically across two runs. Both timing gates passed while this was red (scaling_ratio 0.711 vs the 1.8 budget, elapsed_ms_large 58.88ms vs the 1000ms backstop), so no throughput claim is being rebaselined away.", "_rebaselined_2856_property_is_detail": "Third and last of the bench guards this branch left red. The Property node table gained an `isDetail` BOOLEAN column (see PROPERTY_SCHEMA in src/core/lbug/schema.ts), so `streamAllCSVsToDisk` writes one more header field and one more cell per Property row — csv-generator.ts `propertyHeader` and the `node.label === 'Property'` tail. Verified to be header-only drift rather than a change in what is emitted: dumping every CSV this bench produces on `origin/main` and on this branch and diffing per-file (filename, byte length, sha256) shows the file SET is identical at 35 CSVs on both sides, 34 of the 35 are byte-identical, and the sole difference is `property.csv` growing 68 -> 77 bytes, `id,name,filePath,startLine,endLine,content,description,declaredType` -> `...,declaredType,isDetail`. The synthetic graph has no Property nodes, so no ROW moved at all. That is the check that matters here: a row routed to the wrong pair file, or a within-file reordering, is what this fingerprint exists to catch, and neither happened. Prior 69e9182ae205183ade24c3d8ad5d7292aea677144b1cbe443dd631bc25b0cafe -> 4ee15e742a9839671a900df4f57c1c91196c64256c8cab2ac445bec605a092d5. Both timing gates passed unchanged while this was red (scaling_ratio 0.783 vs budget 1.8, elapsed_ms_large 229ms vs the 1000ms backstop), so no throughput claim is being rebaselined away.", "_rebaselined_3040_convex_endpoint_factory": "Const and Function gained a trailing convexEndpointFactory column. A deterministic 2,400-entity emit produced the same 35 CSV files and fingerprint c4d799c5336d616955b3530ba051b7dca300d1a0e412a66741cf2f27e04c533e. Removing the new Const and Function header fields plus the new trailing empty Function cell from each of 4,800 Function rows restored the exact prior fingerprint 4ee15e742a9839671a900df4f57c1c91196c64256c8cab2ac445bec605a092d5. No file or row moved or reordered. The measured scaling ratio remained 0.826 against the 1.8 budget and elapsed_ms_large was 307.75ms against the 1000ms backstop.", + "_rebaselined_3107_route_runtime_evidence": "Route gained trailing runtimeConfirmed BOOLEAN, runtimeSource STRING, and runtimeStatus STRING columns in its schema, CSV header/rows, COPY statement, and graph API projection. The deterministic emit still produces the same 35 CSV files; the synthetic benchmark graph has no Route rows, so the only byte drift is the Route CSV header and no row moved or reordered. Prior c4d799c5336d616955b3530ba051b7dca300d1a0e412a66741cf2f27e04c533e -> 7b2ec01a110dcbc66868fba2c97714aaece8c3864c2eba00df68b5c027f034d2. While the guard was red, scaling_ratio was 1.044 against the 1.8 budget and elapsed_ms_large was 121.08ms against the 1000ms backstop.", "_note": "fingerprint = sha256 over per-file digests (filename + sha256(file bytes)), entry list sorted — binds each emitted line to its file so a row routed to the WRONG pair file changes the hash, AND catches within-file row reordering (file bytes hashed as-written). Byte-identity gate for #2203 U2/U3. NOTE: a future change that legitimately reorders emit (without changing the node/edge SET) will trip --check; regenerate then, and record WHY in a `_rebaselined_` key alongside — bench/scope-capture/baselines.json sets that convention and it is what makes a regenerated hash reviewable. scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). max_ms_large=1000ms is a coarse absolute backstop (observed ~200ms) that catches a gross uniform slowdown the ratio gate misses; generous so CI host noise won't flake it. Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`." } diff --git a/gitnexus/bench/java-lombok-synthesis/baselines.json b/gitnexus/bench/java-lombok-synthesis/baselines.json new file mode 100644 index 000000000..9e14f8f3c --- /dev/null +++ b/gitnexus/bench/java-lombok-synthesis/baselines.json @@ -0,0 +1,8 @@ +{ + "_comment": "Baselines for bench/java-lombok-synthesis/measure.mjs --check (#2885). fingerprint is sha256 over synthetic Method node ids on the lombok_large corpus (800 @Data entities × 4 fields × 2 accessors = 6400 methods). no_lombok arm must emit 0 methods. Budgets are timing gates with CI headroom.", + "fingerprint": "b935d6894d32de7594d5887bb62af6ade2b66b19d6846700a05ef3baf1ed1eb1", + "scaling_budget": 1.6, + "_scaling_note": "(t_large/t_small)/(800/250) on the lombok arm. Measured ~1.01.", + "widening_overhead_budget": 2.5, + "_widening_overhead_note": "lombok_large_ms / no_lombok_large_ms using an unannotated, shape-equivalent four-field control. Measured about 1.24; budget guards against a pathological feature-arm regression." +} diff --git a/gitnexus/bench/java-lombok-synthesis/measure.mjs b/gitnexus/bench/java-lombok-synthesis/measure.mjs new file mode 100644 index 000000000..512af1e12 --- /dev/null +++ b/gitnexus/bench/java-lombok-synthesis/measure.mjs @@ -0,0 +1,124 @@ +/** + * Build-free throughput + identity bench for Java Lombok accessor synthesis. + * + * Arms: + * - no_lombok: unannotated fields (shape-equivalent control) — synthesizer no-ops + * - lombok_heavy: @Data classes (feature path) + * + * Times synthesizeLombokAccessors over N separate files (not one giant buffer). + * + * Usage: + * node --import tsx bench/java-lombok-synthesis/measure.mjs + * node --import tsx bench/java-lombok-synthesis/measure.mjs --check + */ +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import Parser from 'tree-sitter'; +import Java from 'tree-sitter-java'; +import { synthesizeLombokAccessors } from '../../src/core/ingestion/languages/java/lombok-synthesizer.ts'; +import { + fingerprintIds, + minSample, + runBaselineCheck, + runMethodCountCheck, +} from '../lib/identity-guard.mjs'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); + +const SMALL = 250; +const LARGE = 800; +const REPS = 15; +const WARMUP = 5; + +function entitySource(i, mode) { + if (mode === 'lombok') { + return `import lombok.Data; +@Data +public class Entity${i} { + private String id; + private String name; + private boolean active; + private Long amount; +} +`; + } + return `public class Entity${i} { + private String id; + private String name; + private boolean active; + private Long amount; +} +`; +} + +function ownerMap(tree, filePath) { + const map = new Map(); + const walk = (node) => { + if (node.type === 'class_declaration') { + const name = node.childForFieldName('name')?.text; + if (name) map.set(node.id, `Class:${filePath}:${name}`); + } + for (const c of node.children) walk(c); + }; + walk(tree.rootNode); + return map; +} + +function prepare(mode, fileCount) { + const files = []; + for (let i = 0; i < fileCount; i++) { + const parser = new Parser(); + parser.setLanguage(Java); + const filePath = `bench/${mode}/Entity${i}.java`; + const tree = parser.parse(entitySource(i, mode)); + files.push({ tree, filePath, owners: ownerMap(tree, filePath) }); + } + return files; +} + +function runAll(files) { + const nodes = []; + for (const f of files) { + const result = synthesizeLombokAccessors(f.tree, f.filePath, f.owners); + for (const n of result.nodes) nodes.push(n.id); + } + return nodes; +} + +function measure(mode, fileCount) { + const files = prepare(mode, fileCount); + const { last, ms } = minSample(() => runAll(files), WARMUP, REPS); + return { + files: fileCount, + ms, + methods: last.length, + fingerprint: fingerprintIds(last), + }; +} + +const report = { + no_lombok_small: measure('bare', SMALL), + no_lombok_large: measure('bare', LARGE), + lombok_small: measure('lombok', SMALL), + lombok_large: measure('lombok', LARGE), +}; +report.scaling_ratio = Number( + (report.lombok_large.ms / report.lombok_small.ms / (LARGE / SMALL)).toFixed(3), +); +report.widening_overhead = Number( + (report.lombok_large.ms / Math.max(report.no_lombok_large.ms, 0.001)).toFixed(3), +); +report.fingerprint = report.lombok_large.fingerprint; + +runMethodCountCheck(report, { + no_lombok_large: 0, + lombok_large: 6400, +}); + +if (!process.argv.includes('--check')) { + console.log(JSON.stringify(report, null, 2)); + process.exit(0); +} + +runBaselineCheck(report, BASELINE_PATH); diff --git a/gitnexus/bench/java-wildcard-route-constants/baselines.json b/gitnexus/bench/java-wildcard-route-constants/baselines.json new file mode 100644 index 000000000..961d9a883 --- /dev/null +++ b/gitnexus/bench/java-wildcard-route-constants/baselines.json @@ -0,0 +1,8 @@ +{ + "_comment": "Baselines for bench/java-wildcard-route-constants/measure.mjs --check (#3110). fingerprint is sha256 over 800 materialized route bindings from 800 constant files and must match the named-import control. The benchmark builds the constant import index once per repo pass, matching ingestion and group wiring.", + "fingerprint": "8114e613e93ce0ef6220b810850888592e0e822fe04bf2eb5d8fc4ec3dbba5ef", + "scaling_budget": 1.6, + "_scaling_note": "(t_large/t_small)/(800/250) while both constant files and wildcard importers scale. Measured about 1.14 with the suffix index; repeated candidate scans are quadratic.", + "absolute_ms_budget": 10, + "_absolute_ms_note": "Wildcard materialization for 800 controllers. Measured about 1.1 ms; the generous ceiling catches gross regressions without treating the near-zero named-import control as a stable ratio denominator." +} diff --git a/gitnexus/bench/java-wildcard-route-constants/measure.mjs b/gitnexus/bench/java-wildcard-route-constants/measure.mjs new file mode 100644 index 000000000..50eda340c --- /dev/null +++ b/gitnexus/bench/java-wildcard-route-constants/measure.mjs @@ -0,0 +1,158 @@ +/** + * Build-free throughput + identity benchmark for Java wildcard-static route constants. + * + * Arms: + * - named: explicit `import static ...ApiPaths.ROUTE_n` control + * - wildcard: `import static ...ApiPaths.*` feature path + * + * Parsing is prepared outside the timer. The measured path mirrors ingestion: + * build the constant-key index once, materialize pending wildcard imports, then + * read the resulting binding. Route folding itself has separate integration + * coverage and an older per-fold index cost shared by both arms. + * + * Usage: + * node --import tsx bench/java-wildcard-route-constants/measure.mjs + * node --import tsx bench/java-wildcard-route-constants/measure.mjs --check + */ +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import Parser from 'tree-sitter'; +import Java from 'tree-sitter-java'; +import { + extractJavaModuleConstants, + prepareJavaRouteConstants, +} from '../../src/core/ingestion/route-extractors/java-const-resolver.ts'; +import { + fingerprintIds, + minSampleFresh, + runBaselineCheck, + runCountCheck, + runFingerprintParityCheck, +} from '../lib/route-constant-guard.mjs'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); +const SMALL = 250; +const LARGE = 800; +const REPS = 15; +const WARMUP = 5; + +const parser = new Parser(); +parser.setLanguage(Java); + +function constantsSource(i) { + return `package bench.constants; +public final class ApiPaths${i} { + public static final String ROUTE = "/api/routes/${i}"; +} +`; +} + +function controllerSource(i, mode) { + const fqn = `bench.constants.ApiPaths${i}`; + const imported = mode === 'wildcard' ? `import static ${fqn}.*;` : `import static ${fqn}.ROUTE;`; + return `package bench.web; +${imported} +class Controller${i} {} +`; +} + +function cloneConstants(mc) { + return { + literals: new Map(mc.literals), + exprs: new Map(mc.exprs), + imports: new Map(mc.imports), + wildcardImports: mc.wildcardImports ? [...mc.wildcardImports] : undefined, + unfoldableDeclarations: new Set(mc.unfoldableDeclarations ?? []), + }; +} + +function prepare(mode, fileCount) { + const constants = []; + const controllers = []; + for (let i = 0; i < fileCount; i++) { + constants.push({ + key: `bench/constants/ApiPaths${i}.java`, + constants: extractJavaModuleConstants(parser.parse(constantsSource(i))), + }); + controllers.push({ + key: `bench/web/Controller${i}.java`, + route: 'ROUTE', + constants: extractJavaModuleConstants(parser.parse(controllerSource(i, mode))), + }); + } + return { constants, controllers }; +} + +function instantiate(prepared) { + const repo = new Map(); + for (const constant of prepared.constants) { + repo.set(constant.key, cloneConstants(constant.constants)); + } + const controllers = []; + for (const controller of prepared.controllers) { + repo.set(controller.key, cloneConstants(controller.constants)); + controllers.push({ key: controller.key, route: controller.route }); + } + return { repo, controllers }; +} + +function runAll(instance) { + const { repo, controllers } = instance; + prepareJavaRouteConstants(repo); + const bindings = []; + for (const controller of controllers) { + const mc = repo.get(controller.key); + const binding = mc.imports.get(controller.route); + if (binding) { + bindings.push( + `${controller.key}:${controller.route}:${binding.module}:${binding.originalName}`, + ); + } + } + return bindings; +} + +function measure(mode, fileCount) { + const prepared = prepare(mode, fileCount); + // Expansion mutates each importing file's `imports` map. Give every timed + // sample a fresh repo, but build those clones outside the timer. + const { last, ms } = minSampleFresh(() => instantiate(prepared), runAll, WARMUP, REPS); + return { + files: fileCount, + ms, + bindings: last.length, + fingerprint: fingerprintIds(last), + }; +} + +const report = { + named_small: measure('named', SMALL), + named_large: measure('named', LARGE), + wildcard_small: measure('wildcard', SMALL), + wildcard_large: measure('wildcard', LARGE), +}; +report.scaling_ratio = Number( + (report.wildcard_large.ms / report.wildcard_small.ms / (LARGE / SMALL)).toFixed(3), +); +report.overhead_us_per_binding = Number( + ( + ((report.wildcard_large.ms - report.named_large.ms) * 1000) / + report.wildcard_large.bindings + ).toFixed(3), +); +report.absolute_ms = report.wildcard_large.ms; +report.fingerprint = report.wildcard_large.fingerprint; + +runCountCheck(report, 'bindings', { + named_large: LARGE, + wildcard_large: LARGE, +}); +runFingerprintParityCheck(report, 'named_large', 'wildcard_large'); + +if (!process.argv.includes('--check')) { + console.log(JSON.stringify(report, null, 2)); + process.exit(0); +} + +runBaselineCheck(report, BASELINE_PATH); diff --git a/gitnexus/bench/kotlin-jvm-accessors/baselines.json b/gitnexus/bench/kotlin-jvm-accessors/baselines.json new file mode 100644 index 000000000..6ef877719 --- /dev/null +++ b/gitnexus/bench/kotlin-jvm-accessors/baselines.json @@ -0,0 +1,8 @@ +{ + "_comment": "Baselines for bench/kotlin-jvm-accessors/measure.mjs --check (#2885). fingerprint is sha256 over synthetic Method node ids on the data_large corpus (800 data classes × 4 vars × 2 accessors = 6400 methods). no_props arm uses @JvmField so kotlinc and the synthesizer emit 0 accessor methods. Budgets are timing gates with CI headroom.", + "fingerprint": "18e4f295a437a747c486699e8ec5d310d9bde54437d9a96356a1b1bf8442b0ef", + "scaling_budget": 1.6, + "_scaling_note": "(t_large/t_small)/(800/250) on the data-class arm. Measured ~1.02.", + "widening_overhead_budget": 2.5, + "_widening_overhead_note": "data_large_ms / no_props_large_ms. The @JvmField control preserves four property declarations without accessors; budget guards against a pathological synthesis-arm regression." +} diff --git a/gitnexus/bench/kotlin-jvm-accessors/measure.mjs b/gitnexus/bench/kotlin-jvm-accessors/measure.mjs new file mode 100644 index 000000000..0edbfb1e4 --- /dev/null +++ b/gitnexus/bench/kotlin-jvm-accessors/measure.mjs @@ -0,0 +1,121 @@ +/** + * Build-free throughput + identity bench for Kotlin JVM accessor synthesis. + * + * Arms: + * - no_props: @JvmField properties with no JVM accessors (control) + * - data_class: data class constructor properties (feature path) + * + * Usage: + * node --import tsx bench/kotlin-jvm-accessors/measure.mjs + * node --import tsx bench/kotlin-jvm-accessors/measure.mjs --check + */ +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import Parser from 'tree-sitter'; +import { SupportedLanguages } from 'gitnexus-shared'; +import { getLanguageGrammar } from '../../src/core/tree-sitter/parser-loader.ts'; +import { synthesizeLombokAccessors } from '../../src/core/ingestion/languages/kotlin/lombok-synthesizer.ts'; +import { + fingerprintIds, + minSample, + runBaselineCheck, + runMethodCountCheck, +} from '../lib/identity-guard.mjs'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); + +const SMALL = 250; +const LARGE = 800; +const REPS = 15; +const WARMUP = 5; + +function entitySource(i, mode) { + if (mode === 'data') { + return `data class Entity${i}(var id: String, var name: String, var active: Boolean, var amount: Long) +`; + } + // @JvmField suppresses accessors in kotlinc and in the synthesizer while + // retaining the same four property declarations as the feature arm. + return `class Entity${i} { + @JvmField var id: String = "" + @JvmField var name: String = "" + @JvmField var active: Boolean = false + @JvmField var amount: Long = 0 +} +`; +} + +function ownerMap(tree, filePath) { + const map = new Map(); + const walk = (node) => { + if (node.type === 'class_declaration' || node.type === 'object_declaration') { + const name = + node.childForFieldName('name')?.text ?? + node.namedChildren.find((c) => c.type === 'type_identifier')?.text; + if (name) map.set(node.id, `Class:${filePath}:${name}`); + } + for (const c of node.children) walk(c); + }; + walk(tree.rootNode); + return map; +} + +function prepare(mode, fileCount) { + const files = []; + const lang = getLanguageGrammar(SupportedLanguages.Kotlin); + for (let i = 0; i < fileCount; i++) { + const parser = new Parser(); + parser.setLanguage(lang); + const filePath = `bench/${mode}/Entity${i}.kt`; + const tree = parser.parse(entitySource(i, mode)); + files.push({ tree, filePath, owners: ownerMap(tree, filePath), parser }); + } + return files; +} + +function runAll(files) { + const nodes = []; + for (const f of files) { + const result = synthesizeLombokAccessors(f.tree, f.filePath, f.owners); + for (const n of result.nodes) nodes.push(n.id); + } + return nodes; +} + +function measure(mode, fileCount) { + const files = prepare(mode, fileCount); + const { last, ms } = minSample(() => runAll(files), WARMUP, REPS); + return { + files: fileCount, + ms, + methods: last.length, + fingerprint: fingerprintIds(last), + }; +} + +const report = { + no_props_small: measure('hand', SMALL), + no_props_large: measure('hand', LARGE), + data_small: measure('data', SMALL), + data_large: measure('data', LARGE), +}; +report.scaling_ratio = Number( + (report.data_large.ms / report.data_small.ms / (LARGE / SMALL)).toFixed(3), +); +report.widening_overhead = Number( + (report.data_large.ms / Math.max(report.no_props_large.ms, 0.001)).toFixed(3), +); +report.fingerprint = report.data_large.fingerprint; + +runMethodCountCheck(report, { + no_props_large: 0, + data_large: 6400, +}); + +if (!process.argv.includes('--check')) { + console.log(JSON.stringify(report, null, 2)); + process.exit(0); +} + +runBaselineCheck(report, BASELINE_PATH); diff --git a/gitnexus/bench/kotlin-star-route-constants/baselines.json b/gitnexus/bench/kotlin-star-route-constants/baselines.json new file mode 100644 index 000000000..a68d3ba9f --- /dev/null +++ b/gitnexus/bench/kotlin-star-route-constants/baselines.json @@ -0,0 +1,10 @@ +{ + "_comment": "Baselines for bench/kotlin-star-route-constants/measure.mjs --check (#3110). fingerprint is sha256 over 800 folded route facts from 800 constant files and must match the explicit-import control. The feature arm resolves package-star names through one prepared KotlinConstantIndex.", + "fingerprint": "881101236c511d73d3894d3c9bd2e4a166e3437329149dd99fd16c256435482e", + "scaling_budget": 1.6, + "_scaling_note": "(t_large/t_small)/(800/250) while both constant files and importing controllers scale. Measured about 1.07-1.14.", + "widening_overhead_budget": 2.5, + "_widening_overhead_note": "star_large_ms / named_large_ms. Measured below 1.0; budget guards against a pathological star-lookup regression.", + "absolute_ms_budget": 5, + "_absolute_ms_note": "Package-star folding for 800 controllers. Measured below 0.6 ms; budget includes substantial CI headroom." +} diff --git a/gitnexus/bench/kotlin-star-route-constants/measure.mjs b/gitnexus/bench/kotlin-star-route-constants/measure.mjs new file mode 100644 index 000000000..299accdb8 --- /dev/null +++ b/gitnexus/bench/kotlin-star-route-constants/measure.mjs @@ -0,0 +1,156 @@ +/** + * Build-free throughput + identity benchmark for Kotlin package-star route constants. + * + * Arms: + * - named: explicit `import bench.constants.ROUTE_n` control + * - star: `import bench.constants.*` feature path + * + * Parsing is prepared outside the timer. The measured path mirrors the Kotlin + * group plugin: overlay one importing controller on the prepared constant + * index, then fold its route. + * + * Usage: + * node --import tsx bench/kotlin-star-route-constants/measure.mjs + * node --import tsx bench/kotlin-star-route-constants/measure.mjs --check + */ +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import Parser from 'tree-sitter'; +import { requireVendoredGrammar } from '../../src/core/tree-sitter/vendored-grammars.ts'; +import { + buildKotlinConstantIndex, + extractKotlinModuleConstants, + foldKotlinOperands, + overlayKotlinConstantIndex, +} from '../../src/core/ingestion/route-extractors/kotlin-const-resolver.ts'; +import { + fingerprintIds, + minSampleFresh, + runBaselineCheck, + runCountCheck, + runFingerprintParityCheck, +} from '../lib/route-constant-guard.mjs'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); +const SMALL = 250; +const LARGE = 800; +const REPS = 15; +const WARMUP = 5; + +const parser = new Parser(); +parser.setLanguage(requireVendoredGrammar('tree-sitter-kotlin')); + +function constantsSource(i) { + return `package bench.constants +const val ROUTE_${i} = "/api/routes/${i}" +`; +} + +function controllerSource(i, mode) { + const route = `ROUTE_${i}`; + const imported = mode === 'star' ? 'import bench.constants.*' : `import bench.constants.${route}`; + return `package bench.web +${imported} +class Controller${i} +`; +} + +function cloneConstants(mc) { + return { + literals: new Map(mc.literals), + exprs: new Map(mc.exprs), + imports: new Map(mc.imports), + wildcardImports: mc.wildcardImports ? [...mc.wildcardImports] : undefined, + packageName: mc.packageName, + unfoldableDeclarations: new Set(mc.unfoldableDeclarations), + topLevelDeclarations: new Set(mc.topLevelDeclarations), + }; +} + +function prepare(mode, fileCount) { + const constants = []; + const controllers = []; + for (let i = 0; i < fileCount; i++) { + constants.push({ + key: `bench/constants/ApiPaths${i}.kt`, + constants: extractKotlinModuleConstants(parser.parse(constantsSource(i))), + }); + controllers.push({ + key: `bench/web/Controller${i}.kt`, + route: `ROUTE_${i}`, + constants: extractKotlinModuleConstants(parser.parse(controllerSource(i, mode))), + }); + } + return { constants, controllers }; +} + +function instantiate(prepared) { + const baseRepo = new Map(); + for (const constant of prepared.constants) { + baseRepo.set(constant.key, cloneConstants(constant.constants)); + } + const controllers = prepared.controllers.map((controller) => ({ + key: controller.key, + route: controller.route, + constants: cloneConstants(controller.constants), + })); + return { baseRepo, controllers }; +} + +function runAll(instance) { + const { baseRepo, controllers } = instance; + const baseIndex = buildKotlinConstantIndex(baseRepo); + const routes = []; + for (const controller of controllers) { + const index = overlayKotlinConstantIndex(baseIndex, controller.key, controller.constants); + const route = foldKotlinOperands( + controller.key, + [{ kind: 'ref', name: controller.route }], + index.repo, + [], + index, + ); + if (route !== null) routes.push(`${controller.key}:${route}`); + } + return routes; +} + +function measure(mode, fileCount) { + const prepared = prepare(mode, fileCount); + const { last, ms } = minSampleFresh(() => instantiate(prepared), runAll, WARMUP, REPS); + return { + files: fileCount, + ms, + routes: last.length, + fingerprint: fingerprintIds(last), + }; +} + +const report = { + named_small: measure('named', SMALL), + named_large: measure('named', LARGE), + star_small: measure('star', SMALL), + star_large: measure('star', LARGE), +}; +report.scaling_ratio = Number( + (report.star_large.ms / report.star_small.ms / (LARGE / SMALL)).toFixed(3), +); +report.widening_overhead = Number( + (report.star_large.ms / Math.max(report.named_large.ms, 0.001)).toFixed(3), +); +report.absolute_ms = report.star_large.ms; +report.fingerprint = report.star_large.fingerprint; + +runCountCheck(report, 'routes', { + named_large: LARGE, + star_large: LARGE, +}); +runFingerprintParityCheck(report, 'named_large', 'star_large'); + +if (!process.argv.includes('--check')) { + console.log(JSON.stringify(report, null, 2)); + process.exit(0); +} + +runBaselineCheck(report, BASELINE_PATH); diff --git a/gitnexus/bench/lib/identity-guard.mjs b/gitnexus/bench/lib/identity-guard.mjs new file mode 100644 index 000000000..b73a1a241 --- /dev/null +++ b/gitnexus/bench/lib/identity-guard.mjs @@ -0,0 +1,62 @@ +/** + * Shared fingerprint + --check for JVM accessor synthesis benches. + */ +import fs from 'node:fs'; +import crypto from 'node:crypto'; + +export function fingerprintIds(ids) { + return crypto + .createHash('sha256') + .update([...ids].sort().join('\n')) + .digest('hex'); +} + +export function minSample(run, warmup, reps) { + for (let w = 0; w < warmup; w++) run(); + const samples = []; + let last; + for (let r = 0; r < reps; r++) { + const t0 = performance.now(); + last = run(); + samples.push(performance.now() - t0); + } + return { last, ms: Math.min(...samples) }; +} + +export function runMethodCountCheck(report, expectedCounts) { + const errors = []; + for (const [arm, expected] of Object.entries(expectedCounts)) { + const actual = report[arm]?.methods; + if (actual !== expected) { + errors.push(`${arm}.methods ${String(actual)} != ${expected}`); + } + } + if (errors.length) { + console.error(JSON.stringify({ report, errors }, null, 2)); + process.exit(1); + } +} + +export function runBaselineCheck(report, baselinePath) { + const baseline = JSON.parse(fs.readFileSync(baselinePath, 'utf-8')); + const errors = []; + if (report.fingerprint !== baseline.fingerprint) { + errors.push(`fingerprint drift: ${report.fingerprint} != ${baseline.fingerprint}`); + } + if (report.scaling_ratio > baseline.scaling_budget) { + errors.push(`scaling_ratio ${report.scaling_ratio} > ${baseline.scaling_budget}`); + } + if ( + baseline.widening_overhead_budget !== undefined && + report.widening_overhead > baseline.widening_overhead_budget + ) { + errors.push( + `widening_overhead ${report.widening_overhead} > ${baseline.widening_overhead_budget}`, + ); + } + if (errors.length) { + console.error(JSON.stringify({ report, errors }, null, 2)); + process.exit(1); + } + console.log(JSON.stringify({ ok: true, report }, null, 2)); +} diff --git a/gitnexus/bench/lib/route-constant-guard.mjs b/gitnexus/bench/lib/route-constant-guard.mjs new file mode 100644 index 000000000..cae8b95fc --- /dev/null +++ b/gitnexus/bench/lib/route-constant-guard.mjs @@ -0,0 +1,77 @@ +/** Shared fingerprint + --check helpers for route-constant benchmarks. */ +import fs from 'node:fs'; +import crypto from 'node:crypto'; + +export function fingerprintIds(ids) { + return crypto + .createHash('sha256') + .update([...ids].sort().join('\n')) + .digest('hex'); +} + +/** Min sample for mutating benchmarks that need fresh state per repetition. */ +export function minSampleFresh(create, run, warmup, reps) { + for (let w = 0; w < warmup; w++) run(create()); + const samples = []; + let last; + for (let r = 0; r < reps; r++) { + const state = create(); + const t0 = performance.now(); + last = run(state); + samples.push(performance.now() - t0); + } + return { last, ms: Math.min(...samples) }; +} + +export function runCountCheck(report, field, expectedCounts) { + const errors = []; + for (const [arm, expected] of Object.entries(expectedCounts)) { + const actual = report[arm]?.[field]; + if (actual !== expected) { + errors.push(`${arm}.${field} ${String(actual)} != ${expected}`); + } + } + failIfNeeded(report, errors); +} + +export function runFingerprintParityCheck(report, leftArm, rightArm) { + const left = report[leftArm]?.fingerprint; + const right = report[rightArm]?.fingerprint; + failIfNeeded( + report, + left === right ? [] : [`${leftArm}.fingerprint ${left} != ${rightArm}.fingerprint ${right}`], + ); +} + +export function runBaselineCheck(report, baselinePath) { + const baseline = JSON.parse(fs.readFileSync(baselinePath, 'utf-8')); + const errors = []; + if (report.fingerprint !== baseline.fingerprint) { + errors.push(`fingerprint drift: ${report.fingerprint} != ${baseline.fingerprint}`); + } + if (report.scaling_ratio > baseline.scaling_budget) { + errors.push(`scaling_ratio ${report.scaling_ratio} > ${baseline.scaling_budget}`); + } + if ( + baseline.absolute_ms_budget !== undefined && + report.absolute_ms > baseline.absolute_ms_budget + ) { + errors.push(`absolute_ms ${report.absolute_ms} > ${baseline.absolute_ms_budget}`); + } + if ( + baseline.widening_overhead_budget !== undefined && + report.widening_overhead > baseline.widening_overhead_budget + ) { + errors.push( + `widening_overhead ${report.widening_overhead} > ${baseline.widening_overhead_budget}`, + ); + } + failIfNeeded(report, errors); + console.log(JSON.stringify({ ok: true, report }, null, 2)); +} + +function failIfNeeded(report, errors) { + if (errors.length === 0) return; + console.error(JSON.stringify({ report, errors }, null, 2)); + process.exit(1); +} diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index d8831f822..7dcd85c15 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -209,8 +209,10 @@ "_rebaselined_blind_spots_2856": "#2856 blind-spots series: the JS/TS SCOPE queries gained capture rules, so fingerprint drift is expected and additive. Verified before re-baselining by diffing the capture-name sets in both scope queries against origin/main: TypeScript gained exactly @reference.read.identifier (A2 bare-identifier reads in value positions) and @reference.type (R2-2 type references, so a declared contract stops reporting incoming:{}); JavaScript gained exactly @reference.read.identifier, @reference.read.destructured (R2-1c) and @reference.write.property-key (R2-1b record-construction writes). NOTHING was removed on either side \u2014 the delta is a pure superset, which is the check that no existing capture moved. capture_groups_small/large are unchanged (4503/14403) because those measure the SYNTHETIC scaling source, which this branch does not touch; only the fixture-corpus count moves. capture_groups_fp 2097 -> 2338 and fixture_count 146 -> 151 from 21 new lang-resolution fixtures. Scaling stayed linear and inside budget: typescript 1.116 < 1.5, javascript 1.010 < 1.5. Prior typescript ed92588e0fc7b28b3a0174339ac378b4dd85965fe007db1208dea97a65ce0571 -> f66a3e6f1e096431e7046505129a627deaa00ca0de5bc846b080591b397248f7; prior javascript 806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594 -> 2026993b81b873839dd2ef8797d9c14d9c48516b2b57b05ac17d8d43f2f4eba3." }, "kotlin": { - "fingerprint": "f98e7e936afbce0e99588285cfc603bf945fd58c5de45271860509a5d90eb832", + "fingerprint": "aeafc7a87402c933786ef582b7c98683b1822b78fa909e605cb97552867fa0d5", "scaling_budget": 1.5, + "_rebaselined_interface_abstract_2885": "#2885: Kotlin interface property accessors stay in the capture set (groups still 5753/18403 and capture_groups_fp 2563) but Method isAbstract is now true for body-less interface properties, which changes accessor-plan identity in the fixture digest. Prior 82ae5e1f750580383344d4c84c400a290474528cd502be4af8cd56705819a683 -> aeafc7a87402c933786ef582b7c98683b1822b78fa909e605cb97552867fa0d5; CI scaling 0.838 < 1.5.", + "_rebaselined_jvm_property_accessors_2885": "#2885: Kotlin val/var properties now emit JVM getter/setter scope and declaration captures, including data-class constructor properties and custom accessors. Synthetic scaling counts move 4753/15203 -> 5753/18403; fixture-corpus groups move 2367 -> 2563. Accessor declaration sidecars use the canonical @declaration.qualified_name key, preserve same-name owner identity, follow JvmAbi is-prefix naming, and suppress @JvmName-renamed accessors until their custom names are modeled. Prior f98e7e936afbce0e99588285cfc603bf945fd58c5de45271860509a5d90eb832 -> 82ae5e1f750580383344d4c84c400a290474528cd502be4af8cd56705819a683; scaling 0.869 < 1.5.", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior bddba25d5a88152bbbee8d70e82c944b5302accb4b625df782adb1d4f7a7ac12 -> e856951c2a779163d555dadc8e1bf59304a86caed78ac1f450d9caa2b50f63d1; scaling 1.090 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Kotlin callable-reference flow facts with invocation-result suppression. Prior 4900431791f2b9280009deb2b82659c26ead8aa6fb8731190a7c505dec5a9041 -> bddba25d5a88152bbbee8d70e82c944b5302accb4b625df782adb1d4f7a7ac12; scaling 0.880 < 1.5.", "_added": "#1951: bench coverage added (was ungated); scale source heritage-bearing (: Base()); js/kotlin O(n^2) findNodeAtRange-per-match fixed to threaded captured node, now linear.", @@ -223,9 +225,9 @@ "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior d3c4d2fa0d82d248a2299cfc888b067187ad1faf2c87a97f93c6ed835eefc3f1 -> c1f0cc9058ab11b7cd6fc8b440deb6db2b2f530f2eb21178923e68a3d0796c4b.", "_rebaselined_2766_await_subscript_emission": "#2766: extractMixedChain now walks THROUGH await and subscript nodes and peels transparent wrappers at loop entry, so sites whose receiver is `repos[0]` or `(await f())` mint a receiver chain where they previously minted none. EMISSION CHANGE: more sites carry `@reference.receiver-chain`; no existing chain changed shape. Only go and kotlin drifted of 15 \u2014 the two whose fixture corpora contain such receivers. Prior c1f0cc9058ab11b7cd6fc8b440deb6db2b2f530f2eb21178923e68a3d0796c4b -> efd5dbf80ffcd3bab2834d1010f6fe2b239dcc5d58229938dea9cff8d0f380f2.", "_rebaselined_2960_declared_package_fixture": "#2960 adds four Kotlin declared-package import-resolution fixture files. This is fixture-corpus growth only: fixture_count 137 -> 141 and capture_groups_fp 2334 -> 2367; the synthetic capture counts remain 4753/15203, no Kotlin scope-capture query or implementation changed, and package resolution runs after capture. Other language fingerprints matched their baselines in the same CI run. Prior a184f8ff0ae40d246db855b63f7ff26bda3afac03e5f4c76e4593c7e2cefce54 -> f98e7e936afbce0e99588285cfc603bf945fd58c5de45271860509a5d90eb832.", - "capture_groups_small": 4753, - "capture_groups_large": 15203, - "capture_groups_fp": 2367, + "capture_groups_small": 5753, + "capture_groups_large": 18403, + "capture_groups_fp": 2563, "fixture_count": 141 } } diff --git a/gitnexus/bench/spring-config-bindings/baselines.json b/gitnexus/bench/spring-config-bindings/baselines.json new file mode 100644 index 000000000..97f881dd4 --- /dev/null +++ b/gitnexus/bench/spring-config-bindings/baselines.json @@ -0,0 +1,8 @@ +{ + "_comment": "Baselines for bench/spring-config-bindings/measure.mjs --check (#2412). fingerprint is sha256 over position-free Kotlin config-consumer fact ids on the wildcard_large corpus (800 files × 2 @Value properties + 1 @ConfigurationProperties class = 2400 facts). Both arms must fingerprint identically: the wildcard arm adds a sibling nested type named `Value`, which must not suppress the imported Spring annotation. Budgets are timing gates with CI headroom.", + "fingerprint": "34776f883427479befbeb3c09eaae2260ba778e769bff195044d3cb8f5ad9889", + "scaling_budget": 1.6, + "_scaling_note": "(t_large/t_small)/(800/250) on the wildcard arm. Measured ~0.99.", + "widening_overhead_budget": 1.8, + "_widening_overhead_note": "wildcard_large_ms / exact_large_ms. The exact-import control resolves each annotation from imports.exact before any shadow check, so this isolates the wildcard path's per-annotation lexical shadow walk. Measured ~1.09; budget guards against a per-annotation rescan of the file's declarations." +} diff --git a/gitnexus/bench/spring-config-bindings/measure.mjs b/gitnexus/bench/spring-config-bindings/measure.mjs new file mode 100644 index 000000000..94ac94bf2 --- /dev/null +++ b/gitnexus/bench/spring-config-bindings/measure.mjs @@ -0,0 +1,164 @@ +/** + * Build-free throughput + identity bench for Kotlin Spring config-consumer + * capture (#2412). + * + * Arms (identical corpora except the import style): + * - exact: explicit `import ...annotation.Value` control, which resolves the + * annotation from `imports.exact` before any shadow check runs + * - wildcard: `import ...annotation.*` feature path, where every simple-name + * annotation pays the lexical local-type shadow walk. Each file also + * declares a sibling nested type named `Value` that must NOT suppress the + * Spring annotation — the file-wide-shadow regression fixed on this branch. + * + * Parsing is prepared outside the timer; the measured path is the capture + * function the Kotlin worker calls on its own AST. + * + * Usage: + * node --import tsx bench/spring-config-bindings/measure.mjs + * node --import tsx bench/spring-config-bindings/measure.mjs --check + */ +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import Parser from 'tree-sitter'; +import { SupportedLanguages } from 'gitnexus-shared'; +import { getLanguageGrammar } from '../../src/core/tree-sitter/parser-loader.ts'; +import { captureKotlinSpringConfigConsumerFacts } from '../../src/core/ingestion/languages/kotlin/spring-config-bindings.ts'; +import { fingerprintIds, minSample, runBaselineCheck } from '../lib/identity-guard.mjs'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); + +const SMALL = 250; +const LARGE = 800; +const REPS = 15; +const WARMUP = 5; +/** Two @Value properties plus one @ConfigurationProperties class per file. */ +const FACTS_PER_FILE = 3; + +function consumerSource(i, mode) { + const imports = + mode === 'wildcard' + ? `import org.springframework.beans.factory.annotation.* +import org.springframework.boot.context.properties.*` + : `import org.springframework.beans.factory.annotation.Value +import org.springframework.boot.context.properties.ConfigurationProperties`; + + return `package bench.config +${imports} + +class Shadowing${i} { + class Value +} + +@ConfigurationProperties(prefix = "svc.${i}") +class Props${i} { + var endpoint: String? = null +} + +class Consumer${i} { + @Value("\\\${app.key${i}}") + var timeout: Int = 0 + + @Value("\\\${app.other${i}:5}") + var other: String? = null + + fun decoy() {} +} +`; +} + +/** Position-free fact identity, so both arms are directly comparable. */ +function factId(fact) { + const consumer = fact.consumer; + return consumer.kind === 'value' + ? `value|${consumer.fieldName}|${[...consumer.keys].sort().join(',')}` + : `configuration-properties|${consumer.className}|${consumer.prefix}`; +} + +function prepare(mode, fileCount) { + const files = []; + const lang = getLanguageGrammar(SupportedLanguages.Kotlin); + for (let i = 0; i < fileCount; i++) { + const parser = new Parser(); + parser.setLanguage(lang); + const filePath = `bench/${mode}/Consumer${i}.kt`; + files.push({ tree: parser.parse(consumerSource(i, mode)), filePath, parser }); + } + return files; +} + +function runAll(files) { + const ids = []; + for (const f of files) { + for (const fact of captureKotlinSpringConfigConsumerFacts(f.tree.rootNode, f.filePath)) { + ids.push(factId(fact)); + } + } + return ids; +} + +function measure(mode, fileCount) { + const files = prepare(mode, fileCount); + const { last, ms } = minSample(() => runAll(files), WARMUP, REPS); + return { + files: fileCount, + ms, + facts: last.length, + fingerprint: fingerprintIds(last), + }; +} + +function failIfNeeded(current, errors) { + if (errors.length === 0) return; + console.error(JSON.stringify({ report: current, errors }, null, 2)); + process.exit(1); +} + +function runFactCountCheck(current, expectedCounts) { + const errors = []; + for (const [arm, expected] of Object.entries(expectedCounts)) { + const actual = current[arm]?.facts; + if (actual !== expected) errors.push(`${arm}.facts ${String(actual)} != ${expected}`); + } + failIfNeeded(current, errors); +} + +/** + * A wildcard import plus a sibling `Value` declaration must capture exactly the + * facts the explicit-import control captures. + */ +function runFingerprintParityCheck(current, leftArm, rightArm) { + const left = current[leftArm]?.fingerprint; + const right = current[rightArm]?.fingerprint; + failIfNeeded( + current, + left === right ? [] : [`${leftArm}.fingerprint ${left} != ${rightArm}.fingerprint ${right}`], + ); +} + +const report = { + exact_small: measure('exact', SMALL), + exact_large: measure('exact', LARGE), + wildcard_small: measure('wildcard', SMALL), + wildcard_large: measure('wildcard', LARGE), +}; +report.scaling_ratio = Number( + (report.wildcard_large.ms / report.wildcard_small.ms / (LARGE / SMALL)).toFixed(3), +); +report.widening_overhead = Number( + (report.wildcard_large.ms / Math.max(report.exact_large.ms, 0.001)).toFixed(3), +); +report.fingerprint = report.wildcard_large.fingerprint; + +runFactCountCheck(report, { + exact_large: LARGE * FACTS_PER_FILE, + wildcard_large: LARGE * FACTS_PER_FILE, +}); +runFingerprintParityCheck(report, 'exact_large', 'wildcard_large'); + +if (!process.argv.includes('--check')) { + console.log(JSON.stringify(report, null, 2)); + process.exit(0); +} + +runBaselineCheck(report, BASELINE_PATH); diff --git a/gitnexus/bench/v8-sidecar/measure.mjs b/gitnexus/bench/v8-sidecar/measure.mjs new file mode 100644 index 000000000..68867b9ba --- /dev/null +++ b/gitnexus/bench/v8-sidecar/measure.mjs @@ -0,0 +1,91 @@ +#!/usr/bin/env node +/** + * Optional V8 sidecar warm-load bench (#3089). + * + * Not part of `npm test`. Measures repeated warm loads of the `.v8` ParsedFile + * shards already on disk through the production loader. Replay of identical + * shards is throughput-only — it is not unique-object scale. + * + * Copies the store into a temporary workspace first. The source cache is + * never mutated. + * + * Usage (from gitnexus/): + * node --expose-gc --import tsx bench/v8-sidecar/measure.mjs + */ +import { cp, mkdtemp, readdir, rm } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import path from 'node:path'; +import { performance } from 'node:perf_hooks'; +import { loadParsedFilesForPaths } from '../../src/storage/parsedfile-store.ts'; +import { inspectV8Cache } from '../../src/storage/v8-sidecar.ts'; + +const srcStorage = process.argv[2]; +if (!srcStorage) { + console.error('usage: node --expose-gc --import tsx bench/v8-sidecar/measure.mjs '); + process.exit(2); +} + +const srcStoreDir = path.join(srcStorage, 'parsedfile-store'); +const benchRoot = await mkdtemp(path.join(tmpdir(), 'gnx-v8-bench-')); +const storeDir = path.join(benchRoot, 'parsedfile-store'); +const PATH_SOURCE_SHARDS = 8; +const RUNS = 3; + +try { + await cp(srcStoreDir, storeDir, { recursive: true }); + + const names = (await readdir(storeDir)) + .filter((f) => f.endsWith('.v8') && !f.includes('.v8.')) + .sort(); + if (names.length === 0) { + throw new Error( + `no .v8 ParsedFile shards in ${srcStoreDir} — run an analyze that populates the store first`, + ); + } + + const want = new Set(); + let sourceShards = 0; + for (const name of names) { + const inspected = await inspectV8Cache(path.join(storeDir, name)); + if (!inspected) continue; + sourceShards++; + for (const filePath of inspected.paths) want.add(filePath); + if (sourceShards >= PATH_SOURCE_SHARDS) break; + } + if (want.size === 0) { + throw new Error( + `no file paths readable from ${names.length} shard(s) in ${srcStoreDir} — shards may be from another Node/V8 runtime, so re-analyze with this runtime`, + ); + } + + const rss = () => Math.round(process.memoryUsage().rss / 1024 / 1024); + const heap = () => Math.round(process.memoryUsage().heapUsed / 1024 / 1024); + + const run = async (label) => { + if (typeof globalThis.gc === 'function') globalThis.gc(); + const t0 = performance.now(); + const loaded = await loadParsedFilesForPaths(benchRoot, want); + const ms = Math.round(performance.now() - t0); + if (loaded.size !== want.size) { + throw new Error(`incomplete V8 load: requested ${want.size} paths but loaded ${loaded.size}`); + } + if (typeof globalThis.gc === 'function') globalThis.gc(); + console.log( + JSON.stringify({ + label, + shards: names.length, + wantPaths: want.size, + files: loaded.size, + ms, + rssMiB: rss(), + heapUsedMiB: heap(), + }), + ); + }; + + for (let i = 1; i <= RUNS; i++) { + await run(`v8-load-${i}`); + } +} finally { + await rm(benchRoot, { recursive: true, force: true }); +} diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index bd3ebd088..e94a72426 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -14,17 +14,18 @@ "@modelcontextprotocol/sdk": "^1.0.0", "@scarf/scarf": "^1.4.0", "busboy": "^1.6.0", - "chokidar": "^4.0.3", + "chokidar": "^5.0.0", "cli-progress": "^3.12.0", "commander": "^15.0.0", "cors": "^2.8.5", "express": "^5.2.1", "express-rate-limit": "^8.4.1", + "fast-xml-parser": "^5.11.1", "glob": "^13.0.6", "graphology": "^0.26.0", "graphology-indices": "^0.17.0", "graphology-utils": "^2.3.0", - "graphql": "^16.14.2", + "graphql": "^17.0.2", "ignore": "^7.0.5", "js-yaml": "^5.0.0", "jsonc-parser": "^3.3.1", @@ -1397,6 +1398,18 @@ } } }, + "node_modules/@nodable/entities": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/@nodable/entities/-/entities-3.0.0.tgz", + "integrity": "sha512-8L9xFeTYKhm49xfIypoe2W5wV1m/3Z58kT+7kR9A8OyFxcPduI4VmxaUMQyKYrRjUoLLSXv6EKKID5Tvj9cUVw==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/nodable" + } + ], + "license": "MIT" + }, "node_modules/@oxc-project/types": { "version": "0.144.0", "resolved": "https://registry.npmjs.org/@oxc-project/types/-/types-0.144.0.tgz", @@ -1875,9 +1888,9 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "26.2.0", - "resolved": "https://registry.npmjs.org/@types/node/-/node-26.2.0.tgz", - "integrity": "sha512-5IviulTZeRNp2vAJ514cc/HUlY5nZ9fCbq9DMyC52BrhFZACo3nI0R7qBxhQmo/d27NFe96ur/b7Wwxklda+kg==", + "version": "26.3.0", + "resolved": "https://registry.npmjs.org/@types/node/-/node-26.3.0.tgz", + "integrity": "sha512-L3fgrnchriRC2ExBflb8j4uZZURHZfQsmQeyVzhjcHW4kkwVyo8/0h1B2MVzMTrYUJYu6G7EWs14hW/L9putqw==", "devOptional": true, "license": "MIT", "dependencies": { @@ -2153,6 +2166,18 @@ "url": "https://github.com/chalk/ansi-styles?sponsor=1" } }, + "node_modules/anynum": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/anynum/-/anynum-1.0.1.tgz", + "integrity": "sha512-N6//FLET/tXYNM/F6ABca1oH6fWB+KlTt909Le28WMDBk8oaT4vY17DCrwg2MvmuqUKt3Ni4N5dGJ/EoBgcO6A==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT" + }, "node_modules/apache-arrow": { "version": "21.1.0", "resolved": "https://registry.npmjs.org/apache-arrow/-/apache-arrow-21.1.0.tgz", @@ -2383,15 +2408,15 @@ } }, "node_modules/chokidar": { - "version": "4.0.3", - "resolved": "https://registry.npmjs.org/chokidar/-/chokidar-4.0.3.tgz", - "integrity": "sha512-Qgzu8kfBvo+cA4962jnP1KkS6Dop5NS6g7R5LFYJr4b8Ub94PPQXUksCw9PvXoeXPRRddRNC5C1JQUR2SMGtnA==", + "version": "5.0.0", + "resolved": "https://registry.npmjs.org/chokidar/-/chokidar-5.0.0.tgz", + "integrity": "sha512-TQMmc3w+5AxjpL8iIiwebF73dRDF4fBIieAqGn9RGCWaEVwQ6Fb2cGe31Yns0RRIzii5goJ1Y7xbMwo1TxMplw==", "license": "MIT", "dependencies": { - "readdirp": "^4.0.1" + "readdirp": "^5.0.0" }, "engines": { - "node": ">= 14.16.0" + "node": ">= 20.19.0" }, "funding": { "url": "https://paulmillr.com/funding/" @@ -3021,6 +3046,45 @@ ], "license": "BSD-3-Clause" }, + "node_modules/fast-xml-builder": { + "version": "1.3.1", + "resolved": "https://registry.npmjs.org/fast-xml-builder/-/fast-xml-builder-1.3.1.tgz", + "integrity": "sha512-pIM/1n3ntFXKYrUZwW7QCK0gAW7XY+wzj1YMIV3tLDvPj/V+zTGJK5e3/4WJfwj0qWw2ElNXiTixda/R+3YSug==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "dependencies": { + "path-expression-matcher": "^1.6.2", + "xml-naming": "^0.3.0" + } + }, + "node_modules/fast-xml-parser": { + "version": "5.11.1", + "resolved": "https://registry.npmjs.org/fast-xml-parser/-/fast-xml-parser-5.11.1.tgz", + "integrity": "sha512-TBw6K/fxoQGGjCmZDw9w/ZwP3uDcnTM4YH/g+PFRWr8sbe5idXtxNN6vITh4+1ruCZaho6uBFurElsA7F0zzgw==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "dependencies": { + "@nodable/entities": "^3.0.0", + "fast-xml-builder": "^1.2.0", + "is-unsafe": "^2.0.0", + "path-expression-matcher": "^1.6.2", + "strnum": "^2.4.2", + "xml-naming": "^0.3.0" + }, + "bin": { + "fxparser": "src/cli/cli.js" + } + }, "node_modules/fdir": { "version": "6.5.0", "resolved": "https://registry.npmjs.org/fdir/-/fdir-6.5.0.tgz", @@ -3308,12 +3372,12 @@ } }, "node_modules/graphql": { - "version": "16.14.2", - "resolved": "https://registry.npmjs.org/graphql/-/graphql-16.14.2.tgz", - "integrity": "sha512-Chq1s4CY7jmh8gO2qvLIJyfCDIN+EHLFW/9iShnp1z8FjBQMoodWP1kDC36VAMXXIvAjj4ARa7ntfAV2BrjsbA==", + "version": "17.0.2", + "resolved": "https://registry.npmjs.org/graphql/-/graphql-17.0.2.tgz", + "integrity": "sha512-FRWbddMxfkjiB7z+aQDWIR+E34xo9I8c9mtK2RPv8PmMzKRvrdsreHL/Ui/TmwHJfhHChEtsFPyMHKI+xuarQQ==", "license": "MIT", "engines": { - "node": "^12.22.0 || ^14.16.0 || ^16.0.0 || >=17.0.0" + "node": "^22.0.0 || ^24.0.0 || ^25.0.0 || >=26.0.0" } }, "node_modules/guid-typescript": { @@ -3481,6 +3545,18 @@ "integrity": "sha512-hvpoI6korhJMnej285dSg6nu1+e6uxs7zG3BYAm5byqDsgJNWwxzM6z6iZiAgQR4TJ30JmBTOwqZUw3WlyH3AQ==", "license": "MIT" }, + "node_modules/is-unsafe": { + "version": "2.0.2", + "resolved": "https://registry.npmjs.org/is-unsafe/-/is-unsafe-2.0.2.tgz", + "integrity": "sha512-HgbIHPBH0KHHCcjLfGsCvhtPTVxjaAZlXjwdz7/GQC40SjSe4sfQsar8J5VFo8JOSbarkpV0OLG95bbaNd9aAQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT" + }, "node_modules/isexe": { "version": "4.0.0", "resolved": "https://registry.npmjs.org/isexe/-/isexe-4.0.0.tgz", @@ -3555,9 +3631,9 @@ "license": "MIT" }, "node_modules/js-yaml": { - "version": "5.3.0", - "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-5.3.0.tgz", - "integrity": "sha512-muutsYr+e2+d3rTgUGslq5rxbBlUy3cJ61IsHag2QNDQV+7zXWjkUpmALIajhrlLlrgRUiymj6U3zUr/TMK84Q==", + "version": "5.4.0", + "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-5.4.0.tgz", + "integrity": "sha512-jE7vUJIebKzYQI5xu4co5CRBDlDEYnHrdzsxs4O2giCz4v2SbVMYKpmt1D9L38OKQAeCWmrOTRiCV93u0UkaJA==", "funding": [ { "type": "github", @@ -4262,15 +4338,15 @@ } }, "node_modules/onnxruntime-common": { - "version": "1.27.0", - "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.27.0.tgz", - "integrity": "sha512-3KxL5wIVqa8Ex08jxSzncm9CMgw8CjOFyOQ7SxvG9o0cVLlhTNKXyIQuTbtX4tGPJEf73OER2xrjt4HJSBL4ow==", + "version": "1.29.0", + "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.29.0.tgz", + "integrity": "sha512-/F63/e2VJoaVXGGNu6S5QH7jivBThGO95OzAVXXQ8hTta/b1QxI8udHa6cI3+3mAb5WWIIaMMwfZw01oivjJ1g==", "license": "MIT" }, "node_modules/onnxruntime-node": { - "version": "1.27.0", - "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.27.0.tgz", - "integrity": "sha512-QEzGwrvNBgv4uPVdnbHsOGG4G6T96mdlcFI8aAKPjMU8wOPpVocPXb6k3QGkaZagVTv2G9Bnnbo6Z3JdXr1fQw==", + "version": "1.29.0", + "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.29.0.tgz", + "integrity": "sha512-WjiVVB72riILz8HbYvxvmjKyE/WmkYoSfKY++axo5jAR609HQg8MwiG/HhShpTcJfmmAdzxxmB+MMST3A+SiPA==", "hasInstallScript": true, "license": "MIT", "optional": true, @@ -4280,9 +4356,9 @@ "linux" ], "dependencies": { - "adm-zip": "^0.5.16", + "adm-zip": "^0.6.0", "global-agent": "^4.1.3", - "onnxruntime-common": "1.27.0" + "onnxruntime-common": "1.29.0" } }, "node_modules/onnxruntime-web": { @@ -4334,6 +4410,21 @@ "node": ">= 0.8" } }, + "node_modules/path-expression-matcher": { + "version": "1.6.2", + "resolved": "https://registry.npmjs.org/path-expression-matcher/-/path-expression-matcher-1.6.2.tgz", + "integrity": "sha512-enSlaiat05iasnzmgNxRj8reFdj3puY2QpNgP1aPIaVfT6nn9ICuPoFlKHk8EN22HcwewshO+mN2DGbkCEOtqQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "engines": { + "node": ">=14.0.0" + } + }, "node_modules/path-key": { "version": "3.1.1", "resolved": "https://registry.npmjs.org/path-key/-/path-key-3.1.1.tgz", @@ -4628,12 +4719,12 @@ } }, "node_modules/readdirp": { - "version": "4.1.2", - "resolved": "https://registry.npmjs.org/readdirp/-/readdirp-4.1.2.tgz", - "integrity": "sha512-GDhwkLfywWL2s6vEjyhri+eXmfH6j1L7JE27WhqLeYzoh/A3DBaYGEj2H/HFZCn/kMfim73FXxEJTw06WtxQwg==", + "version": "5.1.1", + "resolved": "https://registry.npmjs.org/readdirp/-/readdirp-5.1.1.tgz", + "integrity": "sha512-Kko+Y5XQ6fM+Ce3dq3m9YGxnacYZYl9cA1wZjaF3Vbry2L3i1qVg8+CAgNPsXRArPMUMCaOR7oa9Nqntc43JKA==", "license": "MIT", "engines": { - "node": ">= 14.18.0" + "node": ">= 20.19.0" }, "funding": { "type": "individual", @@ -5080,6 +5171,21 @@ "node": ">=0.10.0" } }, + "node_modules/strnum": { + "version": "2.4.2", + "resolved": "https://registry.npmjs.org/strnum/-/strnum-2.4.2.tgz", + "integrity": "sha512-rDG3Ah4TV0k1hWvLSzkZtMmLN9+eS+h3knq4MP6A42Y3Yh5qGNnOUs1jJkoSr8FG5dsL28c7KgkIBzSEykqtuw==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "dependencies": { + "anynum": "^1.0.1" + } + }, "node_modules/supports-color": { "version": "7.2.0", "resolved": "https://registry.npmjs.org/supports-color/-/supports-color-7.2.0.tgz", @@ -5765,6 +5871,21 @@ "integrity": "sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ==", "license": "ISC" }, + "node_modules/xml-naming": { + "version": "0.3.0", + "resolved": "https://registry.npmjs.org/xml-naming/-/xml-naming-0.3.0.tgz", + "integrity": "sha512-ghig2TBE/H11aOVgmahA3MhimvkBr6JIYknH/Dhdk10nXwdbIqBJsbfMxpvFPG8bAw77gN29aQWvKpmVoPlvPQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "engines": { + "node": ">=16.0.0" + } + }, "node_modules/y18n": { "version": "5.0.8", "resolved": "https://registry.npmjs.org/y18n/-/y18n-5.0.8.tgz", diff --git a/gitnexus/package.json b/gitnexus/package.json index cf511b17b..1e15ad9b4 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -60,17 +60,18 @@ "@modelcontextprotocol/sdk": "^1.0.0", "@scarf/scarf": "^1.4.0", "busboy": "^1.6.0", - "chokidar": "^4.0.3", + "chokidar": "^5.0.0", "cli-progress": "^3.12.0", "commander": "^15.0.0", "cors": "^2.8.5", "express": "^5.2.1", "express-rate-limit": "^8.4.1", + "fast-xml-parser": "^5.11.1", "glob": "^13.0.6", "graphology": "^0.26.0", "graphology-indices": "^0.17.0", "graphology-utils": "^2.3.0", - "graphql": "^16.14.2", + "graphql": "^17.0.2", "ignore": "^7.0.5", "js-yaml": "^5.0.0", "jsonc-parser": "^3.3.1", diff --git a/gitnexus/skills/gitnexus-cli.md b/gitnexus/skills/gitnexus-cli.md index 1b1f733b6..170703439 100644 --- a/gitnexus/skills/gitnexus-cli.md +++ b/gitnexus/skills/gitnexus-cli.md @@ -27,9 +27,12 @@ Run from the project root. This parses all source files, builds the knowledge gr | `--embeddings` | Enable embedding generation for semantic search (off by default) | | `--drop-embeddings` | Drop existing embeddings on rebuild. By default, an `analyze` without `--embeddings` preserves them. | | `--pdg` | Build the program-dependence layers used by `explain` and `pdg_query` (taint, CDG, and REACHING_DEF). | +| `--spring-actuator ` | Import opt-in Spring Boot Actuator mappings, beans, conditions, configprops, and env snapshots. Forces a full rebuild; unsupported with `--watch`. | **When to run:** First time in a project, after major code changes, or when `gitnexus://repo/{name}/context` reports the index is stale. In Claude Code, a PostToolUse hook detects staleness after `git commit` and `git merge` and notifies the agent to run `analyze` — the hook does not run analyze itself, to avoid blocking the agent for up to 120s and risking KuzuDB corruption on timeout. +For Spring runtime enrichment, pass a JSON bundle, one endpoint JSON file, or a directory containing endpoint files. Route evidence is authoritative only when `runtimeConfirmed === true`; `runtimeSource` records provenance and may also accompany `handler-conflict`. Env/configprops values are never persisted. + Use `node .gitnexus/run.cjs analyze --watch` for a long-lived local Git repository. It performs an initial analysis, queues scanner-admitted file changes, and retries intact failed batches with bounded backoff. Watch refreshes update only the graph: they skip AGENTS.md / CLAUDE.md injection and standard skill installation, so run a one-shot `analyze` when those generated files need updating. Watch rejects one-shot or context-output flags including `--force`, embedding flags, `--skills`, `--default-branch`, `--skip-agents-md`, `--skip-skills`, `--no-stats`, `--self-commit`, `--index-only`, and `--skip-git`. It never pulls remotes. Running MCP and `serve` processes periodically check for a published replacement and reopen it without a restart. MCP checks are throttled to once every five seconds, so a tool call before the next check can briefly use the previous index. ### status — Check index freshness diff --git a/gitnexus/src/cli/ai-context.ts b/gitnexus/src/cli/ai-context.ts index cb7b42c60..c7b3a934f 100644 --- a/gitnexus/src/cli/ai-context.ts +++ b/gitnexus/src/cli/ai-context.ts @@ -11,6 +11,7 @@ import path from 'path'; import { fileURLToPath } from 'url'; import { type GeneratedSkillInfo } from './generated-skill.js'; import { STANDARD_SKILL_CATALOG } from './standard-skills.js'; +import { isEnoent } from './editor-targets.js'; import { logger } from '../core/logger.js'; // ESM equivalent of __dirname @@ -42,6 +43,8 @@ export interface AIContextOptions { * "no PDG layer" note, so advertising it on a non-`--pdg` index is noise. */ hasPdg?: boolean; + /** Whether this index includes opt-in Spring Actuator runtime evidence. */ + hasSpringActuator?: boolean; } const GITNEXUS_START_MARKER = ''; @@ -136,6 +139,8 @@ export interface GitNexusContentOptions { * line below — false (default) omits it, so a non-pdg index doesn't advertise * a tool that only returns a "no PDG layer" note. */ hasPdg?: boolean; + /** Whether Route nodes may carry Spring Actuator runtime evidence. */ + hasSpringActuator?: boolean; } export function generateGitNexusContent( @@ -151,6 +156,7 @@ export function generateGitNexusContent( runnerPath = '.gitnexus/run.cjs', defaultBranch = 'main', hasPdg = false, + hasSpringActuator = false, } = opts; const generatedRows = generatedSkills && generatedSkills.length > 0 @@ -226,8 +232,11 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s - **MUST analyze graph changes before committing.** Use \`detect_changes({scope: "all"})\` (MCP) or \`${runner} detect-changes --scope all --repo .\` (CLI fallback). \`partial: true\` or \`truncated: true\` is not a clean check — a zero means unseen, not unaffected; re-run it. For regression review: \`detect_changes({scope: "compare", base_ref: ${JSON.stringify(markdownSafeBranch(defaultBranch))}})\` or \`${runner} detect-changes --scope compare --base-ref ${JSON.stringify(markdownSafeBranch(defaultBranch))} --repo .\`. - MUST warn on HIGH/CRITICAL \`risk\` pre-edit; never use \`riskSharedAxes\` to waive a HIGH/CRITICAL \`risk\` warning. Compare File/symbol: MCP File omits axes; Graph-RAG expands File. - **MUST treat \`risk: UNKNOWN\` as unresolved, not as low.** An empty caller set is not evidence the symbol is unused — it can also mean the callers are not resolvable by the index (plain-object property access, dynamic dispatch, cross-language calls). \`impact\` pairs \`UNKNOWN\` with a \`riskNote\` saying so. Confirm with a text search before treating the symbol as safe to change or delete; do not proceed on the strength of a zero. -- Explore with \`query({search_query: "concept"})\` for process-grouped flows. -- Use \`context({name: "symbolName"})\` for callers, callees, and flows. +- **MUST use \`query({search_query: "concept"})\` for concepts/flows, \`context({name: "symbolName"})\` for a named symbol, or \`impact\` for blast radius, on read-only callers, dependencies, imports, or execution flow.** Graph first; text search only for empty/\`UNKNOWN\`/literals.${ + hasSpringActuator + ? '\n- Spring Actuator runtime evidence is enabled. A Route is authoritative only when `runtimeConfirmed === true`; `runtimeSource` is provenance and may also describe conflicts. Snapshot values are never persisted.' + : '' + } - For security review, \`explain({target: "fileOrSymbol"})\` lists taint findings (source→sink flows; needs \`analyze --pdg\`).${ hasPdg ? `\n- For control/data dependence, \`pdg_query({mode: "controls", target: "fileOrSymbol"})\` answers "under what condition does X run?" (CDG, incl. guard clauses) and \`pdg_query({mode: "flows", target, variable})\` traces "where does variable Y flow?" (REACHING_DEF). \`--pdg\` layer.` @@ -432,17 +441,84 @@ export async function shouldMirrorSkillsToAgents(repoPath: string): Promise { + try { + return await fs.readFile(filePath, 'utf-8'); + } catch (err) { + if (isEnoent(err)) return null; + throw err; + } +} + +function skillBytesDiverge(existing: string | null, bundled: string): boolean { + return existing !== null && existing !== bundled; +} + +/** Write bundled skill bytes unless an existing file already differs. */ +async function writeSkillUnlessDivergent(filePath: string, content: string): Promise { + const existing = await readUtf8IfPresent(filePath); + if (skillBytesDiverge(existing, content)) { + logger.warn(`Preserved customized skill ${filePath}; ${SKILL_PRESERVE_HINT}.`); + return true; + } + await fs.mkdir(path.dirname(filePath), { recursive: true }); + await fs.writeFile(filePath, content, 'utf-8'); + return false; +} + +async function inspectLegacySkillDir( + legacyDir: string, +): Promise<{ nestedExisting: string | null; hasSiblings: boolean } | null> { + let entries: string[]; + try { + entries = await fs.readdir(legacyDir); + } catch (err) { + if (isEnoent(err)) return null; + throw err; + } + const nestedExisting = entries.includes('SKILL.md') + ? await fs.readFile(path.join(legacyDir, 'SKILL.md'), 'utf-8') + : null; + return { + nestedExisting, + hasSiblings: entries.some((entry) => entry !== 'SKILL.md'), + }; +} + +function formatSkillInstallLine( + prefix: string, + total: number, + preserved: number, + allWrittenSuffix: string, + partialSuffix: string, +): string { + if (preserved > 0) { + return `${prefix} (${total - preserved} written, ${preserved} ${partialSuffix})`; + } + return `${prefix} (${total} ${allWrittenSuffix})`; +} + /** * Install GitNexus skills as direct children of .claude/skills/ * Works natively with Claude Code, Cursor, and GitHub Copilot. * Mirrored to .agents/skills/ when .agents/ exists. */ -async function installSkills( - repoPath: string, -): Promise<{ skills: string[]; agentsMirror: boolean }> { +async function installSkills(repoPath: string): Promise<{ + skills: string[]; + agentsMirror: boolean; + claudePreserved: number; + agentsPreserved: number; + legacyPreserved: number; +}> { const skillsDir = path.join(repoPath, '.claude', 'skills'); const legacySkillsDir = path.join(skillsDir, 'gitnexus'); const installedSkills: string[] = []; + let claudePreserved = 0; + let agentsPreserved = 0; + let legacyPreserved = 0; const agentsMirror = await shouldMirrorSkillsToAgents(repoPath); for (const skill of STANDARD_SKILL_CATALOG.filter( @@ -452,9 +528,6 @@ async function installSkills( const skillPath = path.join(skillDir, 'SKILL.md'); try { - // Create skill directory - await fs.mkdir(skillDir, { recursive: true }); - // Try to read from package skills directory const packageSkillPath = path.join(__dirname, '..', '..', 'skills', `${skill.name}.md`); let skillContent: string; @@ -476,14 +549,13 @@ Use GitNexus tools to accomplish this task. `; } - await fs.writeFile(skillPath, skillContent, 'utf-8'); + if (await writeSkillUnlessDivergent(skillPath, skillContent)) claudePreserved += 1; // Mirror to .agents/skills/ for agents that read repo-local skills if (agentsMirror) { try { - const agentsSkillDir = path.join(repoPath, '.agents', 'skills', skill.name); - await fs.mkdir(agentsSkillDir, { recursive: true }); - await fs.writeFile(path.join(agentsSkillDir, 'SKILL.md'), skillContent, 'utf-8'); + const agentsSkillPath = path.join(repoPath, '.agents', 'skills', skill.name, 'SKILL.md'); + if (await writeSkillUnlessDivergent(agentsSkillPath, skillContent)) agentsPreserved += 1; } catch (err) { logger.warn({ err }, `Warning: Could not mirror skill ${skill.name} to .agents/skills:`); } @@ -495,7 +567,20 @@ Use GitNexus tools to accomplish this task. // deep. Remove only the child owned by this installer; unknown siblings // under the legacy grouping directory may be user-authored and survive. try { - await fs.rm(path.join(legacySkillsDir, skill.name), { recursive: true, force: true }); + const legacyDir = path.join(legacySkillsDir, skill.name); + const nestedSkill = path.join(legacyDir, 'SKILL.md'); + const leftover = await inspectLegacySkillDir(legacyDir); + if (leftover !== null && skillBytesDiverge(leftover.nestedExisting, skillContent)) { + logger.warn(`Preserved customized skill ${nestedSkill}; ${SKILL_PRESERVE_HINT}.`); + legacyPreserved += 1; + } else if (leftover?.hasSiblings) { + logger.warn( + `Preserved legacy skill directory ${legacyDir} because it contains operator-owned files.`, + ); + legacyPreserved += 1; + } else if (leftover !== null) { + await fs.rm(legacyDir, { recursive: true, force: true }); + } } catch (err) { logger.warn({ err }, `Warning: Could not remove legacy skill ${skill.name}:`); } @@ -505,7 +590,13 @@ Use GitNexus tools to accomplish this task. } } - return { skills: installedSkills, agentsMirror }; + return { + skills: installedSkills, + agentsMirror, + claudePreserved, + agentsPreserved, + legacyPreserved, + }; } /** @@ -551,6 +642,7 @@ export async function generateAIContextFiles( runnerPath, defaultBranch: options?.defaultBranch ?? 'main', hasPdg: options?.hasPdg ?? false, + hasSpringActuator: options?.hasSpringActuator ?? false, }); const createdFiles: string[] = []; @@ -583,12 +675,37 @@ export async function generateAIContextFiles( // Install standard skills directly under .claude/skills/ (unless --skip-skills) if (!options?.skipSkills) { - const { skills: installedSkills, agentsMirror } = await installSkills(repoPath); + const { + skills: installedSkills, + agentsMirror, + claudePreserved, + agentsPreserved, + legacyPreserved, + } = await installSkills(repoPath); if (installedSkills.length > 0) { - createdFiles.push(`.claude/skills/gitnexus-*/ (${installedSkills.length} skills)`); + createdFiles.push( + formatSkillInstallLine( + '.claude/skills/gitnexus-*/', + installedSkills.length, + claudePreserved, + 'skills', + 'preserved', + ), + ); if (agentsMirror) { createdFiles.push( - `.agents/skills/gitnexus-*/ (${installedSkills.length} skills mirrored for .agents)`, + formatSkillInstallLine( + '.agents/skills/gitnexus-*/', + installedSkills.length, + agentsPreserved, + 'skills mirrored for .agents', + 'preserved for .agents', + ), + ); + } + if (legacyPreserved > 0) { + createdFiles.push( + `.claude/skills/gitnexus// (legacy directories preserved: ${legacyPreserved})`, ); } } diff --git a/gitnexus/src/cli/analyze-config.ts b/gitnexus/src/cli/analyze-config.ts index 64b7c1573..3d040fac1 100644 --- a/gitnexus/src/cli/analyze-config.ts +++ b/gitnexus/src/cli/analyze-config.ts @@ -60,7 +60,8 @@ type ValueKind = | 'string-array' | 'numeric-string' | 'embeddings' - | 'branch'; + | 'branch' + | 'path'; interface KeySpec { /** The `AnalyzeOptions` field this config key normalizes into. */ @@ -108,6 +109,9 @@ const KEY_SPECS: Record = { // built-in convention set, is otherwise invisible to route_map consumers. // Listing it here adds it to the cross-file consumer scan. fetchWrappers: { target: 'fetchWrappers', kind: 'string-array' }, + // Explicit local Actuator snapshot input (#2418). The path itself is safe in + // project config; payload contents are never copied into the graph wholesale. + springActuator: { target: 'springActuator', kind: 'path' }, // Auth token AND dims are intentionally CLI/env-only — no embeddingAuthToken // or embeddingDims key here: // - the token keeps secrets out of a committed .gitnexusrc; @@ -226,6 +230,17 @@ const normalizeValue = (kind: ValueKind, value: unknown, key: string): unknown = throw new GitNexusRcError(`${source} must be a string branch name.`); } return validateBranchName(value, source); + case 'path': { + if (typeof value !== 'string') { + throw new GitNexusRcError(`${source} must be a file or directory path.`); + } + const trimmed = value.trim(); + if (!trimmed) { + throw new GitNexusRcError(`${source} must not be empty.`); + } + assertNoHiddenChars(trimmed, source); + return trimmed; + } case 'string': { if (typeof value !== 'string') { throw new GitNexusRcError(`${source} must be a string.`); diff --git a/gitnexus/src/cli/analyze-options.ts b/gitnexus/src/cli/analyze-options.ts index 460218954..a749589f1 100644 --- a/gitnexus/src/cli/analyze-options.ts +++ b/gitnexus/src/cli/analyze-options.ts @@ -124,6 +124,11 @@ export interface AnalyzeOptions { * outside the built-in convention still produces `route_map` consumers. */ fetchWrappers?: string[]; + /** + * Explicit local Spring Boot Actuator snapshot input (#2418). Accepts a JSON + * bundle or a directory containing endpoint JSON files. Disabled by default. + */ + springActuator?: string; /** OpenAI-compatible embeddings base URL (incl. /v1). Overrides GITNEXUS_EMBEDDING_URL. */ embeddingBaseUrl?: string; /** Embedding model name. Overrides GITNEXUS_EMBEDDING_MODEL. */ diff --git a/gitnexus/src/cli/analyze.ts b/gitnexus/src/cli/analyze.ts index 2719b8f40..1fa987320 100644 --- a/gitnexus/src/cli/analyze.ts +++ b/gitnexus/src/cli/analyze.ts @@ -668,6 +668,8 @@ const ANALYZE_CLI_ENV_KEYS = [ 'GITNEXUS_EMBEDDING_SUB_BATCH_SIZE', 'GITNEXUS_EMBEDDING_DEVICE', 'GITNEXUS_ANALYZE_PROGRESS_ACTIVE', + 'GITNEXUS_ANALYZER_IDENTITY_IN_PROCESS_GUARDS', + 'GITNEXUS_RESOLVE_DEF_GRAPH_ID_MEMO', 'GITNEXUS_EMBEDDING_URL', 'GITNEXUS_EMBEDDING_MODEL', 'GITNEXUS_EMBEDDING_API_KEY', @@ -1372,6 +1374,7 @@ const analyzeCommandImpl = async ( // Extra fetch-wrapper names from `.gitnexusrc` (#1589/#1852 residual); // forwarded to the routes phase consumer scan. fetchWrappers: options.fetchWrappers, + springActuatorPath: options.springActuator, // The CLI always process.exit()s after this returns (success path at the // end of analyzeCommandImpl, error/interrupt paths via process.exit too), // so the finalize close skips the native conn/db close — it can double-free @@ -1530,6 +1533,7 @@ const analyzeCommandImpl = async ( // exercised on the `--skills` path by analyze-no-stats-bridge.test.ts. noStats: options.stats === false, hasPdg: options.pdg === true, + hasSpringActuator: options.springActuator !== undefined, }, ); } diff --git a/gitnexus/src/cli/i18n/en.ts b/gitnexus/src/cli/i18n/en.ts index 698e27d4e..aba8e78f0 100644 --- a/gitnexus/src/cli/i18n/en.ts +++ b/gitnexus/src/cli/i18n/en.ts @@ -201,7 +201,7 @@ export const en = { 'help.option.analyze.skills': 'Generate repo-specific skill files from detected communities (no-op when --index-only is also set).', 'help.option.analyze.skipAgentsMd': - 'Skip updating the gitnexus section in AGENTS.md and CLAUDE.md', + 'Skip updating the gitnexus section in AGENTS.md and CLAUDE.md. Does not skip standard skills in .claude/skills or .agents/skills; use --skip-skills for those. Community skills from --skills are unaffected.', 'help.option.analyze.noStats': 'Omit volatile file/symbol counts from AGENTS.md and CLAUDE.md', 'help.option.analyze.selfCommit': 'Auto-commit AGENTS.md/CLAUDE.md changes after analyze (opt-in, off by default). Scoped to only those two files (never `git add -A`); no-ops if neither exists, neither changed, or the repo has no git identity configured.', @@ -238,7 +238,7 @@ export const en = { 'help.option.mcp.host': 'HTTP bind address (only with --http). Default: 127.0.0.1 (loopback). Use 0.0.0.0 to expose to all interfaces.', 'help.option.mcp.authToken': - 'Require this bearer token in the Authorization header (only with --http); may also be set via the GITNEXUS_MCP_AUTH_TOKEN env var. Required for a non-loopback bind (--host 0.0.0.0/::), which otherwise refuses to start.', + "Require this bearer token in the Authorization header (only with --http); may also be set via the GITNEXUS_MCP_AUTH_TOKEN env var, which also enables MCP Bearer auth on gitnexus serve's /api/mcp route. Required for a non-loopback bind (--host 0.0.0.0/::), which otherwise refuses to start.", 'help.option.force.confirmation': 'Skip confirmation prompt', 'help.option.uninstall.force': 'Apply the changes (default is a dry-run preview)', 'help.option.clean.all': 'Clean all indexed repos', diff --git a/gitnexus/src/cli/i18n/zh-CN.ts b/gitnexus/src/cli/i18n/zh-CN.ts index 24a6a2ad6..5f3b52040 100644 --- a/gitnexus/src/cli/i18n/zh-CN.ts +++ b/gitnexus/src/cli/i18n/zh-CN.ts @@ -188,7 +188,8 @@ export const zhCN = { '重建时删除现有嵌入。默认情况下,未传 `--embeddings` 的 `analyze` 会保留索引中已有嵌入。', 'help.option.analyze.skills': '根据检测到的社区生成仓库专属 skill 文件(同时设置 --index-only 时无效)。', - 'help.option.analyze.skipAgentsMd': '跳过更新 AGENTS.md 和 CLAUDE.md 中的 gitnexus 区块', + 'help.option.analyze.skipAgentsMd': + '跳过更新 AGENTS.md 和 CLAUDE.md 中的 gitnexus 区块。不会跳过 .claude/skills 或 .agents/skills 下的标准 skill;如需跳过那些请使用 --skip-skills。--skills 生成的社区 skill 不受影响。', 'help.option.analyze.noStats': '从 AGENTS.md 和 CLAUDE.md 中省略易变的文件/符号计数', 'help.option.analyze.selfCommit': '在 analyze 后自动提交 AGENTS.md/CLAUDE.md 的变更(默认关闭,需显式开启)。仅限这两个文件(绝不使用 `git add -A`);若两者均不存在、均未变更,或仓库未配置 git 身份,则不执行任何操作。', @@ -222,7 +223,7 @@ export const zhCN = { 'help.option.mcp.host': 'HTTP 绑定地址(仅与 --http 搭配使用)。默认:127.0.0.1(回环)。使用 0.0.0.0 向所有接口开放。', 'help.option.mcp.authToken': - '要求 Authorization 头携带此 Bearer Token(仅与 --http 搭配使用);也可通过 GITNEXUS_MCP_AUTH_TOKEN 环境变量设置。非回环绑定(--host 0.0.0.0/::)时必填,否则拒绝启动。', + '要求 Authorization 头携带此 Bearer Token(仅与 --http 搭配使用);也可通过 GITNEXUS_MCP_AUTH_TOKEN 环境变量设置,该变量同时为 gitnexus serve 的 /api/mcp 路由启用 MCP Bearer 认证。非回环绑定(--host 0.0.0.0/::)时必填,否则拒绝启动。', 'help.option.force.confirmation': '跳过确认提示', 'help.option.uninstall.force': '应用更改(默认仅为预演预览)', 'help.option.clean.all': '清理所有已索引仓库', diff --git a/gitnexus/src/cli/index.ts b/gitnexus/src/cli/index.ts index 8296787af..e53312d62 100644 --- a/gitnexus/src/cli/index.ts +++ b/gitnexus/src/cli/index.ts @@ -76,7 +76,10 @@ program 'Generate repo-specific skill files from detected communities ' + '(no-op when --index-only is also set).', ) - .option('--skip-agents-md', 'Skip updating the gitnexus section in AGENTS.md and CLAUDE.md') + .option( + '--skip-agents-md', + 'Skip updating the gitnexus section in AGENTS.md and CLAUDE.md. Does not skip standard skills in .claude/skills or .agents/skills; use --skip-skills for those. Community skills from --skills are unaffected.', + ) .option( '--pdg', 'Build the control-flow-graph / PDG substrate (BasicBlock nodes + CFG edges) ' + @@ -139,6 +142,11 @@ program '--workers ', 'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.', ) + .option( + '--spring-actuator ', + 'Import local Spring Boot Actuator JSON snapshots (mappings, beans, conditions, ' + + 'configprops, env). Explicit opt-in; disabled by default.', + ) .option('--embedding-threads ', 'Limit local ONNX embedding CPU threads') .option('--embedding-batch-size ', 'Number of nodes per embedding batch') .option('--embedding-sub-batch-size ', 'Number of chunks per embedding model call') @@ -245,7 +253,7 @@ program ) .option( '--auth-token ', - 'Require this bearer token in the Authorization header (only with --http); may also be set via the GITNEXUS_MCP_AUTH_TOKEN env var. Required for a non-loopback bind (--host 0.0.0.0/::), which otherwise refuses to start.', + "Require this bearer token in the Authorization header (only with --http); may also be set via the GITNEXUS_MCP_AUTH_TOKEN env var, which also enables MCP Bearer auth on gitnexus serve's /api/mcp route. Required for a non-loopback bind (--host 0.0.0.0/::), which otherwise refuses to start.", ) .action(createLbugLazyAction(() => import('./mcp.js'), 'mcpCommand')); diff --git a/gitnexus/src/cli/setup.ts b/gitnexus/src/cli/setup.ts index 003e548a9..6dd0f9b85 100644 --- a/gitnexus/src/cli/setup.ts +++ b/gitnexus/src/cli/setup.ts @@ -1090,14 +1090,30 @@ async function installSkillsTo(targetDir: string): Promise { const skillDir = path.join(targetDir, skillName); try { - if (source.isDirectory) { - const dirSource = path.join(skillsRoot, skillName); - await copyDirRecursive(dirSource, skillDir); - } else { - const flatSource = path.join(skillsRoot, `${skillName}.md`); - const content = await fs.readFile(flatSource, 'utf-8'); + const sourceSkillPath = source.isDirectory + ? path.join(skillsRoot, skillName, 'SKILL.md') + : path.join(skillsRoot, `${skillName}.md`); + const destinationSkillPath = path.join(skillDir, 'SKILL.md'); + const [sourceSkillContent, destinationSkillContent] = await Promise.all([ + fs.readFile(sourceSkillPath, 'utf-8'), + fs.readFile(destinationSkillPath, 'utf-8').catch((err) => { + if (!isEnoent(err)) throw err; + return null; + }), + ]); + + const preserved = + destinationSkillContent !== null && destinationSkillContent !== sourceSkillContent; + if (preserved && !source.isDirectory) { + console.log( + `[gitnexus] preserved customized skill ${destinationSkillPath}; ` + + 'delete the file and rerun setup to refresh it.', + ); + } else if (source.isDirectory) { + await copyDirRecursive(path.join(skillsRoot, skillName), skillDir); + } else if (!preserved) { await fs.mkdir(skillDir, { recursive: true }); - await fs.writeFile(path.join(skillDir, 'SKILL.md'), content, 'utf-8'); + await fs.writeFile(destinationSkillPath, sourceSkillContent, 'utf-8'); } // A directory superseded by a shipped rename is warned about, never @@ -1113,7 +1129,7 @@ async function installSkillsTo(targetDir: string): Promise { ); } } - installed.push(skillName); + if (!preserved) installed.push(skillName); } catch { // Source skill not found — skip } @@ -1133,9 +1149,23 @@ async function copyDirRecursive(src: string, dest: string): Promise { const destPath = path.join(dest, entry.name); if (entry.isDirectory()) { await copyDirRecursive(srcPath, destPath); - } else { - await fs.copyFile(srcPath, destPath); + continue; } + const [srcBuf, destBuf] = await Promise.all([ + fs.readFile(srcPath), + fs.readFile(destPath).catch((err) => { + if (!isEnoent(err)) throw err; + return null; + }), + ]); + if (destBuf !== null && !destBuf.equals(srcBuf)) { + console.log( + `[gitnexus] preserved customized skill ${destPath}; ` + + 'delete the file and rerun setup to refresh it.', + ); + continue; + } + await fs.writeFile(destPath, srcBuf); } } diff --git a/gitnexus/src/cli/watch.ts b/gitnexus/src/cli/watch.ts index 3cbec636f..13212a56c 100644 --- a/gitnexus/src/cli/watch.ts +++ b/gitnexus/src/cli/watch.ts @@ -114,6 +114,7 @@ export async function resolveWatchOptions( ['--self-commit', cli.selfCommit], ['--index-only', cli.indexOnly], ['--skip-git', cli.skipGit], + ['--spring-actuator', cli.springActuator], ['walCheckpointThreshold', cli.walCheckpointThreshold], ['embeddingThreads', cli.embeddingThreads], ['embeddingBatchSize', cli.embeddingBatchSize], @@ -137,6 +138,7 @@ export async function resolveWatchOptions( ['skipAgentsMd', config.skipAgentsMd !== undefined], ['skipSkills', config.skipSkills !== undefined], ['stats', config.stats !== undefined], + ['springActuator', config.springActuator], ['walCheckpointThreshold', config.walCheckpointThreshold], ['embeddingThreads', config.embeddingThreads], ['embeddingBatchSize', config.embeddingBatchSize], diff --git a/gitnexus/src/core/analyzer-identity.ts b/gitnexus/src/core/analyzer-identity.ts index 6bc80848f..628f27716 100644 --- a/gitnexus/src/core/analyzer-identity.ts +++ b/gitnexus/src/core/analyzer-identity.ts @@ -15,6 +15,7 @@ */ import { + accessSync, closeSync, constants as fsConstants, existsSync, @@ -38,6 +39,7 @@ import { spawnSync } from 'node:child_process'; import { isDeepStrictEqual } from 'node:util'; import os from 'node:os'; import path from 'node:path'; +import { parseTruthyEnv } from './ingestion/utils/env.js'; import { fileURLToPath } from 'node:url'; import type { AnalyzerRunnerIdentity } from '../storage/repo-manager.js'; @@ -2335,8 +2337,30 @@ function snapshotCacheGuardDirect(request: CacheGuardRequest): CacheGuardResult } } -function snapshotCacheGuards(requests: CacheGuardRequest[]): CacheGuardResult[] { +function installTreeUnwritable(packageRoot: string, buildRoot: string): boolean { + for (const dir of [packageRoot, buildRoot]) { + try { + accessSync(dir, fsConstants.W_OK); + } catch (err) { + const code = (err as NodeJS.ErrnoException).code; + if (code === 'EACCES' || code === 'EROFS') return true; + } + } + return false; +} + +function snapshotCacheGuards( + requests: CacheGuardRequest[], + packageRoot: string, + buildRoot: string, +): CacheGuardResult[] { if (requests.length < 128) return requests.map(snapshotCacheGuardDirect); + if ( + parseTruthyEnv(process.env.GITNEXUS_ANALYZER_IDENTITY_IN_PROCESS_GUARDS) || + installTreeUnwritable(packageRoot, buildRoot) + ) { + return requests.map(snapshotCacheGuardDirect); + } try { const probe = spawnSync( process.execPath, @@ -2468,7 +2492,7 @@ function validateIdentityCache( return { mode, absolutePath }; }); options.onCacheValidationPass?.({ guardCount: requests.length }); - const actual = snapshotCacheGuards(requests); + const actual = snapshotCacheGuards(requests, cache.packageRoot, cache.buildRoot); const mismatch = actual.findIndex( (result, index) => !isDeepStrictEqual(result, entries[index][1]), ); diff --git a/gitnexus/src/core/group/cross-impact.ts b/gitnexus/src/core/group/cross-impact.ts index fd2ae0779..da9ba0de4 100644 --- a/gitnexus/src/core/group/cross-impact.ts +++ b/gitnexus/src/core/group/cross-impact.ts @@ -159,6 +159,15 @@ export function validateGroupImpactParams(params: Record): name: string; repoPath: string; target: string; + // Target selectors, same names/semantics as the single-repo impact tool + // (target_uid = zero-ambiguity lookup that wins over the name; + // file_path/kind narrow a name shared by same-named symbols). Threading + // them through HERE is what makes the MCP boundary's forwarding live — + // dropping them at this boundary silently re-broke the group-mode + // disambiguation loop once already. + target_uid?: string; + file_path?: string; + kind?: string; direction: 'upstream' | 'downstream'; maxDepth: number; crossDepth: number; @@ -173,11 +182,21 @@ export function validateGroupImpactParams(params: Record): | { ok: false; error: string } { const name = String(params.name ?? '').trim(); const repoPath = String(params.repo ?? '').trim(); - const target = String(params.target ?? '').trim(); + // Optional string, same helper shape as cross-trace's `str()`: empty/blank + // counts as absent so `target_uid: ''` degrades to the name lookup rather + // than a zero-ambiguity lookup of the empty uid. Parsed before the required + // check so UID-only callers (MCP impact schema requires `direction`, not + // `target`) are accepted. + const str = (v: unknown): string | undefined => + typeof v === 'string' && v.trim() !== '' ? v : undefined; + const targetName = String(params.target ?? '').trim(); + const target_uidEarly = str(params.target_uid); if (!name) return { ok: false, error: 'name is required' }; if (!repoPath) return { ok: false, error: 'repo is required (group repo path, e.g. app/backend)' }; - if (!target) return { ok: false, error: 'target is required' }; + if (!targetName && !target_uidEarly) + return { ok: false, error: 'target or target_uid is required' }; + const target = targetName || target_uidEarly!; if ( params.service !== undefined && params.service !== null && @@ -205,6 +224,10 @@ export function validateGroupImpactParams(params: Record): const service = normalizeServicePrefix(params.service); const subgroup = typeof params.subgroup === 'string' ? params.subgroup : undefined; + const target_uid = target_uidEarly; + const file_path = str(params.file_path); + const kind = str(params.kind); + // Clamp at the validate boundary so the downstream `deadline` (line // ~366) and `safeLocalImpact`'s `setTimeout` both see a single // bounded value. Without this, the outer deadline budgeted Phase-2 @@ -224,6 +247,9 @@ export function validateGroupImpactParams(params: Record): name, repoPath, target, + target_uid, + file_path, + kind, direction, maxDepth, crossDepth, @@ -579,6 +605,9 @@ export async function runGroupImpact( name, repoPath, target, + target_uid, + file_path, + kind, direction, maxDepth, crossDepth: _crossDepth, @@ -606,6 +635,14 @@ export async function runGroupImpact( const impactParams: Parameters[1] = { target, + // Selector params pass through to the member repo's impact (the port + // contract in service.ts documents them), so the single-repo tool's + // "re-call with target_uid to disambiguate" loop works unchanged in + // group mode. `undefined` keeps the call shape flat — same convention + // as the relationTypes line below. + target_uid, + file_path, + kind, direction, maxDepth, relationTypes: relationTypes && relationTypes.length > 0 ? relationTypes : undefined, diff --git a/gitnexus/src/core/group/extractors/http-patterns/java.ts b/gitnexus/src/core/group/extractors/http-patterns/java.ts index 081b71805..8d216c50a 100644 --- a/gitnexus/src/core/group/extractors/http-patterns/java.ts +++ b/gitnexus/src/core/group/extractors/http-patterns/java.ts @@ -29,10 +29,13 @@ import { EXCHANGE_CONFIDENCE, } from './spring-consumer-shared.js'; import { + expandJavaWildcardStaticImports, extractJavaModuleConstants, foldJavaOperands, isJavaConstantFile, parseJavaConstOperands, + prepareJavaRouteConstants, + type JavaConstantIndex, type RepoConstants, } from '../../../ingestion/route-extractors/java-const-resolver.js'; import { @@ -917,7 +920,12 @@ export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = { const tree = args.parseSource(args.parser, src); if (!tree) continue; const mc = extractJavaModuleConstants(tree); - if (mc.literals.size > 0 || mc.exprs.size > 0 || mc.imports.size > 0) { + if ( + mc.literals.size > 0 || + mc.exprs.size > 0 || + mc.imports.size > 0 || + (mc.wildcardImports?.length ?? 0) > 0 + ) { constants.set(rel, mc); } } catch { @@ -927,11 +935,20 @@ export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = { continue; } } - return { constants }; + // On-demand static imports (`import static a.b.C.*`) were recorded as + // pending class FQNs during extraction; materialize their bare-name + // bindings now that the whole map exists. A wildcard's target is itself + // a constants file, so it is necessarily a map entry — anything else + // degrades to the fold's skip floor. In-place: each entry is owned by + // this map, and every file is expanded exactly once. + const constantIndex = prepareJavaRouteConstants(constants); + return { constants, constantIndex }; }, scan(tree, repoContext, fileRel) { const out: HttpDetection[] = []; - const javaCtx = repoContext as { constants: RepoConstants } | undefined; + const javaCtx = repoContext as + | { constants: RepoConstants; constantIndex: JavaConstantIndex } + | undefined; // ─── Spring providers + OpenFeign consumers (one query pass) ──── // `scanRouteAnnotations` resolves every route-defining annotation — @@ -966,8 +983,12 @@ export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = { if (javaCtx.constants.has(fileRel)) return foldConstants; try { const mc = extractJavaModuleConstants(tree); - if (mc.imports.size > 0) { + // A file carrying ONLY wildcard static imports has an empty import + // table pre-expansion — overlay it too, then materialize the promised + // bindings against the repo map before it becomes a fold target. + if (mc.imports.size > 0 || (mc.wildcardImports?.length ?? 0) > 0) { const merged = new Map(javaCtx.constants); + expandJavaWildcardStaticImports(mc, fileRel, merged, javaCtx.constantIndex); merged.set(fileRel, mc); foldConstants = merged; } diff --git a/gitnexus/src/core/group/extractors/http-patterns/kotlin.ts b/gitnexus/src/core/group/extractors/http-patterns/kotlin.ts index d9a7c1705..ec695a9b6 100644 --- a/gitnexus/src/core/group/extractors/http-patterns/kotlin.ts +++ b/gitnexus/src/core/group/extractors/http-patterns/kotlin.ts @@ -1328,6 +1328,7 @@ function buildKotlinPlugin(language: unknown): HttpLanguagePlugin { mc.literals.size > 0 || mc.exprs.size > 0 || mc.imports.size > 0 || + (mc.wildcardImports?.length ?? 0) > 0 || unfoldableDeclarationsOf(mc).size > 0 ) { // POSIX key (see `normalizeRel`); `readFile` above got the raw `rel`. @@ -1373,6 +1374,7 @@ function buildKotlinPlugin(language: unknown): HttpLanguagePlugin { mc.literals.size > 0 || mc.exprs.size > 0 || mc.imports.size > 0 || + (mc.wildcardImports?.length ?? 0) > 0 || unfoldableDeclarationsOf(mc).size > 0 ) { foldIndex = overlayKotlinConstantIndex(kotlinCtx.index, fileKey, mc); diff --git a/gitnexus/src/core/group/extractors/http-patterns/python.ts b/gitnexus/src/core/group/extractors/http-patterns/python.ts index 7d8bd99af..7a0fa2f7f 100644 --- a/gitnexus/src/core/group/extractors/http-patterns/python.ts +++ b/gitnexus/src/core/group/extractors/http-patterns/python.ts @@ -1137,13 +1137,17 @@ export const PYTHON_HTTP_PLUGIN: HttpLanguagePlugin = { name: 'python-http', language: Python, // routeCoverage intentionally LEFT at the default 'partial' (#2138 Part 2). - // It would be a no-op even if set to 'complete': FastAPI decorator routes set - // no handlerName (generic worker path) and Django sets methodName: null, so no - // Python file ever resolves a handlerSymbolId and none would be parse-skipped. - // Declaring 'complete' now is only a latent trap for the moment a follow-up - // gives FastAPI routes a handlerName. `hasConsumerSignals` is kept (and is a - // true superset of scan()'s consumer shapes) so the precondition already holds - // when Python is later flipped to 'complete'. + // 'complete' is now an active data-loss risk rather than a no-op: FastAPI and + // Flask decorator routes do carry a handlerName (Python's + // `decoratorRouteHandlerName` hook reads the `decorated_definition`), so their + // files can resolve every handlerSymbolId and become parse-skip candidates. + // The flag asserts more than that — it asserts ingestion emits a Route node + // for EVERY provider route this scan() finds, and it does not: Flask's + // imperative `add_url_rule('/p', view_func=handler)` registration below has no + // ingestion counterpart, so skipping a file that mixes it with resolved + // decorator routes would drop those providers. `hasConsumerSignals` is kept + // (and is a true superset of scan()'s consumer shapes) so the consumer half of + // the precondition already holds once provider parity is closed. // Consumer signals scan() can detect: `requests.`/`requests.request`, // `httpx` (sync/async client), the `uri=`/`url=` keyword/variable wrapper // calls, plus aiohttp/urllib. Conservative — over-matching only costs a parse. diff --git a/gitnexus/src/core/group/extractors/http-route-extractor.ts b/gitnexus/src/core/group/extractors/http-route-extractor.ts index 6aa4c8476..5def6a9a9 100644 --- a/gitnexus/src/core/group/extractors/http-route-extractor.ts +++ b/gitnexus/src/core/group/extractors/http-route-extractor.ts @@ -621,6 +621,7 @@ export class HttpRouteExtractor implements ContractExtractor { dbExecutor, getDetections, resolveDetectionSymbol, + loadFileSymbols, coveredFiles, ) : []; @@ -690,6 +691,7 @@ export class HttpRouteExtractor implements ContractExtractor { db: CypherExecutor, getDetections: (rel: string) => Promise, resolveSymbol: (filePath: string, d: HttpDetection) => Promise, + loadFileSymbols: (filePath: string) => Promise[]>, coveredFiles?: Set, ): Promise { const out: ExtractedContract[] = []; @@ -749,15 +751,11 @@ export class HttpRouteExtractor implements ContractExtractor { if (!method) method = 'GET'; symbolUid = handlerSymbolId; if (filePath) { - try { - const syms = await db(CONTAINING_QUERY, { filePath }); - const hit = syms.find((s) => String(s.uid ?? s[0]) === handlerSymbolId); - if (hit) { - symbolName = String(hit.name ?? hit[1]) || symbolName; - symPath = String(hit.filePath ?? hit[2]) || filePath; - } - } catch { - /* keep the authoritative uid + basename fallback */ + const syms = await loadFileSymbols(filePath); + const hit = syms.find((s) => String(s.uid ?? s[0]) === handlerSymbolId); + if (hit) { + symbolName = String(hit.name ?? hit[1]) || symbolName; + symPath = String(hit.filePath ?? hit[2]) || filePath; } } } else { diff --git a/gitnexus/src/core/group/extractors/java-workspace-extractor.ts b/gitnexus/src/core/group/extractors/java-workspace-extractor.ts index b6beed71c..fcbb06507 100644 --- a/gitnexus/src/core/group/extractors/java-workspace-extractor.ts +++ b/gitnexus/src/core/group/extractors/java-workspace-extractor.ts @@ -1,5 +1,6 @@ import fs from 'node:fs/promises'; import path from 'node:path'; +import { XMLParser } from 'fast-xml-parser'; import type { CypherExecutor } from '../contract-extractor.js'; import type { GroupManifestLink, ContractRole } from '../types.js'; import { shouldIgnorePath, loadIgnoreRules } from '../../../config/ignore-service.js'; @@ -20,6 +21,21 @@ interface ImportedSymbol { filePath: string; } +type XmlNode = Record; + +// POMs are static metadata. Parse hierarchy with a real XML parser, but do not +// invoke Maven or resolve the effective model. Properties, profiles, and remote +// parent resolution remain outside this extractor's deterministic boundary. +const pomParser = new XMLParser({ + ignoreAttributes: true, + removeNSPrefix: true, + trimValues: true, + parseTagValue: false, + processEntities: false, + ignoreDeclaration: true, + ignorePiTags: true, +}); + async function parseJavaManifest( repoPath: string, ): Promise<{ groupId: string; artifactId: string; deps: string[] } | null> { @@ -28,14 +44,15 @@ async function parseJavaManifest( const content = await fs.readFile(pomPath, 'utf-8'); return parsePom(content); } catch { - // fall through to Gradle + // Missing pom.xml — fall through to Gradle. } + const gradleSidecars = await readGradleSidecars(repoPath); for (const name of ['build.gradle.kts', 'build.gradle']) { const gradlePath = path.join(repoPath, name); try { const content = await fs.readFile(gradlePath, 'utf-8'); - return parseGradle(content, repoPath); + return parseGradle(content, repoPath, gradleSidecars); } catch { continue; } @@ -44,59 +61,286 @@ async function parseJavaManifest( return null; } -function parsePom(content: string): { groupId: string; artifactId: string; deps: string[] } | null { - const projectGroupMatch = content.match(/]*>[\s\S]*?([^<]+)<\/groupId>/); - const projectArtifactMatch = content.match( - /]*>[\s\S]*?([^<]+)<\/artifactId>/, - ); - if (!projectGroupMatch || !projectArtifactMatch) return null; +interface GradleSidecars { + propertiesGroup?: string; + rootProjectName?: string; + catalogLibraries: Map; + catalogBundles: Map; +} - const groupId = projectGroupMatch[1].trim(); - const artifactId = projectArtifactMatch[1].trim(); +async function readIfPresent(filePath: string): Promise { + try { + return await fs.readFile(filePath, 'utf-8'); + } catch { + return undefined; + } +} - const deps: string[] = []; - const depBlocks = content.matchAll(/\s*([\s\S]*?)<\/dependency>/g); - for (const block of depBlocks) { - const gMatch = block[1].match(/([^<]+)<\/groupId>/); - const aMatch = block[1].match(/([^<]+)<\/artifactId>/); - if (gMatch && aMatch) { - deps.push(`${gMatch[1].trim()}:${aMatch[1].trim()}`); +async function readGradleSidecars(repoPath: string): Promise { + const [properties, settingsKts, settingsGroovy, catalog] = await Promise.all([ + readIfPresent(path.join(repoPath, 'gradle.properties')), + readIfPresent(path.join(repoPath, 'settings.gradle.kts')), + readIfPresent(path.join(repoPath, 'settings.gradle')), + readIfPresent(path.join(repoPath, 'gradle', 'libs.versions.toml')), + ]); + + const sidecars: GradleSidecars = { + catalogLibraries: new Map(), + catalogBundles: new Map(), + }; + + const groupMatch = properties?.match(/(?:^|\n)\s*group\s*=\s*([^\s#]+)/); + if (groupMatch) sidecars.propertiesGroup = groupMatch[1]; + + const settings = settingsKts ?? settingsGroovy; + const nameMatch = settings?.match(/rootProject\.name\s*=\s*['"]([^'"]+)['"]/); + if (nameMatch) sidecars.rootProjectName = nameMatch[1]; + + if (catalog) { + const parsed = parseGradleVersionCatalog(catalog); + sidecars.catalogLibraries = parsed.libraries; + sidecars.catalogBundles = parsed.bundles; + } + + return sidecars; +} + +function catalogAccessors(alias: string): string[] { + const dotted = alias.replace(/[-_]/g, '.'); + const camel = alias.replace(/[-_]+([A-Za-z0-9])/g, (_, char: string) => char.toUpperCase()); + return [...new Set([alias, dotted, camel])]; +} + +function projectAccessorToArtifactId(accessor: string): string { + const last = accessor.split('.').pop()!; + return last.replace(/[A-Z]/g, (char) => `-${char.toLowerCase()}`).replace(/^-/, ''); +} + +function moduleToGa(module: string): string | undefined { + const parts = module.split(':'); + return parts.length >= 2 ? `${parts[0]}:${parts[1]}` : undefined; +} + +function parseInlineTomlTable(rhs: string): Record { + const fields: Record = {}; + for (const match of rhs.matchAll(/([A-Za-z0-9_-]+)\s*=\s*['"]([^'"]+)['"]/g)) { + fields[match[1]] = match[2]; + } + return fields; +} + +/** Default Gradle catalog (`gradle/libs.versions.toml`) — aliases only, no version resolution. */ +function parseGradleVersionCatalog(toml: string): { + libraries: Map; + bundles: Map; +} { + const libraries = new Map(); + const bundles = new Map(); + let section: 'libraries' | 'bundles' | 'other' = 'other'; + + const addLibrary = (alias: string, ga: string) => { + for (const accessor of catalogAccessors(alias)) libraries.set(accessor, ga); + }; + + for (const raw of toml.split(/\r?\n/)) { + const line = raw.replace(/#.*$/, '').trim(); + if (!line) continue; + const header = line.match(/^\[([^\]]+)\]$/); + if (header) { + const name = header[1]; + section = + name === 'libraries' || name.endsWith('.libraries') + ? 'libraries' + : name === 'bundles' || name.endsWith('.bundles') + ? 'bundles' + : 'other'; + continue; + } + + if (section === 'libraries') { + const dottedModule = line.match(/^([A-Za-z0-9._-]+)\.module\s*=\s*['"]([^'"]+)['"]$/); + if (dottedModule) { + const ga = moduleToGa(dottedModule[2]); + if (ga) addLibrary(dottedModule[1], ga); + continue; + } + const assignment = line.match(/^([A-Za-z0-9._-]+)\s*=\s*(.+)$/); + if (!assignment) continue; + const alias = assignment[1]; + const rhs = assignment[2].trim(); + const quoted = rhs.match(/^['"]([^'"]+)['"]$/); + if (quoted) { + const ga = moduleToGa(quoted[1]); + if (ga) addLibrary(alias, ga); + continue; + } + const table = parseInlineTomlTable(rhs); + const ga = table.module + ? moduleToGa(table.module) + : table.group && table.name + ? `${table.group}:${table.name}` + : undefined; + if (ga) addLibrary(alias, ga); + continue; + } + + if (section === 'bundles') { + const assignment = line.match(/^([A-Za-z0-9._-]+)\s*=\s*\[([^\]]*)\]$/); + if (!assignment) continue; + const members = [...assignment[2].matchAll(/['"]([^'"]+)['"]/g)].map((match) => match[1]); + for (const accessor of catalogAccessors(assignment[1])) bundles.set(accessor, members); } } + return { libraries, bundles }; +} + +const GRADLE_GROUP_PATTERNS = [ + /(?:^|[\n{;])\s*(?:rootProject\.)?group\s*=\s*['"]([^'"]+)['"]/, + /(?:^|[\n{;])\s*group\s+['"]([^'"]+)['"]/, +]; + +const GRADLE_COORD_CONFIGS = + 'implementation|api|compileOnly|runtimeOnly|testImplementation|testApi|testCompileOnly|compile|kapt|ksp|commonMainImplementation|commonMainApi'; + +const CATALOG_ALIAS = '([A-Za-z0-9_]+(?:\\.[A-Za-z0-9_]+)*)(?:\\.get\\(\\)|\\.asProvider\\(\\))?'; + +function gradleDepRe(suffix: string): RegExp { + return new RegExp(`(?:${GRADLE_COORD_CONFIGS})\\s*${suffix}`, 'g'); +} + +function parseGradleGroup(content: string): string | undefined { + for (const pattern of GRADLE_GROUP_PATTERNS) { + const match = content.match(pattern); + if (match?.[1]) return match[1]; + } + return undefined; +} + +function asXmlNode(value: unknown): XmlNode | undefined { + return value !== null && typeof value === 'object' && !Array.isArray(value) + ? (value as XmlNode) + : undefined; +} + +function xmlText(value: unknown): string | undefined { + if (typeof value === 'string' || typeof value === 'number') { + const text = String(value).trim(); + return text || undefined; + } + const nested = asXmlNode(value)?.['#text']; + if (nested === undefined) return undefined; + return xmlText(nested); +} + +function xmlChildText(node: XmlNode | undefined, name: string): string | undefined { + return node ? xmlText(node[name]) : undefined; +} + +function asList(value: unknown): unknown[] { + if (value === undefined || value === null) return []; + return Array.isArray(value) ? value : [value]; +} + +/** Direct project dependencies only — not BOM, profiles, or plugin classpath. */ +function collectProjectDependencies(project: XmlNode, deps: string[]): void { + const dependencies = asXmlNode(project.dependencies); + if (!dependencies) return; + for (const dep of asList(dependencies.dependency)) { + const depNode = asXmlNode(dep); + const groupId = xmlChildText(depNode, 'groupId'); + const artifactId = xmlChildText(depNode, 'artifactId'); + if (groupId && artifactId) deps.push(`${groupId}:${artifactId}`); + } +} + +function parsePom(content: string): { groupId: string; artifactId: string; deps: string[] } | null { + let parsed: unknown; + try { + // parseSourceSafe guards tree-sitter's Windows SIGSEGV by switching to a + // chunked input callback above 16 KB; XMLParser only accepts XML text, so + // routing POMs through it silently yields an empty document. + // eslint-disable-next-line gitnexus/require-safe-parse + parsed = pomParser.parse(content); + } catch { + return null; + } + + const project = asXmlNode(asXmlNode(parsed)?.project); + if (!project) return null; + + // Maven inherits groupId from , but artifactId is always the + // project's own direct child and must never fall back to parent.artifactId. + const groupId = + xmlChildText(project, 'groupId') ?? xmlChildText(asXmlNode(project.parent), 'groupId'); + const artifactId = xmlChildText(project, 'artifactId'); + if (!groupId || !artifactId) return null; + + const deps: string[] = []; + collectProjectDependencies(project, deps); return { groupId, artifactId, deps: [...new Set(deps)] }; } function parseGradle( content: string, repoPath: string, + sidecars: GradleSidecars = { catalogLibraries: new Map(), catalogBundles: new Map() }, ): { groupId: string; artifactId: string; deps: string[] } | null { - const groupMatch = content.match(/group\s*=\s*['"]([^'"]+)['"]/); - const dirName = path.basename(repoPath); - const groupId = groupMatch ? groupMatch[1] : ''; + // Static text + default catalog file. Do not execute Gradle. + const groupId = parseGradleGroup(content) ?? sidecars.propertiesGroup ?? ''; if (!groupId) return null; - const artifactId = dirName; + const artifactId = sidecars.rootProjectName ?? path.basename(repoPath); + const { catalogLibraries, catalogBundles } = sidecars; const deps: string[] = []; - // implementation("group:artifact:version") or api("group:artifact:version") - const depMatches = content.matchAll( - /(?:implementation|api|compileOnly|runtimeOnly)\s*\(\s*['"]([^'"]+)['"]\s*\)/g, + const pushCatalogAlias = (alias: string) => { + const ga = catalogLibraries.get(alias); + if (ga) deps.push(ga); + }; + + const namedPattern = gradleDepRe( + `(?:\\(\\s*)?(?:group\\s*=\\s*['"](?[^'"]+)['"]\\s*,\\s*name\\s*=\\s*['"](?[^'"]+)['"]|name\\s*=\\s*['"](?[^'"]+)['"]\\s*,\\s*group\\s*=\\s*['"](?[^'"]+)['"]|group:\\s*['"](?[^'"]+)['"]\\s*,\\s*name:\\s*['"](?[^'"]+)['"]|name:\\s*['"](?[^'"]+)['"]\\s*,\\s*group:\\s*['"](?[^'"]+)['"])`, ); - for (const m of depMatches) { - const parts = m[1].split(':'); - if (parts.length >= 2) { - deps.push(`${parts[0]}:${parts[1]}`); + for (const match of content.matchAll(namedPattern)) { + const group = + match.groups?.group1 ?? match.groups?.group2 ?? match.groups?.group3 ?? match.groups?.group4; + const name = + match.groups?.name1 ?? match.groups?.name2 ?? match.groups?.name3 ?? match.groups?.name4; + if (group && name) deps.push(`${group}:${name}`); + } + + for (const match of content.matchAll( + gradleDepRe(`(?:\\(\\s*)?libs(?:\\.libraries)?\\.(?!bundles\\.|plugins\\.)${CATALOG_ALIAS}`), + )) { + pushCatalogAlias(match[1]); + } + + for (const match of content.matchAll( + gradleDepRe(`(?:\\(\\s*)?libs\\.bundles\\.${CATALOG_ALIAS}`), + )) { + for (const member of catalogBundles.get(match[1]) ?? []) { + for (const accessor of catalogAccessors(member)) pushCatalogAlias(accessor); } } - // implementation(project(":subproject")) - const projDeps = content.matchAll( - /(?:implementation|api)\s*\(\s*project\s*\(\s*['"]([^'"]+)['"]\s*\)\s*\)/g, - ); - for (const m of projDeps) { - const subName = m[1].replace(/^:/, ''); - deps.push(`${groupId}:${subName}`); + for (const match of content.matchAll(gradleDepRe(`\\(\\s*projects\\.([A-Za-z][A-Za-z0-9.]*)`))) { + deps.push(`${groupId}:${projectAccessorToArtifactId(match[1])}`); + } + + for (const match of content.matchAll( + gradleDepRe(`(?:\\(\\s*['"]([^'"]+)['"]\\s*\\)|['"]([^'"]+)['"])`), + )) { + const coord = match[1] ?? match[2]; + if (!coord) continue; + const parts = coord.split(':'); + if (parts.length >= 2) deps.push(`${parts[0]}:${parts[1]}`); + } + + for (const match of content.matchAll( + gradleDepRe(`(?:\\(\\s*)?project\\s*\\(\\s*['"]([^'"]+)['"]\\s*\\)`), + )) { + deps.push(`${groupId}:${match[1].replace(/^:/, '')}`); } return { groupId, artifactId, deps: [...new Set(deps)] }; diff --git a/gitnexus/src/core/group/normalization.ts b/gitnexus/src/core/group/normalization.ts index c99d36850..50102415a 100644 --- a/gitnexus/src/core/group/normalization.ts +++ b/gitnexus/src/core/group/normalization.ts @@ -91,6 +91,61 @@ function crossLinkKey(link: CrossLink): string { ].join('\0'); } +/** + * True when a link endpoint carries no resolved graph symbol — empty + * `symbolUid` or a missing/empty `symbolRef`. + * + * Sync marks a cross-link `degraded: true` when this holds for the PROVIDER + * endpoint (`to`): the contract boundary is proven, but the empty uid can + * never match a Phase-1 impact symbol id, so cross-repo fan-out across the + * link silently yields nothing (the classic case is a provider whose handler + * failed to resolve, leaving `symbolName` degraded to the file name with one + * pseudo-symbol carrying every route in that file). Consumer-side (`from`) + * emptiness is deliberately NOT degraded — several extractors (topics, grpc) + * legitimately emit consumer contracts without a per-call symbol, and the + * anchor that matters for far-side fan-out is the provider's. + * + * Kept next to the endpoint merge logic because `dedupeCrossLinks` must + * re-derive the flag after a merge: `mergeEndpoints` backfills `symbolUid` + * from the losing twin, which can invalidate a flag carried in from the winner. + * + * NOT unresolved: a deterministic `manifest::::` synthetic + * uid (see `manifestSymbolUid`). Manifest endpoints fall back to it precisely + * when the graph has no symbol for them — its empty `symbolRef.filePath` would + * otherwise trip the check below — yet cross-impact anchors those links by + * design (#2722: the crossing is preserved with `fanout_status: + * 'not_attempted'` instead of silently yielding cross=0). The prefix is the + * canonical discriminator — real indexer uids never start with `manifest::` + * — and `cross-impact.ts` branches on the same test. Encoding the exemption + * HERE (not at the sync marking call site) keeps marking and the post-merge + * re-derivation from drifting apart, and keeps the flag's meaning exactly what + * `types.ts` documents: "distinct from manifest::… synthetic UIDs". + */ +export function isUnresolvedEndpoint(endpoint: CrossLinkEndpoint): boolean { + if (endpoint.symbolUid.startsWith('manifest::')) return false; + return ( + !endpoint.symbolUid || + !endpoint.symbolRef || + !endpoint.symbolRef.filePath || + !endpoint.symbolRef.name + ); +} + +/** + * Derive `degraded` from the provider endpoint. Present (`true`) only when + * unresolved; deleted otherwise so contracts.json stays "carried only when + * meaningful" (`'degraded' in link === false` for anchored links). + */ +export function applyDegradedFlag(link: CrossLink): CrossLink { + const next: CrossLink = { ...link }; + if (isUnresolvedEndpoint(next.to)) { + next.degraded = true; + } else { + delete next.degraded; + } + return next; +} + export function dedupeContracts(items: StoredContract[]): StoredContract[] { const deduped = new Map(); for (const contract of items) { @@ -113,12 +168,15 @@ export function dedupeCrossLinks(items: CrossLink[]): CrossLink[] { const keepIncoming = link.confidence > existing.confidence; const primary = keepIncoming ? link : existing; const secondary = keepIncoming ? existing : link; - deduped.set(key, { + const merged: CrossLink = { ...primary, confidence: Math.max(existing.confidence, link.confidence), from: mergeEndpoints(primary.from, secondary.from), to: mergeEndpoints(primary.to, secondary.to), - }); + }; + // Re-derive after mergeEndpoints: a richer twin can backfill `to.symbolUid` + // and must not leave a stale `degraded` flag on an now-anchored link. + deduped.set(key, applyDegradedFlag(merged)); } - return [...deduped.values()]; + return [...deduped.values()].map(applyDegradedFlag); } diff --git a/gitnexus/src/core/group/service.ts b/gitnexus/src/core/group/service.ts index d0aa4882c..6f41d6586 100644 --- a/gitnexus/src/core/group/service.ts +++ b/gitnexus/src/core/group/service.ts @@ -51,6 +51,17 @@ export interface GroupToolPort { repo: GroupRepoHandle, params: { target: string; + /** + * Target-selector params, same semantics as the single-repo `impact` + * tool: `target_uid` is the zero-ambiguity lookup (it wins over the + * name), `file_path`/`kind` narrow a name shared by several symbols + * (e.g. same-named Api/Impl/Controller layers). The port implementation + * consumes them directly; the Phase-1 caller in cross-impact.ts is + * responsible for threading them from the MCP `impact` args. + */ + target_uid?: string; + file_path?: string; + kind?: string; direction: 'upstream' | 'downstream'; maxDepth?: number; relationTypes?: string[]; @@ -517,6 +528,14 @@ export class GroupService { // can otherwise see contract counts that disagree with this payload, with // nothing here explaining why the write was skipped. registryOutcome: result.registryOutcome, + // Data-quality signals surfaced from the sync run: links whose provider + // endpoint never resolved to a graph symbol, per-repo extraction + // failures with reasons, and operator warnings (e.g. bridge.lbug write + // failed after contracts.json was written). Always present so MCP + // consumers can branch on them without existence checks. + degradedLinks: result.degradedLinks, + failedRepos: result.failedRepos, + warnings: result.warnings, }; } diff --git a/gitnexus/src/core/group/sync.ts b/gitnexus/src/core/group/sync.ts index 7981a3d0c..de7ce5c9e 100644 --- a/gitnexus/src/core/group/sync.ts +++ b/gitnexus/src/core/group/sync.ts @@ -37,6 +37,7 @@ import { buildProviderIndex, runExactMatch, runWildcardMatch } from './matching. import type { WildcardMatchResult } from './matching.js'; import { detectServiceBoundaries, assignService } from './service-boundary-detector.js'; import type { CypherExecutor } from './contract-extractor.js'; +import { applyDegradedFlag } from './normalization.js'; import { getContractRegistryPath, readContractRegistry, writeContractRegistry } from './storage.js'; import { markBridgeProvenanceUnknown, @@ -103,6 +104,24 @@ export interface SyncResult { * none of that repo's contracts are in `contracts`. */ unreadableRepos: string[]; + /** + * Cross-links whose provider endpoint has no resolved graph symbol + * (`degraded: true` on the link — see `isUnresolvedEndpoint`). The boundary + * is proven but cross-impact fan-out cannot anchor it; the usual remedy is + * re-analyzing the provider repo so its handlers resolve. + */ + degradedLinks: number; + /** + * Repos whose per-repo extraction threw (init, an extractor, or the + * snapshot read). Each still lands in `unreadableRepos` (group path) — + * unchanged downstream semantics — but carries its failure reason here: the + * catch used to swallow the exception, leaving contracts already pushed by + * earlier extractors in this iteration as silent half-repo data. `repo` is + * that same group path (e.g. `app/backend`), not the registry display name. + */ + failedRepos: Array<{ repo: string; reason: string }>; + /** Operator-facing run warnings (e.g. bridge.lbug write failed after contracts.json was written). */ + warnings: string[]; repoSnapshots: Record; /** * Matching stages this run was asked to skip. Populated on EVERY outcome, @@ -275,6 +294,8 @@ export function partitionManifestWindows( export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promise { const missingRepos: string[] = []; + const failedRepos: Array<{ repo: string; reason: string }> = []; + const warnings: string[] = []; // Repos that ARE registered but that we could not extract from — the index // would not open, or an extractor threw partway and the repo's staged // contracts were dropped. Kept separate from `missingRepos` because the two @@ -472,6 +493,10 @@ export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promis "⚠️ Could not read this repo's index; its contracts are omitted from this sync.", ); unreadableRepos.push(groupPath); + failedRepos.push({ + repo: groupPath, + reason: err instanceof Error ? err.message : String(err), + }); // Forget the handle recorded above (present only if the failure came // after initLbug). Deferred manifest resolution derives its known-repo // set from this map, so leaving the entry here re-opens a repo this @@ -639,7 +664,9 @@ export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promis // manifest-declared link can also emit a matchType:'exact' CrossLink with the // same endpoints. Prefer the manifest version — it reflects operator intent // and carries matchType:'manifest' which downstream consumers may rely on. - const crossLinks = dedupeCrossLinks([...manifestCrossLinks, ...matched, ...wildcard.matched]); + const crossLinks = dedupeCrossLinks([...manifestCrossLinks, ...matched, ...wildcard.matched]).map( + applyDegradedFlag, + ); const allContracts: StoredContract[] = autoContracts; const registry: ContractRegistry = { @@ -867,13 +894,16 @@ export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promis 'a lower bound rather than as complete.' : 'Its metadata could NOT be marked provenance-unknown, so those answers may still ' + 'report as complete despite describing an older sync.'; + const writeBridgeWarn = + '⚠️ writeBridge failed; contracts.json is intact and is the canonical copy, ' + + 'but bridge.lbug was not replaced: cross-repo queries may still answer from ' + + `the previous sync's contracts. ${provenanceNote} ` + + 'Re-run `gitnexus group sync` to retry.'; logger.warn( { err: msg, groupDir, bridgeProvenanceWithdrawn: withdrawn }, - '⚠️ writeBridge failed; contracts.json is intact and is the canonical copy, ' + - 'but bridge.lbug was not replaced: cross-repo queries may still answer from ' + - `the previous sync's contracts. ${provenanceNote} ` + - 'Re-run `gitnexus group sync` to retry.', + writeBridgeWarn, ); + warnings.push(writeBridgeWarn); } } }); @@ -886,6 +916,9 @@ export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promis unmatched: wildcard.remaining, missingRepos, unreadableRepos, + failedRepos, + warnings, + degradedLinks: crossLinks.filter((l) => l.degraded === true).length, repoSnapshots, registryOutcome, }; diff --git a/gitnexus/src/core/group/types.ts b/gitnexus/src/core/group/types.ts index 023629506..b0ba1a1dc 100644 --- a/gitnexus/src/core/group/types.ts +++ b/gitnexus/src/core/group/types.ts @@ -97,6 +97,19 @@ export interface CrossLink { contractId: string; matchType: MatchType; confidence: number; + /** + * `true` when the PROVIDER endpoint (`to`) has no resolved graph symbol — + * empty `symbolUid` / `symbolRef` at sync time (e.g. the handler failed to + * resolve and `symbolName` degraded to the file name). The contract boundary + * is still proven, but the link cannot anchor a cross-impact fan-out: an + * empty provider uid never matches a Phase-1 symbol id, and a downstream + * fan-out into it has no neighbor symbol to resolve. Derived once at the + * sync persistence boundary (`isUnresolvedEndpoint` in normalization.ts) and + * re-derived by `dedupeCrossLinks` when a merge backfills the uid. Absent on + * fully-anchored links. Distinct from manifest `manifest::…` synthetic UIDs, + * which have their own `fanout_status: 'not_attempted'` channel downstream. + */ + degraded?: boolean; } export interface RepoSnapshot { diff --git a/gitnexus/src/core/incremental/derived-writeback.ts b/gitnexus/src/core/incremental/derived-writeback.ts new file mode 100644 index 000000000..9cec191eb --- /dev/null +++ b/gitnexus/src/core/incremental/derived-writeback.ts @@ -0,0 +1,83 @@ +/** + * Incremental derived-layer writeback helpers (#3016). + * + * The derived layers — Leiden communities, execution flows, and the FTS + * indexes — are graph-wide, so every analyze run rebuilt all three in full no + * matter how small the diff. A surgical incremental write can instead: + * - drop and rebuild only the FTS indexes whose tables hold rows in the + * write set (LadybugDB still cannot DML a table with a live FTS index — + * #2589 — so a table being written must still lose its index first); + * - leave the untouched tables' rows alone, so their indexes stay live; + * - reuse persisted Community/Process rows only when the file-hash diff is + * empty (no added, changed, or deleted files). Any content change can + * add, rename, or retarget symbols that Leiden and flow extraction + * consume — a no-deletion edit is not a validity proof. + */ +import { FTS_INDEXES } from '../search/fts-schema.js'; +import type { KnowledgeGraph } from '../graph/types.js'; +import type { FileHashDiff } from '../../storage/file-hash.js'; + +const FTS_TABLE_NAMES: ReadonlySet = new Set(FTS_INDEXES.map((i) => i.table)); + +/** The FTS-backed members of `tables`. */ +export const ftsTablesAmong = (tables: Iterable): Set => { + const out = new Set(); + for (const table of tables) { + if (FTS_TABLE_NAMES.has(table)) out.add(table); + } + return out; +}; + +/** + * Whether a surgical incremental write may reuse the persisted derived layer. + * + * Deletions disqualify it: the persisted Community/Process rows and their + * MEMBER_OF / STEP_IN_PROCESS edges can reference nodes that no longer exist + * after this run, and nothing short of re-deriving can tell which. + * + * Added or content-changed files also disqualify it: they can introduce, + * rename, or retarget symbols and CALLS edges that Leiden and flow extraction + * consume. File-deletion-only was too weak a proof that the derived graph is + * still valid. + */ +export const shouldPreservePersistedDerivedGraph = ( + diff: Pick, +): boolean => diff.deleted.length === 0 && diff.added.length === 0 && diff.changed.length === 0; + +/** + * FTS-backed node tables that the fresh graph will WRITE rows into for + * `fileSet` — the inserting half of the DML. + * + * Callers must union this with a DB probe for the deleting half + * (`nodeTablesWithRowsForFiles`): a table whose last row in these files was + * just removed by the edit has nothing here, but still holds a stale row that + * the writeback must delete, and deleting it means taking its index down too. + */ +export const incrementalFtsTablesFromGraph = ( + graph: KnowledgeGraph, + fileSet: ReadonlySet, +): Set => { + const touched = new Set(); + graph.forEachNode((n) => { + const filePath = n.properties?.filePath as string | undefined; + if (!filePath || !fileSet.has(filePath)) return; + if (FTS_TABLE_NAMES.has(n.label)) touched.add(n.label); + }); + return touched; +}; + +/** + * The node tables an incremental DETACH DELETE should target, given the FTS + * tables this run is rebuilding. + * + * Every non-FTS table (Folder, CodeElement, …) deletes as before. An FTS-backed + * table only deletes when its index is being rebuilt anyway, because deleting + * from it otherwise would mean DML against a live FTS index (#2589). + */ +export const nodeTablesForIncrementalDelete = ( + allNodeTables: readonly string[], + rebuildingFtsTables: ReadonlySet, +): string[] => + allNodeTables.filter( + (tableName) => !FTS_TABLE_NAMES.has(tableName) || rebuildingFtsTables.has(tableName), + ); diff --git a/gitnexus/src/core/incremental/spring-config-drift.ts b/gitnexus/src/core/incremental/spring-config-drift.ts new file mode 100644 index 000000000..958f33924 --- /dev/null +++ b/gitnexus/src/core/incremental/spring-config-drift.ts @@ -0,0 +1,57 @@ +import type { KnowledgeGraph } from '../graph/types.js'; +import { SPRING_CONFIG_UNRESOLVED_PREFIX } from '../ingestion/frameworks/spring/config-bindings.js'; + +export interface PersistedSpringConfigConsumerRow { + readonly id?: unknown; + readonly description?: unknown; +} + +const CONSUMER_LABELS = new Set(['Property', 'Class', 'Record']); + +function unresolvedKeys(description: unknown): readonly string[] { + if (typeof description !== 'string') return []; + return description + .split(';') + .map((part) => part.trim()) + .filter((part) => part.startsWith(SPRING_CONFIG_UNRESOLVED_PREFIX)) + .map((part) => part.slice(SPRING_CONFIG_UNRESOLVED_PREFIX.length)) + .sort(); +} + +/** + * Find unchanged Spring consumer files whose unresolved markers changed. + * + * A removed config key also removes the old USES edge from the fresh graph, so + * ordinary new-graph boundary expansion cannot discover the consumer file. + */ +export function collectSpringConfigConsumerDriftFiles( + graph: KnowledgeGraph, + persistedRows: readonly PersistedSpringConfigConsumerRow[], +): Set { + const persistedById = new Map(); + for (const row of persistedRows) { + if (typeof row.id !== 'string') continue; + persistedById.set(row.id, unresolvedKeys(row.description)); + } + + const driftFiles = new Set(); + graph.forEachNode((node) => { + if (!CONSUMER_LABELS.has(node.label)) return; + const filePath = node.properties.filePath; + if (typeof filePath !== 'string') return; + const description = node.properties.description; + const persisted = persistedById.get(node.id); + if ( + persisted === undefined && + (typeof description !== 'string' || !description.includes(SPRING_CONFIG_UNRESOLVED_PREFIX)) + ) { + return; + } + const current = unresolvedKeys(description); + const prior = persisted ?? []; + if (current.length !== prior.length || current.some((key, index) => key !== prior[index])) { + driftFiles.add(filePath); + } + }); + return driftFiles; +} diff --git a/gitnexus/src/core/incremental/subgraph-extract.ts b/gitnexus/src/core/incremental/subgraph-extract.ts index e0f0e41eb..64751d99e 100644 --- a/gitnexus/src/core/incremental/subgraph-extract.ts +++ b/gitnexus/src/core/incremental/subgraph-extract.ts @@ -6,9 +6,10 @@ * replaced, produce a smaller KnowledgeGraph that contains: * * - Every node whose `properties.filePath` is in `toWriteSet`. - * - Every graph-wide node (Community, Process, and Spring metadata - * placeholders) — these are regenerated each run and must be fully - * rewritten. + * - Graph-wide Community/Process nodes unless `includeDerivedGraphWide` + * is false (#3016 incremental preserve). Spring metadata placeholders + * and `Destination` nodes are always included — their owning phase + * delete-alls them unconditionally before the writeback. * - Every relationship where AT LEAST ONE endpoint is in the writable * set above. Relationships entirely between unchanged-file nodes * are skipped — their rows are still in the DB and re-inserting @@ -57,9 +58,31 @@ import { } from '../ingestion/frameworks/spring/auto-configuration.js'; import { isSpringAopEvidenceNode } from '../ingestion/frameworks/spring/aop.js'; +/** + * `Destination` is graph-wide for the same reason as the Spring AOP evidence + * nodes: the layer is recomputed in full on every run and deleted in full + * before the writeback (`deleteAllDestinations`), so it must be re-included in + * full or it is simply lost. + * + * The endpoint-writability rule cannot carry it. A RESOLVED destination stores + * no `filePath` at all — deliberately, so an incremental delete keyed on + * `filePath IN [...]` cannot cut a node shared across files — and the include + * test below starts from exactly that property. The result was a defect in both + * directions: a newly added file publishing to a new topic reported + * `added=1, exit 0` and silently put neither the destination nor the + * publisher's edge into the graph, so after the first index every new topic was + * invisible until a full rebuild; and a destination whose last referrer stopped + * referring to it survived forever as an edgeless orphan still carrying + * `address`, the cross-repository join key. + * + * Unresolved destinations DO carry a file path and would ride the ordinary + * rule, but they are included here too: the delete-all removes them as well, so + * anything not re-included would be dropped rather than merely stale. + */ const isGraphWideNode = (node: GraphNode): boolean => node.label === 'Community' || node.label === 'Process' || + node.label === 'Destination' || isSpringAopEvidenceNode(node) || isSpringAutoConfigurationSyntheticClass(node); @@ -122,13 +145,18 @@ const indexNodeFilePaths = (fullGraph: KnowledgeGraph): Map => { export const extractChangedSubgraph = ( fullGraph: KnowledgeGraph, toWriteSet: ReadonlySet, + options?: { includeDerivedGraphWide?: boolean }, ): KnowledgeGraph => { const sub = createKnowledgeGraph(); const writableNodeIds = new Set(); + const includeDerivedGraphWide = options?.includeDerivedGraphWide !== false; + fullGraph.forEachNode((n: GraphNode) => { const filePath = n.properties?.filePath as string | undefined; - const include = (filePath && toWriteSet.has(filePath)) || isGraphWideNode(n); + const derivedWide = + includeDerivedGraphWide || (n.label !== 'Community' && n.label !== 'Process'); + const include = (filePath && toWriteSet.has(filePath)) || (isGraphWideNode(n) && derivedWide); if (include) { sub.addNode(n); writableNodeIds.add(n.id); diff --git a/gitnexus/src/core/ingestion/call-processor.ts b/gitnexus/src/core/ingestion/call-processor.ts index 2a7f450d7..610280c4f 100644 --- a/gitnexus/src/core/ingestion/call-processor.ts +++ b/gitnexus/src/core/ingestion/call-processor.ts @@ -406,7 +406,9 @@ export function resolveRouteHandlerSymbols( httpMethod: string | null | undefined, symbolId: string | undefined, ) => { - if (!routePath) return; + // An empty path is a valid, pathless mapping and normalizes to either `/` + // or its class/router prefix. Only null means the extractor had no route. + if (routePath === null) return; const url = normalizeExtractedRoutePath(routePath, prefix); const key = routeNodeKey(normalizeRouteMethod(httpMethod), url); if (claimed.has(key)) return; // first-writer-wins: later same-key routes can't override diff --git a/gitnexus/src/core/ingestion/destination-key.ts b/gitnexus/src/core/ingestion/destination-key.ts new file mode 100644 index 000000000..31a76c307 --- /dev/null +++ b/gitnexus/src/core/ingestion/destination-key.ts @@ -0,0 +1,62 @@ +/** + * Shared destination-identity keying — the async counterpart of `routeNodeKey` + * in `route-extractors/route-path.ts`. + * + * Deliberately OUTSIDE `frameworks/spring/`, and for the same reason + * `routeNodeKey` sits outside the routes phase: the identity has to be mintable + * by anything that names a broker address, so a Node Kafka client or a Celery + * task queue can land on the very node a Spring publisher minted. A key that + * lived in the Spring module would force every other producer to import Spring, + * or — worse — let each one invent its own spelling, and two spellings of one + * address is precisely the missed connection this overlay exists to make. + * + * `broker` is a plain `string`, NOT the Spring `SpringDestinationBroker` union. + * Importing that union here is the dependency this module exists to avoid, and + * widening it costs nothing that matters: the union is a subtype of `string`, + * so a Spring caller passes its own values unchanged, while a future + * non-Spring caller stays free to attest to a broker Spring has no name for. + * The trade is real but small — this signature cannot reject a misspelled + * broker — and it is the same trade `routeNodeKey` makes by taking `method` as + * a `string` rather than an HTTP-verb union. Pure string logic, no + * dependencies. + */ + +/** + * The `Destination` node identity: `(broker, address)` when the broker is + * known, falling back to the address alone when it is not. + * + * The broker belongs IN the key, exactly as the HTTP verb belongs in + * `routeNodeKey`. `GET /x` and `POST /x` are two nodes, both fully joinable, + * and neither is punished for the other's existence; `kafka orders` and + * `rabbit orders` are two nodes on the same terms. A Kafka topic and a Rabbit + * queue that happen to share a name are two places, and one node for both would + * report a publisher and a subscriber as connected when nothing connects them. + * + * The known objection is that the broker is INFERRED — from a receiver's name, + * from an annotation table — so a wrong guess splits a pair that is really one. + * That is true and it is the cost. It is worth paying because the alternative + * tried first was worse: withdrawing the address from every site that named it + * split the pair even when the guess was RIGHT, since one unrelated third party + * writing the same word anywhere in the repository was enough to disconnect + * everybody on that spelling. Putting the broker in the key bounds the damage + * of a wrong guess to the one pair it was wrong about, instead of spreading it + * to every pair that shares an address with a stranger. + * + * ── THE ADDRESS-ONLY FALLBACK IS UNREACHABLE TODAY ────────────────────── + * + * `SpringDestinationCandidate.broker` is REQUIRED, and every annotation rule + * and every producer template supplies one, so no Spring caller can reach the + * `undefined` branch. It is written anyway, and on purpose: the parameter shape + * is the contract this module offers the next language, and the next language + * may well capture an address without being able to attest to a broker (a bare + * `queue.publish(name)` in a dynamic language, a binding that names only a + * channel). Degrading to address-only is the right answer there — silence about + * the broker is not a claim about it, and refusing to key such a site at all + * would lose a real destination over a value nobody disagreed about. + * + * Because the branch is dead, it is covered by testing THIS function directly + * rather than by a pipeline test staged to look as though a phase reached it. + */ +export function destinationNodeKey(broker: string | undefined, address: string): string { + return broker ? `${broker} ${address}` : address; +} diff --git a/gitnexus/src/core/ingestion/entry-point-scoring.ts b/gitnexus/src/core/ingestion/entry-point-scoring.ts index 58cf9389c..30a9c78c1 100644 --- a/gitnexus/src/core/ingestion/entry-point-scoring.ts +++ b/gitnexus/src/core/ingestion/entry-point-scoring.ts @@ -13,6 +13,7 @@ import { detectFrameworkFromPath } from './framework-detection.js'; import { SupportedLanguages } from 'gitnexus-shared'; import { providers } from './languages/index.js'; +import { isTestFilePath } from './utils/test-file-path.js'; // ============================================================================ // NAME PATTERNS @@ -164,54 +165,15 @@ export function calculateEntryPointScore( // ============================================================================ /** - * Check if a file path is a test file (should be excluded from entry points) - * Covers common test file patterns across all supported languages + * Check if a file path is a test file (should be excluded from entry points). + * + * Delegates to the shared predicate in `utils/test-file-path.ts`. This used to be + * a second, hand-maintained copy that had drifted from the one backing the MCP + * `includeTests` flag — see that module's header. Re-exported under this name so + * existing importers are unaffected. */ export function isTestFile(filePath: string): boolean { - const p = filePath.toLowerCase().replace(/\\/g, '/'); - - return ( - // JavaScript/TypeScript test patterns - p.includes('.test.') || - p.includes('.spec.') || - p.includes('__tests__/') || - p.includes('__mocks__/') || - // Generic test folders - p.includes('/test/') || - p.includes('/tests/') || - p.includes('/testing/') || - // Python test patterns - p.endsWith('_test.py') || - p.includes('/test_') || - // Go test patterns - p.endsWith('_test.go') || - // Java test patterns - p.includes('/src/test/') || - // Rust test patterns (inline tests are different, but test files) - p.includes('/tests/') || - // Swift/iOS test patterns - p.endsWith('tests.swift') || - p.endsWith('test.swift') || - p.includes('uitests/') || - // C# test patterns - p.endsWith('tests.cs') || - p.endsWith('test.cs') || - p.includes('.tests/') || - p.includes('.test/') || - p.includes('.integrationtests/') || - p.includes('.unittests/') || - p.includes('/testproject/') || - // PHP/Laravel test patterns - p.endsWith('test.php') || - p.endsWith('spec.php') || - p.includes('/tests/feature/') || - p.includes('/tests/unit/') || - // Ruby test patterns - p.endsWith('_spec.rb') || - p.endsWith('_test.rb') || - p.includes('/spec/') || - p.includes('/test/fixtures/') - ); + return isTestFilePath(filePath); } /** diff --git a/gitnexus/src/core/ingestion/frameworks/spring/actuator-runtime.ts b/gitnexus/src/core/ingestion/frameworks/spring/actuator-runtime.ts new file mode 100644 index 000000000..4af7073cb --- /dev/null +++ b/gitnexus/src/core/ingestion/frameworks/spring/actuator-runtime.ts @@ -0,0 +1,968 @@ +import fs from 'node:fs/promises'; +import path from 'node:path'; +import type { GraphNode } from 'gitnexus-shared'; +import { generateId } from '../../../../lib/utils.js'; +import type { KnowledgeGraph } from '../../../graph/types.js'; +import { SPRING_DI_PROVIDER_PROPERTY } from '../../di-extractors/spring.js'; +import { + normalizeExtractedRoutePath, + normalizeRouteMethod, + routeNodeKey, +} from '../../route-extractors/route-path.js'; +import { stripBidiAndZeroWidth } from '../../utils/ast-helpers.js'; +import { SPRING_CONFIG_DESCRIPTION } from './config-bindings.js'; +import { getProviderForFile } from '../../languages/index.js'; +import type { RuntimeCallableIdentity } from '../../language-provider.js'; + +export const ACTUATOR_ENDPOINTS = [ + 'mappings', + 'beans', + 'conditions', + 'configprops', + 'env', +] as const; +type ActuatorEndpoint = (typeof ACTUATOR_ENDPOINTS)[number]; + +const MAX_ACTUATOR_PAYLOAD_BYTES = 16 * 1024 * 1024; +export const MAX_RUNTIME_RECORDS = 50_000; +const MAX_RUNTIME_DEPTH = 64; +const RUNTIME_FILE_PREFIX = 'spring-actuator:'; + +type JsonObject = Record; + +export interface SpringActuatorImportStats { + readonly payloads: number; + readonly mappings: number; + readonly beans: number; + readonly conditions: number; + readonly configProperties: number; + readonly environmentProperties: number; + /** Endpoint categories that exceeded the bounded import size. */ + readonly truncatedEndpoints: readonly ActuatorEndpoint[]; +} + +interface MutableImportStats { + payloads: number; + mappings: number; + beans: number; + conditions: number; + configProperties: number; + environmentProperties: number; + truncatedEndpoints: ActuatorEndpoint[]; +} + +interface ImportResult { + readonly count: number; + readonly truncated: boolean; +} + +export class SpringActuatorImportError extends Error { + constructor(message: string) { + super(message); + this.name = 'SpringActuatorImportError'; + } +} + +function objectValue(value: unknown): JsonObject | undefined { + return value !== null && typeof value === 'object' && !Array.isArray(value) + ? (value as JsonObject) + : undefined; +} + +function safeText(value: unknown, maxLength = 1024): string | undefined { + if (typeof value !== 'string') return undefined; + const sanitized = stripBidiAndZeroWidth(value) + .replace(/[\u0000-\u001f\u007f]/g, ' ') + .replace(/\s+/g, ' ') + .trim(); + return sanitized.length === 0 ? undefined : sanitized.slice(0, maxLength); +} + +function safeStrings(value: unknown, limit = 100): string[] { + if (!Array.isArray(value)) return []; + const strings: string[] = []; + for (const item of value.slice(0, limit)) { + const text = safeText(item); + if (text !== undefined) strings.push(text); + } + return strings; +} + +async function readPayloadFile(filePath: string, label: string): Promise { + // Size gate and read share one handle so both observe the same inode. + // Re-resolving the path for the read would let a swapped file bypass the + // payload cap (CodeQL js/file-system-race). + let handle: Awaited> | undefined; + let raw: string; + try { + handle = await fs.open(filePath, 'r'); + const stat = await handle.stat(); + if (!stat.isFile()) { + throw new SpringActuatorImportError(`Spring Actuator ${label} input must be a JSON file.`); + } + if (stat.size > MAX_ACTUATOR_PAYLOAD_BYTES) { + throw new SpringActuatorImportError( + `Spring Actuator ${label} payload exceeds the ${MAX_ACTUATOR_PAYLOAD_BYTES / 1024 / 1024} MiB limit.`, + ); + } + const buffer = Buffer.alloc(MAX_ACTUATOR_PAYLOAD_BYTES + 1); + let bytesRead = 0; + while (bytesRead < buffer.length) { + const result = await handle.read(buffer, bytesRead, buffer.length - bytesRead, bytesRead); + if (result.bytesRead === 0) break; + bytesRead += result.bytesRead; + } + if (bytesRead > MAX_ACTUATOR_PAYLOAD_BYTES) { + throw new SpringActuatorImportError( + `Spring Actuator ${label} payload exceeds the ${MAX_ACTUATOR_PAYLOAD_BYTES / 1024 / 1024} MiB limit.`, + ); + } + raw = buffer.subarray(0, bytesRead).toString('utf8'); + } catch (err) { + if (err instanceof SpringActuatorImportError) throw err; + throw new SpringActuatorImportError(`Spring Actuator ${label} input could not be read.`); + } finally { + await handle?.close().catch(() => {}); + } + let parsed: unknown; + try { + parsed = JSON.parse(raw); + } catch { + // Do not include JSON.parse's message: newer runtimes may quote source text, + // which could disclose an env/configprops value in CLI output. + throw new SpringActuatorImportError(`Spring Actuator ${label} payload is not valid JSON.`); + } + const object = objectValue(parsed); + if (object === undefined) { + throw new SpringActuatorImportError(`Spring Actuator ${label} payload must be a JSON object.`); + } + return object; +} + +async function loadPayloads( + repoPath: string, + configuredPath: string, +): Promise> { + const inputPath = path.resolve(repoPath, configuredPath); + let stat; + try { + stat = await fs.stat(inputPath); + } catch { + throw new SpringActuatorImportError( + 'Spring Actuator input path does not exist or is unreadable.', + ); + } + + const payloads = new Map(); + if (stat.isDirectory()) { + for (const endpoint of ACTUATOR_ENDPOINTS) { + const filePath = path.join(inputPath, `${endpoint}.json`); + try { + const endpointStat = await fs.stat(filePath); + if (!endpointStat.isFile()) continue; + } catch { + continue; + } + payloads.set(endpoint, await readPayloadFile(filePath, endpoint)); + } + } else if (stat.isFile()) { + const parsed = await readPayloadFile(inputPath, 'bundle'); + const endpointFromName = ACTUATOR_ENDPOINTS.find( + (endpoint) => path.basename(inputPath).toLowerCase() === `${endpoint}.json`, + ); + if (endpointFromName !== undefined) { + payloads.set(endpointFromName, parsed); + } else { + for (const endpoint of ACTUATOR_ENDPOINTS) { + const payload = objectValue(parsed[endpoint]); + if (payload !== undefined) payloads.set(endpoint, payload); + } + } + } else { + throw new SpringActuatorImportError( + 'Spring Actuator input must be a JSON bundle or a directory of endpoint JSON files.', + ); + } + + if (payloads.size === 0) { + throw new SpringActuatorImportError( + 'Spring Actuator input contains none of mappings, beans, conditions, configprops, or env.', + ); + } + return payloads; +} + +function evidenceFile(graph: KnowledgeGraph, endpoint: ActuatorEndpoint): GraphNode { + const filePath = `${RUNTIME_FILE_PREFIX}${endpoint}`; + const id = generateId('File', filePath); + const existing = graph.getNode(id); + if (existing !== undefined) return existing; + const node: GraphNode = { + id, + label: 'File', + properties: { name: `${endpoint}.json`, filePath }, + }; + graph.addNode(node); + return node; +} + +function appendRuntimeMarker(node: GraphNode, marker: string): void { + const current = + typeof node.properties.description === 'string' ? node.properties.description : ''; + if (current.includes(marker)) return; + node.properties.description = current.length === 0 ? marker : `${current}; ${marker}`; +} + +function markRuntimeEvidence( + graph: KnowledgeGraph, + endpoint: ActuatorEndpoint, + target: GraphNode, + status: string = 'runtime-confirmed', + confirmed: boolean = true, +): void { + // Only Route declares structured runtime columns in the persisted schema. + // Other labels retain the same evidence durably through their description + // plus the DECLARES edge below; setting undeclared properties would make the + // in-memory graph promise data that CSV/LadybugDB silently drops. + if (target.label === 'Route') { + // Confirmation is conflict-dominant. Once any runtime observation + // disagrees with static or runtime ownership, a later duplicate must not + // restore authoritative status. + target.properties.runtimeConfirmed = + target.properties.runtimeConfirmed === false ? false : confirmed; + // Source records provenance, not authority. Consumers MUST use + // runtimeConfirmed === true before treating runtime evidence as confirmed. + target.properties.runtimeSource = 'spring-actuator'; + const previousStatus = safeText(target.properties.runtimeStatus); + target.properties.runtimeStatus = [...new Set([...(previousStatus?.split(',') ?? []), status])] + .sort() + .join(','); + } + const marker = `Spring Actuator ${endpoint} ${status}`; + appendRuntimeMarker(target, marker); + + const evidence = evidenceFile(graph, endpoint); + graph.addRelationship({ + id: generateId('DECLARES', `${evidence.id}->${target.id}:${status}`), + sourceId: evidence.id, + targetId: target.id, + type: 'DECLARES', + confidence: 1, + reason: `spring-actuator:${endpoint}:${status}`, + }); +} + +function normalizedQualifiedName(value: string): string { + return value + .replace(/\$\$(?:SpringCGLIB|EnhancerBySpringCGLIB|FastClassBySpringCGLIB).*$/, '') + .replaceAll('$', '.'); +} + +function uniqueIndexAdd(index: Map, key: string, node: GraphNode): void { + const existing = index.get(key); + if (existing === undefined) index.set(key, node); + else if (existing !== null && existing.id !== node.id) index.set(key, null); +} + +interface RuntimeNodeIndexes { + readonly classesByQualifiedName: Map; + readonly classesByRuntimeAlias: Map; + readonly classesBySimpleName: Map; + readonly beanProvidersByName: Map; + readonly methodsByOwnerId: Map; + readonly callablesByRuntimeOwner: Map; + readonly routeOwnerFileIdsByRouteId: Map>; +} + +function addRuntimeCallable( + index: Map, + ownerName: string, + node: GraphNode, +): void { + const normalizedOwner = normalizedQualifiedName(ownerName); + const nodes = index.get(normalizedOwner) ?? []; + if (!nodes.some((candidate) => candidate.id === node.id)) nodes.push(node); + index.set(normalizedOwner, nodes); +} + +function buildRuntimeNodeIndexes(graph: KnowledgeGraph): RuntimeNodeIndexes { + const allNodes = [...graph.iterNodes()]; + const classesByQualifiedName = new Map(); + const classesByRuntimeAlias = new Map(); + const classesBySimpleName = new Map(); + const beanProvidersByName = new Map(); + const nodesById = new Map(allNodes.map((node) => [node.id, node])); + const methodsByOwnerId = new Map(); + const callablesByRuntimeOwner = new Map(); + const routeOwnerFileIdsByRouteId = new Map>(); + for (const node of allNodes) { + if (node.label === 'Class' || node.label === 'Record') { + const qualified = safeText(node.properties.qualifiedName); + if (qualified !== undefined) { + uniqueIndexAdd(classesByQualifiedName, normalizedQualifiedName(qualified), node); + } + uniqueIndexAdd(classesBySimpleName, String(node.properties.name), node); + } + const provider = objectValue(node.properties[SPRING_DI_PROVIDER_PROPERTY]); + for (const name of safeStrings(provider?.names)) + uniqueIndexAdd(beanProvidersByName, name, node); + } + const ownedNodeIds = new Set(); + for (const relationshipType of ['HAS_METHOD', 'HAS_PROPERTY'] as const) { + for (const relationship of graph.iterRelationshipsByType(relationshipType)) { + const member = nodesById.get(relationship.targetId); + const owner = nodesById.get(relationship.sourceId); + if ( + member === undefined || + !['Method', 'Function', 'Property'].includes(member.label) || + owner === undefined + ) { + continue; + } + ownedNodeIds.add(member.id); + if (member.label === 'Method' || member.label === 'Function') { + const methods = methodsByOwnerId.get(relationship.sourceId) ?? []; + methods.push(member); + methodsByOwnerId.set(relationship.sourceId, methods); + } + const ownerQualifiedName = safeText(owner.properties.qualifiedName); + if (ownerQualifiedName !== undefined) { + addRuntimeCallable(callablesByRuntimeOwner, ownerQualifiedName, member); + } + const strategy = getProviderForFile( + String(member.properties.filePath), + )?.runtimeSymbolStrategy; + for (const alias of strategy?.callableOwnerAliases?.(member, owner) ?? []) { + addRuntimeCallable(callablesByRuntimeOwner, alias, member); + if ( + (owner.label === 'Class' || owner.label === 'Record') && + ownerQualifiedName !== undefined && + normalizedQualifiedName(alias) !== normalizedQualifiedName(ownerQualifiedName) + ) { + uniqueIndexAdd(classesByRuntimeAlias, normalizedQualifiedName(alias), owner); + } + } + } + } + for (const node of allNodes) { + if ( + ownedNodeIds.has(node.id) || + (node.label !== 'Function' && node.label !== 'Method' && node.label !== 'Property') + ) { + continue; + } + const strategy = getProviderForFile(String(node.properties.filePath))?.runtimeSymbolStrategy; + for (const alias of strategy?.callableOwnerAliases?.(node, undefined) ?? []) { + addRuntimeCallable(callablesByRuntimeOwner, alias, node); + } + } + for (const relationship of graph.iterRelationshipsByType('HANDLES_ROUTE')) { + const owners = routeOwnerFileIdsByRouteId.get(relationship.targetId) ?? new Set(); + owners.add(relationship.sourceId); + routeOwnerFileIdsByRouteId.set(relationship.targetId, owners); + } + return { + classesByQualifiedName, + classesByRuntimeAlias, + classesBySimpleName, + beanProvidersByName, + methodsByOwnerId, + callablesByRuntimeOwner, + routeOwnerFileIdsByRouteId, + }; +} + +function resolveClass( + indexes: RuntimeNodeIndexes, + rawType: string | undefined, +): GraphNode | undefined { + if (rawType === undefined) return undefined; + const type = normalizedQualifiedName(rawType.replace(/\[\]$/, '')); + const exact = indexes.classesByQualifiedName.get(type); + if (exact !== null && exact !== undefined) return exact; + const alias = indexes.classesByRuntimeAlias.get(type); + if (alias !== null && alias !== undefined) return alias; + // A qualified runtime name is authoritative. Falling back to a unique class + // with the same simple name can bind a stale snapshot to a different package + // and then mint confidence-1 handler evidence for the wrong source. + if (type.includes('.')) return undefined; + const simple = type.slice(type.lastIndexOf('.') + 1); + const fallback = indexes.classesBySimpleName.get(simple); + return fallback === null ? undefined : fallback; +} + +function providerMatchesRuntimeType( + indexes: RuntimeNodeIndexes, + providerNode: GraphNode, + runtimeType: string | undefined, +): boolean { + if (runtimeType === undefined) return true; + const provider = objectValue(providerNode.properties[SPRING_DI_PROVIDER_PROPERTY]); + const providerType = + safeText(provider?.providedTypeName) ?? + (providerNode.label === 'Class' || providerNode.label === 'Record' + ? safeText(providerNode.properties.qualifiedName) + : undefined); + if (providerType === undefined) return true; + + const providerClass = resolveClass(indexes, providerType); + const runtimeClass = resolveClass(indexes, runtimeType); + if (providerClass !== undefined && runtimeClass !== undefined) { + return providerClass.id === runtimeClass.id; + } + + const normalizedProvider = normalizedQualifiedName(providerType); + const normalizedRuntime = normalizedQualifiedName(runtimeType); + if (normalizedProvider.includes('.')) return normalizedProvider === normalizedRuntime; + return normalizedProvider === normalizedRuntime.slice(normalizedRuntime.lastIndexOf('.') + 1); +} + +function descriptorParameterTypes(descriptor: string | undefined): string[] | undefined { + if (descriptor === undefined || descriptor.charAt(0) !== '(') return undefined; + const types: string[] = []; + for (let index = 1; index < descriptor.length && descriptor.charAt(index) !== ')'; ) { + let arrayDimensions = 0; + while (descriptor.charAt(index) === '[') { + arrayDimensions++; + index++; + } + const arraySuffix = '[]'.repeat(arrayDimensions); + if (descriptor.charAt(index) === 'L') { + const end = descriptor.indexOf(';', index); + if (end === -1) return undefined; + types.push(`${descriptor.slice(index + 1, end)}${arraySuffix}`); + index = end + 1; + } else { + const primitive = descriptor.charAt(index); + if (!'BCDFIJSZ'.includes(primitive)) return undefined; + types.push(`${primitive}${arraySuffix}`); + index++; + } + } + return descriptor.includes(')') ? types : undefined; +} + +function matchesRuntimeCallable(node: GraphNode, runtime: RuntimeCallableIdentity): boolean { + const strategy = getProviderForFile(String(node.properties.filePath))?.runtimeSymbolStrategy; + if (strategy !== undefined) return strategy.matchesCallable(node, runtime); + return ( + (node.label === 'Method' || node.label === 'Function') && + node.properties.name === runtime.name && + (runtime.descriptorParameterTypes === undefined || + node.properties.parameterCount === runtime.descriptorParameterTypes.length) + ); +} + +function resolveHandlerNode( + indexes: RuntimeNodeIndexes, + handlerMethod: JsonObject | undefined, +): GraphNode | undefined { + const className = safeText(handlerMethod?.className); + const methodName = safeText(handlerMethod?.name); + if (methodName === undefined) return resolveClass(indexes, className); + if (className === undefined) return undefined; + const owner = resolveClass(indexes, className); + const runtime: RuntimeCallableIdentity = { + name: methodName, + descriptorParameterTypes: descriptorParameterTypes(safeText(handlerMethod?.descriptor)), + }; + const ownerCandidates = owner === undefined ? [] : (indexes.methodsByOwnerId.get(owner.id) ?? []); + const aliasCandidates = + indexes.callablesByRuntimeOwner.get(normalizedQualifiedName(className)) ?? []; + const candidates = [...ownerCandidates, ...aliasCandidates] + .filter((node, index, all) => all.findIndex((candidate) => candidate.id === node.id) === index) + .filter((node) => matchesRuntimeCallable(node, runtime)); + return candidates.length === 1 ? candidates[0] : undefined; +} + +function predicateParts(predicate: string | undefined): { + readonly methods: string[]; + readonly patterns: string[]; +} { + if (predicate === undefined) return { methods: [], patterns: [] }; + const methodListEnd = predicate.indexOf('['); + const methodRegion = methodListEnd === -1 ? predicate : predicate.slice(0, methodListEnd); + const methods = [ + ...methodRegion.matchAll(/\b(GET|POST|PUT|PATCH|DELETE|HEAD|OPTIONS|TRACE|CONNECT)\b/g), + ] + .map((match) => match[1]) + .filter((method): method is string => method !== undefined); + const patterns = [...predicate.matchAll(/(?:^|[\s[(])((?:\/)[^\s\]),}]+)/g)] + .map((match) => safeText(match[1])) + .filter((pattern): pattern is string => pattern !== undefined); + return { methods: [...new Set(methods)], patterns: [...new Set(patterns)] }; +} + +function mappingEntries(payload: JsonObject): { + entries: JsonObject[]; + truncated: boolean; +} { + const entries: JsonObject[] = []; + const contexts = objectValue(payload.contexts); + if (contexts === undefined) return { entries, truncated: false }; + for (const context of Object.values(contexts)) { + const mappings = objectValue(objectValue(context)?.mappings); + if (mappings === undefined) continue; + for (const groupName of ['dispatcherServlets', 'dispatcherHandlers']) { + const groups = objectValue(mappings[groupName]); + if (groups === undefined) continue; + for (const group of Object.values(groups)) { + if (!Array.isArray(group)) continue; + for (const entry of group) { + const object = objectValue(entry); + if (object !== undefined) entries.push(object); + if (entries.length > MAX_RUNTIME_RECORDS) { + entries.pop(); + return { entries, truncated: true }; + } + } + } + } + } + return { entries, truncated: false }; +} + +interface RuntimeMappingCandidate { + readonly key: string; + readonly method: string | undefined; + readonly url: string; + readonly handler: GraphNode | undefined; +} + +function importMappings( + graph: KnowledgeGraph, + payload: JsonObject, + indexes: RuntimeNodeIndexes, +): ImportResult { + let imported = 0; + const payloadEntries = mappingEntries(payload); + let truncated = payloadEntries.truncated; + const candidatesByKey = new Map(); + for (const entry of payloadEntries.entries) { + const details = objectValue(entry.details); + const conditions = objectValue(details?.requestMappingConditions); + const predicate = predicateParts(safeText(entry.predicate)); + const patterns = safeStrings(conditions?.patterns); + const methods = safeStrings(conditions?.methods) + .map(normalizeRouteMethod) + .filter((method): method is string => method !== undefined); + const effectivePatterns = patterns.length > 0 ? patterns : predicate.patterns; + const effectiveMethods = methods.length > 0 ? methods : predicate.methods; + if (effectivePatterns.length === 0) continue; + + const handler = resolveHandlerNode(indexes, objectValue(details?.handlerMethod)); + for (const rawPattern of effectivePatterns) { + const url = normalizeExtractedRoutePath(rawPattern, null); + for (const method of effectiveMethods.length > 0 ? effectiveMethods : [undefined]) { + const normalizedMethod = normalizeRouteMethod(method); + const key = routeNodeKey(normalizedMethod, url); + const candidate = { key, method: normalizedMethod, url, handler }; + const existing = candidatesByKey.get(key); + if (existing === undefined) { + if (candidatesByKey.size >= MAX_RUNTIME_RECORDS) { + truncated = true; + continue; + } + candidatesByKey.set(key, [candidate]); + } else { + existing.push(candidate); + } + } + } + } + + for (const candidates of candidatesByKey.values()) { + const first = candidates[0]; + if (first === undefined) continue; + const { key, method: normalizedMethod, url } = first; + const resolvedHandlers = new Map( + candidates + .map((candidate) => candidate.handler) + .filter((handler): handler is GraphNode => handler !== undefined) + .map((handler) => [handler.id, handler]), + ); + const runtimeHandlerConflict = resolvedHandlers.size > 1; + const handler = runtimeHandlerConflict ? undefined : resolvedHandlers.values().next().value; + const exactId = generateId('Route', key); + const fallbackId = generateId('Route', url); + let route = graph.getNode(exactId) ?? graph.getNode(fallbackId); + if (route?.label !== 'Route') route = undefined; + const routeWasPresent = route !== undefined; + if (route === undefined) { + route = { + id: exactId, + label: 'Route', + properties: { + name: url, + filePath: handler?.properties.filePath ?? `${RUNTIME_FILE_PREFIX}mappings`, + ...(normalizedMethod === undefined ? {} : { method: normalizedMethod }), + ...(handler === undefined ? {} : { handlerSymbolId: handler.id }), + }, + }; + graph.addNode(route); + } + const existingHandlerId = safeText(route.properties.handlerSymbolId); + const handlerFilePath = + handler !== undefined && typeof handler.properties.filePath === 'string' + ? handler.properties.filePath + : undefined; + const handlerFileId = + handlerFilePath === undefined ? undefined : generateId('File', handlerFilePath); + const staticOwnerFileIds = indexes.routeOwnerFileIdsByRouteId.get(route.id); + const conflictsWithStaticOwner = + routeWasPresent && + handlerFileId !== undefined && + staticOwnerFileIds !== undefined && + [...staticOwnerFileIds].some((ownerFileId) => ownerFileId !== handlerFileId); + if ( + runtimeHandlerConflict || + (handler !== undefined && + ((existingHandlerId !== undefined && existingHandlerId !== handler.id) || + conflictsWithStaticOwner)) + ) { + // Static ownership and runtime ownership disagree. Preserve the + // static handler, persist an explicit conflict, and do not mint an + // authoritative HANDLES_ROUTE edge from the runtime candidate. + markRuntimeEvidence(graph, 'mappings', route, 'handler-conflict', false); + imported++; + continue; + } + if (handler !== undefined && existingHandlerId === undefined) { + route.properties.handlerSymbolId = handler.id; + } + markRuntimeEvidence(graph, 'mappings', route); + if (handler !== undefined && handlerFileId !== undefined) { + if (graph.getNode(handlerFileId) !== undefined) { + graph.addRelationship({ + id: generateId('HANDLES_ROUTE', `${handlerFileId}->${route.id}`), + sourceId: handlerFileId, + targetId: route.id, + type: 'HANDLES_ROUTE', + confidence: 1, + reason: 'spring-actuator:runtime-confirmed', + }); + } + } + imported++; + } + return { count: imported, truncated }; +} + +function contextObjects(payload: JsonObject): JsonObject[] { + const contexts = objectValue(payload.contexts); + if (contexts === undefined) return []; + return Object.values(contexts) + .map(objectValue) + .filter((context): context is JsonObject => context !== undefined); +} + +function importBeans( + graph: KnowledgeGraph, + payload: JsonObject, + indexes: RuntimeNodeIndexes, +): ImportResult { + let imported = 0; + const seen = new Set(); + for (const [contextIndex, context] of contextObjects(payload).entries()) { + const beans = objectValue(context.beans); + if (beans === undefined) continue; + for (const [rawBeanName, rawBean] of Object.entries(beans)) { + if (imported >= MAX_RUNTIME_RECORDS) return { count: imported, truncated: true }; + const beanName = safeText(rawBeanName, 512); + const bean = objectValue(rawBean); + if (beanName === undefined || bean === undefined) continue; + const identity = `${contextIndex}:${beanName}`; + if (seen.has(identity)) continue; + seen.add(identity); + const type = safeText(bean.type, 1024); + const named = indexes.beanProvidersByName.get(beanName); + let target = named === null ? undefined : named; + if (target !== undefined && !providerMatchesRuntimeType(indexes, target, type)) { + target = undefined; + } + target ??= resolveClass(indexes, type); + if (target === undefined) { + const id = generateId('CodeElement', `spring-runtime-bean:${identity}`); + target = graph.getNode(id); + if (target === undefined) { + const scope = safeText(bean.scope, 128); + target = { + id, + label: 'CodeElement', + properties: { + name: beanName, + filePath: `${RUNTIME_FILE_PREFIX}beans`, + description: + `Spring runtime Bean ${beanName}` + + (type === undefined ? '' : ` of type ${type}`) + + (scope === undefined ? '' : ` (${scope})`), + ...(type === undefined ? {} : { qualifiedName: normalizedQualifiedName(type) }), + }, + }; + graph.addNode(target); + } + } + markRuntimeEvidence(graph, 'beans', target); + imported++; + } + } + return { count: imported, truncated: false }; +} + +function resolveConditionOwner( + indexes: RuntimeNodeIndexes, + rawName: string, +): GraphNode | undefined { + const separator = rawName.lastIndexOf('#'); + return resolveHandlerNode(indexes, { + className: separator === -1 ? rawName : rawName.slice(0, separator), + ...(separator === -1 ? {} : { name: rawName.slice(separator + 1) }), + }); +} + +function importConditions( + graph: KnowledgeGraph, + payload: JsonObject, + indexes: RuntimeNodeIndexes, +): ImportResult { + let imported = 0; + const seen = new Set(); + for (const context of contextObjects(payload)) { + for (const [field, status] of [ + ['positiveMatches', 'matched'], + ['negativeMatches', 'not-matched'], + ] as const) { + const matches = objectValue(context[field]); + if (matches === undefined) continue; + for (const rawName of Object.keys(matches)) { + if (imported >= MAX_RUNTIME_RECORDS) return { count: imported, truncated: true }; + const name = safeText(rawName); + if (name === undefined || seen.has(`${status}:${name}`)) continue; + seen.add(`${status}:${name}`); + const owner = resolveConditionOwner(indexes, name); + if (owner === undefined) continue; + // Actuator reports this status for the aggregate owner entry. Its child + // details may contain a mix of matched and not-matched conditions, but + // do not carry a stable identifier that maps to our CONDITIONAL_ON + // targets. Keep the aggregate on the owner instead of guessing. + markRuntimeEvidence(graph, 'conditions', owner, status); + imported++; + } + } + } + return { count: imported, truncated: false }; +} + +function relaxedPropertyName(value: string): string { + return value.toLowerCase().replace(/[-_.\[\]]/g, ''); +} + +interface RuntimePropertyIndex { + readonly exact: Map; + readonly relaxed: Map; +} + +function buildRuntimePropertyIndex(graph: KnowledgeGraph): RuntimePropertyIndex { + const exact = new Map(); + const relaxed = new Map(); + for (const node of graph.iterNodes()) { + if (node.label !== 'Property') continue; + const description = safeText(node.properties.description) ?? ''; + if (!description.startsWith(SPRING_CONFIG_DESCRIPTION) && !node.id.includes('spring-runtime')) { + continue; + } + const name = String(node.properties.name); + uniqueIndexAdd(exact, name, node); + uniqueIndexAdd(relaxed, relaxedPropertyName(name), node); + } + return { exact, relaxed }; +} + +function ensureRuntimeProperty( + graph: KnowledgeGraph, + index: RuntimePropertyIndex, + endpoint: 'configprops' | 'env', + rawName: string, +): GraphNode | undefined { + const name = safeText(rawName, 1024); + if (name === undefined) return undefined; + const exact = index.exact.get(name); + let node = exact === null ? undefined : exact; + if (node === undefined) { + const relaxed = index.relaxed.get(relaxedPropertyName(name)); + node = relaxed === null ? undefined : relaxed; + } + if (node === undefined) { + const id = generateId('Property', `spring-runtime-config:${name}`); + node = graph.getNode(id); + if (node === undefined) { + node = { + id, + label: 'Property', + properties: { + name, + filePath: `${RUNTIME_FILE_PREFIX}${endpoint}`, + description: `${SPRING_CONFIG_DESCRIPTION}; imported from Spring Actuator ${endpoint}`, + }, + }; + graph.addNode(node); + } + uniqueIndexAdd(index.exact, name, node); + uniqueIndexAdd(index.relaxed, relaxedPropertyName(name), node); + } + markRuntimeEvidence(graph, endpoint, node); + return node; +} + +function configInputPaths(inputs: unknown): { + paths: string[]; + truncated: boolean; +} { + const out: string[] = []; + const stack: Array<{ value: unknown; prefix: string; depth: number }> = [ + { value: inputs, prefix: '', depth: 0 }, + ]; + while (stack.length > 0 && out.length < MAX_RUNTIME_RECORDS) { + const current = stack.pop(); + if (current === undefined || current.depth > MAX_RUNTIME_DEPTH) continue; + const object = objectValue(current.value); + if (object === undefined) { + if (current.prefix.length > 0) out.push(current.prefix); + continue; + } + const keys = Object.keys(object); + const metadataLeaf = + keys.length === 0 || keys.every((key) => key === 'value' || key === 'origin'); + if (metadataLeaf) { + if (current.prefix.length > 0) out.push(current.prefix); + continue; + } + for (let index = keys.length - 1; index >= 0; index--) { + const rawKey = keys[index]; + if (rawKey === undefined) continue; + const key = safeText(rawKey, 256); + if (key === undefined) continue; + stack.push({ + value: object[rawKey], + prefix: current.prefix.length === 0 ? key : `${current.prefix}.${key}`, + depth: current.depth + 1, + }); + } + } + return { paths: out, truncated: stack.length > 0 }; +} + +function importConfigProperties( + graph: KnowledgeGraph, + payload: JsonObject, + propertyIndex: RuntimePropertyIndex, +): ImportResult { + let imported = 0; + let truncated = false; + const seen = new Set(); + for (const context of contextObjects(payload)) { + const beans = objectValue(context.beans); + if (beans === undefined) continue; + for (const rawBean of Object.values(beans)) { + const bean = objectValue(rawBean); + const prefix = safeText(bean?.prefix, 512)?.replace(/\.+$/, ''); + if (bean === undefined || prefix === undefined) continue; + const inputPaths = configInputPaths(bean.inputs); + truncated ||= inputPaths.truncated; + const names = + inputPaths.paths.length === 0 + ? [prefix] + : inputPaths.paths.map((entry) => `${prefix}.${entry}`); + for (const name of names) { + if (imported >= MAX_RUNTIME_RECORDS) return { count: imported, truncated: true }; + if (seen.has(name)) continue; + seen.add(name); + if (ensureRuntimeProperty(graph, propertyIndex, 'configprops', name) !== undefined) + imported++; + } + } + } + return { count: imported, truncated }; +} + +function importEnvironmentProperties( + graph: KnowledgeGraph, + payload: JsonObject, + propertyIndex: RuntimePropertyIndex, +): ImportResult { + let imported = 0; + const seen = new Set(); + if (!Array.isArray(payload.propertySources)) return { count: imported, truncated: false }; + for (const rawSource of payload.propertySources) { + const properties = objectValue(objectValue(rawSource)?.properties); + if (properties === undefined) continue; + // Deliberately enumerate keys only. Never read, retain, interpolate, or log + // the corresponding {value, origin} objects. + for (const rawName of Object.keys(properties)) { + if (imported >= MAX_RUNTIME_RECORDS) return { count: imported, truncated: true }; + const name = safeText(rawName, 1024); + if (name === undefined || seen.has(name)) continue; + seen.add(name); + if (ensureRuntimeProperty(graph, propertyIndex, 'env', name) !== undefined) imported++; + } + } + return { count: imported, truncated: false }; +} + +/** + * Import explicitly supplied Spring Boot Actuator snapshots. Runtime evidence + * is additive: it confirms existing static nodes where possible and creates + * conservative synthetic Route/Bean/Property nodes otherwise. Raw payloads, + * condition messages, config values, env values, origins, and source names are + * never copied into graph properties or logs. + */ +export async function importSpringActuatorRuntime( + graph: KnowledgeGraph, + repoPath: string, + configuredPath: string, +): Promise { + const payloads = await loadPayloads(repoPath, configuredPath); + const stats: MutableImportStats = { + payloads: payloads.size, + mappings: 0, + beans: 0, + conditions: 0, + configProperties: 0, + environmentProperties: 0, + truncatedEndpoints: [], + }; + const indexes = buildRuntimeNodeIndexes(graph); + const propertyIndex = buildRuntimePropertyIndex(graph); + + const mappings = payloads.get('mappings'); + if (mappings !== undefined) { + const result = importMappings(graph, mappings, indexes); + stats.mappings = result.count; + if (result.truncated) stats.truncatedEndpoints.push('mappings'); + } + const beans = payloads.get('beans'); + if (beans !== undefined) { + const result = importBeans(graph, beans, indexes); + stats.beans = result.count; + if (result.truncated) stats.truncatedEndpoints.push('beans'); + } + const conditions = payloads.get('conditions'); + if (conditions !== undefined) { + const result = importConditions(graph, conditions, indexes); + stats.conditions = result.count; + if (result.truncated) stats.truncatedEndpoints.push('conditions'); + } + const configprops = payloads.get('configprops'); + if (configprops !== undefined) { + const result = importConfigProperties(graph, configprops, propertyIndex); + stats.configProperties = result.count; + if (result.truncated) stats.truncatedEndpoints.push('configprops'); + } + const env = payloads.get('env'); + if (env !== undefined) { + const result = importEnvironmentProperties(graph, env, propertyIndex); + stats.environmentProperties = result.count; + if (result.truncated) stats.truncatedEndpoints.push('env'); + } + return stats; +} diff --git a/gitnexus/src/core/ingestion/frameworks/spring/annotation-arguments.ts b/gitnexus/src/core/ingestion/frameworks/spring/annotation-arguments.ts index 928f3e805..9754d40c8 100644 --- a/gitnexus/src/core/ingestion/frameworks/spring/annotation-arguments.ts +++ b/gitnexus/src/core/ingestion/frameworks/spring/annotation-arguments.ts @@ -124,7 +124,13 @@ export function parseSpringAnnotationArguments( const body = annotationText.slice(open + 1, close).trim(); if (body.length === 0) return []; const rawArguments = splitTopLevel(body, ','); - if (rawArguments === null || rawArguments.some((argument) => argument.length === 0)) return null; + if (rawArguments === null) return null; + // Kotlin (and some formatters) allow a trailing comma. An empty *middle* + // argument is still invalid and fail-closed. + while (rawArguments.at(-1)?.length === 0) { + rawArguments.pop(); + } + if (rawArguments.some((argument) => argument.length === 0)) return null; const parsed: SpringAnnotationArgument[] = []; for (const raw of rawArguments) { diff --git a/gitnexus/src/core/ingestion/frameworks/spring/argument-facts.ts b/gitnexus/src/core/ingestion/frameworks/spring/argument-facts.ts new file mode 100644 index 000000000..cb7c9cf9d --- /dev/null +++ b/gitnexus/src/core/ingestion/frameworks/spring/argument-facts.ts @@ -0,0 +1,140 @@ +/** + * One argument of a Spring annotation or of a messaging-template call, captured + * exactly as it is written in source. + * + * Capture-time facts are deliberately UNRESOLVED. When these facts are produced + * the file's imports are not finalized, constants declared in sibling files do + * not exist yet, and no configuration source has been read — so a captured + * `text` may be a string literal, a constant reference (`Destinations.ORDERS`), + * a property placeholder (`"${app.orders.topic}"`), or an arbitrary expression. + * Turning any of those into an address is a separate, later phase; nothing here + * may call a resolver. + * + * NOT the same thing as `SpringAnnotationArgument` in `annotation-arguments.ts`, + * and the two are deliberately not merged: + * + * - Source. This fact is built from AST nodes while the tree is in hand; + * `parseSpringAnnotationArguments` re-parses an annotation's `text` much + * later, from a string, with a hand-written delimiter scanner. + * - Failure. The text parser returns `null` when its scanner cannot balance + * the input, and a caller must decide what that means. There is no such + * state here: the grammar has already decided where each argument begins + * and ends. + * - Absence. The text parser answers `[]` both for `@Scheduled` and for + * `@Scheduled()`, because a string cannot tell "no list" from "empty list" + * without re-deriving it. Capture keeps the two apart — absent versus `[]` — + * so downstream code can rely on the distinction wherever arguments were + * read at all. A capture that reads them for only some of its facts says so + * on its own `args` field. + * - Scope. This fact also describes CALL arguments (`template.send(topic, p)`), + * which the annotation parser has no notion of. + * + * Collapsing them would mean giving the text parser a failure mode it cannot + * produce, or taking the three-state distinction away from capture. + */ +export interface SpringArgumentFact { + /** + * Argument name for a named argument, absent for a positional one. + * + * Both forms occur, and where the destination sits differs by construct. An + * annotation names it (`@KafkaListener(topics = ...)` versus + * `@RabbitListener(queues = ...)`). A call normally gives it by position + * (`kafkaTemplate.send(topic, payload)`) — always so in Java, which has no + * named arguments — but a Kotlin call may name its arguments whenever the + * callee is itself declared in Kotlin, and then the key is captured too. + */ + readonly name?: string; + /** + * Argument value in its source spelling — quotes, braces and casts intact, + * nothing resolved — after `normalizeSpringFactText`. That pass trims the + * text and collapses whitespace around the dots of a multi-line expression, + * so one destination written two ways yields one fact. It is the only + * rewrite; see the function for why formatting must not reach the data. + */ + readonly text: string; +} + +/** + * Join an expression that the source wrapped across lines, so that one + * expression has one spelling no matter where it was written. + * + * A receiver chain written as `outer\n .inner\n .kafkaTemplate`, and an + * argument written as `Destinations\n .ORDERS`, are the same expressions as + * their single-line spellings. Raw node text would carry the newline and the + * ENCLOSING BLOCK's indentation across the worker boundary, so the same + * expression at two nesting depths — or in a CRLF checkout — would not compare + * equal downstream. Receivers and arguments get the identical treatment on + * purpose: an inconsistent rule inside one fact is a trap for the phase that + * has to match a publish against a subscription. + * + * Only a run of whitespace that CONTAINS A NEWLINE and sits next to a dot is + * removed, and only OUTSIDE a string literal. Single-line spacing is left + * alone, so `registry.get("a . b").template` keeps its argument exactly as + * written; literal-awareness extends that to Java text blocks and Kotlin raw + * strings, whose embedded newlines are part of the value and must survive + * (`"""line-a\n.line-b"""` is not the same string as `"""line-a.line-b"""`). + * + * Wraps that are not adjacent to a dot (`"a" +\n "b"`) are left as written: + * normalizing them would have to reason about operators, and the same + * conservatism already applies to receivers. + */ +export function normalizeSpringFactText(text: string): string { + const trimmed = text.trim(); + // Fast path: the overwhelming majority of captured text is single-line. + if (!trimmed.includes('\n') && !trimmed.includes('\r')) return trimmed; + + let out = ''; + let index = 0; + let quote: '"""' | '"' | "'" | null = null; + while (index < trimmed.length) { + const char = trimmed[index] as string; + if (quote === '"""') { + if (trimmed.startsWith('"""', index)) { + out += '"""'; + index += 3; + quote = null; + continue; + } + out += char; + index += 1; + continue; + } + if (quote !== null) { + // A backslash escape is copied whole so that `"\\"` ends the literal and + // `"\""` does not. + if (char === '\\' && index + 1 < trimmed.length) { + out += trimmed.slice(index, index + 2); + index += 2; + continue; + } + if (char === quote) quote = null; + out += char; + index += 1; + continue; + } + if (trimmed.startsWith('"""', index)) { + quote = '"""'; + out += '"""'; + index += 3; + continue; + } + if (char === '"' || char === "'") { + quote = char; + out += char; + index += 1; + continue; + } + if (char === '.' || /\s/.test(char)) { + const separator = /^\s*\.\s*/.exec(trimmed.slice(index)); + if (separator !== null) { + const matched = separator[0]; + out += matched.includes('\n') ? '.' : matched; + index += matched.length; + continue; + } + } + out += char; + index += 1; + } + return out; +} diff --git a/gitnexus/src/core/ingestion/frameworks/spring/config-bindings.ts b/gitnexus/src/core/ingestion/frameworks/spring/config-bindings.ts index 0baba4e88..e08273167 100644 --- a/gitnexus/src/core/ingestion/frameworks/spring/config-bindings.ts +++ b/gitnexus/src/core/ingestion/frameworks/spring/config-bindings.ts @@ -3,6 +3,7 @@ import type { KnowledgeGraph } from '../../../graph/types.js'; import { generateId } from '../../../../lib/utils.js'; export const SPRING_CONFIG_DESCRIPTION = 'Spring configuration property'; +export const SPRING_CONFIG_UNRESOLVED_PREFIX = 'Spring config unresolved: '; export interface SpringValueConsumer { readonly kind: 'value'; @@ -41,7 +42,7 @@ function closestNode( } function markUnresolved(node: GraphNode, key: string): void { - const marker = `Spring config unresolved: ${key}`; + const marker = `${SPRING_CONFIG_UNRESOLVED_PREFIX}${key}`; const existing = typeof node.properties.description === 'string' ? node.properties.description : ''; if (existing.includes(marker)) return; diff --git a/gitnexus/src/core/ingestion/frameworks/spring/destinations.ts b/gitnexus/src/core/ingestion/frameworks/spring/destinations.ts new file mode 100644 index 000000000..f02386192 --- /dev/null +++ b/gitnexus/src/core/ingestion/frameworks/spring/destinations.ts @@ -0,0 +1,1216 @@ +import type { SpringArgumentFact } from './argument-facts.js'; +import type { SpringMessageProducerTemplate } from './message-producers.js'; + +/** + * Resolution of Spring async messaging DESTINATIONS — the broker address a + * `@KafkaListener` reads from or a `kafkaTemplate.send(...)` writes to. + * + * The capture layer records the destination argument exactly as written and + * resolves nothing (see `argument-facts.ts`). This module is the other half: + * it decides WHICH argument names the destination, then walks a four-step + * cascade to turn that argument's source text into an address. It is pure — + * no graph, no filesystem, no parser — so every rule below is unit-testable + * against a string, and `pipeline-phases/spring-destinations.ts` is left with + * only node and edge emission. + * + * ── THE INVARIANT THIS MODULE EXISTS TO PROTECT ────────────────────────── + * + * An address that could NOT be resolved must never become a shared identity. + * Two unrelated services that each merely write + * + * @KafkaListener(topics = "${app.topic}") + * + * have said nothing about each other. If the graph keyed a destination node on + * that placeholder text, they would land on one node and READ AS CONNECTED — + * and a false edge is worse than a missing one, because a missing edge is + * visible as a gap while a false one enters reports as a fact. + * + * So this module never returns a placeholder, a constant name, or any other + * unresolved spelling as an `address`. An unresolved candidate comes back as + * `{ kind: 'unresolved', reason }` with no address at all, and the phase keys + * such a node by its SOURCE LOCATION. A status flag would not have been + * enough: the two services would still share whatever key the node was minted + * from. Only withholding the key prevents the join. + * + * ── REFUSAL IS DATA ────────────────────────────────────────────────────── + * + * Every path that declines to produce an address records WHY, from a closed + * set ({@link SpringDestinationRefusal}). The measure of this feature is the + * unresolved fraction, so a silent `continue` would hide precisely the number + * that says whether it works. + */ + +/** + * Broker family behind a destination, as far as the syntax can attest. + * + * Part of the `Destination` node IDENTITY, not merely a label on it: the phase + * keys a resolved node by `(broker, address)` via the framework-neutral + * `ingestion/destination-key.ts`, so two brokers claiming one address are two + * ordinary nodes. Adding or renaming a member here therefore re-keys every node + * it applies to, which a full re-index absorbs and an incremental one does not + * — the destination layer is delete-alled and rebuilt graph-wide on every + * incremental writeback for exactly this class of reason. + * + * A member is only added when the SYNTAX attests to it. A guess here becomes a + * guess in the identity, and the cost of a wrong one is a real pair split in + * two (see `destinationNodeKey` for why that cost is nonetheless the cheaper + * of the two failures available). + */ +export type SpringDestinationBroker = + | 'kafka' + | 'rabbit' + | 'jms' + | 'pulsar' + | 'sqs' + | 'stream' + | 'integration'; + +/** + * Why a candidate produced no address. Closed set: each member is a distinct, + * countable diagnosis, and no path may decline without naming one. + * + * Members are split rather than merged wherever the two causes are different + * FACTS about the repository. The unresolved fraction is only useful if its + * breakdown says what to go and fix, and a bucket that means "either the + * capture could not read this or the source really did write it that way" + * answers neither question. + */ +export type SpringDestinationRefusal = + /** The annotation is a recognized listener but its arguments were never read + * — a CAPTURE limitation, not a statement about the source. */ + | 'annotation-arguments-unavailable' + /** The annotation's argument list was read and it was EMPTY: `@KafkaListener` + * with no elements at all. A real source-level gap, and deliberately not the + * same bucket as `annotation-arguments-unavailable` — see + * `SpringNonHttpHandlerAnnotationFact.args`, which keeps absent and `[]` + * apart precisely so a consumer of the fact does not have to guess. */ + | 'annotation-arguments-empty' + /** Recognized listener, argument list present, no element names a destination. */ + | 'no-destination-argument' + /** `@KafkaListeners({@KafkaListener(...), ...})` and its siblings. The + * container's single argument is a list of NESTED annotations, and capture + * does not descend into them, so their destinations are unreadable here. + * Recorded rather than skipped: a repository using repeated-listener + * containers loses real destinations, and that has to show up in the count + * instead of looking like a repository with no listeners. */ + | 'repeated-listener-container' + /** `@KafkaListener(topicPattern = ...)` — a regex over topics, not an address. */ + | 'topic-pattern' + /** A destination form this module deliberately does not read, e.g. + * `@RabbitListener(bindings = @QueueBinding(...))` or `topicPartitions`. */ + | 'unsupported-annotation-argument' + /** A Kotlin trailing-lambda call: the publish has no argument list at all. */ + | 'producer-arguments-unavailable' + /** The call's arity matches none of the overloads that carry a destination. */ + | 'producer-arity-unrecognized' + /** The call used NAMED arguments and none of them names a destination + * parameter this module knows. Selecting by position instead would read + * whatever the author happened to write first — see + * {@link selectProducerDestinationArguments}. */ + | 'producer-named-argument-unrecognized' + /** `rabbitTemplate.convertAndSend(message)` — default exchange, empty routing + * key. There is no address in the source to record. */ + | 'rabbit-default-exchange' + /** The argument in the destination position is not shaped like an address + * (not a string literal, not a constant reference) — most often because the + * overload actually taken has the payload there. */ + | 'producer-argument-not-address-shaped' + /** Two overloads fit the call, they disagree about which slot is the address, + * and the argument is spelled the same way under both readings. The + * archetype is `convertAndSend("orders.rk", "body", correlationData)`: it is + * `(exchange, routingKey, message)` with the address `"body"`, or + * `(routingKey, message, correlationData)` with the address `"orders.rk"` + * and `"body"` as a String PAYLOAD. Both are real overloads spelled + * (String, String, ref). + * + * Distinct from `producer-argument-not-address-shaped`, which says the slot + * cannot hold an address at all. This one says it can, twice, and the module + * will not pick — a payload published as an address joins a consumer of a + * queue that happens to be named after the payload's text. */ + | 'ambiguous-producer-overload' + /** `topics = {}` / `topics = []` / `arrayOf()`. */ + | 'empty-destination-list' + /** The element is an expression this module will not evaluate — a + * concatenation, a call, a ternary. */ + | 'not-a-literal-or-constant' + /** A constant reference no constant resolver could fold to a string. */ + | 'unresolved-constant' + /** `#{...}` — a SpEL expression, evaluated by the container against beans and + * the environment at RUNTIME. `#{@kafkaProps.ordersTopic}` is the archetypal + * unresolvable address: nothing in the source says what it evaluates to, and + * two services that merely wrote the same expression have said nothing about + * each other. */ + | 'spel-expression' + /** An unescaped `$` interpolation in a language whose string literals + * interpolate. In Kotlin `"orders-$env"` and `"orders-${env}"` are STRING + * TEMPLATES evaluated at runtime, not addresses and not Spring placeholders + * — the escaped `"\${app.topic}"` is how a Spring placeholder has to be + * written there. Java does not interpolate, so `$` is an ordinary character + * and this never fires for it. */ + | 'unescaped-interpolation' + /** `${key}` with no default. The KEY is recorded; the VALUE is deliberately + * absent from the graph (config values may hold credentials — see the header + * of `pipeline-phases/spring-config.ts`), so this can never resolve here. */ + | 'unresolved-config-key' + /** `${key:default}`. The default IS written in the source, and it is kept on + * the node — but it is not an IDENTITY. It holds only while the key is not + * overridden in configuration, and configuration VALUES are deliberately + * absent from this graph, so the code cannot know whether it holds. Keying + * on it merges every service that copy-pasted the same fallback: `${a:events}` + * and `${b:events}` are two different addresses that happen to share a + * default. Both the key and the default text survive as properties, so the + * case stays countable and distinguishable from a bare `${key}`. */ + | 'overridable-config-default' + /** `${}` — a placeholder that names no key. There is nothing to record and + * nothing to look up; kept separate so an empty key never reaches the + * `Property` lookup as if it were a real one. */ + | 'empty-config-key' + /** A string literal that is empty or nothing but whitespace. An empty address + * addresses nothing, and letting it through would give every such site one + * shared `''` identity — the same false join the placeholder rule prevents. */ + | 'empty-literal-address' + /** A constant reference that folded to an empty or whitespace-only string. + * Same outcome as `empty-literal-address`, different repository fact: there + * the source wrote `""`, here a constant declaration did. */ + | 'empty-constant-address'; + +/** + * How an address was arrived at, kept on the node for provenance. + * + * There is deliberately no `config-default` member. A `${key:default}` does not + * resolve — see `overridable-config-default` — so no address can be reached + * that way. + */ +export type SpringDestinationVia = 'literal' | 'constant' | 'specification'; + +export type SpringDestinationRole = 'consumer' | 'producer'; + +/** + * One argument element that has been ACCEPTED as naming a destination, before + * any attempt to resolve it. An array-valued argument yields one candidate per + * element: `topics = ["a", "b"]` really is two destinations, and each gets its + * own node and its own edge (see the phase for why no group node is minted). + */ +export interface SpringDestinationCandidate { + readonly role: SpringDestinationRole; + /** Annotation simple name (`KafkaListener`) or producer template (`kafka`). */ + readonly source: string; + readonly broker: SpringDestinationBroker; + /** Index of the argument this element came from, in source order. */ + readonly argIndex: number; + /** Argument name when the call/annotation named it (`topics`, `queues`). */ + readonly argName?: string; + /** Index within an array-valued argument; `0` for a scalar. */ + readonly elementIndex: number; + /** The element's source text, exactly as captured. */ + readonly rawText: string; + /** Companion provenance that is not itself an address — currently only the + * Rabbit exchange that accompanies a routing key. */ + readonly exchange?: string; +} + +/** A candidate that was declined before resolution was even attempted. */ +export interface SpringDestinationRefusalRecord { + readonly role: SpringDestinationRole; + readonly source: string; + readonly broker: SpringDestinationBroker; + readonly reason: SpringDestinationRefusal; + /** Source text that provoked the refusal, when there was one. */ + readonly rawText?: string; + readonly argIndex?: number; + readonly argName?: string; +} + +export interface SpringDestinationSelection { + readonly candidates: readonly SpringDestinationCandidate[]; + readonly refusals: readonly SpringDestinationRefusalRecord[]; +} + +export type SpringDestinationResolution = + | { readonly kind: 'resolved'; readonly address: string; readonly via: SpringDestinationVia } + | { + readonly kind: 'unresolved'; + readonly reason: SpringDestinationRefusal; + /** Configuration key named by an unresolvable `${...}` placeholder. Lets + * the phase link the node to the `Property` nodes for that key without + * ever learning the key's value. */ + readonly configKey?: string; + /** Default text of a `${key:default}`, exactly as the source wrote it. + * Kept as PROVENANCE only — it is never an address and never a key, for + * the reason `overridable-config-default` gives. */ + readonly configDefault?: string; + }; + +/** + * The cascade's pluggable steps plus the one language capability it needs. + * + * The steps are supplied by the phase, which owns the language-specific + * machinery; keeping them as callbacks is what lets this module stay + * language-neutral and testable with a plain map. + */ +export interface SpringDestinationResolvers { + /** + * Whether the owning language INTERPOLATES string literals — Kotlin does, + * Java does not. A capability, deliberately not a language name: shared + * ingestion code may not branch on a language (see AGENTS.md), and the + * capability is also the thing that actually matters. Supplied alongside + * `getSpringMessagingFacts` by the provider and threaded in by the phase. + * + * When true, an unescaped `$` inside a literal is a runtime template and the + * candidate is refused. When false (the default) `$` is an ordinary + * character and `"${app.topic}"` is a Spring placeholder. + */ + readonly interpolatesStringLiterals?: boolean; + /** + * Step 2 — fold a constant reference (`Topics.ORDERS`, `ORDERS`) to its + * string value, or `null` when it cannot be folded. Backed by + * `resolveJavaConstant` / `resolveKotlinConstant`. + */ + readonly constant?: (name: string) => string | null; + /** + * Step 4 — SEAM, DELIBERATELY NOT IMPLEMENTED. + * + * Some destinations are named nowhere in the source: the address lives in a + * published API specification (AsyncAPI / springwolf) that the service + * generates, and the code only names a binding. Resolving those means reading + * an artifact that is not a source file, deciding which specification belongs + * to which module, and trusting a generated document — a different problem + * from the three syntactic steps above, with a different failure mode. + * + * The hook exists so that work has a defined place to land and so the cascade + * order is fixed now rather than renegotiated later. Nothing supplies it + * today, so step 4 is a no-op and such destinations stay unresolved with the + * reason the earlier step recorded. + */ + readonly specification?: (candidate: SpringDestinationCandidate) => string | null; +} + +// ── Consumer side: which annotation argument names the destination ───────── + +interface ConsumerAnnotationRule { + readonly broker: SpringDestinationBroker; + /** Argument names that carry an address, in preference order. */ + readonly addressArgs: readonly string[]; + /** + * A bare positional argument is the annotation's `value` element. Accepted + * only where `value` really is the destination: `@SqsListener("q")` and + * `@StreamListener("ch")`. `@KafkaListener`, `@RabbitListener`, `@JmsListener` + * and `@ServiceActivator` declare no `value` alias for their destination, so + * a positional argument on one of those is something else entirely and is + * refused rather than guessed at. + */ + readonly positionalIsAddress: boolean; + /** Arguments that are patterns over addresses, not addresses. */ + readonly patternArgs?: readonly string[]; + /** Arguments that name a destination in a shape this module will not read. */ + readonly unsupportedArgs?: readonly string[]; +} + +/** + * Recognized listener annotations, keyed by SIMPLE name. + * + * Simple names, not fully-qualified ones, because a pipeline phase runs after + * scope resolution has finished and no longer has the import tables that + * `createSpringAnnotationNameResolver` needs. The capture layer already gates + * on simple names for the same reason (`CAPTURE_RELEVANT_SIMPLE_NAMES` in + * `non-http-handlers.ts`), so nothing reaches this map that was not already + * admitted on that basis; matching on the FQN here would only reject facts the + * capture had already accepted, never admit more. + * + * DELIBERATELY ABSENT: `@MessageMapping` and `@SubscribeMapping`. Both are + * recognized by `non-http-handlers.ts` as message handlers, and both are + * WebSocket/STOMP routes — an application-level destination inside a + * server-managed session, not an address on a broker. Modelling `/topic/prices` + * as a `Destination` would put a STOMP path in the same namespace as a Kafka + * topic and let the cross-service joiner match them. + */ +const CONSUMER_ANNOTATIONS: ReadonlyMap = new Map([ + [ + 'KafkaListener', + { + broker: 'kafka' as const, + addressArgs: ['topics'], + positionalIsAddress: false, + patternArgs: ['topicPattern'], + unsupportedArgs: ['topicPartitions'], + }, + ], + [ + 'PulsarListener', + { + broker: 'pulsar' as const, + addressArgs: ['topics'], + positionalIsAddress: false, + patternArgs: ['topicPattern'], + }, + ], + [ + 'RabbitListener', + { + broker: 'rabbit' as const, + addressArgs: ['queues'], + positionalIsAddress: false, + unsupportedArgs: ['bindings', 'queuesToDeclare'], + }, + ], + [ + 'JmsListener', + { broker: 'jms' as const, addressArgs: ['destination'], positionalIsAddress: false }, + ], + [ + 'ServiceActivator', + { broker: 'integration' as const, addressArgs: ['inputChannel'], positionalIsAddress: false }, + ], + ['SqsListener', { broker: 'sqs' as const, addressArgs: ['value'], positionalIsAddress: true }], + [ + 'StreamListener', + { broker: 'stream' as const, addressArgs: ['value'], positionalIsAddress: true }, + ], +]); + +/** + * Plural container annotations (`@KafkaListeners`, `@RabbitListeners`, …) wrap + * repeated listeners. Their single argument is a list of nested annotations, + * whose own arguments the capture does not descend into, so there is nothing + * here to read. + * + * They are recognized rather than ignored so the loss is COUNTED. A repository + * that declares its listeners this way really does lose those destinations, and + * returning an empty selection would make it indistinguishable from a + * repository with no listeners at all — the module header promises that every + * path which declines to produce an address records why, and an empty + * `refusals` array records nothing. The broker comes from the container's own + * name, which is the one thing the annotation does state. + */ +const CONSUMER_CONTAINER_ANNOTATIONS: ReadonlyMap = new Map([ + ['KafkaListeners', 'kafka' as const], + ['RabbitListeners', 'rabbit' as const], + ['JmsListeners', 'jms' as const], + ['PulsarListeners', 'pulsar' as const], +]); + +function simpleName(name: string): string { + const separator = name.lastIndexOf('.'); + return separator === -1 ? name : name.slice(separator + 1); +} + +/** + * Choose the destination-bearing arguments of one listener annotation. + * + * Returns `null` when the annotation is not a broker listener at all — that is + * not a refusal, there was nothing to refuse. A recognized annotation always + * returns a selection, even when every path in it declined, so the caller can + * count what was seen against what resolved. + */ +export function selectConsumerDestinationArguments( + annotationName: string, + args: readonly SpringArgumentFact[] | undefined, +): SpringDestinationSelection | null { + const name = simpleName(annotationName); + const containerBroker = CONSUMER_CONTAINER_ANNOTATIONS.get(name); + if (containerBroker !== undefined) { + return { + candidates: [], + refusals: [ + { + role: 'consumer', + source: name, + broker: containerBroker, + reason: 'repeated-listener-container', + ...(args === undefined || args[0] === undefined ? {} : { rawText: args[0].text }), + }, + ], + }; + } + const rule = CONSUMER_ANNOTATIONS.get(name); + if (rule === undefined) return null; + + const refusals: SpringDestinationRefusalRecord[] = []; + const refuse = ( + reason: SpringDestinationRefusal, + extra: Omit = {}, + ): void => { + refusals.push({ role: 'consumer', source: name, broker: rule.broker, reason, ...extra }); + }; + + // ABSENT arguments are a capture limitation: the annotation was recognized + // but its argument list was never read (see + // `SpringNonHttpHandlerAnnotationFact.args`). An empty ARRAY is a different + // fact entirely — an argument list WAS read and it was empty, so the source + // really does declare a listener that names no destination. Capture keeps the + // two apart on purpose, the producer side of this module already does, and + // merging them here would file a source-level gap under a tooling gap and + // corrupt the one breakdown this feature is measured on. + if (args === undefined) { + refuse('annotation-arguments-unavailable'); + return { candidates: [], refusals }; + } + if (args.length === 0) { + refuse('annotation-arguments-empty'); + return { candidates: [], refusals }; + } + + const candidates: SpringDestinationCandidate[] = []; + let sawDestinationArgument = false; + for (const [argIndex, arg] of args.entries()) { + const argName = arg.name; + if (argName === undefined) { + // Positional. Only the annotations whose `value` element IS the + // destination accept it; on the others a positional argument is a + // different element entirely and gets no guess. + if (!rule.positionalIsAddress) continue; + sawDestinationArgument = true; + pushElements(candidates, refusals, { + role: 'consumer', + source: name, + broker: rule.broker, + argIndex, + rawText: arg.text, + }); + continue; + } + if (rule.patternArgs?.includes(argName)) { + sawDestinationArgument = true; + refuse('topic-pattern', { rawText: arg.text, argIndex, argName }); + continue; + } + if (rule.unsupportedArgs?.includes(argName)) { + sawDestinationArgument = true; + refuse('unsupported-annotation-argument', { rawText: arg.text, argIndex, argName }); + continue; + } + if (!rule.addressArgs.includes(argName)) continue; + sawDestinationArgument = true; + pushElements(candidates, refusals, { + role: 'consumer', + source: name, + broker: rule.broker, + argIndex, + argName, + rawText: arg.text, + }); + } + + // A listener whose arguments were read and named `groupId` and + // `containerFactory` but no destination is a real, countable gap — most often + // a form this module has not learned. It must not be silent. + if (!sawDestinationArgument) refuse('no-destination-argument'); + return { candidates, refusals }; +} + +// ── Producer side: which call argument names the destination ─────────────── + +/** + * Parameter names that carry a destination, per template, for calls that pass + * their arguments BY NAME. + * + * Kotlin call sites may name arguments, and a named argument list is in source + * order, not parameter order — `send(data = payload, topic = "orders")` is + * legal and puts the payload in slot 0. Reading slot 0 there publishes the + * PAYLOAD as an address. The name is captured + * ({@link SpringArgumentFact.name}), so the honest rule is to use it: select by + * name when there is one, and refuse when the names present say nothing this + * module recognizes. Selecting by position while ignoring a name that + * contradicts it is the one option that is never defensible. + * + * `exchange` is listed for rabbit but is NOT an address — it is the companion + * provenance the routing key carries (see the arity notes below). + */ +const PRODUCER_DESTINATION_PARAMETERS: Readonly< + Record +> = { + kafka: ['topic'], + // `RabbitTemplate.convertAndSend(String exchange, String routingKey, Object message, …)`. + rabbit: ['routingKey'], + // `JmsTemplate.convertAndSend(Destination destination, …)` and the + // `String destinationName` overloads. + jms: ['destination', 'destinationName'], + // `StreamBridge.send(String bindingName, Object data, …)`. + 'stream-bridge': ['bindingName'], +}; + +/** Rabbit's exchange parameter, carried as provenance rather than as an address. */ +const RABBIT_EXCHANGE_PARAMETER = 'exchange'; + +/** + * Choose the destination-bearing arguments of one messaging-template publish. + * + * A NAME beats a position, arity decides where it can decide, and shape decides + * where it cannot. + * + * When any argument is passed by name, {@link PRODUCER_DESTINATION_PARAMETERS} + * decides — position is not consulted at all, because a named argument list + * need not be in parameter order. When the slot this module would have read + * positionally is itself named with something it does not recognize, that is a + * contradiction and the publish is refused rather than read. + * + * `KafkaTemplate.send` and `StreamBridge.send` put the destination first in + * every multi-argument positional overload they have, so once such a call has + * two or more arguments its slot 0 is the destination and nothing further needs + * deciding. Those slots use the PERMISSIVE gate ({@link isAddressShaped}): a + * bare identifier is let through to the cascade, which refuses it by name if no + * constant folds. That keeps `unresolved-constant` — a thing we tried to + * resolve — distinct from `producer-argument-not-address-shaped`, a thing we + * declined to read at all. + * + * The `convertAndSend` families are different. Both admit trailing + * `MessagePostProcessor` and `CorrelationData` parameters, and arity does not + * separate the overloads in EITHER direction: + * + * jms (destination, message) 2 vs (message, postProcessor) 2 + * rabbit (routingKey, message) 2 vs (message, postProcessor) 2 + * rabbit (exchange, routingKey, message) 3 vs (routingKey, message, pp) 3 + * vs (routingKey, message, correlation) 3 + * rabbit (exchange, routingKey, message, pp) 4 vs (routingKey, message, pp, corr) 4 + * + * So the tie is broken by the STRICT gate ({@link isConfidentAddressShape}) — a + * string literal, a qualified reference, or a screaming-snake constant, all of + * which a payload variable is not. A lowercase bare identifier is NOT confident + * evidence, so `convertAndSend(topic, payload)` is refused rather than read: + * the same spelling is how a payload variable looks, and nothing in the syntax + * separates them. That refusal is the deliberate cost. A refusal is counted and + * recoverable; a wrong address enters reports as a fact. + * + * There is NO positional fallback at rabbit arity 3+. An earlier revision fell + * back to accepting slot 0 when slot 1 was not confident, which turned the + * ordinary `convertAndSend(EXCHANGE, routingKey, event)` — routing key in a + * variable — into a destination whose address was the EXCHANGE NAME. That is + * the worst possible outcome: an address that looks entirely plausible and can + * join a `@RabbitListener(queues = "orders")` that has nothing to do with it. + * + * ── THE ONE AMBIGUITY, REFUSED RATHER THAN GUESSED ─────────────────────── + * + * `convertAndSend("orders.rk", "body", correlationData)` fits two overloads at + * once and they disagree about which slot is the address: + * + * (exchange, routingKey, message) → the address is `"body"` + * (routingKey, message, correlationData) → the address is `"orders.rk"` + * + * Both are real, both are spelled (String, String, ref), and no rule over the + * syntax separates them. Picking either one publishes the OTHER reading's + * payload as an address, where a consumer of a queue named after that text + * joins a publisher that never wrote to it. So neither is picked: the call is + * refused as `ambiguous-producer-overload` and yields no candidate and no edge. + * + * The refusal is narrow on purpose, because over-refusing here costs the + * ordinary case, and a suppression that eats correct results is the more + * expensive mistake. It fires ONLY on a STRING LITERAL in slot 1, at the + * arities where a competing overload exists: + * + * - A literal is no evidence at all. An address and a payload are BOTH + * ordinarily written as literals, so the spelling distinguishes nothing. + * - A CONSTANT or QUALIFIED reference is evidence, which is the whole premise + * of {@link isConfidentAddressShape}: `ORDERS_ROUTING_KEY` and + * `Topics.ORDERS_KEY` are how a configured NAME is written, not how a + * payload computed at the call site is. Those keep resolving. + * - Arity 5 has no competing overload at all — + * `(exchange, routingKey, message, pp, correlationData)` is the only + * five-argument form — so slot 1 there is the routing key whatever it is + * spelled like, and the refusal must not reach it. + * - A NAMED argument settles the reading outright, and the name-beats-position + * pre-pass above has already returned by then. + */ +export function selectProducerDestinationArguments(fact: { + readonly template: SpringMessageProducerTemplate; + readonly methodName: string; + readonly args?: readonly SpringArgumentFact[]; +}): SpringDestinationSelection { + const broker: SpringDestinationBroker = + fact.template === 'stream-bridge' ? 'stream' : fact.template; + const source = fact.template; + const refusals: SpringDestinationRefusalRecord[] = []; + const refuse = ( + reason: SpringDestinationRefusal, + extra: Omit = {}, + ): void => { + refusals.push({ role: 'producer', source, broker, reason, ...extra }); + }; + + const args = fact.args; + if (args === undefined) { + refuse('producer-arguments-unavailable'); + return { candidates: [], refusals }; + } + if (args.length === 0) { + refuse('producer-arity-unrecognized'); + return { candidates: [], refusals }; + } + + const candidates: SpringDestinationCandidate[] = []; + const accept = (argIndex: number, exchange?: string): void => { + const arg = args[argIndex] as SpringArgumentFact; + pushElements(candidates, refusals, { + role: 'producer', + source, + broker, + argIndex, + ...(arg.name === undefined ? {} : { argName: arg.name }), + rawText: arg.text, + ...(exchange === undefined ? {} : { exchange }), + }); + }; + /** + * Accept a slot chosen by POSITION. + * + * Refuses when the argument in that slot carries a name — a named list need + * not be in parameter order, so a name in the destination slot that is not a + * destination parameter contradicts the position, and reading the position + * anyway is how `send(data = "payload", topic = "orders")` published the + * payload. The name-matching pre-pass above has already had its chance. + */ + const acceptPositional = (argIndex: number, exchange?: string): void => { + const arg = args[argIndex] as SpringArgumentFact; + if (arg.name !== undefined) { + refuse('producer-named-argument-unrecognized', { + rawText: arg.text, + argIndex, + argName: arg.name, + }); + return; + } + accept(argIndex, exchange); + }; + const textAt = (index: number): string => (args[index] as SpringArgumentFact).text; + const confident = (index: number): boolean => + index < args.length && isConfidentAddressShape(textAt(index)); + const refuseShape = (index: number): void => { + refuse('producer-argument-not-address-shaped', { rawText: textAt(index), argIndex: index }); + }; + + // ── A name beats a position ───────────────────────────────────────────── + // When an argument names a destination parameter, that argument IS the + // destination wherever it sits in the list. Only when no name matches does + // the positional reasoning below run, and `acceptPositional` then refuses if + // the slot it lands on turns out to be named after something else. + const destinationNames = PRODUCER_DESTINATION_PARAMETERS[fact.template]; + const namedIndex = args.findIndex( + (arg) => arg.name !== undefined && destinationNames.includes(arg.name), + ); + if (namedIndex !== -1) { + const exchangeIndex = + fact.template === 'rabbit' + ? args.findIndex((arg) => arg.name === RABBIT_EXCHANGE_PARAMETER) + : -1; + accept( + namedIndex, + exchangeIndex === -1 ? undefined : unquoteForProvenance(textAt(exchangeIndex)), + ); + return { candidates, refusals }; + } + + if (fact.template === 'rabbit') { + // `convertAndSend` overloads, by what occupies the leading slots: + // (message) → default exchange, no address + // (routingKey, message) → arg0 is the routing key + // (message, postProcessor) → NO address, same arity + // (exchange, routingKey, message) → arg0 + arg1 + // (routingKey, message, postProcessor) → arg0 only, same arity + // (routingKey, message, correlationData) → arg0 only, same arity + // (exchange, routingKey, message, pp) → arg0 + arg1 + // (routingKey, message, pp, correlationData) → arg0 only, same arity + // (exchange, routingKey, message, pp, corr) → arg0 + arg1 + if (args.length === 1) { + refuse('rabbit-default-exchange', { rawText: textAt(0), argIndex: 0 }); + return { candidates, refusals }; + } + if (args.length === 2) { + // (routingKey, message) versus (message, postProcessor). + if (confident(0)) { + acceptPositional(0); + return { candidates, refusals }; + } + refuseShape(0); + return { candidates, refusals }; + } + // Three arguments and up. Arity separates almost nothing here — three and + // four both admit an exchange form and a routing-key form — so the ONLY + // acceptance is confident evidence in slot 1, and there is no positional + // fallback. `convertAndSend(EXCHANGE, routingKey, event)` fails that test + // and is refused; the discarded fallback published `EXCHANGE` as the + // address, which is a wrong answer wearing the costume of a right one. + // + // And confident evidence in slot 1 is not enough when the evidence is a + // STRING LITERAL: under the competing overload that same literal is the + // String PAYLOAD, and the two readings are spelled identically. See the + // ambiguity section in this function's doc comment for why this is a + // refusal rather than a choice, and for each of the three cases it must not + // touch — a constant in slot 1 (spelling that IS evidence), arity 5 (no + // competing overload exists), and a named argument (already returned + // above, and its name settles the reading). + const competingOverload = args.length === 3 || args.length === 4; + if ( + competingOverload && + args[1]?.name === undefined && + parseSpringStringLiteral(textAt(1)) !== null + ) { + refuse('ambiguous-producer-overload', { rawText: textAt(1), argIndex: 1 }); + return { candidates, refusals }; + } + if (confident(1)) { + // The ADDRESS is the routing key. The exchange rides along as provenance + // on the edge rather than becoming part of the address: composing + // `exchange/routingKey` would invent a spelling no consumer ever writes, + // and a `@RabbitListener` names a QUEUE, so the two sides do not join on + // the exchange anyway. Which queue an exchange/key pair reaches is decided + // by bindings this index does not read. + acceptPositional(1, unquoteForProvenance(textAt(0))); + return { candidates, refusals }; + } + refuseShape(1); + return { candidates, refusals }; + } + + // kafka `send(topic, …)`, jms `convertAndSend(destination, message, …)` and + // stream-bridge `send(binding, …)` all put the destination first and all + // require at least one further argument for the payload. A single-argument + // call is therefore one of the payload-only overloads — + // `send(ProducerRecord)`, `send(Message)`, `convertAndSend(Object)` — which + // carries its destination inside an object this module does not open. + if (args.length < 2) { + refuse('producer-arity-unrecognized', { rawText: textAt(0), argIndex: 0 }); + return { candidates, refusals }; + } + // Two-argument `convertAndSend` is the one JMS arity that collides with the + // post-processor overload, so only there does slot 0 need confident evidence. + const strict = fact.template === 'jms' && args.length === 2; + if (strict ? !confident(0) : !isAddressShaped(textAt(0))) { + refuseShape(0); + return { candidates, refusals }; + } + acceptPositional(0); + return { candidates, refusals }; +} + +// ── Array / literal / placeholder text handling ──────────────────────────── + +/** + * Split an array-valued destination argument into its elements. + * + * Both languages hand this module ONE unsplit string per argument: capture + * records an argument's source text, and `topics = {"a", "b"}` is a single + * argument whose text happens to be a list. So the list is parsed here, in the + * three spellings the two languages use — Java `{…}`, Kotlin `[…]`, and Kotlin + * `arrayOf(…)`. + * + * Anything else comes back as a single element, unchanged: a scalar argument, + * and equally an expression that merely starts with a brace. The split tracks + * nesting and string literals, so a comma inside a literal or inside a nested + * call does not split the list. + * + * Returns `[]` for an empty list, which the caller must distinguish from a + * one-element list — `topics = {}` names no destination at all. + */ +export function splitSpringDestinationList(text: string): readonly string[] { + const trimmed = text.trim(); + let inner: string | null = null; + if (trimmed.startsWith('{') && trimmed.endsWith('}')) inner = trimmed.slice(1, -1); + else if (trimmed.startsWith('[') && trimmed.endsWith(']')) inner = trimmed.slice(1, -1); + else if (/^arrayOf\s*\(/.test(trimmed) && trimmed.endsWith(')')) { + inner = trimmed.slice(trimmed.indexOf('(') + 1, -1); + } + if (inner === null) return [trimmed]; + if (inner.trim() === '') return []; + + const elements: string[] = []; + let current = ''; + let depth = 0; + let quote: '"""' | '"' | "'" | null = null; + for (let index = 0; index < inner.length; index += 1) { + const char = inner[index] as string; + if (quote === '"""') { + current += char; + if (inner.startsWith('"""', index)) { + current += '""'; + index += 2; + quote = null; + } + continue; + } + if (quote !== null) { + if (char === '\\' && index + 1 < inner.length) { + current += inner.slice(index, index + 2); + index += 1; + continue; + } + current += char; + if (char === quote) quote = null; + continue; + } + if (inner.startsWith('"""', index)) { + current += '"""'; + index += 2; + quote = '"""'; + continue; + } + if (char === '"' || char === "'") { + quote = char; + current += char; + continue; + } + if (char === '(' || char === '[' || char === '{') depth += 1; + else if (char === ')' || char === ']' || char === '}') depth -= 1; + if (char === ',' && depth === 0) { + elements.push(current.trim()); + current = ''; + continue; + } + current += char; + } + elements.push(current.trim()); + return elements.filter((element) => element !== ''); +} + +/** + * ONE string literal, whole. + * + * The triple-quoted alternative excludes `"""` from its body rather than + * matching greedily: `"""a""" + """b"""` is a concatenation, not a literal, and + * a greedy body swallowed the operator and folded it to the single address + * `a""" + """b`. Excluding the terminator makes the whole-string anchor fail + * there, so the text falls through to the constant test and is refused as + * `not-a-literal-or-constant`, which is what it is. + */ +const STRING_LITERAL = + /^(?:"""((?:(?!""")[\s\S])*)"""|"((?:[^"\\]|\\[\s\S])*)"|'((?:[^'\\]|\\[\s\S])*)')$/; + +/** + * Unquote a string literal to its value, or `null` when the text is not a + * single literal. + * + * Escapes are undone only for the sequences that can appear inside a + * destination: `\"`, `\\`, and Kotlin's `\$`. That last one matters more than it + * looks — a Spring placeholder written in Kotlin MUST escape the dollar + * (`"\${app.topic}"`) or the compiler reads it as a string template, so without + * undoing it every Kotlin placeholder would fail the `${` test below and be + * misfiled as a plain literal address named `\${app.topic}`. + * + * The unescaping is also why {@link hasUnescapedStringInterpolation} has to run + * against the RAW text: once `\$` has become `$`, the escaped placeholder and + * the runtime template are the same string. + */ +export function parseSpringStringLiteral(text: string): string | null { + const match = STRING_LITERAL.exec(text.trim()); + if (match === null) return null; + const raw = match[1] ?? match[2] ?? match[3] ?? ''; + return raw.replace(/\\(["'\\$nrt])/g, (_all, escaped: string) => { + if (escaped === 'n') return '\n'; + if (escaped === 'r') return '\r'; + if (escaped === 't') return '\t'; + return escaped; + }); +} + +/** `$` followed by a brace or an identifier start — the two template forms. */ +const INTERPOLATION_START = /^[{A-Za-z_]/; + +/** + * Whether a string literal contains an UNESCAPED interpolation, for a language + * whose literals interpolate. + * + * Only meaningful for such a language; Java never calls it. In Kotlin: + * + * "orders-$env" template — the value is decided at runtime + * "orders-${env}" template — NOT a Spring placeholder + * "\${app.topic}" escaped — this is how a Spring placeholder is written + * """orders-$env""" template — raw strings interpolate and cannot escape + * + * Reads the RAW literal text on purpose: {@link parseSpringStringLiteral} + * resolves `\$` to `$`, after which the second and third rows above are the + * same string and the distinction is gone. A raw (`"""`) literal has no + * backslash escapes at all — `${'$'}` is the only way to write a dollar there — + * so every `$` in one is an interpolation. + */ +export function hasUnescapedStringInterpolation(text: string): boolean { + const trimmed = text.trim(); + const match = STRING_LITERAL.exec(trimmed); + if (match === null) return false; + const raw = match[1] ?? match[2] ?? match[3] ?? ''; + const escapable = match[1] === undefined; + for (let index = 0; index < raw.length; index += 1) { + const char = raw[index] as string; + if (escapable && char === '\\') { + index += 1; + continue; + } + if (char === '$' && INTERPOLATION_START.test(raw.slice(index + 1))) return true; + } + return false; +} + +/** `#{...}` — a SpEL expression the container evaluates at runtime. */ +function containsSpelExpression(value: string): boolean { + return value.includes('#{'); +} + +/** A dotted or bare identifier — the only non-literal shape read as a constant. */ +const CONSTANT_REFERENCE = /^[A-Za-z_$][A-Za-z0-9_$]*(?:\s*\.\s*[A-Za-z_$][A-Za-z0-9_$]*)*$/; + +/** + * PERMISSIVE gate — true when the text could name an address: a string literal, + * or any reference a constant resolver could plausibly fold. Says nothing about + * whether that reference actually resolves; the cascade decides that and + * records `unresolved-constant` when it does not. + * + * Used where the overload set already fixes which slot holds the destination. + */ +export function isAddressShaped(text: string): boolean { + const trimmed = text.trim(); + if (parseSpringStringLiteral(trimmed) !== null) return true; + return CONSTANT_REFERENCE.test(trimmed); +} + +/** A reference whose spelling is evidence in itself: qualified (`Topics.ORDERS`) + * or a screaming-snake constant (`ORDERS_TOPIC`). `this.x` is excluded — the + * qualifier says nothing about the member. */ +const CONFIDENT_REFERENCE = /^(?!this\s*\.)[A-Za-z_$][A-Za-z0-9_$]*\s*\.\s*[A-Za-z0-9_$.\s]+$/; +const SCREAMING_SNAKE = /^[A-Z][A-Z0-9_$]*$/; + +/** + * STRICT gate — true only when the spelling is confident evidence of an + * address, not merely compatible with one. + * + * The difference from {@link isAddressShaped} is the lowercase bare identifier. + * `convertAndSend(topic, payload)` and `convertAndSend(message, processor)` are + * the same syntax; only a human reading the names can tell which slot is the + * destination, and a name is not something this module is willing to rank as + * evidence. So a bare `topic` fails here and the publish is refused, while + * `"orders"`, `Topics.ORDERS` and `ORDERS_TOPIC` pass. + * + * Used ONLY at the arities where a trailing `MessagePostProcessor` overload + * collides with the destination-carrying one. Everywhere else the permissive + * gate applies, so this stricter rule costs nothing outside the ambiguity. + */ +export function isConfidentAddressShape(text: string): boolean { + const trimmed = text.trim(); + if (parseSpringStringLiteral(trimmed) !== null) return true; + if (!CONSTANT_REFERENCE.test(trimmed)) return false; + return CONFIDENT_REFERENCE.test(trimmed) || SCREAMING_SNAKE.test(trimmed); +} + +/** Best-effort display form for provenance text; never used as an identity. */ +function unquoteForProvenance(text: string): string { + return parseSpringStringLiteral(text) ?? text.trim(); +} + +export interface SpringPlaceholderResult { + /** True when the text contained no `${…}` at all. */ + readonly plain: boolean; + /** Key of the FIRST placeholder, in source order. Present whenever `plain` + * is false; the empty string when the placeholder named no key (`${}`). */ + readonly key?: string; + /** Default text of that placeholder, exactly as written, when it had one. + * Absent for a bare `${key}`. The empty string for `${key:}`, which is a + * default that was written and is empty. */ + readonly defaultValue?: string; +} + +/** + * Read the FIRST Spring property placeholder out of an already-unquoted value. + * + * NOTHING IS SUBSTITUTED, and that is the rule, not an omission. + * + * `${key}` cannot resolve: the value lives in a configuration file this index + * deliberately does not read into the graph (values may hold credentials — see + * `pipeline-phases/spring-config.ts`). The KEY comes back instead, so the + * caller can link the node to the `Property` nodes for that key without ever + * learning its value. + * + * `${key:default}` cannot resolve EITHER, which is a correction to this + * module's original rule. The default is written in the source, so reading it + * is legitimate and it is returned — but it is provenance, never an identity. + * A default holds only while the key is not overridden, and whether it is + * overridden is a fact about configuration VALUES, which are absent from this + * graph by design. Substituting it made `${a.topic:events}` and + * `${b.topic:events}` one node and reported a producer/consumer pair between + * two services that shared nothing but a copy-pasted fallback. The same + * reasoning applies to any placeholder-derived value: a value the configuration + * can override is not an identity. + * + * Only the first placeholder is read because there is nothing to do with the + * rest — the value is already unresolvable, and the first key is the one a + * reader would look up. A nested default (`${a:${b}}`) needs no special case + * under this rule: `a` is the key and `${b}` is the default text, both reported + * as written. + */ +export function resolveSpringPlaceholders(value: string): SpringPlaceholderResult { + const start = value.indexOf('${'); + if (start === -1) return { plain: true }; + let depth = 1; + let cursor = start + 2; + while (cursor < value.length && depth > 0) { + if (value.startsWith('${', cursor)) { + depth += 1; + cursor += 2; + continue; + } + if (value[cursor] === '}') depth -= 1; + cursor += 1; + } + // An unterminated `${` is not a placeholder this module can read. Treating + // the tail as a literal would mint an address containing `${`; treating it as + // a key at least names the thing the author was reaching for. + if (depth > 0) return { plain: false, key: value.slice(start + 2).trim() }; + const body = value.slice(start + 2, cursor - 1); + const separator = body.indexOf(':'); + // Spring splits on the FIRST colon, so `${a:b:c}` defaults to `b:c`. + if (separator === -1) return { plain: false, key: body.trim() }; + return { + plain: false, + key: body.slice(0, separator).trim(), + defaultValue: body.slice(separator + 1), + }; +} + +// ── The cascade ──────────────────────────────────────────────────────────── + +/** + * Resolve one candidate to an address, or to a named refusal. + * + * Four steps, in this order, each of which may decline: + * + * 1. literal — `"orders.v1"`, including one element of an array form. + * 2. constant — `Topics.ORDERS`, through the supplied constant resolver. + * 3. configuration — neither `${app.topic}` nor `${app.topic:orders}` + * resolves; the key, and the default text when there is + * one, are reported instead. + * 4. specification — the deferred seam; see {@link SpringDestinationResolvers}. + * + * Steps 1 and 2 both feed step 3: a literal may be a placeholder, and so may + * the value a constant folds to (`static final String TOPIC = "${app.topic}"` + * is an ordinary way to write one). Skipping step 3 after step 2 would file + * that constant's placeholder text as a resolved address — the exact false + * identity this module exists to prevent, arrived at one step later. + * + * Two classes of text are rejected BEFORE step 3, because they are not + * addresses in any configuration: a SpEL expression, which the container + * evaluates against live beans, and an unescaped string-template interpolation + * in a language that interpolates. Order matters between them and the + * placeholder rule — `"#{'${app.topics}'.split(',')}"` contains a `${` and + * would otherwise be filed under a configuration key that is not really what + * it is. + * + * WHITESPACE. An address is kept exactly as the source wrote it, `" orders "` + * included, so `" orders "` is its own node and does not join `"orders"`. That + * is a missing connection rather than a false one, which is the trade this + * module makes everywhere. The emptiness test below trims, because a + * whitespace-only address addresses nothing — the two rules disagree on + * purpose, and this is the statement of it. + */ +export function resolveSpringDestination( + candidate: SpringDestinationCandidate, + resolvers: SpringDestinationResolvers = {}, +): SpringDestinationResolution { + const specification = (): SpringDestinationResolution | null => { + const resolved = resolvers.specification?.(candidate); + if (resolved === undefined || resolved === null || resolved === '') return null; + return { kind: 'resolved', address: resolved, via: 'specification' }; + }; + + const literal = parseSpringStringLiteral(candidate.rawText); + if (literal !== null) { + // The raw spelling, not the unquoted value: unquoting has already turned + // `\$` into `$` and the escaped placeholder into the runtime template. + if ( + resolvers.interpolatesStringLiterals === true && + hasUnescapedStringInterpolation(candidate.rawText) + ) { + return specification() ?? { kind: 'unresolved', reason: 'unescaped-interpolation' }; + } + return finish(literal, 'literal', specification); + } + + const trimmed = candidate.rawText.trim(); + if (CONSTANT_REFERENCE.test(trimmed)) { + const folded = resolvers.constant?.(trimmed.replace(/\s*\.\s*/g, '.')) ?? null; + if (folded === null) + return specification() ?? { kind: 'unresolved', reason: 'unresolved-constant' }; + // A folded value has already lost its escapes, so an interpolating language + // cannot tell `"\${app.topic}"` from `"${app.topic}"` here the way the + // literal branch can. Both are unresolved either way, so the cost is a + // reason filed under `unescaped-interpolation` that might have belonged + // under `unresolved-config-key` — never a false address. + // + // A LIVE PATH, not a guard for the future. Kotlin both interpolates and + // supplies a constant fold — `languages/kotlin.ts` declares + // `extractModuleConstants` and `foldRoutePathOperands`, and + // `spring-destinations.ts` hands the fold to this cascade — so a Kotlin + // constant reaching this branch is an ordinary occurrence and the misfiled + // reason above is a cost actually paid. Fixing it means teaching the fold + // to report whether the value it returned was escaped at its declaration, + // which the shared `ModuleConstants` shape does not carry. + if (resolvers.interpolatesStringLiterals === true && /\$[{A-Za-z_]/.test(folded)) { + return specification() ?? { kind: 'unresolved', reason: 'unescaped-interpolation' }; + } + return finish(folded, 'constant', specification); + } + + return specification() ?? { kind: 'unresolved', reason: 'not-a-literal-or-constant' }; +} + +function finish( + value: string, + via: 'literal' | 'constant', + specification: () => SpringDestinationResolution | null, +): SpringDestinationResolution { + const decline = ( + reason: SpringDestinationRefusal, + extra: { configKey?: string; configDefault?: string } = {}, + ): SpringDestinationResolution => + specification() ?? { + kind: 'unresolved', + reason, + ...(extra.configKey === undefined ? {} : { configKey: extra.configKey }), + ...(extra.configDefault === undefined ? {} : { configDefault: extra.configDefault }), + }; + + // Before the placeholder rule: a SpEL expression may CONTAIN a `${…}`, and + // calling that a configuration key would name the wrong diagnosis. + if (containsSpelExpression(value)) return decline('spel-expression'); + + const placeholders = resolveSpringPlaceholders(value); + if (!placeholders.plain) { + const key = placeholders.key ?? ''; + if (key === '') return decline('empty-config-key'); + if (placeholders.defaultValue !== undefined) { + return decline('overridable-config-default', { + configKey: key, + configDefault: placeholders.defaultValue, + }); + } + return decline('unresolved-config-key', { configKey: key }); + } + + if (value.trim() === '') { + return decline(via === 'literal' ? 'empty-literal-address' : 'empty-constant-address'); + } + return { kind: 'resolved', address: value, via }; +} + +/** + * Expand one accepted argument into per-element candidates. + * + * An empty list is a refusal rather than zero silent candidates: `topics = {}` + * is a listener that names nothing, which is a finding, not an absence. + */ +function pushElements( + candidates: SpringDestinationCandidate[], + refusals: SpringDestinationRefusalRecord[], + base: Omit, +): void { + const elements = splitSpringDestinationList(base.rawText); + if (elements.length === 0) { + refusals.push({ + role: base.role, + source: base.source, + broker: base.broker, + reason: 'empty-destination-list', + rawText: base.rawText, + argIndex: base.argIndex, + ...(base.argName === undefined ? {} : { argName: base.argName }), + }); + return; + } + for (const [elementIndex, element] of elements.entries()) { + candidates.push({ ...base, rawText: element, elementIndex }); + } +} diff --git a/gitnexus/src/core/ingestion/frameworks/spring/dynamic-lookups.ts b/gitnexus/src/core/ingestion/frameworks/spring/dynamic-lookups.ts new file mode 100644 index 000000000..dfcfbb7c0 --- /dev/null +++ b/gitnexus/src/core/ingestion/frameworks/spring/dynamic-lookups.ts @@ -0,0 +1,177 @@ +import type { ParsedFile, Range, ScopeId, SymbolDefinition } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../../../graph/types.js'; +import type { DiInjectionMatch } from '../../di-extractors/index.js'; +import { SPRING_DI_INJECTION_SITES_PROPERTY } from '../../di-extractors/spring.js'; +import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; +import { + resolveCallerGraphId, + resolveDefGraphId, +} from '../../scope-resolution/graph-bridge/ids.js'; +import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js'; +import { isClassLike, lookupBindingsAt } from '../../scope-resolution/scope/walkers.js'; + +const COLLECTION_LOOKUP_METHODS = new Set(['getBeans', 'getBeansOfType']); +const SINGLE_LOOKUP_METHODS = new Set(['getBean']); + +/** + * Distinctive utility names plus conventional Spring context variable names. + * Generic locals remain recall-oriented because repositories often omit the + * third-party context type from the index; AST call/class-literal gates and + * import-aware target resolution prevent the raw-text false-positive class. + */ +const KNOWN_RECEIVERS = new Set([ + 'SpringContextUtil', + 'SpringContextHolder', + 'SpringBeanUtil', + 'ApplicationContextProvider', + 'BeanFactoryProvider', + 'ApplicationContext', + 'BeanFactory', + 'ListableBeanFactory', + 'applicationContext', + 'context', + 'ctx', + 'appContext', + 'beanFactory', +]); + +export interface SpringDynamicLookupFact { + readonly ownerScopeId: ScopeId; + readonly ownerRange: Range; + readonly receiverName: string; + readonly methodName: string; + readonly targetTypeName: string; +} + +export function springDynamicLookupCardinality( + receiverName: string, + methodName: string, +): DiInjectionMatch['cardinality'] | null { + const receiverSimpleName = receiverName.slice(receiverName.lastIndexOf('.') + 1); + if (!KNOWN_RECEIVERS.has(receiverSimpleName)) return null; + if (COLLECTION_LOOKUP_METHODS.has(methodName)) return 'collection'; + if (SINGLE_LOOKUP_METHODS.has(methodName)) return 'single'; + return null; +} + +function visibleTypeDefinitions( + fact: SpringDynamicLookupFact, + indexes: ScopeResolutionIndexes, +): readonly SymbolDefinition[] { + const simpleName = fact.targetTypeName.slice(fact.targetTypeName.lastIndexOf('.') + 1); + let scopeId: ScopeId | null = fact.ownerScopeId; + + while (scopeId !== null) { + const visible = lookupBindingsAt(scopeId, simpleName, indexes) + .map(({ def }) => def) + .filter((def) => isClassLike(def.type)) + .filter( + (def) => !fact.targetTypeName.includes('.') || def.qualifiedName === fact.targetTypeName, + ); + if (visible.length > 0) { + const unique = new Map(visible.map((def) => [def.nodeId, def])); + return [...unique.values()]; + } + scopeId = indexes.scopeTree.getScope(scopeId)?.parent ?? null; + } + + return []; +} + +function resolveTargetTypeName( + graph: KnowledgeGraph, + fact: SpringDynamicLookupFact, + callerLanguage: string | undefined, + nodeLookup: GraphNodeLookup, + indexes: ScopeResolutionIndexes, +): string | undefined { + const graphIds = new Set(); + for (const definition of visibleTypeDefinitions(fact, indexes)) { + const graphId = resolveDefGraphId(definition.filePath, definition, nodeLookup); + if (graphId === undefined) continue; + const node = graph.getNode(graphId); + if ( + (node?.label === 'Class' || + node?.label === 'Interface' || + node?.label === 'Record' || + node?.label === 'Enum') && + node.properties.language === callerLanguage + ) { + graphIds.add(graphId); + } + } + if (graphIds.size !== 1) return undefined; + + const targetId = graphIds.values().next().value; + if (targetId === undefined) return undefined; + const target = graph.getNode(targetId); + if (target === undefined) return undefined; + const qualifiedName = target.properties.qualifiedName; + return typeof qualifiedName === 'string' ? qualifiedName : target.properties.name; +} + +export interface SpringDynamicLookupMetadataAdapter { + getFacts(filePath: string): readonly SpringDynamicLookupFact[]; +} + +/** + * Attach AST-captured programmatic Spring lookups to the framework-neutral DI + * resolver. Java/Kotlin own syntax capture; this shared JVM/Spring seam owns + * import-aware type binding and metadata attachment. + */ +export function createSpringDynamicLookupMetadataAttacher( + adapter: SpringDynamicLookupMetadataAdapter, +) { + return ( + graph: KnowledgeGraph, + parsedFiles: readonly ParsedFile[], + nodeLookup: GraphNodeLookup, + indexes: ScopeResolutionIndexes, + ): void => { + for (const parsed of parsedFiles) { + for (const fact of adapter.getFacts(parsed.filePath)) { + const cardinality = springDynamicLookupCardinality(fact.receiverName, fact.methodName); + if (cardinality === null) continue; + + const callerId = resolveCallerGraphId(fact.ownerScopeId, indexes, nodeLookup, { + startLine: fact.ownerRange.startLine, + startCol: fact.ownerRange.startCol, + }); + if (callerId === undefined) continue; + const caller = graph.getNode(callerId); + if ( + caller === undefined || + (caller.label !== 'Function' && + caller.label !== 'Method' && + caller.label !== 'Constructor') + ) { + continue; + } + + const targetTypeName = resolveTargetTypeName( + graph, + fact, + caller.properties.language, + nodeLookup, + indexes, + ); + if (targetTypeName === undefined) continue; + + const match: DiInjectionMatch = { + targetTypeName, + cardinality, + edgeSource: 'site', + reason: `Spring dynamic lookup: ${fact.receiverName}.${fact.methodName}(${fact.targetTypeName})`, + }; + // Singular lookups intentionally use the shared DI selection policy: + // a unique/@Primary candidate wins; unresolved multiplicity is an + // explicit 0.5-confidence fan-out rather than a guessed runtime winner. + const existing = caller.properties[SPRING_DI_INJECTION_SITES_PROPERTY]; + caller.properties[SPRING_DI_INJECTION_SITES_PROPERTY] = [ + ...(Array.isArray(existing) ? existing : []), + match, + ]; + } + } + }; +} diff --git a/gitnexus/src/core/ingestion/frameworks/spring/message-producers.ts b/gitnexus/src/core/ingestion/frameworks/spring/message-producers.ts new file mode 100644 index 000000000..4a4a731dc --- /dev/null +++ b/gitnexus/src/core/ingestion/frameworks/spring/message-producers.ts @@ -0,0 +1,140 @@ +import type { Range, ScopeId } from 'gitnexus-shared'; +import type { SpringArgumentFact } from './argument-facts.js'; + +/** + * Outbound side of Spring messaging: the template calls that publish to a + * broker destination, mirroring the inbound `@KafkaListener` / `@RabbitListener` + * family already recognized in `non-http-handlers.ts`. + * + * Recognition is purely syntactic and happens while the language's own scope + * query already has the call node in hand. The receiver's declared type is NOT + * consulted: at capture time the field may be inherited, injected from another + * file, or typed through an import that is not finalized yet. Matching on the + * receiver's simple name instead keeps the capture cheap and resolver-free; a + * later phase that owns type information can refine or discard a fact. + */ +export type SpringMessageProducerTemplate = 'kafka' | 'rabbit' | 'jms' | 'stream-bridge'; + +interface ProducerSignature { + readonly template: SpringMessageProducerTemplate; + /** + * Simple type name of the template bean, matched case-insensitively as a + * SUBSTRING of the receiver's folded simple name. The classifier below states + * which decorations that accepts, and what it does when one receiver name + * contains the type names of two different templates. + */ + readonly typeName: string; + readonly methodName: string; +} + +const PRODUCER_SIGNATURES: readonly ProducerSignature[] = [ + { template: 'kafka', typeName: 'KafkaTemplate', methodName: 'send' }, + { template: 'rabbit', typeName: 'RabbitTemplate', methodName: 'convertAndSend' }, + { template: 'jms', typeName: 'JmsTemplate', methodName: 'convertAndSend' }, + { template: 'stream-bridge', typeName: 'StreamBridge', methodName: 'send' }, +]; + +const PRODUCER_METHOD_NAMES: ReadonlySet = new Set( + PRODUCER_SIGNATURES.map((signature) => signature.methodName), +); + +/** A receiver we can attribute; `templates["k"]` or `getTemplate()` cannot be. */ +const PLAIN_IDENTIFIER = /^[A-Za-z_$][A-Za-z0-9_$]*$/; + +/** + * Fold a receiver's simple name to the form the type-name match runs against. + * + * `_` and `$` are word separators in the spellings this has to accept, not part + * of the words: `KAFKA_TEMPLATE` and `kafka_template` are the same bean name as + * `kafkaTemplate`, written to the constant and snake conventions. Digits stay, + * because they are part of a name (`kafkaTemplate2`), never a separator. + */ +function foldReceiverName(receiverSimpleName: string): string { + return receiverSimpleName.replace(/[_$]/g, '').toLowerCase(); +} + +/** + * Cheap pre-filter usable before any receiver text is materialized. Both + * languages visit every member call, so the common case must cost one set + * lookup on the method name. + */ +export function isSpringMessageProducerMethod(methodName: string): boolean { + return PRODUCER_METHOD_NAMES.has(methodName); +} + +/** + * Classify a `receiver.method(...)` call as a messaging producer, or `null`. + * + * `receiverName` is the receiver expression as written; only its last + * dot-separated segment participates, so `this.kafkaTemplate` and + * `outer.inner.kafkaTemplate` match while `templates.get("k")` does not. + * + * The PLAIN_IDENTIFIER gate runs BEFORE the fold and is load-bearing, because + * the last-dot split is textual: in `config.get("a.kafkaTemplate")` it yields + * `kafkaTemplate")`, which folds to something a name match would accept. Only + * an identifier survives the gate, which is also what rejects `templates["k"]`, + * `getTemplate()`, and a receiver whose dot is separated by a comment. + * + * The folded segment then matches case-insensitively when it CONTAINS the + * template type name, so every convention a template bean is really declared + * with is recognized — decorated by prefix (`orderKafkaTemplate`), by suffix + * (`kafkaTemplateDlq`, `kafkaTemplateV2`, `rabbitTemplate1`), or written as a + * constant (`KAFKA_TEMPLATE`, `STREAM_BRIDGE`). A suffix-only rule accepted + * one of those and silently dropped the rest, which are exactly the publishes + * this capture exists to find. A receiver named only `template` still does not + * match: without type information that would attribute any `send` in the + * repository to Kafka. + * + * The bare type name (`KafkaTemplate.send(...)`) contains itself and so is + * accepted. That is left as it is: the match is by NAME, a name equal to the + * type is the strongest evidence the rule has, and a later phase that owns type + * information can discard a static-looking receiver. + * + * A substring rule also lets ONE receiver satisfy TWO signatures, which a + * suffix rule could not: `KafkaTemplate` and `StreamBridge` both publish + * through `send`, and `RabbitTemplate` and `JmsTemplate` both through + * `convertAndSend`, so `streamBridgeKafkaTemplate.send(...)` matches two + * templates at once. Such a receiver yields NO fact. Nothing here can break the + * tie honestly: the receiver's TYPE is deliberately not resolved, and the name + * is not ranked evidence — neither the longest match, nor the last one, nor the + * order of this list says whether that bean is a KafkaTemplate fronted by a + * stream binding or a StreamBridge named after the broker behind it. Returning + * the first match published an arbitrary choice as a definite broker + * attribution, the one outcome a consumer cannot tell from a fact. Silence + * costs a rare publish and stays recoverable by a phase that owns types. + */ +export function springMessageProducerTemplateOf( + receiverName: string, + methodName: string, +): SpringMessageProducerTemplate | null { + if (!isSpringMessageProducerMethod(methodName)) return null; + const receiverSimpleName = receiverName.slice(receiverName.lastIndexOf('.') + 1).trim(); + if (!PLAIN_IDENTIFIER.test(receiverSimpleName)) return null; + const folded = foldReceiverName(receiverSimpleName); + let matched: SpringMessageProducerTemplate | null = null; + for (const signature of PRODUCER_SIGNATURES) { + if (signature.methodName !== methodName) continue; + if (!folded.includes(signature.typeName.toLowerCase())) continue; + // A second match makes the receiver ambiguous; see above for why it is not + // resolved by preferring one of them. + if (matched !== null) return null; + matched = signature.template; + } + return matched; +} + +export interface SpringMessageProducerFact { + /** Callable that performs the publish; the enclosing method or function. */ + readonly ownerScopeId: ScopeId; + readonly ownerRange: Range; + readonly template: SpringMessageProducerTemplate; + /** Receiver expression as written, for example `this.orderKafkaTemplate`. */ + readonly receiverName: string; + readonly methodName: string; + /** + * Call arguments in source order, or absent when the call site has no + * argument list at all (a Kotlin trailing-lambda call). An empty array means + * an empty argument list was written — a different fact from no list. + */ + readonly args?: readonly SpringArgumentFact[]; +} diff --git a/gitnexus/src/core/ingestion/frameworks/spring/non-http-handlers.ts b/gitnexus/src/core/ingestion/frameworks/spring/non-http-handlers.ts index 8f3144d01..ae43f3eac 100644 --- a/gitnexus/src/core/ingestion/frameworks/spring/non-http-handlers.ts +++ b/gitnexus/src/core/ingestion/frameworks/spring/non-http-handlers.ts @@ -3,6 +3,7 @@ import type { KnowledgeGraph } from '../../../graph/types.js'; import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; import { resolveCallerGraphId } from '../../scope-resolution/graph-bridge/ids.js'; import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js'; +import type { SpringArgumentFact } from './argument-facts.js'; import { createSpringAnnotationNameResolver } from './bean-candidates.js'; import { SPRING_BEAN_ANNOTATION } from './bean-factories.js'; @@ -14,6 +15,34 @@ export interface SpringNonHttpHandlerAnnotationFact { readonly name: string; /** Kotlin use-site targets describe generated/property elements, not the callable. */ readonly useSiteTarget?: string; + /** + * Annotation arguments in source order. An empty array always means an empty + * list was written (`@Scheduled()`), which is a different fact from absence — + * but absence has TWO causes, and only one of them is a statement about the + * source. Either the annotation was written without an argument list + * (`@Scheduled`), or arguments were never read for this callable. + * + * They are read only for a callable that carries a handler annotation. Java + * produces facts for no other callable, so there absence does mean "no list + * was written". Kotlin also produces a fact for a merely annotated function — + * it captures those without a name prefilter so an import alias cannot hide a + * handler — and on those facts arguments are absent however the annotation + * was written. + * + * The values keep their source spelling, with one deliberate exception: + * `normalizeSpringFactText` trims them and collapses whitespace around the + * dots of a multi-line expression, so `Destinations.ORDERS` and the same + * reference wrapped across lines produce equal facts. Without that, source + * formatting — including the enclosing block's indentation, which is not a + * property of the expression at all — would leak into the data and make two + * spellings of one destination compare unequal downstream. + * + * Nothing else is touched. `@KafkaListener(topics = ...)` and + * `@RabbitListener(queues = ...)` name the destination differently, and a + * destination may be a literal, a constant reference, or a `${...}` + * placeholder; resolving any of those belongs to a later phase. + */ + readonly args?: readonly SpringArgumentFact[]; } export interface SpringNonHttpHandlerFact< diff --git a/gitnexus/src/core/ingestion/language-provider.ts b/gitnexus/src/core/ingestion/language-provider.ts index 2775cf07b..d5d8048ef 100644 --- a/gitnexus/src/core/ingestion/language-provider.ts +++ b/gitnexus/src/core/ingestion/language-provider.ts @@ -36,7 +36,7 @@ import type { VariableExtractor } from './variable-types.js'; import type { ImportResolverFn } from './import-resolvers/types.js'; import type { SyntaxNode } from './utils/ast-helpers.js'; import type { CfgVisitor } from './cfg/types.js'; -import type { NodeLabel } from 'gitnexus-shared'; +import type { GraphNode, NodeLabel } from 'gitnexus-shared'; import type { ExtractedRoute } from './route-extractors/laravel.js'; import type { SharedSpringType } from './route-extractors/spring-shared.js'; import type { @@ -46,6 +46,16 @@ import type { } from './route-extractors/constant-resolver.js'; import type Parser from 'tree-sitter'; import type { ExtractedDecoratorRoute } from './workers/parse-worker.js'; +import type { SpringNonHttpHandlerFact } from './frameworks/spring/non-http-handlers.js'; +import type { SpringMessageProducerFact } from './frameworks/spring/message-producers.js'; + +/** One file's captured Spring async messaging facts, in both directions. */ +export interface SpringMessagingFacts { + /** Callables carrying a listener annotation — the inbound side. */ + readonly handlers: readonly SpringNonHttpHandlerFact[]; + /** Messaging-template publishes — the outbound side. */ + readonly producers: readonly SpringMessageProducerFact[]; +} // ── Shared type aliases ──────────────────────────────────────────────────── /** Tree-sitter query captures: capture name → AST node (or undefined if not captured). */ @@ -54,6 +64,7 @@ export type CaptureMap = Record; export interface DefinitionPropertiesContext { readonly nodeLabel: NodeLabel; readonly nodeName: string; + readonly filePath: string; readonly definitionNode: SyntaxNode; readonly parsedImports: readonly ParsedImport[]; readonly isExported: boolean; @@ -63,6 +74,26 @@ export type DefinitionPropertiesExtractor = ( context: DefinitionPropertiesContext, ) => Readonly> | undefined; +export interface RuntimeCallableIdentity { + readonly name: string; + readonly descriptorParameterTypes: readonly string[] | undefined; +} + +/** + * Optional language-owned bridge from runtime/compiler symbol identities to + * source graph symbols. Framework importers use this instead of naming + * languages or reproducing compiler conventions in shared ingestion code. + */ +export interface RuntimeSymbolStrategy { + /** Runtime owner names that may contain this callable/property. */ + readonly callableOwnerAliases?: ( + node: GraphNode, + owner: GraphNode | undefined, + ) => readonly string[]; + /** Whether a runtime callable identity can conservatively identify a node. */ + readonly matchesCallable: (node: GraphNode, runtime: RuntimeCallableIdentity) => boolean; +} + /** Run optional provider enrichment without allowing one hook failure to drop * the rest of the worker's language batch. */ export function runDefinitionPropertiesExtractor( @@ -192,6 +223,13 @@ interface LanguageProviderConfig { */ readonly preprocessSource?: (sourceText: string, filePath: string) => string; + /** + * Runtime/compiler identity reconciliation for framework metadata. The + * central importer owns ambiguity handling; providers only supply aliases + * and language-specific callable compatibility. + */ + readonly runtimeSymbolStrategy?: RuntimeSymbolStrategy; + // ── Core (required) ─────────────────────────────────────────────── /** Type extraction: declarations, initializers, for-loop bindings */ readonly typeConfig: LanguageTypeConfig; @@ -385,6 +423,28 @@ interface LanguageProviderConfig { lineOffset: number, ) => ExtractedDecoratorRoute[]; + /** + * Name of the function a route decorator captured by the worker's generic + * `@decorator` query applies to, given the decorator's own AST node. + * + * The worker knows a decorator is a route decorator but not how this + * language's grammar attaches it to a definition, so it hands the node over + * unchanged and takes whatever the language returns. Only languages that + * declare route handlers through the generic decorator captures need this; + * languages with a dedicated {@link extractDecoratorRoutes} extractor + * (JS/TS via `nest.ts`, Java via `spring.ts`) already set + * `ExtractedDecoratorRoute.handlerName` there and should leave this undefined. + * + * Implementations must read their own decorated-definition shape directly and + * return undefined for anything else — never climb ancestors to find a name, + * since a decorator that is not attached to a function has no handler and a + * borrowed enclosing name resolves `handlerSymbolId` to the wrong symbol. The + * routes phase treats undefined as "fall back to the file-level edge". + * + * Default: undefined (no handler name from generic decorator captures). + */ + readonly decoratorRouteHandlerName?: (decoratorNode: SyntaxNode) => string | undefined; + /** * Collect a project-wide, language-agnostic view of route-defining * class/interface declarations (`SharedSpringType`) from a parsed file. @@ -402,6 +462,54 @@ interface LanguageProviderConfig { filePath: string, ) => SharedSpringType[]; + /** + * Optional post-capture emission of synthetic structure members (nodes, + * symbols, ownership edges) that have no AST method node — e.g. Lombok + * accessors. Called once per file after the capture loop, at the same + * post-capture site as {@link extractDecoratorRoutes}. + * + * `classOwnersByNodeId` maps in-memory tree-sitter node ids of type + * declarations materialized in THIS file's capture loop to their graph + * node ids. Keys are never persisted; they exist only for the duration + * of the worker pass. + * + * Default: undefined (no synthetic structure members). + */ + readonly synthesizeStructureMembers?: ( + tree: Parser.Tree, + filePath: string, + classOwnersByNodeId: ReadonlyMap, + ) => { + nodes: ReadonlyArray<{ + id: string; + label: string; + properties: Record; + }>; + symbols: ReadonlyArray<{ + filePath: string; + name: string; + nodeId: string; + type: string; + ownerId?: string; + parameterCount?: number; + requiredParameterCount?: number; + parameterTypes?: string[]; + returnType?: string; + visibility?: string; + isStatic?: boolean; + isAbstract?: boolean; + isFinal?: boolean; + }>; + relationships: ReadonlyArray<{ + id: string; + sourceId: string; + targetId: string; + type: string; + confidence: number; + reason: string; + }>; + }; + /** * Harvest this file's module-level string constants (#2391 core, #2980 Java * parity) into the language-agnostic {@link ModuleConstants} shape, so the @@ -436,6 +544,50 @@ interface LanguageProviderConfig { */ readonly moduleConstantHeuristic?: (content: string) => boolean; + /** + * Prepare this language's harvested constants once the complete repo map is + * available and before route operands are folded. The parse phase passes only + * entries owned by this provider, so implementations can build one reusable + * language-specific index and may materialize deferred bindings in place. + * + * Default: undefined (the harvested constants are already fold-ready). + */ + readonly prepareRouteConstants?: (repo: RepoConstants) => void; + + /** + * Spring async messaging facts captured for one file — the listener + * annotations that subscribe to a broker destination and the template calls + * that publish to one. + * + * Both families are collected during capture and restored on the main thread + * by {@link LanguageProviderConfig.applyCaptureSideChannel}, so they are only + * readable AFTER scope resolution has run. The `springDestinations` phase is + * the caller; routing through a provider hook is what keeps that phase from + * naming a language to reach a per-language fact store. + * + * Default: undefined — this language captures no Spring messaging facts, and + * the phase contributes nothing for its files. + */ + readonly getSpringMessagingFacts?: (filePath: string) => SpringMessagingFacts; + + /** + * Whether this language INTERPOLATES its string literals — Kotlin's + * `"orders-$env"` and `"orders-${env}"` are string templates evaluated at + * runtime, while Java's are ordinary characters. + * + * A capability rather than a language name, because shared ingestion code may + * not branch on a language (see AGENTS.md) and because the capability is what + * the consumer actually needs. Spring destination resolution is the caller: + * in an interpolating language an unescaped `$` in a destination literal is a + * runtime value and must be refused, and `"${app.topic}"` is a TEMPLATE, not + * a Spring property placeholder — the placeholder has to be written + * `"\${app.topic}"` there. Reading either as an address gives two unrelated + * services one shared destination node. + * + * Default: false — literals are literal, `$` is a character. + */ + readonly interpolatesStringLiterals?: boolean; + /** * Fold one file's non-literal route-path operand list * (`routePathExpr`/`routePathOperands` of an `ExtractedDecoratorRoute`) @@ -854,6 +1006,34 @@ export interface LanguageProvider extends Omit boolean; } +/** + * Run each provider's repo-constant preparation hook once over only the files + * that provider owns. Values are shared with `repo`, so in-place preparation + * is visible to the subsequent fold without copying the complete map. + */ +export function prepareRouteConstantsByProvider( + repo: RepoConstants, + providerForFile: (filePath: string) => Pick | null, +): void { + const slices = new Map< + Pick, + Map + >(); + for (const [filePath, constants] of repo) { + const provider = providerForFile(filePath); + if (!provider?.prepareRouteConstants) continue; + let slice = slices.get(provider); + if (!slice) { + slice = new Map(); + slices.set(provider, slice); + } + slice.set(filePath, constants); + } + for (const [provider, slice] of slices) { + provider.prepareRouteConstants?.(slice); + } +} + const DEFAULTS: Pick = { mroStrategy: 'first-wins', }; diff --git a/gitnexus/src/core/ingestion/languages/csharp/razor-view-components.ts b/gitnexus/src/core/ingestion/languages/csharp/razor-view-components.ts new file mode 100644 index 000000000..0e7ea55a4 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/csharp/razor-view-components.ts @@ -0,0 +1,955 @@ +/** + * ASP.NET Core ViewComponent convention support. + * + * Same bound as Spring Boot DI in Java/Kotlin: do not resolve into the SDK + * (`Microsoft.AspNetCore.Mvc.ViewComponent`, `IViewComponentHelper`, + * `Component.InvokeAsync` itself). Those types live outside the workspace. + * The only hop worth taking is the framework convention that lands on an + * **in-repo** class — `InvokeAsync("Foo")` → workspace `FooViewComponent`, + * just as a Spring `@Autowired IFoo` fans out to an in-repo `@Service`, + * not to `ApplicationContext`. + * + * Razor templates are not parsed as C# (markup + code would poison + * tree-sitter-c-sharp). A small Razor state machine extracts C# islands and + * markup tag helpers; C# files use a string/comment-aware lexer so attributes + * and literals are not mistaken for helper calls. Literal names are enough + * because the target catalog is already built from parsed `.cs` classes. + */ + +import fs from 'node:fs/promises'; +import path from 'node:path'; +import { glob } from 'glob'; +import type { ParsedFile } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../../../graph/types.js'; +import { createIgnoreFilter } from '../../../../config/ignore-service.js'; +import { generateId } from '../../../../lib/utils.js'; +import { getMaxFileSizeBytes } from '../../utils/max-file-size.js'; +import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js'; +import { resolveDefGraphId } from '../../scope-resolution/graph-bridge/ids.js'; +import { definitionIdPosition } from '../../scope-resolution/utils/definition-id.js'; + +const VIEW_COMPONENT_SUFFIX = 'ViewComponent'; +const VIEW_COMPONENT_TAG_RE = /<\s*vc:([a-z][a-z0-9-]*)\b/gi; +const COMPONENT_NAME_RE = /^[A-Za-z_][A-Za-z0-9_.-]*$/; +const TYPE_MODIFIERS = new Set([ + 'public', + 'internal', + 'protected', + 'private', + 'abstract', + 'sealed', + 'partial', + 'static', + 'new', + 'file', + 'required', + 'unsafe', + 'readonly', +]); +const RAZOR_BLOCK_KEYWORDS = new Set([ + 'if', + 'for', + 'foreach', + 'while', + 'using', + 'switch', + 'try', + 'lock', + 'functions', + 'helper', + 'code', + 'section', + 'do', +]); + +export interface RazorViewComponentConfig { + /** Repo-relative `.cshtml` path → extracted invocation names. */ + readonly views: ReadonlyMap; +} + +export interface ViewComponentAliasBind { + readonly className: string; + /** 1-based line of the type declaration (including leading attributes). */ + readonly startLine: number; + /** 0-based column of the type declaration (including leading attributes). */ + readonly startCol: number; + readonly aliases: readonly string[]; +} + +class SourceCursor { + i = 0; + line = 1; + col = 0; + + constructor(readonly source: string) {} + + get length(): number { + return this.source.length; + } + + get done(): boolean { + return this.i >= this.source.length; + } + + peek(n = 0): string { + return this.source[this.i + n] ?? ''; + } + + startsWith(value: string): boolean { + return this.source.startsWith(value, this.i); + } + + snapshot(): { i: number; line: number; col: number } { + return { i: this.i, line: this.line, col: this.col }; + } + + restore(pos: { i: number; line: number; col: number }): void { + this.i = pos.i; + this.line = pos.line; + this.col = pos.col; + } + + advance(count = 1): void { + const end = Math.min(this.i + count, this.source.length); + while (this.i < end) { + const ch = this.source[this.i]!; + this.i += 1; + if (ch === '\n') { + this.line += 1; + this.col = 0; + } else { + this.col += 1; + } + } + } +} + +function isIdentStart(ch: string): boolean { + return (ch >= 'A' && ch <= 'Z') || (ch >= 'a' && ch <= 'z') || ch === '_' || ch === '@'; +} + +function isIdentPart(ch: string): boolean { + return isIdentStart(ch) || (ch >= '0' && ch <= '9'); +} + +function skipWhitespace(cur: SourceCursor): void { + while (!cur.done) { + const ch = cur.peek(); + if (ch !== ' ' && ch !== '\t' && ch !== '\n' && ch !== '\r' && ch !== '\f' && ch !== '\v') + break; + cur.advance(); + } +} + +/** Skip line comments and block comments. Returns true if a comment was consumed. */ +function skipCsharpComment(cur: SourceCursor): boolean { + if (cur.startsWith('//')) { + while (!cur.done && cur.peek() !== '\n') cur.advance(); + return true; + } + if (cur.startsWith('/*')) { + cur.advance(2); + while (!cur.done && !cur.startsWith('*/')) cur.advance(); + if (cur.startsWith('*/')) cur.advance(2); + return true; + } + return false; +} + +function skipCsharpTrivia(cur: SourceCursor): void { + for (;;) { + skipWhitespace(cur); + if (!skipCsharpComment(cur)) return; + } +} + +function skipRegularString(cur: SourceCursor, interpolated: boolean): void { + cur.advance(); // opening " + while (!cur.done) { + const ch = cur.peek(); + if (ch === '\\') { + cur.advance(2); + continue; + } + if (interpolated && ch === '{') { + if (cur.peek(1) === '{') { + cur.advance(2); + continue; + } + skipInterpolation(cur); + continue; + } + cur.advance(); + if (ch === '"') return; + } +} + +function skipVerbatimString(cur: SourceCursor, interpolated: boolean): void { + cur.advance(2); // @" + while (!cur.done) { + const ch = cur.peek(); + if (ch === '"') { + if (cur.peek(1) === '"') { + cur.advance(2); + continue; + } + cur.advance(); + return; + } + if (interpolated && ch === '{') { + if (cur.peek(1) === '{') { + cur.advance(2); + continue; + } + skipInterpolation(cur); + continue; + } + cur.advance(); + } +} + +function skipRawString(cur: SourceCursor): void { + let quoteCount = 0; + while (cur.peek() === '"') { + quoteCount += 1; + cur.advance(); + } + while (!cur.done) { + if (cur.peek() !== '"') { + cur.advance(); + continue; + } + let seen = 0; + while (cur.peek() === '"') { + seen += 1; + cur.advance(); + } + if (seen >= quoteCount) return; + } +} + +function skipInterpolation(cur: SourceCursor): void { + cur.advance(); // { + let depth = 1; + while (!cur.done && depth > 0) { + skipCsharpTrivia(cur); + if (cur.done) return; + if (skipCsharpString(cur)) continue; + const ch = cur.peek(); + if (ch === '{') depth += 1; + else if (ch === '}') depth -= 1; + cur.advance(); + } +} + +function skipCsharpString(cur: SourceCursor): boolean { + const ch = cur.peek(); + if (ch === "'") { + cur.advance(); + if (cur.peek() === '\\') cur.advance(2); + else cur.advance(); + if (cur.peek() === "'") cur.advance(); + return true; + } + if (ch === '"') { + if (cur.peek(1) === '"' && cur.peek(2) === '"') skipRawString(cur); + else skipRegularString(cur, false); + return true; + } + if (ch === '$' && cur.peek(1) === '@' && cur.peek(2) === '"') { + cur.advance(); + skipVerbatimString(cur, true); + return true; + } + if (ch === '@' && cur.peek(1) === '$' && cur.peek(2) === '"') { + cur.advance(2); + skipVerbatimString(cur, true); + return true; + } + if (ch === '@' && cur.peek(1) === '"') { + skipVerbatimString(cur, false); + return true; + } + if (ch === '$' && cur.peek(1) === '"') { + if (cur.peek(2) === '"' && cur.peek(3) === '"') { + cur.advance(); + skipRawString(cur); + } else { + cur.advance(); + skipRegularString(cur, true); + } + return true; + } + return false; +} + +function readIdent(cur: SourceCursor): string | undefined { + if (!isIdentStart(cur.peek())) return undefined; + const start = cur.i; + if (cur.peek() === '@') cur.advance(); + if (!isIdentStart(cur.peek()) && !(cur.peek() >= 'A' && cur.peek() <= 'z')) { + cur.i = start; + return undefined; + } + while (isIdentPart(cur.peek()) && cur.peek() !== '@') cur.advance(); + const raw = cur.source.slice(start, cur.i); + return raw.startsWith('@') ? raw.slice(1) : raw; +} + +function tryReadIdent(cur: SourceCursor): string | undefined { + skipCsharpTrivia(cur); + return readIdent(cur); +} + +function decodeCsharpString(cur: SourceCursor): string | undefined { + skipCsharpTrivia(cur); + const start = cur.snapshot(); + const ch = cur.peek(); + if (ch === '$') return undefined; + if (ch === '@' && cur.peek(1) === '"') { + cur.advance(2); + let value = ''; + while (!cur.done) { + if (cur.peek() === '"') { + if (cur.peek(1) === '"') { + value += '"'; + cur.advance(2); + continue; + } + cur.advance(); + return value; + } + value += cur.peek(); + cur.advance(); + } + cur.restore(start); + return undefined; + } + if (ch === '"' && cur.peek(1) === '"' && cur.peek(2) === '"') { + let quoteCount = 0; + while (cur.peek() === '"') { + quoteCount += 1; + cur.advance(); + } + const bodyStart = cur.i; + while (!cur.done) { + if (cur.peek() !== '"') { + cur.advance(); + continue; + } + const closeStart = cur.i; + let seen = 0; + while (cur.peek() === '"') { + seen += 1; + cur.advance(); + } + if (seen >= quoteCount) { + return cur.source.slice(bodyStart, closeStart); + } + } + cur.restore(start); + return undefined; + } + if (ch === '"') { + cur.advance(); + let value = ''; + while (!cur.done) { + const next = cur.peek(); + if (next === '\\') { + cur.advance(); + const esc = cur.peek(); + cur.advance(); + const map: Record = { + n: '\n', + r: '\r', + t: '\t', + '"': '"', + '\\': '\\', + '0': '\0', + }; + value += map[esc] ?? esc; + continue; + } + if (next === '"') { + cur.advance(); + return value; + } + value += next; + cur.advance(); + } + cur.restore(start); + return undefined; + } + return undefined; +} + +function skipBalanced(cur: SourceCursor, open: string, close: string): boolean { + skipCsharpTrivia(cur); + if (cur.peek() !== open) return false; + let depth = 0; + while (!cur.done) { + skipCsharpTrivia(cur); + if (cur.done) return false; + if (skipCsharpString(cur)) continue; + const ch = cur.peek(); + if (ch === open) depth += 1; + else if (ch === close) { + depth -= 1; + cur.advance(); + if (depth === 0) return true; + continue; + } + cur.advance(); + } + return false; +} + +function componentNameFromLiteral(value: string | undefined): string | undefined { + if (value === undefined || !COMPONENT_NAME_RE.test(value)) return undefined; + return value; +} + +function isViewComponentAttributeName(name: string): boolean { + return name === 'ViewComponent' || name === 'ViewComponentAttribute'; +} + +function readQualifiedTail(cur: SourceCursor): string | undefined { + skipCsharpTrivia(cur); + let name = readIdent(cur); + if (name === undefined) return undefined; + for (;;) { + skipCsharpTrivia(cur); + if (cur.peek() === '.' || (cur.peek() === ':' && cur.peek(1) === ':')) { + cur.advance(cur.peek() === ':' ? 2 : 1); + skipCsharpTrivia(cur); + const next = readIdent(cur); + if (next === undefined) return name; + name = next; + continue; + } + return name; + } +} + +function readViewComponentNameArgument(cur: SourceCursor): string | undefined { + skipCsharpTrivia(cur); + if (cur.peek() !== '(') return undefined; + cur.advance(); + let alias: string | undefined; + while (!cur.done && cur.peek() !== ')') { + skipCsharpTrivia(cur); + if (cur.peek() === ')') break; + const beforeArg = cur.snapshot(); + const ident = readIdent(cur); + skipCsharpTrivia(cur); + if (ident === 'Name' && cur.peek() === '=') { + cur.advance(); + alias = componentNameFromLiteral(decodeCsharpString(cur)); + } else { + cur.restore(beforeArg); + skipCsharpTrivia(cur); + if (cur.peek() === '"' || cur.peek() === '@') { + // Positional string arguments are not ViewComponentAttribute.Name. + skipCsharpString(cur); + } else if (cur.peek() === '(' || cur.peek() === '[' || cur.peek() === '{') { + const open = cur.peek(); + const close = open === '(' ? ')' : open === '[' ? ']' : '}'; + skipBalanced(cur, open, close); + } else { + while (!cur.done && cur.peek() !== ',' && cur.peek() !== ')') { + if (skipCsharpString(cur)) continue; + if (skipCsharpComment(cur)) continue; + cur.advance(); + } + } + } + skipCsharpTrivia(cur); + if (cur.peek() === ',') cur.advance(); + } + if (cur.peek() === ')') cur.advance(); + return alias; +} + +function collectInvokeAfterIdent( + ident: string, + cur: SourceCursor, + previous: string | undefined, + memberReceiver: string | undefined, + names: Set, +): void { + skipCsharpTrivia(cur); + const hasMvcReceiver = previous !== '.' || memberReceiver === 'this' || memberReceiver === 'base'; + if (ident === 'ViewComponent' && cur.peek() === '(') { + if (previous === '[' || previous === ',' || !hasMvcReceiver) return; + cur.advance(); + const name = componentNameFromLiteral(decodeCsharpString(cur)); + if (name !== undefined) names.add(name); + return; + } + if (ident !== 'Component' || cur.peek() !== '.' || !hasMvcReceiver) return; + const afterDot = cur.snapshot(); + cur.advance(); + skipCsharpTrivia(cur); + if (readIdent(cur) !== 'InvokeAsync') { + cur.restore(afterDot); + return; + } + skipCsharpTrivia(cur); + if (cur.peek() !== '(') return; + cur.advance(); + const name = componentNameFromLiteral(decodeCsharpString(cur)); + if (name !== undefined) names.add(name); +} + +/** In-repo C# `Component.InvokeAsync("X")` / `ViewComponent("X")` literals. */ +export function extractCsharpViewComponentInvocations(source: string): string[] { + if (!source.includes('ViewComponent') && !source.includes('InvokeAsync')) return []; + const names = new Set(); + const cur = new SourceCursor(source); + let previous: string | undefined; + let memberReceiver: string | undefined; + let squareDepth = 0; + while (!cur.done) { + skipCsharpTrivia(cur); + if (cur.done) break; + if (skipCsharpString(cur)) { + previous = 'string'; + continue; + } + const ident = readIdent(cur); + if (ident !== undefined) { + const inAttribute = squareDepth > 0; + collectInvokeAfterIdent(ident, cur, inAttribute ? '[' : previous, memberReceiver, names); + previous = ident; + memberReceiver = undefined; + continue; + } + const ch = cur.peek(); + if (ch === '[') squareDepth += 1; + else if (ch === ']' && squareDepth > 0) squareDepth -= 1; + memberReceiver = ch === '.' ? previous : undefined; + previous = ch; + cur.advance(); + } + return [...names]; +} + +function parseAttributeListBody(cur: SourceCursor): string[] { + const aliases: string[] = []; + skipCsharpTrivia(cur); + const specifier = cur.snapshot(); + const specifierName = readIdent(cur); + skipCsharpTrivia(cur); + if (specifierName !== undefined && cur.peek() === ':' && cur.peek(1) !== ':') { + cur.advance(); + } else { + cur.restore(specifier); + } + while (!cur.done && cur.peek() !== ']') { + skipCsharpTrivia(cur); + if (cur.peek() === ']') break; + const tail = readQualifiedTail(cur); + skipCsharpTrivia(cur); + if (tail !== undefined && isViewComponentAttributeName(tail) && cur.peek() === '(') { + const alias = readViewComponentNameArgument(cur); + if (alias !== undefined) aliases.push(alias); + } else if (cur.peek() === '(') { + skipBalanced(cur, '(', ')'); + } + skipCsharpTrivia(cur); + if (cur.peek() === ',') cur.advance(); + else break; + } + if (cur.peek() === ']') cur.advance(); + return aliases; +} + +/** + * Explicit `[ViewComponent(Name = "...")]` aliases keyed to the following + * class declaration. Positional constructor arguments are ignored: the MVC + * attribute only exposes `Name` as a property. + */ +export function extractViewComponentAliasBinds(source: string): ViewComponentAliasBind[] { + if (!source.includes('ViewComponent')) return []; + const binds: ViewComponentAliasBind[] = []; + const cur = new SourceCursor(source); + const pending: { startLine: number; startCol: number; aliases: string[] }[] = []; + + const flushPending = (className: string, startLine: number, startCol: number): void => { + const aliases = pending.flatMap((entry) => entry.aliases); + const start = pending[0]; + binds.push({ + className, + startLine: start?.startLine ?? startLine, + startCol: start?.startCol ?? startCol, + aliases: [...new Set(aliases)], + }); + pending.length = 0; + }; + + while (!cur.done) { + skipCsharpTrivia(cur); + if (cur.done) break; + if (skipCsharpString(cur)) continue; + const startLine = cur.line; + const startCol = cur.col; + if (cur.peek() === '[') { + cur.advance(); + const aliases = parseAttributeListBody(cur); + pending.push({ startLine, startCol, aliases }); + continue; + } + const ident = readIdent(cur); + if (ident === undefined) { + pending.length = 0; + cur.advance(); + continue; + } + if (TYPE_MODIFIERS.has(ident)) continue; + if (ident === 'class' || ident === 'record') { + let className = tryReadIdent(cur); + if (ident === 'record' && (className === 'class' || className === 'struct')) { + className = tryReadIdent(cur); + } + if (className !== undefined && pending.some((entry) => entry.aliases.length > 0)) { + flushPending(className, startLine, startCol); + } else { + pending.length = 0; + } + continue; + } + pending.length = 0; + } + return binds; +} + +/** Extract explicit `[ViewComponent(Name = "...")]` aliases by class name. */ +export function extractViewComponentAliases( + source: string, +): ReadonlyMap { + const aliases = new Map(); + for (const bind of extractViewComponentAliasBinds(source)) { + if (bind.aliases.length === 0) continue; + const existing = aliases.get(bind.className); + if (existing) { + for (const alias of bind.aliases) { + if (!existing.includes(alias)) existing.push(alias); + } + } else { + aliases.set(bind.className, [...bind.aliases]); + } + } + return aliases; +} + +function tagNameToComponentName(tagName: string): string { + return tagName + .split('-') + .filter(Boolean) + .map((part) => part[0]!.toUpperCase() + part.slice(1)) + .join(''); +} + +function collectVcTags(span: string, names: Set): void { + VIEW_COMPONENT_TAG_RE.lastIndex = 0; + for (const match of span.matchAll(VIEW_COMPONENT_TAG_RE)) { + names.add(tagNameToComponentName(match[1]!)); + } +} + +function skipRazorComment(cur: SourceCursor): boolean { + if (!cur.startsWith('@*')) return false; + cur.advance(2); + while (!cur.done && !cur.startsWith('*@')) cur.advance(); + if (cur.startsWith('*@')) cur.advance(2); + return true; +} + +function countAtRun(cur: SourceCursor): number { + let count = 0; + while (cur.peek() === '@') { + count += 1; + cur.advance(); + } + return count; +} + +function scanCsharpSpan(span: string, names: Set): void { + for (const name of extractCsharpViewComponentInvocations(span)) names.add(name); +} + +function skipOptionalParens(cur: SourceCursor): void { + skipWhitespace(cur); + if (cur.peek() === '(') skipBalanced(cur, '(', ')'); +} + +function consumeRazorCodeBlock(cur: SourceCursor, names: Set): void { + skipCsharpTrivia(cur); + skipOptionalParens(cur); + skipCsharpTrivia(cur); + if (cur.peek() !== '{') { + const start = cur.i; + while (!cur.done && cur.peek() !== '\n' && cur.peek() !== '{') { + if (skipCsharpString(cur) || skipCsharpComment(cur)) continue; + cur.advance(); + } + scanCsharpSpan(cur.source.slice(start, cur.i), names); + if (cur.peek() === '{') consumeRazorCodeBlock(cur, names); + return; + } + const bodyStart = cur.i + 1; + if (!skipBalanced(cur, '{', '}')) return; + scanCsharpSpan(cur.source.slice(bodyStart, cur.i - 1), names); +} + +function consumeImplicitExpression(cur: SourceCursor, names: Set): void { + const start = cur.i; + skipCsharpTrivia(cur); + if (cur.peek() === '(') { + const innerStart = cur.i + 1; + if (skipBalanced(cur, '(', ')')) { + scanCsharpSpan(cur.source.slice(innerStart, cur.i - 1), names); + } + return; + } + // Implicit expressions: `@await Component.InvokeAsync("X")` / `@Component.InvokeAsync(...)`. + while (!cur.done) { + skipCsharpTrivia(cur); + if (cur.done) break; + if (skipCsharpString(cur)) continue; + if (cur.peek() === '(') { + skipBalanced(cur, '(', ')'); + continue; + } + if (cur.peek() === '{') { + skipBalanced(cur, '{', '}'); + continue; + } + const ch = cur.peek(); + if (ch === '<' || ch === '\n') break; + if (ch === '@') break; + if (!isIdentPart(ch) && ch !== '.' && ch !== '?') { + if (ch === ';') cur.advance(); + break; + } + cur.advance(); + } + scanCsharpSpan(cur.source.slice(start, cur.i), names); +} + +function consumeRazorTransition(cur: SourceCursor, names: Set): void { + skipWhitespace(cur); + if (cur.peek() === '{') { + consumeRazorCodeBlock(cur, names); + return; + } + if (cur.peek() === '(') { + consumeImplicitExpression(cur, names); + return; + } + const identStart = cur.snapshot(); + const ident = readIdent(cur); + if (ident === undefined) { + consumeImplicitExpression(cur, names); + return; + } + if (ident === 'await' || ident === 'Component') { + cur.restore(identStart); + consumeImplicitExpression(cur, names); + return; + } + if (RAZOR_BLOCK_KEYWORDS.has(ident)) { + if (ident === 'section' || ident === 'helper') tryReadIdent(cur); + consumeRazorCodeBlock(cur, names); + return; + } + cur.restore(identStart); + consumeImplicitExpression(cur, names); +} + +/** Extract statically resolvable ViewComponent names from one Razor template. */ +export function extractRazorViewComponentInvocations(source: string): string[] { + // Most views do not invoke a ViewComponent. Avoid the character-by-character + // Razor scan unless one of the two supported invocation spellings is present. + // This is only a coarse gate; the state machine below still decides whether a + // token is executable markup/C# or a comment/string/escaped transition. + if (!source.includes('InvokeAsync') && !/<\s*vc:/i.test(source)) return []; + + const names = new Set(); + const cur = new SourceCursor(source); + let markupStart = 0; + const flushMarkup = (): void => { + if (cur.i > markupStart) collectVcTags(source.slice(markupStart, cur.i), names); + }; + + while (!cur.done) { + if (cur.peek() !== '@') { + cur.advance(); + continue; + } + flushMarkup(); + if (skipRazorComment(cur)) { + markupStart = cur.i; + continue; + } + const atCount = countAtRun(cur); + const leftover = atCount % 2; + if (leftover === 0) { + markupStart = cur.i; + continue; + } + consumeRazorTransition(cur, names); + markupStart = cur.i; + } + flushMarkup(); + return [...names]; +} + +/** + * Read Razor views once per C# resolution pass. The same ignore rules and file + * size ceiling as repository scanning are applied, and edge emission later + * additionally requires a live File node. This prevents ignored, oversized, + * or concurrently removed templates from entering the graph. + */ +export async function loadRazorViewComponentConfig( + repoRoot: string, +): Promise { + const ignore = await createIgnoreFilter(repoRoot); + const paths = await glob('**/*.cshtml', { + cwd: repoRoot, + nodir: true, + dot: false, + ignore, + }); + paths.sort(); + + const maxBytes = getMaxFileSizeBytes(); + const views = new Map(); + for (const rawPath of paths) { + const filePath = rawPath.replace(/\\/g, '/'); + // The size gate and the read go through one handle so both observe the same + // inode. Re-resolving the path for the read would let a template swapped in + // between them be read unchecked (CodeQL js/file-system-race). + let handle: fs.FileHandle | undefined; + try { + handle = await fs.open(path.join(repoRoot, filePath), 'r'); + const stat = await handle.stat(); + if (!stat.isFile() || stat.size > maxBytes) continue; + const source = await handle.readFile('utf8'); + views.set(filePath, extractRazorViewComponentInvocations(source)); + } catch { + // A view may disappear between glob/open/read during watch mode. + } finally { + await handle?.close().catch(() => {}); + } + } + return { views }; +} + +function addCandidate( + candidates: Map>, + invocationName: string, + targetId: string, +): void { + const key = invocationName.toLocaleLowerCase('en-US'); + const existing = candidates.get(key); + if (existing) { + existing.add(targetId); + } else { + candidates.set(key, new Set([targetId])); + } +} + +function bindAliasesForClass( + binds: readonly ViewComponentAliasBind[], + className: string, + nodeId: string, + filePath: string, +): readonly string[] | undefined { + const matches = binds.filter((bind) => bind.className === className); + if (matches.length === 0) return undefined; + if (matches.length === 1) return matches[0]!.aliases; + const pos = definitionIdPosition(nodeId, filePath); + if (pos === undefined) return undefined; + const atPosition = matches.filter( + (bind) => bind.startLine === pos.line && bind.startCol === pos.column, + ); + if (atPosition.length === 1) return atPosition[0]!.aliases; + return undefined; +} + +/** + * Emit workspace File → in-repo ViewComponent Class CALLS edges. + * + * Targets are only Class nodes produced from this repo's `.cs` files. There is + * no lookup of ASP.NET SDK types; `: ViewComponent` in source is a naming + * hint, not a resolved EXTENDS edge to `Microsoft.AspNetCore.Mvc.ViewComponent`. + * + * Ambiguous component names fail closed: two in-repo classes claiming the + * same name is not evidence for picking either one. + */ +export function emitRazorViewComponentEdges( + graph: KnowledgeGraph, + parsedFiles: readonly ParsedFile[], + nodeLookup: GraphNodeLookup, + config: RazorViewComponentConfig | undefined, + csharpSources: ReadonlyMap, +): void { + if (!config) return; + + const candidates = new Map>(); + for (const parsed of parsedFiles) { + if (!parsed.filePath.endsWith('.cs')) continue; + const source = csharpSources.get(parsed.filePath) ?? ''; + const binds = source.includes('ViewComponent') ? extractViewComponentAliasBinds(source) : []; + for (const def of parsed.localDefs) { + if (def.type !== 'Class') continue; + const className = def.qualifiedName?.split('.').pop() ?? def.nodeId.split(':').pop() ?? ''; + const conventionalName = className.endsWith(VIEW_COMPONENT_SUFFIX) + ? className.slice(0, -VIEW_COMPONENT_SUFFIX.length) + : undefined; + const explicitAliases = bindAliasesForClass(binds, className, def.nodeId, parsed.filePath); + if (!conventionalName && (explicitAliases === undefined || explicitAliases.length === 0)) { + continue; + } + + const targetId = resolveDefGraphId(parsed.filePath, def, nodeLookup); + if (!targetId || !graph.getNode(targetId)) continue; + // An explicit [ViewComponent(Name = "...")] replaces the suffix name, + // matching ASP.NET. Never register the SDK base type as a candidate. + if (explicitAliases !== undefined && explicitAliases.length > 0) { + for (const alias of explicitAliases) addCandidate(candidates, alias, targetId); + } else if (conventionalName) { + addCandidate(candidates, conventionalName, targetId); + } + } + } + + const emitFromFile = (filePath: string, invocationNames: readonly string[]): void => { + const sourceId = generateId('File', filePath); + if (!graph.getNode(sourceId)) return; + for (const invocationName of invocationNames) { + const matches = candidates.get(invocationName.toLocaleLowerCase('en-US')); + if (!matches || matches.size !== 1) continue; + const targetId = matches.values().next().value; + if (typeof targetId !== 'string' || !graph.getNode(targetId)) continue; + graph.addRelationship({ + id: generateId('CALLS', `${sourceId}:razor-view-component:${targetId}`), + sourceId, + targetId, + type: 'CALLS', + confidence: 0.9, + reason: 'aspnet-razor-view-component', + }); + } + }; + + for (const [viewPath, invocationNames] of config.views) { + emitFromFile(viewPath, invocationNames); + } + for (const [filePath, source] of csharpSources) { + if (!filePath.endsWith('.cs')) continue; + if (!source.includes('ViewComponent') && !source.includes('InvokeAsync')) continue; + emitFromFile(filePath, extractCsharpViewComponentInvocations(source)); + } +} diff --git a/gitnexus/src/core/ingestion/languages/csharp/resolution-config.ts b/gitnexus/src/core/ingestion/languages/csharp/resolution-config.ts index 9ea232c05..714be8c1f 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/resolution-config.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/resolution-config.ts @@ -12,19 +12,29 @@ import { type CSharpProjectConfig, type CSharpNamespaceEvidence, } from '../../language-config.js'; +import { + loadRazorViewComponentConfig, + type RazorViewComponentConfig, +} from './razor-view-components.js'; export interface CsharpResolutionConfig { readonly csharpConfigs: readonly CSharpProjectConfig[]; /** In-repo declared-namespace evidence gating suffix-fallback resolution (#1881). */ readonly namespaces?: CSharpNamespaceEvidence; + /** Razor views scanned for ASP.NET ViewComponent invocation conventions. */ + readonly razorViewComponents?: RazorViewComponentConfig; } export async function loadCsharpResolutionConfig( repoRoot: string, ): Promise { - const scan = await scanCSharpProject(repoRoot); + const [scan, razorViewComponents] = await Promise.all([ + scanCSharpProject(repoRoot), + loadRazorViewComponentConfig(repoRoot), + ]); return { csharpConfigs: scan.configs, namespaces: csharpScanToEvidence(scan), + razorViewComponents, }; } diff --git a/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts index 4b50efc67..6d200de9b 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts @@ -22,6 +22,7 @@ import { import { populateCsharpNamespaceSiblings } from './namespace-siblings.js'; import { loadCsharpResolutionConfig, type CsharpResolutionConfig } from './resolution-config.js'; import { unwrapCsharpElementType } from './accessor-unwrap.js'; +import { emitRazorViewComponentEdges } from './razor-view-components.js'; const csharpScopeResolver: ScopeResolver = { // Construction is keyword-prefixed: `new Service(db).doWork()` (#2708). @@ -106,6 +107,20 @@ const csharpScopeResolver: ScopeResolver = { // `IValidator` and `IValidator` are one instantiation, so the // dispatch fan-out must not read them as two (#2912). See the alias table. normalizeTypeArgument: normalizeCsharpTypeArgument, + + // Razor views stay out of the C# parser. Bind literal ViewComponent names + // only onto in-repo classes (Spring-style: skip the SDK type, hop to the + // workspace implementor). + emitPostResolutionEdges: (graph, parsedFiles, nodeLookup, _indexes, ctx) => { + const config = ctx.resolutionConfig as CsharpResolutionConfig | undefined; + emitRazorViewComponentEdges( + graph, + parsedFiles, + nodeLookup, + config?.razorViewComponents, + ctx.fileContents, + ); + }, }; /** diff --git a/gitnexus/src/core/ingestion/languages/java.ts b/gitnexus/src/core/ingestion/languages/java.ts index 0fddb6657..45386dd7b 100644 --- a/gitnexus/src/core/ingestion/languages/java.ts +++ b/gitnexus/src/core/ingestion/languages/java.ts @@ -19,6 +19,7 @@ import { extractJavaModuleConstants, foldJavaOperands, isJavaConstantFile, + prepareJavaRouteConstants, } from '../route-extractors/java-const-resolver.js'; import { javaExportChecker } from '../export-detection.js'; import { createImportResolver } from '../import-resolvers/resolver-factory.js'; @@ -32,12 +33,17 @@ import { createVariableExtractor } from '../variable-extractors/generic.js'; import { javaVariableConfig } from '../variable-extractors/configs/jvm.js'; import { createJavaCfgVisitor } from '../cfg/visitors/java.js'; import { assertCloneable } from '../workers/clone-safety.js'; -import { collectJavaCaptureSideChannel } from './java/capture-side-channel.js'; +import { + collectJavaCaptureSideChannel, + getJavaSpringMessageProducerFacts, + getJavaSpringNonHttpHandlerFacts, +} from './java/capture-side-channel.js'; import type { SymbolDefinition } from 'gitnexus-shared'; import { javaRecordMethodExtractor, shouldSkipJavaRecordComponentDefinition, } from './java/record-components.js'; +import { synthesizeLombokAccessors } from './java/lombok-synthesizer.js'; import { emitJavaScopeCaptures, interpretJavaImport, @@ -49,6 +55,7 @@ import { javaArityCompatibility, resolveJavaImportTarget, } from './java/index.js'; +import { javaRuntimeSymbolStrategy } from './java/spring-actuator.js'; /** * Java names the platform owns, matched against a BARE IDENTIFIER — a dropped @@ -197,6 +204,7 @@ export const javaProvider = defineLanguage({ shouldSkipDefinitionCapture: shouldSkipJavaRecordComponentDefinition, variableExtractor: createVariableExtractor(javaVariableConfig), classExtractor: createClassExtractor(javaClassConfig), + runtimeSymbolStrategy: javaRuntimeSymbolStrategy, // ── Javadoc → description (issue #2270) ── descriptionExtractor: createLeadingDocDescriptionExtractor(), @@ -222,6 +230,8 @@ export const javaProvider = defineLanguage({ extractDecoratorRoutes: extractSpringRoutes, extractRouteInheritanceTypes: extractSpringTypes, + synthesizeStructureMembers: synthesizeLombokAccessors, + // ── #2980: constant harvest + qualified-ref fold for non-literal mapping // paths (`@PostMapping(ApiPaths.SAVE_V1)`) — kept behind provider hooks so // the shared ingestion layers stay language-agnostic. The heuristic is @@ -236,11 +246,17 @@ export const javaProvider = defineLanguage({ // the group still published the contract). moduleConstantHeuristic: (content) => isJavaConstantFile(content) || - // `import com.winning.opt.common.ApiPaths;` — ANY class import can bind a - // constant ref (`ApiPaths.X` at an annotation site), so gate on the - // general import shape, not on the imported name. Ingestion-only: this - // side needs the importing controller's own import table, which the group - // side instead derives lazily from the tree it already holds. - /\bimport\s+(?:static\s+)?[\w.]+\s*;/.test(content), + // Class imports and static (including on-demand) imports can bind a + // constant ref. Ordinary `import a.b.*;` is not a Java type import and is + // not expanded by extractJavaModuleConstants, so it must not harvest. + /\bimport\s+(?:static\s+[\w.]+(?:\.\*)?|[\w.]+)\s*;/.test(content), + prepareRouteConstants: prepareJavaRouteConstants, foldRoutePathOperands: foldJavaOperands, + // Async messaging facts for the `springDestinations` phase. Both stores are + // repopulated on the main thread by `applyJavaCaptureSideChannel`, so this + // answers for cache hits and misses alike. + getSpringMessagingFacts: (filePath) => ({ + handlers: getJavaSpringNonHttpHandlerFacts(filePath), + producers: getJavaSpringMessageProducerFacts(filePath), + }), }); diff --git a/gitnexus/src/core/ingestion/languages/java/analysis-features.ts b/gitnexus/src/core/ingestion/languages/java/analysis-features.ts index 73969670d..64ff16294 100644 --- a/gitnexus/src/core/ingestion/languages/java/analysis-features.ts +++ b/gitnexus/src/core/ingestion/languages/java/analysis-features.ts @@ -5,17 +5,24 @@ function isSpringApplicationConfig(filePath: string): boolean { return /^application(?:-[^.]+)?\.(?:properties|ya?ml)$/i.test(base); } -/** Durable completeness contract for Java Spring configuration bindings. */ +/** Durable completeness contract for Java and Kotlin Spring configuration bindings. */ export const SPRING_CONFIG_BINDINGS_FEATURE: AnalysisFeatureDescriptor = { id: 'spring.config-bindings', - version: 1, - // Java sources need consumer extraction even without config files (missing - // placeholders still get unresolved markers). Config-only repositories also - // need a one-time rebuild to backfill language-agnostic Property nodes. + version: 2, + // Java and Kotlin sources need consumer extraction even without config files + // (missing placeholders still get unresolved markers). Config-only + // repositories also need a one-time rebuild to backfill language-agnostic + // Property nodes. Gradle Kotlin DSL is not a consumer source. appliesTo: (filePaths) => - filePaths.some( - (filePath) => filePath.toLowerCase().endsWith('.java') || isSpringApplicationConfig(filePath), - ), + filePaths.some((filePath) => { + const normalized = filePath.replaceAll('\\', '/').toLowerCase(); + if (normalized.endsWith('.gradle.kts')) return false; + return ( + normalized.endsWith('.java') || + normalized.endsWith('.kt') || + isSpringApplicationConfig(filePath) + ); + }), }; /** Durable completeness contract for implicit Java record-component accessors. */ diff --git a/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts index 91a910fa5..a8c5e268e 100644 --- a/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts +++ b/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts @@ -13,6 +13,8 @@ import type { JavaSpringConfigConsumerFact } from './spring-config-bindings.js'; import type { JavaSpringAopFact } from './spring-aop.js'; import type { JavaSpringConditionalFact } from './spring-conditionals.js'; import type { JavaSpringDiClassFact } from './spring-di.js'; +import type { SpringDynamicLookupFact } from '../../frameworks/spring/dynamic-lookups.js'; +import type { SpringMessageProducerFact } from '../../frameworks/spring/message-producers.js'; import type { JavaSpringNonHttpHandlerFact } from './spring-non-http-handlers.js'; export type JavaClassAnnotationFact = ClassAnnotationFact; @@ -25,7 +27,9 @@ export interface JavaCaptureSideChannel { readonly springConfigConsumers?: readonly JavaSpringConfigConsumerFact[]; readonly springConditionalFacts?: readonly JavaSpringConditionalFact[]; readonly springDiFacts?: readonly JavaSpringDiClassFact[]; + readonly springDynamicLookupFacts?: readonly SpringDynamicLookupFact[]; readonly springNonHttpHandlerFacts?: readonly JavaSpringNonHttpHandlerFact[]; + readonly springMessageProducerFacts?: readonly SpringMessageProducerFact[]; } const classAnnotations = createClassAnnotationFactStore(); @@ -33,7 +37,9 @@ const springAopFacts = new Map(); const springConfigConsumers = new Map(); const springConditionalFacts = new Map(); const springDiFacts = new Map(); +const springDynamicLookupFacts = new Map(); const springNonHttpHandlerFacts = new Map(); +const springMessageProducerFacts = new Map(); /** Clear facts retained by a prior workspace pass in a long-lived process. */ export function clearJavaClassAnnotationFacts(): void { @@ -42,7 +48,9 @@ export function clearJavaClassAnnotationFacts(): void { springConfigConsumers.clear(); springConditionalFacts.clear(); springDiFacts.clear(); + springDynamicLookupFacts.clear(); springNonHttpHandlerFacts.clear(); + springMessageProducerFacts.clear(); } export function setJavaSpringAopFacts(filePath: string, facts: readonly JavaSpringAopFact[]): void { @@ -102,6 +110,20 @@ export function getJavaSpringDiFacts(filePath: string): readonly JavaSpringDiCla return springDiFacts.get(filePath) ?? []; } +export function setJavaSpringDynamicLookupFacts( + filePath: string, + facts: readonly SpringDynamicLookupFact[], +): void { + if (facts.length === 0) springDynamicLookupFacts.delete(filePath); + else springDynamicLookupFacts.set(filePath, facts); +} + +export function getJavaSpringDynamicLookupFacts( + filePath: string, +): readonly SpringDynamicLookupFact[] { + return springDynamicLookupFacts.get(filePath) ?? []; +} + export function setJavaSpringNonHttpHandlerFacts( filePath: string, facts: readonly JavaSpringNonHttpHandlerFact[], @@ -116,6 +138,20 @@ export function getJavaSpringNonHttpHandlerFacts( return springNonHttpHandlerFacts.get(filePath) ?? []; } +export function setJavaSpringMessageProducerFacts( + filePath: string, + facts: readonly SpringMessageProducerFact[], +): void { + if (facts.length === 0) springMessageProducerFacts.delete(filePath); + else springMessageProducerFacts.set(filePath, facts); +} + +export function getJavaSpringMessageProducerFacts( + filePath: string, +): readonly SpringMessageProducerFact[] { + return springMessageProducerFacts.get(filePath) ?? []; +} + /** Snapshot worker-local Java annotation facts for ParsedFile serialization. */ export function collectJavaCaptureSideChannel( filePath: string, @@ -125,7 +161,9 @@ export function collectJavaCaptureSideChannel( const configConsumers = springConfigConsumers.get(filePath) ?? []; const conditionFacts = springConditionalFacts.get(filePath) ?? []; const diFacts = springDiFacts.get(filePath) ?? []; + const dynamicLookupFacts = springDynamicLookupFacts.get(filePath) ?? []; const nonHttpHandlerFacts = springNonHttpHandlerFacts.get(filePath) ?? []; + const messageProducerFacts = springMessageProducerFacts.get(filePath) ?? []; const packageFact = getJavaPackageFact(filePath); if ( facts.length === 0 && @@ -133,7 +171,9 @@ export function collectJavaCaptureSideChannel( configConsumers.length === 0 && conditionFacts.length === 0 && diFacts.length === 0 && + dynamicLookupFacts.length === 0 && nonHttpHandlerFacts.length === 0 && + messageProducerFacts.length === 0 && packageFact === undefined ) { return undefined; @@ -146,7 +186,11 @@ export function collectJavaCaptureSideChannel( ...(configConsumers.length > 0 ? { springConfigConsumers: configConsumers } : {}), ...(conditionFacts.length > 0 ? { springConditionalFacts: conditionFacts } : {}), ...(diFacts.length > 0 ? { springDiFacts: diFacts } : {}), + ...(dynamicLookupFacts.length > 0 ? { springDynamicLookupFacts: dynamicLookupFacts } : {}), ...(nonHttpHandlerFacts.length > 0 ? { springNonHttpHandlerFacts: nonHttpHandlerFacts } : {}), + ...(messageProducerFacts.length > 0 + ? { springMessageProducerFacts: messageProducerFacts } + : {}), }; } @@ -169,7 +213,9 @@ export function applyJavaCaptureSideChannel(parsed: ParsedFile): void { setJavaSpringConfigConsumerFacts(parsed.filePath, []); setJavaSpringConditionalFacts(parsed.filePath, []); setJavaSpringDiFacts(parsed.filePath, []); + setJavaSpringDynamicLookupFacts(parsed.filePath, []); setJavaSpringNonHttpHandlerFacts(parsed.filePath, []); + setJavaSpringMessageProducerFacts(parsed.filePath, []); setJavaPackageFact(parsed.filePath, UNKNOWN_JVM_PACKAGE_FACT); return; } @@ -190,10 +236,18 @@ export function applyJavaCaptureSideChannel(parsed: ParsedFile): void { parsed.filePath, Array.isArray(data.springDiFacts) ? data.springDiFacts : [], ); + setJavaSpringDynamicLookupFacts( + parsed.filePath, + Array.isArray(data.springDynamicLookupFacts) ? data.springDynamicLookupFacts : [], + ); setJavaSpringNonHttpHandlerFacts( parsed.filePath, Array.isArray(data.springNonHttpHandlerFacts) ? data.springNonHttpHandlerFacts : [], ); + setJavaSpringMessageProducerFacts( + parsed.filePath, + Array.isArray(data.springMessageProducerFacts) ? data.springMessageProducerFacts : [], + ); setJavaPackageFact( parsed.filePath, isJvmPackageFact(data.packageFact) ? data.packageFact : UNKNOWN_JVM_PACKAGE_FACT, diff --git a/gitnexus/src/core/ingestion/languages/java/captures.ts b/gitnexus/src/core/ingestion/languages/java/captures.ts index ce5ed4b93..5d31aed4a 100644 --- a/gitnexus/src/core/ingestion/languages/java/captures.ts +++ b/gitnexus/src/core/ingestion/languages/java/captures.ts @@ -39,12 +39,18 @@ import { setJavaSpringConfigConsumerFacts, setJavaSpringConditionalFacts, setJavaSpringDiFacts, + setJavaSpringDynamicLookupFacts, + setJavaSpringMessageProducerFacts, setJavaSpringNonHttpHandlerFacts, } from './capture-side-channel.js'; import { captureJavaPackageFact } from './package-facts.js'; import { synthesizeCallableFlowCaptures } from '../../utils/callable-flow-captures.js'; import { captureJavaSpringConfigConsumerFacts } from './spring-config-bindings.js'; import { captureJavaSpringDiClassFact, type JavaSpringDiClassFact } from './spring-di.js'; +import type { SpringDynamicLookupFact } from '../../frameworks/spring/dynamic-lookups.js'; +import { captureJavaSpringDynamicLookupFact } from './spring-dynamic-lookup.js'; +import type { SpringMessageProducerFact } from '../../frameworks/spring/message-producers.js'; +import { captureJavaSpringMessageProducerFact } from './spring-message-producers.js'; import { synthesizeReceiverChainCapture } from '../../utils/receiver-chain-captures.js'; import { captureJavaSpringAopFacts, type JavaSpringAopFact } from './spring-aop.js'; import { @@ -56,6 +62,7 @@ import { type JavaSpringNonHttpHandlerFact, } from './spring-non-http-handlers.js'; import { synthesizeJavaRecordComponentAccessorCaptures } from './record-components.js'; +import { synthesizeLombokAccessorCaptures } from './lombok-synthesizer.js'; /** Declaration anchors that carry function-like arity metadata. */ const FUNCTION_DECL_TAGS = ['@declaration.method', '@declaration.constructor'] as const; @@ -146,6 +153,9 @@ export function emitJavaScopeCaptures( const springDiFacts: JavaSpringDiClassFact[] = []; const springNonHttpHandlerFacts: JavaSpringNonHttpHandlerFact[] = []; const springDiClassNodeIds = new Set(); + const springDynamicLookupFacts: SpringDynamicLookupFact[] = []; + const springMessageProducerFacts: SpringMessageProducerFact[] = []; + const springMemberCallNodeIds = new Set(); for (const m of rawMatches) { const grouped: Record = {}; @@ -165,6 +175,17 @@ export function emitJavaScopeCaptures( } if (Object.keys(grouped).length === 0) continue; + // One visit per member call node: the same invocation can back several + // query matches, and both Spring call-shape captures must see it once. + const memberCallNode = nodeIfType(nodeMap['@reference.call.member'], 'method_invocation'); + if (memberCallNode !== null && !springMemberCallNodeIds.has(memberCallNode.id)) { + springMemberCallNodeIds.add(memberCallNode.id); + const lookupFact = captureJavaSpringDynamicLookupFact(memberCallNode, filePath); + if (lookupFact !== null) springDynamicLookupFacts.push(lookupFact); + const producerFact = captureJavaSpringMessageProducerFact(memberCallNode, filePath); + if (producerFact !== null) springMessageProducerFacts.push(producerFact); + } + const springAopTypeNode = [ nodeIfType(nodeMap['@scope.class'], 'class_declaration'), nodeIfType(nodeMap['@scope.class'], 'interface_declaration'), @@ -401,7 +422,9 @@ export function emitJavaScopeCaptures( setJavaSpringAopFacts(filePath, springAopFacts); setJavaSpringConditionalFacts(filePath, springConditionalFacts); setJavaSpringDiFacts(filePath, springDiFacts); + setJavaSpringDynamicLookupFacts(filePath, springDynamicLookupFacts); setJavaSpringNonHttpHandlerFacts(filePath, springNonHttpHandlerFacts); + setJavaSpringMessageProducerFacts(filePath, springMessageProducerFacts); return [ ...resolveVarTypeBindings(out), @@ -409,6 +432,7 @@ export function emitJavaScopeCaptures( ...synthesizeJavaExplicitConstructorReferences(tree.rootNode), ...synthesizeJavaAnonymousClassDeclarations(tree.rootNode), ...synthesizeJavaRecordComponentAccessorCaptures(tree.rootNode), + ...synthesizeLombokAccessorCaptures(tree.rootNode), ...synthesizeCallableFlowCaptures(tree.rootNode, JAVA_CALLABLE_CAPTURE_OPTIONS), ]; } diff --git a/gitnexus/src/core/ingestion/languages/java/lombok-synthesizer.ts b/gitnexus/src/core/ingestion/languages/java/lombok-synthesizer.ts new file mode 100644 index 000000000..3f2eeab9a --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/lombok-synthesizer.ts @@ -0,0 +1,539 @@ +/** + * Lombok accessor synthesizer for Java. + * + * Lombok generates getters/setters at compile time. They are absent from the + * AST, so calls like `obj.getOrderId()` on a `@Data` class would otherwise + * leave unresolved CALLS edges. This module walks the tree-sitter Java AST + * and synthesizes Method graph members for the accessors Lombok would emit + * under the supported subset. + * + * ## Supported subset (v1) + * - Proven `lombok.Data` / `lombok.Getter` / `lombok.Setter` (FQN or import). + * - Class- or field-level enable; `AccessLevel.NONE` disables. + * - Default JavaBeans naming; primitive `boolean isX` → `isX` / `setX`. + * - Access levels PUBLIC/PROTECTED/PRIVATE/PACKAGE. + * - `@Accessors(chain=true)` modeled as setter return = declaring type. + * - `@Accessors(fluent=true)` / `prefix=…`: omit affected accessors (names + * cannot be proven without full Lombok config). + * - External `lombok.config`: unsupported (may change semantics invisibly). + * + * ## Identity + * Owner lookup uses in-memory AST node ids only. Method ids are derived from + * the stable declaring-owner graph key (the Class node id's name segment), + * never from persisted tree-sitter node ids. + */ + +import type Parser from 'tree-sitter'; +import type { CaptureMatch } from 'gitnexus-shared'; +import { jvmGetterName, jvmSetterName } from '../jvm/beanspec.js'; +import { + createExistingMethodIndex, + createJvmAccessorSynthesis, + hasExistingMethod, + rememberExistingMethodRange, + type ExistingMethodIndex, + type PlannedJvmAccessor, + type PlannedJvmAccessorOwner, + type SyntheticAccessorResult, + type SyntheticVisibility, +} from '../jvm/accessor-synthesis.js'; + +const JAVA_TYPE_DECLS = new Set([ + 'class_declaration', + 'enum_declaration', + 'interface_declaration', + 'record_declaration', +]); + +// ── Public result types (ParsedSymbol / ParsedNode compatible) ──────────── + +export type LombokVisibility = SyntheticVisibility; +export type SyntheticSymbol = SyntheticAccessorResult['symbols'][number]; +export type SyntheticNode = SyntheticAccessorResult['nodes'][number]; +export type SyntheticRelationship = SyntheticAccessorResult['relationships'][number]; +export type LombokSynthesisResult = SyntheticAccessorResult; +export type PlannedLombokAccessor = PlannedJvmAccessor; + +export interface AccessorConfig { + enabled: boolean; + visibility: LombokVisibility; +} + +interface AccessorsOptions { + /** When true, JavaBeans get/set/is prefixes are not used — omit (unsupported). */ + fluent: boolean; + /** When true, field prefixes alter base names — omit (unsupported). */ + hasPrefix: boolean; + /** When true, setters return the declaring type instead of void. */ + chain: boolean; +} + +interface LombokField { + name: string; + type: string; + isStatic: boolean; + isFinal: boolean; + startLine: number; + endLine: number; + declaratorNode: Parser.SyntaxNode; + fieldGetter: AccessorConfig | null; + fieldSetter: AccessorConfig | null; + accessors: AccessorsOptions; + accessorsPresent: boolean; +} + +interface LombokClass { + node: Parser.SyntaxNode; + name: string; + classGetter: AccessorConfig | null; + classSetter: AccessorConfig | null; + classAccessors: AccessorsOptions; + fields: LombokField[]; + existingMethods: ExistingMethodIndex; +} + +const LOMBOK_ANNOTATION_PACKAGE = new Map([ + ['Data', 'lombok'], + ['Getter', 'lombok'], + ['Setter', 'lombok'], + ['Accessors', 'lombok.experimental'], + ['Tolerate', 'lombok.experimental'], +]); + +export function getterName(fieldName: string, fieldType: string): string { + return jvmGetterName(fieldName, fieldType === 'boolean'); +} + +export function setterName(fieldName: string, fieldType: string): string { + return jvmSetterName(fieldName, fieldType === 'boolean'); +} + +// ── Provenance / imports ────────────────────────────────────────────────── + +function annotationSimpleName(nameText: string): string { + return nameText.split('.').pop() ?? nameText; +} + +interface LombokImportIndex { + bySimple: Map; + starPackages: Set; + shadowedSimpleNames: Set; +} + +/** + * Compilation-unit imports only — Java `import` is never nested in a type body. + */ +function collectLombokImports(root: Parser.SyntaxNode): LombokImportIndex { + const bySimple = new Map(); + const starPackages = new Set(); + const shadowedSimpleNames = new Set(); + for (const child of root.children) { + if (!JAVA_TYPE_DECLS.has(child.type) && child.type !== 'annotation_type_declaration') continue; + const name = child.childForFieldName('name')?.text; + if (name) shadowedSimpleNames.add(name); + } + for (const child of root.children) { + if (child.type !== 'import_declaration') continue; + if (/^import\s+static\b/.test(child.text)) continue; + const text = child.text + .replace(/^import\s+/, '') + .replace(/;\s*$/, '') + .replace(/\/\*[\s\S]*?\*\//g, '') + .replace(/\s+/g, '') + .trim(); + if (text === 'lombok.*') { + starPackages.add('lombok'); + } else if (text === 'lombok.experimental.*') { + starPackages.add('lombok.experimental'); + } else if (!text.endsWith('.*')) { + bySimple.set(annotationSimpleName(text), text); + } + } + return { bySimple, starPackages, shadowedSimpleNames }; +} + +function isProvenLombokAnnotation(nameText: string, imports: LombokImportIndex): boolean { + const simple = annotationSimpleName(nameText); + const packageName = LOMBOK_ANNOTATION_PACKAGE.get(simple); + if (packageName === undefined) return false; + if (nameText.includes('.')) return nameText === `${packageName}.${simple}`; + const imported = imports.bySimple.get(simple); + if (imported !== undefined) return imported === `${packageName}.${simple}`; + if (imports.shadowedSimpleNames.has(simple)) return false; + return imports.starPackages.has(packageName); +} + +// ── AccessLevel / Accessors structural parse ────────────────────────────── + +function parseAccessLevelToken(text: string): LombokVisibility | 'none' | null { + const simple = annotationSimpleName(text.trim()); + switch (simple) { + case 'PUBLIC': + return 'public'; + case 'PROTECTED': + return 'protected'; + case 'PRIVATE': + return 'private'; + case 'PACKAGE': + case 'MODULE': // treated as package-private for graph metadata + return 'package'; + case 'NONE': + return 'none'; + default: + return null; + } +} + +function findAccessLevelInAnnotation(ann: Parser.SyntaxNode): LombokVisibility | 'none' | null { + // Positional: @Getter(AccessLevel.PROTECTED) or @Getter(lombok.AccessLevel.NONE) + // Named: @Getter(value = AccessLevel.PRIVATE) + const stack: Parser.SyntaxNode[] = [...ann.children]; + while (stack.length > 0) { + const n = stack.pop(); + if (!n) break; + if (n.type === 'field_access' || n.type === 'identifier') { + const level = parseAccessLevelToken(n.text); + if (level !== null) return level; + } + for (const c of n.children) stack.push(c); + } + return null; +} + +function defaultAccessors(): AccessorsOptions { + return { fluent: false, hasPrefix: false, chain: false }; +} + +function parseAccessorsAnnotation(ann: Parser.SyntaxNode): AccessorsOptions { + const opts = defaultAccessors(); + const stack: Parser.SyntaxNode[] = [...ann.children]; + while (stack.length > 0) { + const n = stack.pop(); + if (!n) break; + if (n.type === 'element_value_pair') { + const key = + n.childForFieldName('key')?.text ?? n.children.find((c) => c.type === 'identifier')?.text; + const valueNode = + n.childForFieldName('value') ?? + n.children.find( + (c) => + c.type === 'true' || c.type === 'false' || c.type === 'element_value_array_initializer', + ); + if (key === 'fluent' && (valueNode?.type === 'true' || valueNode?.type === 'false')) { + opts.fluent = valueNode.type === 'true'; + } + if (key === 'chain' && (valueNode?.type === 'true' || valueNode?.type === 'false')) { + opts.chain = valueNode.type === 'true'; + } + if (key === 'prefix') opts.hasPrefix = true; + } + for (const c of n.children) stack.push(c); + } + const text = ann.text; + if (/\bprefix\s*=/.test(text)) opts.hasPrefix = true; + if (/\bfluent\s*=\s*true\b/.test(text)) opts.fluent = true; + if (/\bfluent\s*=\s*false\b/.test(text)) opts.fluent = false; + if (/\bchain\s*=\s*true\b/.test(text)) opts.chain = true; + if (/\bchain\s*=\s*false\b/.test(text)) opts.chain = false; + return opts; +} + +interface ParsedAnnotations { + getter: AccessorConfig | null; + setter: AccessorConfig | null; + accessors: AccessorsOptions; + accessorsPresent: boolean; + tolerate: boolean; +} + +function parseModifierAnnotations( + modifiersNode: Parser.SyntaxNode | null, + imports: LombokImportIndex, +): ParsedAnnotations { + const result: ParsedAnnotations = { + getter: null, + setter: null, + accessors: defaultAccessors(), + accessorsPresent: false, + tolerate: false, + }; + if (!modifiersNode) return result; + + for (const child of modifiersNode.children) { + if (child.type !== 'marker_annotation' && child.type !== 'annotation') continue; + const nameNode = child.childForFieldName('name'); + const nameText = nameNode?.text ?? ''; + if (!isProvenLombokAnnotation(nameText, imports)) continue; + const simple = annotationSimpleName(nameText); + + if (simple === 'Tolerate') { + result.tolerate = true; + continue; + } + if (simple === 'Accessors') { + result.accessors = parseAccessorsAnnotation(child); + result.accessorsPresent = true; + continue; + } + if (simple === 'Data') { + result.getter ??= { enabled: true, visibility: 'public' }; + result.setter ??= { enabled: true, visibility: 'public' }; + continue; + } + if (simple === 'Getter' || simple === 'Setter') { + const level = child.type === 'annotation' ? findAccessLevelInAnnotation(child) : null; + const cfg: AccessorConfig = + level === 'none' + ? { enabled: false, visibility: 'public' } + : { enabled: true, visibility: level ?? 'public' }; + if (simple === 'Getter') result.getter = cfg; + else result.setter = cfg; + } + } + return result; +} + +function mergeAccessors( + classOpts: AccessorsOptions, + fieldOpts: AccessorsOptions, + fieldAccessorsPresent: boolean, +): AccessorsOptions { + return fieldAccessorsPresent ? fieldOpts : classOpts; +} + +function effectiveAccessor( + classCfg: AccessorConfig | null, + fieldCfg: AccessorConfig | null, +): AccessorConfig | null { + if (fieldCfg !== null) return fieldCfg; + return classCfg; +} + +// ── Field / method collection ───────────────────────────────────────────── + +function parseFieldDeclaration( + fieldNode: Parser.SyntaxNode, + imports: LombokImportIndex, +): LombokField[] { + const typeNode = fieldNode.childForFieldName('type'); + const fieldType = typeNode?.text ?? 'Object'; + const modifiers = fieldNode.children.find((c) => c.type === 'modifiers') ?? null; + let isStatic = false; + let isFinal = false; + if (modifiers) { + for (const mod of modifiers.children) { + if (mod.text === 'static') isStatic = true; + else if (mod.text === 'final') isFinal = true; + } + } + const fieldAnn = parseModifierAnnotations(modifiers, imports); + + const declarators: Parser.SyntaxNode[] = []; + const declaratorField = fieldNode.childForFieldName('declarator'); + if (declaratorField) declarators.push(declaratorField); + for (const child of fieldNode.children) { + if (child.type === 'variable_declarator' && child !== declaratorField) { + declarators.push(child); + } + } + + const startLine = fieldNode.startPosition.row + 1; + const endLine = fieldNode.endPosition.row + 1; + const out: LombokField[] = []; + for (const declaratorNode of declarators) { + const nameNode = declaratorNode.childForFieldName('name'); + if (!nameNode) continue; + out.push({ + name: nameNode.text, + type: fieldType, + isStatic, + isFinal, + startLine, + endLine, + declaratorNode, + fieldGetter: fieldAnn.getter, + fieldSetter: fieldAnn.setter, + accessors: fieldAnn.accessors, + accessorsPresent: fieldAnn.accessorsPresent, + }); + } + return out; +} + +function methodArityRange(methodNode: Parser.SyntaxNode): { min: number; max: number } { + const params = methodNode.childForFieldName('parameters'); + if (!params) return { min: 0, max: 0 }; + let count = 0; + for (const child of params.namedChildren) { + if (child.type === 'spread_parameter') return { min: count, max: Number.POSITIVE_INFINITY }; + if (child.type === 'formal_parameter') count += 1; + } + return { min: count, max: count }; +} + +function collectExistingMethods( + classBody: Parser.SyntaxNode | null, + imports: LombokImportIndex, +): ExistingMethodIndex { + const index = createExistingMethodIndex('case-folded'); + if (!classBody) return index; + const scan = (container: Parser.SyntaxNode): void => { + for (const child of container.children) { + if (child.type === 'enum_body_declarations') { + scan(child); + continue; + } + if (child.type !== 'method_declaration') continue; + const mods = child.children.find((c) => c.type === 'modifiers') ?? null; + const ann = parseModifierAnnotations(mods, imports); + if (ann.tolerate) continue; + const nameNode = child.childForFieldName('name'); + if (!nameNode) continue; + const arity = methodArityRange(child); + rememberExistingMethodRange(index, nameNode.text, arity.min, arity.max); + } + }; + scan(classBody); + return index; +} + +const TYPE_BODIES = new Set(['class_body', 'enum_body']); + +function findTypeBody(node: Parser.SyntaxNode): Parser.SyntaxNode | null { + return node.children.find((c) => TYPE_BODIES.has(c.type)) ?? null; +} + +function findLombokClasses(root: Parser.SyntaxNode, imports: LombokImportIndex): LombokClass[] { + const classes: LombokClass[] = []; + + function walk(node: Parser.SyntaxNode): void { + if (node.type === 'class_declaration' || node.type === 'enum_declaration') { + const modifiers = node.children.find((c) => c.type === 'modifiers') ?? null; + const classAnn = parseModifierAnnotations(modifiers, imports); + const nameNode = node.childForFieldName('name'); + const className = nameNode?.text ?? ''; + if (className) { + const body = findTypeBody(node); + const fields: LombokField[] = []; + if (body) { + const collectFields = (container: Parser.SyntaxNode): void => { + for (const child of container.children) { + if (child.type === 'field_declaration') { + for (const f of parseFieldDeclaration(child, imports)) { + if (f.isStatic) continue; + fields.push(f); + } + } else if (child.type === 'enum_body_declarations') { + collectFields(child); + } + } + }; + collectFields(body); + } + + const anyFieldEnable = fields.some( + (f) => f.fieldGetter?.enabled === true || f.fieldSetter?.enabled === true, + ); + const classEnable = classAnn.getter?.enabled === true || classAnn.setter?.enabled === true; + + // Class-level NONE alone is not enable — getter/setter configs may be disabled + if (classEnable || anyFieldEnable) { + classes.push({ + node, + name: className, + classGetter: classAnn.getter, + classSetter: classAnn.setter, + classAccessors: classAnn.accessors, + fields, + existingMethods: collectExistingMethods(body, imports), + }); + } + } + } + for (const child of node.children) walk(child); + } + + walk(root); + return classes; +} + +function planAccessors(cls: LombokClass): PlannedLombokAccessor[] { + const planned: PlannedLombokAccessor[] = []; + for (const field of cls.fields) { + const accessors = mergeAccessors(cls.classAccessors, field.accessors, field.accessorsPresent); + // fluent/prefix change names — omit rather than invent wrong names + if (accessors.fluent || accessors.hasPrefix) continue; + + const getterCfg = effectiveAccessor(cls.classGetter, field.fieldGetter); + const setterCfg = effectiveAccessor(cls.classSetter, field.fieldSetter); + + if (getterCfg?.enabled) { + const gName = getterName(field.name, field.type); + if (!hasExistingMethod(cls.existingMethods, gName, 0)) { + planned.push({ + kind: 'getter', + name: gName, + returnType: field.type, + parameterTypes: [], + visibility: getterCfg.visibility, + isStatic: false, + isAbstract: false, + startLine: field.startLine, + endLine: field.endLine, + declaratorNode: field.declaratorNode, + }); + } + } + + if (setterCfg?.enabled && !field.isFinal) { + const sName = setterName(field.name, field.type); + if (!hasExistingMethod(cls.existingMethods, sName, 1)) { + // chain=true → setter returns declaring type; never emit void in that case + const returnType = accessors.chain ? cls.name : 'void'; + planned.push({ + kind: 'setter', + name: sName, + returnType, + parameterTypes: [field.type], + visibility: setterCfg.visibility, + isStatic: false, + isAbstract: false, + startLine: field.startLine, + endLine: field.endLine, + declaratorNode: field.declaratorNode, + }); + } + } + } + return planned; +} + +function planLombokAccessorOwners(root: Parser.SyntaxNode): PlannedJvmAccessorOwner[] { + const imports = collectLombokImports(root); + return findLombokClasses(root, imports).map((cls) => ({ + node: cls.node, + name: cls.name, + accessors: planAccessors(cls), + })); +} + +const lombokAccessorSynthesis = createJvmAccessorSynthesis({ + language: 'java', + synthetic: 'lombok', + planOwners: planLombokAccessorOwners, +}); + +// ── Main API ────────────────────────────────────────────────────────────── + +export function synthesizeLombokAccessors( + tree: Parser.Tree, + filePath: string, + classOwnersById: ReadonlyMap, +): LombokSynthesisResult { + return lombokAccessorSynthesis.synthesize(tree, filePath, classOwnersById); +} + +/** Scope captures for Lombok accessors (dual-path parity with record components). */ +export function synthesizeLombokAccessorCaptures(rootNode: Parser.SyntaxNode): CaptureMatch[] { + return lombokAccessorSynthesis.captures(rootNode); +} diff --git a/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts index 0410baa6c..3f23040a6 100644 --- a/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts @@ -35,6 +35,7 @@ import { attachJavaSpringConfigBindings } from './spring-config-bindings.js'; import { attachJavaSpringConditionalMetadata } from './spring-conditionals.js'; import { attachJavaSpringDiMetadata } from './spring-di.js'; import { attachJavaSpringNonHttpHandlerMetadata } from './spring-non-http-handlers.js'; +import { attachJavaSpringDynamicLookup } from './spring-dynamic-lookup.js'; import { applyJavaCaptureSideChannel, clearJavaClassAnnotationFacts, @@ -97,6 +98,7 @@ const javaScopeResolver: ScopeResolver = { attachJavaSpringDiMetadata(graph, parsedFiles, nodeLookup, indexes); attachJavaSpringNonHttpHandlerMetadata(graph, parsedFiles, nodeLookup, indexes); attachJavaSpringConfigBindings(graph, parsedFiles, nodeLookup, indexes, ctx); + attachJavaSpringDynamicLookup(graph, parsedFiles, nodeLookup, indexes); }, }; diff --git a/gitnexus/src/core/ingestion/languages/java/spring-actuator.ts b/gitnexus/src/core/ingestion/languages/java/spring-actuator.ts new file mode 100644 index 000000000..dd3627315 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/spring-actuator.ts @@ -0,0 +1,72 @@ +import type { GraphNode } from 'gitnexus-shared'; +import type { RuntimeCallableIdentity, RuntimeSymbolStrategy } from '../../language-provider.js'; + +const JVM_PRIMITIVES: Readonly> = { + B: 'byte', + C: 'char', + D: 'double', + F: 'float', + I: 'int', + J: 'long', + S: 'short', + Z: 'boolean', +}; + +function normalizedType(value: string, runtime: boolean): string { + let erased = value.trim(); + let arrayDimensions = 0; + if (erased.endsWith('...')) { + arrayDimensions++; + erased = erased.slice(0, -3); + } + while (erased.endsWith('[]')) { + arrayDimensions++; + erased = erased.slice(0, -2); + } + erased = erased.replace(/<.*>$/, '').replaceAll('$', '.').replaceAll('/', '.'); + const simple = erased.slice(erased.lastIndexOf('.') + 1); + const base = runtime ? (JVM_PRIMITIVES[simple] ?? simple) : simple; + return `${base}${'[]'.repeat(arrayDimensions)}`; +} + +function sourceTypeIsUnknown(value: string): boolean { + const type = normalizedType(value, false).replace(/(?:\[\])+$/, ''); + return type === '?' || /^[A-Z]$/.test(type); +} + +function matchesJavaCallable(node: GraphNode, runtime: RuntimeCallableIdentity): boolean { + if (node.label !== 'Method' || node.properties.name !== runtime.name) return false; + + const descriptorTypes = runtime.descriptorParameterTypes; + if (descriptorTypes === undefined) return true; + + const parameterCount = node.properties.parameterCount; + if (typeof parameterCount === 'number' && parameterCount !== descriptorTypes.length) return false; + + const sourceTypes = node.properties.parameterTypes; + if ( + !Array.isArray(sourceTypes) || + sourceTypes.length !== descriptorTypes.length || + !sourceTypes.every((type): type is string => typeof type === 'string') + ) { + return true; + } + + return sourceTypes.every((sourceType, index) => { + if (sourceTypeIsUnknown(sourceType)) return true; + const source = normalizedType(sourceType, false); + const descriptor = normalizedType(descriptorTypes[index] ?? '', true); + if (source === descriptor) return true; + // Java parser metadata currently drops the ellipsis from varargs and also + // leaves parameterCount open-ended. Only in that shape may T match JVM T[]. + return ( + typeof parameterCount !== 'number' && + descriptor.endsWith('[]') && + source === descriptor.slice(0, -2) + ); + }); +} + +export const javaRuntimeSymbolStrategy: RuntimeSymbolStrategy = { + matchesCallable: matchesJavaCallable, +}; diff --git a/gitnexus/src/core/ingestion/languages/java/spring-di.ts b/gitnexus/src/core/ingestion/languages/java/spring-di.ts index c6dcbe261..121cd1e3b 100644 --- a/gitnexus/src/core/ingestion/languages/java/spring-di.ts +++ b/gitnexus/src/core/ingestion/languages/java/spring-di.ts @@ -12,13 +12,72 @@ import { hasSpringBeanFactorySyntax, type SpringBeanFactoryMethodFact, } from '../../frameworks/spring/bean-factories.js'; +import { + normalizeSpringFactText, + type SpringArgumentFact, +} from '../../frameworks/spring/argument-facts.js'; import { parseSpringInjectionType } from '../../di-extractors/spring.js'; -import { nodeToCapture, type SyntaxNode } from '../../utils/ast-helpers.js'; +import { hasRecoveredSyntax, nodeToCapture, type SyntaxNode } from '../../utils/ast-helpers.js'; import { isJavaPackageSiblingVisibilityIncomplete } from './package-siblings.js'; import { getJavaSpringDiFacts } from './capture-side-channel.js'; export interface JavaAnnotationSyntaxFact extends SpringDiAnnotationFact { readonly line: number; + /** Present only for callers that opt in via `javaSpringAnnotationFacts`. */ + readonly args?: readonly SpringArgumentFact[]; +} + +/** + * Options for `javaSpringAnnotationFacts`. + * + * The STRUCTURED arguments are opt-in because DI captures every annotated + * field, constructor, and method in the repository, and none of its consumers + * reads them. Note what this does and does not save: every fact already carries + * `text`, the annotation's full source, so the argument TEXT crosses the worker + * boundary either way. What the opt-in avoids is a second, parsed copy of that + * same text on facts that would never look at it. + */ +export interface JavaSpringAnnotationFactOptions { + readonly includeArguments?: boolean; +} + +const JAVA_COMMENT_NODE_TYPES = new Set(['line_comment', 'block_comment']); + +/** + * Annotation arguments as written, or `undefined` for a marker annotation. + * + * `@Scheduled` yields `undefined` (no argument list in the syntax) while + * `@Scheduled()` yields `[]` (an empty list was written). Named arguments keep + * their key, single-element ones stay positional, and array initializers are + * kept as one raw `{...}` text — splitting or dereferencing them would be + * resolution, which does not belong at capture time. + * + * An argument list that did not parse also yields `undefined`. Error recovery + * fills gaps with invented nodes — `@KafkaListener(topics = "orders", groupId =` + * hands back a `groupId` whose value is a `{}` that nobody wrote — and there is + * no fourth state here for "unreadable". Collapsing it into the marker case is + * deliberate: both tell a consumer there is nothing here to resolve, which is + * true, whereas a fabricated value would send it somewhere real and wrong. + */ +function javaAnnotationArgumentFacts(annotation: SyntaxNode): SpringArgumentFact[] | undefined { + const argumentList = annotation.childForFieldName('arguments'); + if (argumentList === null || hasRecoveredSyntax(argumentList)) return undefined; + const args: SpringArgumentFact[] = []; + for (const child of argumentList.namedChildren) { + if (JAVA_COMMENT_NODE_TYPES.has(child.type)) continue; + if (child.type === 'element_value_pair') { + const key = child.childForFieldName('key'); + const value = child.childForFieldName('value'); + if (key === null || value === null) { + args.push({ text: normalizeSpringFactText(child.text) }); + continue; + } + args.push({ name: key.text.trim(), text: normalizeSpringFactText(value.text) }); + continue; + } + args.push({ text: normalizeSpringFactText(child.text) }); + } + return args; } export type JavaSpringDependencyFact = SpringDiDependencyFact; @@ -36,7 +95,10 @@ export type JavaSpringDiClassFact = SpringDiClassFact< >; type JavaSpringBeanFactoryMethodFact = SpringBeanFactoryMethodFact; -export function javaSpringAnnotationFacts(node: SyntaxNode): JavaAnnotationSyntaxFact[] { +export function javaSpringAnnotationFacts( + node: SyntaxNode, + options: JavaSpringAnnotationFactOptions = {}, +): JavaAnnotationSyntaxFact[] { const facts: JavaAnnotationSyntaxFact[] = []; for (const child of node.namedChildren) { if (child.type !== 'modifiers') continue; @@ -44,10 +106,13 @@ export function javaSpringAnnotationFacts(node: SyntaxNode): JavaAnnotationSynta if (modifier.type !== 'marker_annotation' && modifier.type !== 'annotation') continue; const nameNode = modifier.childForFieldName('name') ?? modifier.firstNamedChild; if (nameNode === null) continue; + const args = + options.includeArguments === true ? javaAnnotationArgumentFacts(modifier) : undefined; facts.push({ name: nameNode.text.trim(), text: modifier.text.trim(), line: modifier.startPosition.row + 1, + ...(args === undefined ? {} : { args }), }); } } diff --git a/gitnexus/src/core/ingestion/languages/java/spring-dynamic-lookup.ts b/gitnexus/src/core/ingestion/languages/java/spring-dynamic-lookup.ts new file mode 100644 index 000000000..922264bf8 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/spring-dynamic-lookup.ts @@ -0,0 +1,77 @@ +import { makeScopeId } from 'gitnexus-shared'; +import { + createSpringDynamicLookupMetadataAttacher, + springDynamicLookupCardinality, + type SpringDynamicLookupFact, +} from '../../frameworks/spring/dynamic-lookups.js'; +import { + findAncestorBeforeBoundary, + nodeToCapture, + type SyntaxNode, +} from '../../utils/ast-helpers.js'; +import { getJavaSpringDynamicLookupFacts } from './capture-side-channel.js'; + +const CALLABLE_NODE_TYPES = new Set([ + 'method_declaration', + 'constructor_declaration', + 'compact_constructor_declaration', +]); +const NO_CALLABLE_BOUNDARIES = new Set(); + +function classLiteralTypeName(argument: SyntaxNode): string | null { + if (argument.type !== 'class_literal' || argument.namedChildCount !== 1) return null; + return argument.namedChild(0)?.text.trim() ?? null; +} + +/** Capture real Java method invocations; comments and literals are never visited as calls. */ +export function captureJavaSpringDynamicLookupFact( + node: SyntaxNode, + filePath: string, +): SpringDynamicLookupFact | null { + if (node.type !== 'method_invocation') return null; + const receiverName = node.childForFieldName('object')?.text.trim(); + const methodName = node.childForFieldName('name')?.text.trim(); + const argumentsNode = node.childForFieldName('arguments'); + if (receiverName === undefined || methodName === undefined || argumentsNode === null) return null; + if (springDynamicLookupCardinality(receiverName, methodName) === null) return null; + + const argumentsWithoutComments = argumentsNode.namedChildren.filter( + (child) => child.type !== 'line_comment' && child.type !== 'block_comment', + ); + if (argumentsWithoutComments.length !== 1) return null; + const argument = argumentsWithoutComments[0]; + if (argument === undefined) return null; + const targetTypeName = classLiteralTypeName(argument); + if (targetTypeName === null) return null; + + const owner = findAncestorBeforeBoundary(node, CALLABLE_NODE_TYPES, NO_CALLABLE_BOUNDARIES); + if (owner === null) return null; + const ownerCapture = nodeToCapture('@spring-dynamic-lookup.owner', owner); + return { + ownerScopeId: makeScopeId({ + filePath, + range: ownerCapture.range, + kind: 'Function', + }), + ownerRange: ownerCapture.range, + receiverName, + methodName, + targetTypeName, + }; +} + +/** Standalone extractor for focused tests; production reuses scope-query call nodes. */ +export function captureJavaSpringDynamicLookupFacts( + rootNode: SyntaxNode, + filePath: string, +): SpringDynamicLookupFact[] { + return rootNode + .descendantsOfType('method_invocation') + .map((node) => captureJavaSpringDynamicLookupFact(node, filePath)) + .filter((fact): fact is SpringDynamicLookupFact => fact !== null); +} + +/** Attach Java lookup facts for later resolution by the shared DI phase. */ +export const attachJavaSpringDynamicLookup = createSpringDynamicLookupMetadataAttacher({ + getFacts: getJavaSpringDynamicLookupFacts, +}); diff --git a/gitnexus/src/core/ingestion/languages/java/spring-message-producers.ts b/gitnexus/src/core/ingestion/languages/java/spring-message-producers.ts new file mode 100644 index 000000000..ce59015ba --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/spring-message-producers.ts @@ -0,0 +1,100 @@ +import { makeScopeId } from 'gitnexus-shared'; +import { + normalizeSpringFactText, + type SpringArgumentFact, +} from '../../frameworks/spring/argument-facts.js'; +import { + isSpringMessageProducerMethod, + springMessageProducerTemplateOf, + type SpringMessageProducerFact, +} from '../../frameworks/spring/message-producers.js'; +import { + findAncestorBeforeBoundary, + hasRecoveredSyntax, + nodeToCapture, + type SyntaxNode, +} from '../../utils/ast-helpers.js'; + +const CALLABLE_NODE_TYPES = new Set([ + 'method_declaration', + 'constructor_declaration', + 'compact_constructor_declaration', +]); +/** + * A type body ends the search for the publishing callable. + * + * Without it the ancestor walk passes THROUGH the body of a class declared + * inside a method, so a publish in that class's field initializer is attributed + * to the enclosing method, which may never run it. The identical construct at + * the top level of a class already yields no fact — there is no enclosing + * callable to find — and the rule has to read the same at every depth. + */ +const TYPE_BODY_BOUNDARIES = new Set([ + 'class_body', + 'interface_body', + 'enum_body', + 'enum_body_declarations', + 'annotation_type_body', +]); +const COMMENT_NODE_TYPES = new Set(['line_comment', 'block_comment']); + +/** Java has no named call arguments, so every argument is captured positionally. */ +function javaCallArgumentFacts(argumentList: SyntaxNode): SpringArgumentFact[] { + return argumentList.namedChildren + .filter((child) => !COMMENT_NODE_TYPES.has(child.type)) + .map((child) => ({ text: normalizeSpringFactText(child.text) })); +} + +/** + * Capture one messaging-template publish from a Java call already surfaced by + * the scope query, without resolving the destination it names. + * + * The destination argument may be a literal, a reference to a constant that + * lives in another file, or a `${...}` placeholder resolved from configuration; + * all three are recorded as written and left to a later phase. + * + * A call whose argument list did not parse yields NO fact. The fact exists to + * carry a destination, and error recovery invents argument boundaries — an + * unterminated `send(TOPIC,` absorbs the next declaration's source and offers + * it as an argument. There is no state on this fact that means "published + * somewhere unreadable", so the choice is between silence and a plausible lie, + * and silence is recoverable: the file is re-captured when it parses. + */ +export function captureJavaSpringMessageProducerFact( + node: SyntaxNode, + filePath: string, +): SpringMessageProducerFact | null { + if (node.type !== 'method_invocation') return null; + const methodName = node.childForFieldName('name')?.text.trim(); + if (methodName === undefined || !isSpringMessageProducerMethod(methodName)) return null; + const receiverText = node.childForFieldName('object')?.text; + if (receiverText === undefined) return null; + const receiverName = normalizeSpringFactText(receiverText); + const template = springMessageProducerTemplateOf(receiverName, methodName); + if (template === null) return null; + + const argumentList = node.childForFieldName('arguments'); + if (argumentList !== null && hasRecoveredSyntax(argumentList)) return null; + const owner = findAncestorBeforeBoundary(node, CALLABLE_NODE_TYPES, TYPE_BODY_BOUNDARIES); + if (owner === null) return null; + const ownerCapture = nodeToCapture('@spring-message-producer.owner', owner); + return { + ownerScopeId: makeScopeId({ filePath, range: ownerCapture.range, kind: 'Function' }), + ownerRange: ownerCapture.range, + template, + receiverName, + methodName, + ...(argumentList === null ? {} : { args: javaCallArgumentFacts(argumentList) }), + }; +} + +/** Standalone extractor for focused tests; production reuses scope-query call nodes. */ +export function captureJavaSpringMessageProducerFacts( + rootNode: SyntaxNode, + filePath: string, +): SpringMessageProducerFact[] { + return rootNode + .descendantsOfType('method_invocation') + .map((node) => captureJavaSpringMessageProducerFact(node, filePath)) + .filter((fact): fact is SpringMessageProducerFact => fact !== null); +} diff --git a/gitnexus/src/core/ingestion/languages/java/spring-non-http-handlers.ts b/gitnexus/src/core/ingestion/languages/java/spring-non-http-handlers.ts index 259dace3a..a9e7c867e 100644 --- a/gitnexus/src/core/ingestion/languages/java/spring-non-http-handlers.ts +++ b/gitnexus/src/core/ingestion/languages/java/spring-non-http-handlers.ts @@ -11,7 +11,17 @@ import { javaSpringAnnotationFacts, type JavaAnnotationSyntaxFact } from './spri export type JavaSpringNonHttpHandlerFact = SpringNonHttpHandlerFact; -/** Capture callable syntax while the Java class AST is already in hand. */ +/** + * Capture callable syntax while the Java class AST is already in hand. + * + * Annotation arguments are read in a second pass, only for callables that + * already carry a handler annotation, so the destination-bearing arguments + * (`topics`, `queues`, `destination`, `cron`) reach the fact without adding + * structured argument text to every annotation in the repository. Java can + * decide that on the simple name alone; Kotlin runs the same two passes but + * widens the first one with the file's import aliases, because a Kotlin handler + * annotation may be written under a name no list can contain. + */ export function captureJavaSpringNonHttpHandlerFacts( classNode: SyntaxNode, filePath: string, @@ -21,8 +31,8 @@ export function captureJavaSpringNonHttpHandlerFacts( if (body === null) return facts; for (const member of body.namedChildren) { if (member.type !== 'method_declaration') continue; - const annotations = javaSpringAnnotationFacts(member); - if (!hasSpringNonHttpHandlerRelevantAnnotation(annotations)) continue; + if (!hasSpringNonHttpHandlerRelevantAnnotation(javaSpringAnnotationFacts(member))) continue; + const annotations = javaSpringAnnotationFacts(member, { includeArguments: true }); const ownerRange = nodeToCapture('@spring-non-http-handler.owner', member).range; facts.push({ ownerScopeId: makeScopeId({ filePath, range: ownerRange, kind: 'Function' }), diff --git a/gitnexus/src/core/ingestion/languages/jvm/accessor-synthesis.ts b/gitnexus/src/core/ingestion/languages/jvm/accessor-synthesis.ts new file mode 100644 index 000000000..e1cb89253 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/jvm/accessor-synthesis.ts @@ -0,0 +1,316 @@ +/** + * Shared planning orchestration and emission for synthetic JVM accessors. + * + * Language adapters discover accessor plans. This module owns method-collision + * policy, graph emission, and scope captures without naming any language. + */ +import type Parser from 'tree-sitter'; +import type { Capture, CaptureMatch } from 'gitnexus-shared'; +import { toZeroBasedLine } from '../../utils/line-base.js'; + +export type SyntheticVisibility = 'public' | 'protected' | 'private' | 'package'; +export type MethodNameMatching = 'exact' | 'case-folded'; + +export interface ExistingMethodIndex { + readonly matching: MethodNameMatching; + readonly aritiesByName: Map>; + readonly arityRangesByName: Map>; +} + +export function createExistingMethodIndex(matching: MethodNameMatching): ExistingMethodIndex { + return { matching, aritiesByName: new Map(), arityRangesByName: new Map() }; +} + +function methodKey(index: ExistingMethodIndex, name: string): string { + return index.matching === 'case-folded' ? name.toLowerCase() : name; +} + +export function rememberExistingMethod( + index: ExistingMethodIndex, + name: string, + arity: number, +): void { + const key = methodKey(index, name); + let arities = index.aritiesByName.get(key); + if (!arities) { + arities = new Set(); + index.aritiesByName.set(key, arities); + } + arities.add(arity); +} + +export function rememberExistingMethodRange( + index: ExistingMethodIndex, + name: string, + min: number, + max: number, +): void { + if (min === max) { + rememberExistingMethod(index, name, min); + return; + } + const key = methodKey(index, name); + const ranges = index.arityRangesByName.get(key) ?? []; + ranges.push({ min, max }); + index.arityRangesByName.set(key, ranges); +} + +export function hasExistingMethod( + index: ExistingMethodIndex, + name: string, + arity: number, +): boolean { + const key = methodKey(index, name); + if (index.aritiesByName.get(key)?.has(arity) === true) return true; + return ( + index.arityRangesByName.get(key)?.some((range) => range.min <= arity && arity <= range.max) === + true + ); +} + +export interface SyntheticAccessorSymbol { + filePath: string; + name: string; + nodeId: string; + type: 'Method'; + ownerId: string; + qualifiedName: string; + parameterCount: number; + requiredParameterCount: number; + parameterTypes: string[]; + returnType: string; + visibility: SyntheticVisibility; + isStatic: boolean; + isAbstract: boolean; + isFinal: boolean; +} + +export interface SyntheticAccessorNode { + id: string; + label: 'Method'; + properties: { + name: string; + filePath: string; + startLine: number; + endLine: number; + language: string; + isExported: boolean; + synthetic: string; + visibility: SyntheticVisibility; + isStatic: boolean; + returnType: string; + parameterTypes: string[]; + parameterCount: number; + qualifiedName: string; + }; +} + +export interface SyntheticAccessorRelationship { + id: string; + sourceId: string; + targetId: string; + type: 'HAS_METHOD'; + confidence: number; + reason: string; +} + +export interface SyntheticAccessorResult { + symbols: SyntheticAccessorSymbol[]; + nodes: SyntheticAccessorNode[]; + relationships: SyntheticAccessorRelationship[]; +} + +export interface PlannedJvmAccessor { + kind: 'getter' | 'setter'; + name: string; + returnType: string; + parameterTypes: string[]; + visibility: SyntheticVisibility; + isStatic: boolean; + isAbstract: boolean; + startLine: number; + endLine: number; + declaratorNode: Parser.SyntaxNode; +} + +export interface PlannedJvmAccessorOwner { + node: Parser.SyntaxNode; + name: string; + accessors: readonly PlannedJvmAccessor[]; +} + +interface JvmAccessorSynthesisConfig { + language: string; + synthetic: string; + planOwners(rootNode: Parser.SyntaxNode): readonly PlannedJvmAccessorOwner[]; +} + +export interface JvmAccessorSynthesis { + synthesize( + tree: Parser.Tree, + filePath: string, + classOwnersById: ReadonlyMap, + ): SyntheticAccessorResult; + captures(rootNode: Parser.SyntaxNode): CaptureMatch[]; +} + +export function createJvmAccessorSynthesis( + config: JvmAccessorSynthesisConfig, +): JvmAccessorSynthesis { + return { + synthesize(tree, filePath, classOwnersById) { + const result = emptySyntheticAccessorResult(); + for (const owner of config.planOwners(tree.rootNode)) { + const ownerId = classOwnersById.get(owner.node.id); + if (!ownerId) continue; + emitPlannedAccessors({ + planned: owner.accessors, + filePath, + ownerId, + idPrefix: ownerIdNamePrefix(ownerId, filePath, owner.name), + language: config.language, + synthetic: config.synthetic, + result, + }); + } + return result; + }, + captures(rootNode) { + return capturesForPlannedAccessors(config.planOwners(rootNode)); + }, + }; +} + +function emptySyntheticAccessorResult(): SyntheticAccessorResult { + return { symbols: [], nodes: [], relationships: [] }; +} + +function ownerIdNamePrefix(ownerId: string, filePath: string, fallback: string): string { + const needle = `Class:${filePath}:`; + if (ownerId.startsWith(needle)) return ownerId.slice(needle.length); + const enumNeedle = `Enum:${filePath}:`; + if (ownerId.startsWith(enumNeedle)) return ownerId.slice(enumNeedle.length); + const ifaceNeedle = `Interface:${filePath}:`; + if (ownerId.startsWith(ifaceNeedle)) return ownerId.slice(ifaceNeedle.length); + return fallback; +} + +export function jvmTypeSimpleName(node: Parser.SyntaxNode): string | undefined { + const named = node.childForFieldName('name')?.text; + if (named) return named; + for (const child of node.namedChildren) { + if (child.type === 'type_identifier' || child.type === 'simple_identifier') return child.text; + } + return undefined; +} + +function emitPlannedAccessors(args: { + planned: readonly PlannedJvmAccessor[]; + filePath: string; + ownerId: string; + idPrefix: string; + language: string; + synthetic: string; + result: SyntheticAccessorResult; +}): void { + const emittedIds = new Set(); + for (const acc of args.planned) { + const arity = acc.parameterTypes.length; + const qualifiedName = `${args.idPrefix}.${acc.name}`; + const nodeId = `Method:${args.filePath}:${qualifiedName}#${arity}`; + if (emittedIds.has(nodeId)) continue; + emittedIds.add(nodeId); + args.result.nodes.push({ + id: nodeId, + label: 'Method', + properties: { + name: acc.name, + filePath: args.filePath, + startLine: toZeroBasedLine(acc.startLine), + endLine: toZeroBasedLine(acc.endLine), + language: args.language, + isExported: false, + synthetic: args.synthetic, + visibility: acc.visibility, + isStatic: acc.isStatic, + returnType: acc.returnType, + parameterTypes: acc.parameterTypes, + parameterCount: arity, + qualifiedName, + }, + }); + args.result.symbols.push({ + filePath: args.filePath, + name: acc.name, + nodeId, + type: 'Method', + ownerId: args.ownerId, + qualifiedName, + parameterCount: arity, + requiredParameterCount: arity, + parameterTypes: acc.parameterTypes, + returnType: acc.returnType, + visibility: acc.visibility, + isStatic: acc.isStatic, + isAbstract: acc.isAbstract, + isFinal: false, + }); + args.result.relationships.push({ + id: `HAS_METHOD:${args.ownerId}->${nodeId}`, + sourceId: args.ownerId, + targetId: nodeId, + type: 'HAS_METHOD', + confidence: 1.0, + reason: acc.kind === 'getter' ? `${args.synthetic}-getter` : `${args.synthetic}-setter`, + }); + } +} + +function accessorCapture(name: string, acc: PlannedJvmAccessor, text: string): Capture { + const node = acc.declaratorNode; + const startLine = node.startPosition.row + 1; + const startCol = node.startPosition.column; + const endLine = node.endPosition.row + 1; + const endCol = acc.kind === 'getter' ? node.endPosition.column : startCol; + return { name, range: { startLine, startCol, endLine, endCol }, text }; +} + +function capturesForPlannedAccessors(owners: readonly PlannedJvmAccessorOwner[]): CaptureMatch[] { + const captures: CaptureMatch[] = []; + for (const owner of owners) { + const enclosing = owner.name; + const emitted = new Set(); + for (const acc of owner.accessors) { + const arity = String(acc.parameterTypes.length); + const qualifiedName = `${enclosing}.${acc.name}`; + const identity = `${qualifiedName}#${arity}`; + if (emitted.has(identity)) continue; + emitted.add(identity); + captures.push({ + '@scope.function': accessorCapture('@scope.function', acc, acc.name), + }); + captures.push({ + '@declaration.method': accessorCapture('@declaration.method', acc, acc.name), + '@declaration.name': accessorCapture('@declaration.name', acc, acc.name), + '@declaration.qualified_name': accessorCapture( + '@declaration.qualified_name', + acc, + qualifiedName, + ), + '@declaration.parameter-count': accessorCapture('@declaration.parameter-count', acc, arity), + '@declaration.required-parameter-count': accessorCapture( + '@declaration.required-parameter-count', + acc, + arity, + ), + '@declaration.return-type': accessorCapture( + '@declaration.return-type', + acc, + acc.returnType, + ), + '@declaration.is-synthetic': accessorCapture('@declaration.is-synthetic', acc, 'true'), + }); + } + } + return captures; +} diff --git a/gitnexus/src/core/ingestion/languages/jvm/beanspec.ts b/gitnexus/src/core/ingestion/languages/jvm/beanspec.ts new file mode 100644 index 000000000..0a14209f5 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/jvm/beanspec.ts @@ -0,0 +1,49 @@ +/** + * Language-neutral JVM JavaBeans naming primitives. + * + * Language adapters choose whether to invent/preserve an `is` prefix and + * which single-character capitalization policy their compiler uses. + */ + +export function capitalizeBeanName(s: string): string { + if (s.length === 0) return s; + const first = s.charAt(0); + const upper = first.toUpperCase(); + // Java Character case conversion is one UTF-16 code unit. JavaScript + // full-case conversion may expand one unit (`ß` → `SS`), which would invent + // a method name no JVM compiler emits. + return (upper.length === 1 ? upper : first) + s.slice(1); +} + +/** + * Primitive-boolean / Kotlin `is`-prefix fields whose name already starts with + * `is` plus a non-lowercase character keep that name for the getter and drop + * the `is` prefix for the setter base (`isEnabled` → `isEnabled()` / + * `setEnabled(...)`, `is1` → `is1()` / `set1(...)`). Digits and punctuation + * count as non-lowercase, matching Lombok `!Character.isLowerCase` and kotlinc. + */ +export function booleanIsPrefixBase(fieldName: string, useIsPrefix: boolean): string | null { + if (!useIsPrefix || !fieldName.startsWith('is') || fieldName.length < 3) return null; + const third = fieldName.charAt(2); + return third === third.toUpperCase() ? fieldName.slice(2) : null; +} + +export function jvmGetterName( + fieldName: string, + useIsPrefix: boolean, + capitalize: (name: string) => string = capitalizeBeanName, +): string { + if (booleanIsPrefixBase(fieldName, useIsPrefix) !== null) return fieldName; + if (useIsPrefix) return `is${capitalize(fieldName)}`; + return `get${capitalize(fieldName)}`; +} + +export function jvmSetterName( + fieldName: string, + useIsPrefix: boolean, + capitalize: (name: string) => string = capitalizeBeanName, +): string { + const stripped = booleanIsPrefixBase(fieldName, useIsPrefix); + if (stripped !== null) return `set${stripped}`; + return `set${capitalize(fieldName)}`; +} diff --git a/gitnexus/src/core/ingestion/languages/kotlin.ts b/gitnexus/src/core/ingestion/languages/kotlin.ts index 18d4fd9c1..b04a58427 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin.ts @@ -24,6 +24,10 @@ import type { SyntaxNode } from '../utils/ast-helpers.js'; import { createCallExtractor } from '../call-extractors/generic.js'; import { kotlinCallConfig } from '../call-extractors/configs/jvm.js'; import { createKotlinCfgVisitor } from '../cfg/visitors/kotlin.js'; +import { + getKotlinSpringMessageProducerFacts, + getKotlinSpringNonHttpHandlerFacts, +} from './kotlin/capture-side-channel.js'; import { createFieldExtractor } from '../field-extractors/generic.js'; import { kotlinConfig } from '../field-extractors/configs/jvm.js'; import { createMethodExtractor } from '../method-extractors/generic.js'; @@ -41,6 +45,16 @@ import { kotlinMergeBindings, kotlinReceiverBinding, } from './kotlin/index.js'; +import { synthesizeLombokAccessors } from './kotlin/lombok-synthesizer.js'; +import { + extractKotlinRuntimeSymbolProperties, + kotlinRuntimeSymbolStrategy, +} from './kotlin/spring-actuator.js'; +import { extractKotlinSpringRoutes } from '../route-extractors/kotlin-spring.js'; +import { + extractKotlinModuleConstants, + foldKotlinOperands, +} from '../route-extractors/kotlin-const-resolver.js'; /** Check if a Kotlin function_declaration capture is inside a class_body (i.e., a method). * Kotlin grammar uses function_declaration for both top-level functions and class methods. @@ -174,6 +188,8 @@ export const kotlinProvider = defineLanguage({ // ── KDoc → description (issue #2270) ── descriptionExtractor: createLeadingDocDescriptionExtractor(), + definitionPropertiesExtractor: extractKotlinRuntimeSymbolProperties, + runtimeSymbolStrategy: kotlinRuntimeSymbolStrategy, labelOverride: (functionNode, defaultLabel) => { if (defaultLabel !== 'Function') return defaultLabel; @@ -202,4 +218,23 @@ export const kotlinProvider = defineLanguage({ mergeBindings: (_scope, bindings) => kotlinMergeBindings(bindings), receiverBinding: kotlinReceiverBinding, arityCompatibility: kotlinArityCompatibility, + synthesizeStructureMembers: synthesizeLombokAccessors, + + // ── Spring decorator routes + composed path constants (#3130) ── + extractDecoratorRoutes: extractKotlinSpringRoutes, + extractModuleConstants: extractKotlinModuleConstants, + foldRoutePathOperands: foldKotlinOperands, + + // Async messaging facts for the `springDestinations` phase. Both stores are + // repopulated on the main thread by `applyKotlinCaptureSideChannel`, so this + // answers for cache hits and misses alike. + getSpringMessagingFacts: (filePath) => ({ + handlers: getKotlinSpringNonHttpHandlerFacts(filePath), + producers: getKotlinSpringMessageProducerFacts(filePath), + }), + // Kotlin string literals interpolate: `"orders-$env"` and `"orders-${env}"` + // are string templates, and a Spring property placeholder has to escape the + // dollar (`"\${app.topic}"`). Destination resolution needs this to keep a + // runtime template out of the address namespace. + interpolatesStringLiterals: true, }); diff --git a/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts index 6ea9480a8..6983e55ec 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts @@ -49,16 +49,22 @@ import { } from '../jvm/package-facts.js'; import { getCompanionScopesForFile, markCompanionScope } from './companion-scopes.js'; import { getKotlinPackageFact, setKotlinPackageFact } from './package-facts.js'; +import type { SpringDynamicLookupFact } from '../../frameworks/spring/dynamic-lookups.js'; +import type { SpringMessageProducerFact } from '../../frameworks/spring/message-producers.js'; import type { KotlinSpringAopFact } from './spring-aop.js'; import type { KotlinSpringConditionalFact } from './spring-conditionals.js'; import type { KotlinSpringDiClassFact } from './spring-di.js'; import type { KotlinSpringNonHttpHandlerFact } from './spring-non-http-handlers.js'; +import type { KotlinSpringConfigConsumerFact } from './spring-config-bindings.js'; const classAnnotations = createClassAnnotationFactStore(); const springAopFacts = new Map(); const springConditionalFacts = new Map(); const springDiFacts = new Map(); +const springDynamicLookupFacts = new Map(); const springNonHttpHandlerFacts = new Map(); +const springConfigConsumerFacts = new Map(); +const springMessageProducerFacts = new Map(); /** * Plain JSON-serializable snapshot of the per-file Kotlin capture-time @@ -80,8 +86,14 @@ export interface KotlinCaptureSideChannel { readonly springConditionalFacts?: readonly KotlinSpringConditionalFact[]; /** Constructor, property, and method injection syntax captured per class. */ readonly springDiFacts?: readonly KotlinSpringDiClassFact[]; + /** Programmatic Spring bean lookups captured per callable. */ + readonly springDynamicLookupFacts?: readonly SpringDynamicLookupFact[]; /** Scheduled, event, messaging, and managed-job handler syntax captured per callable. */ readonly springNonHttpHandlerFacts?: readonly KotlinSpringNonHttpHandlerFact[]; + /** `@Value` / `@ConfigurationProperties` syntax captured per owner. */ + readonly springConfigConsumerFacts?: readonly KotlinSpringConfigConsumerFact[]; + /** Messaging-template publish syntax captured per callable. */ + readonly springMessageProducerFacts?: readonly SpringMessageProducerFact[]; } export function clearKotlinClassAnnotationFacts(): void { @@ -89,7 +101,10 @@ export function clearKotlinClassAnnotationFacts(): void { springAopFacts.clear(); springConditionalFacts.clear(); springDiFacts.clear(); + springDynamicLookupFacts.clear(); springNonHttpHandlerFacts.clear(); + springConfigConsumerFacts.clear(); + springMessageProducerFacts.clear(); } export function setKotlinSpringAopFacts( @@ -141,6 +156,20 @@ export function getKotlinSpringDiFacts(filePath: string): readonly KotlinSpringD return springDiFacts.get(filePath) ?? []; } +export function setKotlinSpringDynamicLookupFacts( + filePath: string, + facts: readonly SpringDynamicLookupFact[], +): void { + if (facts.length === 0) springDynamicLookupFacts.delete(filePath); + else springDynamicLookupFacts.set(filePath, facts); +} + +export function getKotlinSpringDynamicLookupFacts( + filePath: string, +): readonly SpringDynamicLookupFact[] { + return springDynamicLookupFacts.get(filePath) ?? []; +} + export function setKotlinSpringNonHttpHandlerFacts( filePath: string, facts: readonly KotlinSpringNonHttpHandlerFact[], @@ -155,6 +184,34 @@ export function getKotlinSpringNonHttpHandlerFacts( return springNonHttpHandlerFacts.get(filePath) ?? []; } +export function setKotlinSpringConfigConsumerFacts( + filePath: string, + facts: readonly KotlinSpringConfigConsumerFact[], +): void { + if (facts.length === 0) springConfigConsumerFacts.delete(filePath); + else springConfigConsumerFacts.set(filePath, facts); +} + +export function getKotlinSpringConfigConsumerFacts( + filePath: string, +): readonly KotlinSpringConfigConsumerFact[] { + return springConfigConsumerFacts.get(filePath) ?? []; +} + +export function setKotlinSpringMessageProducerFacts( + filePath: string, + facts: readonly SpringMessageProducerFact[], +): void { + if (facts.length === 0) springMessageProducerFacts.delete(filePath); + else springMessageProducerFacts.set(filePath, facts); +} + +export function getKotlinSpringMessageProducerFacts( + filePath: string, +): readonly SpringMessageProducerFact[] { + return springMessageProducerFacts.get(filePath) ?? []; +} + /** * `LanguageProvider.collectCaptureSideChannel` implementation for Kotlin. * Returns `undefined` when this file recorded no side-channel state at all, so @@ -168,7 +225,10 @@ export function collectKotlinCaptureSideChannel( const aopFacts = springAopFacts.get(filePath) ?? []; const conditionFacts = springConditionalFacts.get(filePath) ?? []; const diFacts = springDiFacts.get(filePath) ?? []; + const dynamicLookupFacts = springDynamicLookupFacts.get(filePath) ?? []; const nonHttpHandlerFacts = springNonHttpHandlerFacts.get(filePath) ?? []; + const configConsumerFacts = springConfigConsumerFacts.get(filePath) ?? []; + const messageProducerFacts = springMessageProducerFacts.get(filePath) ?? []; const packageFact = getKotlinPackageFact(filePath); if ( companionScopes.length === 0 && @@ -176,7 +236,10 @@ export function collectKotlinCaptureSideChannel( aopFacts.length === 0 && conditionFacts.length === 0 && diFacts.length === 0 && + dynamicLookupFacts.length === 0 && nonHttpHandlerFacts.length === 0 && + configConsumerFacts.length === 0 && + messageProducerFacts.length === 0 && packageFact === undefined ) { return undefined; @@ -189,7 +252,12 @@ export function collectKotlinCaptureSideChannel( ...(aopFacts.length > 0 ? { springAopFacts: aopFacts } : {}), ...(conditionFacts.length > 0 ? { springConditionalFacts: conditionFacts } : {}), ...(diFacts.length > 0 ? { springDiFacts: diFacts } : {}), + ...(dynamicLookupFacts.length > 0 ? { springDynamicLookupFacts: dynamicLookupFacts } : {}), ...(nonHttpHandlerFacts.length > 0 ? { springNonHttpHandlerFacts: nonHttpHandlerFacts } : {}), + ...(configConsumerFacts.length > 0 ? { springConfigConsumerFacts: configConsumerFacts } : {}), + ...(messageProducerFacts.length > 0 + ? { springMessageProducerFacts: messageProducerFacts } + : {}), }; } @@ -215,7 +283,10 @@ export function applyKotlinCaptureSideChannel(parsed: ParsedFile): void { setKotlinSpringAopFacts(parsed.filePath, []); setKotlinSpringConditionalFacts(parsed.filePath, []); setKotlinSpringDiFacts(parsed.filePath, []); + setKotlinSpringDynamicLookupFacts(parsed.filePath, []); setKotlinSpringNonHttpHandlerFacts(parsed.filePath, []); + setKotlinSpringConfigConsumerFacts(parsed.filePath, []); + setKotlinSpringMessageProducerFacts(parsed.filePath, []); setKotlinPackageFact(parsed.filePath, UNKNOWN_JVM_PACKAGE_FACT); return; } @@ -235,10 +306,22 @@ export function applyKotlinCaptureSideChannel(parsed: ParsedFile): void { parsed.filePath, Array.isArray(data.springDiFacts) ? data.springDiFacts : [], ); + setKotlinSpringDynamicLookupFacts( + parsed.filePath, + Array.isArray(data.springDynamicLookupFacts) ? data.springDynamicLookupFacts : [], + ); setKotlinSpringNonHttpHandlerFacts( parsed.filePath, Array.isArray(data.springNonHttpHandlerFacts) ? data.springNonHttpHandlerFacts : [], ); + setKotlinSpringConfigConsumerFacts( + parsed.filePath, + Array.isArray(data.springConfigConsumerFacts) ? data.springConfigConsumerFacts : [], + ); + setKotlinSpringMessageProducerFacts( + parsed.filePath, + Array.isArray(data.springMessageProducerFacts) ? data.springMessageProducerFacts : [], + ); setKotlinPackageFact( parsed.filePath, isJvmPackageFact(data.packageFact) ? data.packageFact : UNKNOWN_JVM_PACKAGE_FACT, diff --git a/gitnexus/src/core/ingestion/languages/kotlin/captures.ts b/gitnexus/src/core/ingestion/languages/kotlin/captures.ts index 84ec8dc4d..a24073324 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/captures.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/captures.ts @@ -23,11 +23,20 @@ import { setKotlinSpringAopFacts, setKotlinSpringConditionalFacts, setKotlinSpringDiFacts, + setKotlinSpringDynamicLookupFacts, + setKotlinSpringMessageProducerFacts, setKotlinSpringNonHttpHandlerFacts, + setKotlinSpringConfigConsumerFacts, } from './capture-side-channel.js'; import { captureKotlinPackageFact } from './package-facts.js'; import { synthesizeCallableFlowCaptures } from '../../utils/callable-flow-captures.js'; +import { synthesizeLombokAccessorCaptures } from './lombok-synthesizer.js'; import { captureKotlinSpringDiClassFact, type KotlinSpringDiClassFact } from './spring-di.js'; +import { captureKotlinSpringConfigConsumerFacts } from './spring-config-bindings.js'; +import type { SpringDynamicLookupFact } from '../../frameworks/spring/dynamic-lookups.js'; +import { captureKotlinSpringDynamicLookupFact } from './spring-dynamic-lookup.js'; +import type { SpringMessageProducerFact } from '../../frameworks/spring/message-producers.js'; +import { captureKotlinSpringMessageProducerFact } from './spring-message-producers.js'; import { synthesizeReceiverChainCapture } from '../../utils/receiver-chain-captures.js'; import { captureKotlinSpringAopFacts, type KotlinSpringAopFact } from './spring-aop.js'; import { @@ -107,6 +116,9 @@ export function emitKotlinScopeCaptures( const springNonHttpHandlerFacts: KotlinSpringNonHttpHandlerFact[] = []; const springNonHttpHandlerTypeNodeIds = new Set(); const springDiClassNodeIds = new Set(); + const springDynamicLookupFacts: SpringDynamicLookupFact[] = []; + const springMessageProducerFacts: SpringMessageProducerFact[] = []; + const springMemberCallNodeIds = new Set(); const returnTypes = collectKotlinReturnTypeTexts(tree.rootNode); out.push(...synthesizeKotlinLocalAssignmentBindings(tree.rootNode, returnTypes)); out.push(...synthesizeKotlinLoopBindings(tree.rootNode, returnTypes)); @@ -130,6 +142,17 @@ export function emitKotlinScopeCaptures( } if (Object.keys(grouped).length === 0) continue; + // One visit per member call node: the same invocation can back several + // query matches, and both Spring call-shape captures must see it once. + const memberCallNode = nodeIfType(groupedNodes['@reference.call.member'], 'call_expression'); + if (memberCallNode !== null && !springMemberCallNodeIds.has(memberCallNode.id)) { + springMemberCallNodeIds.add(memberCallNode.id); + const lookupFact = captureKotlinSpringDynamicLookupFact(memberCallNode, filePath); + if (lookupFact !== null) springDynamicLookupFacts.push(lookupFact); + const producerFact = captureKotlinSpringMessageProducerFact(memberCallNode, filePath); + if (producerFact !== null) springMessageProducerFacts.push(producerFact); + } + // tree-sitter-kotlin represents both classes and interfaces with // `class_declaration`; `object_declaration` is the separate object form. const springAopTypeNode = [ @@ -357,7 +380,14 @@ export function emitKotlinScopeCaptures( setKotlinSpringAopFacts(filePath, springAopFacts); setKotlinSpringConditionalFacts(filePath, springConditionalFacts); setKotlinSpringDiFacts(filePath, springDiFacts); + setKotlinSpringDynamicLookupFacts(filePath, springDynamicLookupFacts); setKotlinSpringNonHttpHandlerFacts(filePath, springNonHttpHandlerFacts); + setKotlinSpringConfigConsumerFacts( + filePath, + captureKotlinSpringConfigConsumerFacts(tree.rootNode, filePath), + ); + setKotlinSpringMessageProducerFacts(filePath, springMessageProducerFacts); + out.push(...synthesizeLombokAccessorCaptures(tree.rootNode)); out.push(...synthesizeCallableFlowCaptures(tree.rootNode, KOTLIN_CALLABLE_CAPTURE_OPTIONS)); return out; } diff --git a/gitnexus/src/core/ingestion/languages/kotlin/lombok-synthesizer.ts b/gitnexus/src/core/ingestion/languages/kotlin/lombok-synthesizer.ts new file mode 100644 index 000000000..bb0c3d5b7 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/kotlin/lombok-synthesizer.ts @@ -0,0 +1,512 @@ +/** + * Kotlin accessor synthesizer (same provider-hook role as Java Lombok). + * + * kotlinc emits JavaBeans getters/setters for `val`/`var` properties. Those + * methods are absent from the tree-sitter AST, so Java (and Kotlin) calls + * like `user.getName()` miss CALLS edges. Planning is Kotlin-specific; + * naming and Method emission share `jvm/beanspec` + `jvm/accessor-synthesis`. + * + * ## Supported subset (v1) + * - Class / data class / object / companion / interface `val`/`var` properties + * (interface accessors without a custom body are abstract JVM methods). + * - Primary-constructor `val`/`var` class parameters. + * - Names beginning with `is` + a non-lowercase character keep that getter name; all other + * properties, including `Boolean`, use `get`. + * - Custom `get()`/`set()` bodies still emit their JVM accessor Methods. + * - Explicit `fun getX` / `@JvmField` / `const` skip synthesis. + * - `@JvmName`-renamed accessors are suppressed until custom-name emission lands. + * Unsupported: `@JvmStatic` renaming, file-facade top-level properties. + */ +import type Parser from 'tree-sitter'; +import type { CaptureMatch } from 'gitnexus-shared'; +import { booleanIsPrefixBase, jvmGetterName, jvmSetterName } from '../jvm/beanspec.js'; +import { + createExistingMethodIndex, + createJvmAccessorSynthesis, + hasExistingMethod, + jvmTypeSimpleName, + rememberExistingMethod, + type ExistingMethodIndex, + type PlannedJvmAccessor, + type PlannedJvmAccessorOwner, + type SyntheticAccessorResult, + type SyntheticVisibility, +} from '../jvm/accessor-synthesis.js'; + +const KOTLIN_TYPE_DECLS = new Set(['class_declaration', 'object_declaration', 'companion_object']); + +function capitalizeAscii(name: string): string { + const first = name.charAt(0); + return first >= 'a' && first <= 'z' + ? String.fromCharCode(first.charCodeAt(0) - 32) + name.slice(1) + : name; +} + +export function kotlinGetterName(propertyName: string): string { + return jvmGetterName( + propertyName, + booleanIsPrefixBase(propertyName, true) !== null, + capitalizeAscii, + ); +} + +export function kotlinSetterName(propertyName: string): string { + return jvmSetterName(propertyName, true, capitalizeAscii); +} + +interface KtProperty { + name: string; + type: string; + isVar: boolean; + skipGetter: boolean; + skipSetter: boolean; + getterVisibility: SyntheticVisibility; + setterVisibility: SyntheticVisibility; + startLine: number; + endLine: number; + propertyNode: Parser.SyntaxNode; + declaratorNode: Parser.SyntaxNode; +} + +interface KtClass { + node: Parser.SyntaxNode; + name: string; + isStatic: boolean; + isInterface: boolean; + wasHoisted: boolean; + properties: KtProperty[]; + existingMethods: ExistingMethodIndex; +} + +interface KotlinImportIndex { + byLocalName: Map; + shadowedSimpleNames: Set; +} + +function collectKotlinImports(root: Parser.SyntaxNode): KotlinImportIndex { + const byLocalName = new Map(); + const shadowedSimpleNames = new Set(); + for (const child of root.children) { + if (child.type !== 'class_declaration') continue; + const name = jvmTypeSimpleName(child); + if (name) shadowedSimpleNames.add(name); + } + const importList = root.children.find((child) => child.type === 'import_list'); + for (const child of importList?.children ?? []) { + if (child.type !== 'import_header') continue; + const text = child.text + .replace(/^import\s+/, '') + .replace(/\/\*[\s\S]*?\*\//g, '') + .trim(); + const [pathText, aliasText] = text.split(/\s+as\s+/, 2); + const importPath = pathText?.replace(/\s+/g, ''); + if (!importPath || importPath.endsWith('.*')) continue; + const localName = aliasText?.trim() || importPath.split('.').pop(); + if (localName) byLocalName.set(localName, importPath); + } + return { byLocalName, shadowedSimpleNames }; +} + +function annotationUserTypeText(annotation: Parser.SyntaxNode): string { + const constructor = annotation.namedChildren.find((c) => c.type === 'constructor_invocation'); + const userType = + constructor?.namedChildren.find((c) => c.type === 'user_type') ?? + annotation.namedChildren.find((c) => c.type === 'user_type'); + return userType?.text ?? ''; +} + +function isKotlinJvmAnnotation( + annotation: Parser.SyntaxNode, + name: string, + imports: KotlinImportIndex, +): boolean { + const typeText = annotationUserTypeText(annotation); + const canonical = `kotlin.jvm.${name}`; + if (typeText.includes('.')) return typeText === canonical; + const imported = imports.byLocalName.get(typeText); + if (imported !== undefined) return imported === canonical; + if (imports.shadowedSimpleNames.has(typeText)) return false; + return typeText === name; +} + +function kotlinVisibility(modifiers: Parser.SyntaxNode | undefined): SyntheticVisibility { + if (!modifiers) return 'public'; + for (const child of modifiers.namedChildren) { + if (child.type !== 'visibility_modifier') continue; + if (child.text === 'private') return 'private'; + if (child.text === 'protected') return 'protected'; + if (child.text === 'internal') return 'package'; + } + return 'public'; +} + +function hasJvmField(node: Parser.SyntaxNode, imports: KotlinImportIndex): boolean { + const mods = node.children.find((c) => c.type === 'modifiers'); + return ( + mods?.namedChildren.some( + (child) => child.type === 'annotation' && isKotlinJvmAnnotation(child, 'JvmField', imports), + ) === true + ); +} + +function hasConst(node: Parser.SyntaxNode): boolean { + const mods = node.children.find((c) => c.type === 'modifiers'); + if ( + mods?.namedChildren.some( + (child) => child.type === 'property_modifier' && child.text === 'const', + ) + ) { + return true; + } + return node.namedChildren.some((child) => child.type === 'const'); +} + +function isVarBinding(node: Parser.SyntaxNode): boolean | null { + const kind = node.children.find((c) => c.type === 'binding_pattern_kind'); + const text = kind?.text; + if (text === 'var') return true; + if (text === 'val') return false; + return null; +} + +function inferredInitializerType(node: Parser.SyntaxNode): string | undefined { + switch (node.type) { + case 'string_literal': + case 'line_string_literal': + case 'multi_line_string_literal': + return 'String'; + case 'character_literal': + return 'Char'; + case 'boolean_literal': + case 'true': + case 'false': + return 'Boolean'; + case 'long_literal': + return 'Long'; + case 'unsigned_literal': + return /l$/i.test(node.text) ? 'ULong' : 'UInt'; + case 'integer_literal': + case 'decimal_integer_literal': + case 'hex_integer_literal': + case 'octal_integer_literal': + case 'binary_integer_literal': + return 'Int'; + case 'real_literal': + case 'decimal_floating_point_literal': + return /f$/i.test(node.text) ? 'Float' : 'Double'; + case 'prefix_expression': { + const operand = node.namedChildren.at(-1); + return operand ? inferredInitializerType(operand) : undefined; + } + case 'call_expression': { + const callee = node.namedChildren.find((child) => child.type === 'simple_identifier'); + if (!callee) return undefined; + const first = callee.text.charAt(0); + return first !== '' && first === first.toUpperCase() ? callee.text : undefined; + } + default: + return undefined; + } +} + +function propertyTypeText(node: Parser.SyntaxNode): string { + const declarator = + node.type === 'class_parameter' + ? node + : (node.children.find((c) => c.type === 'variable_declaration') ?? node); + const colon = declarator.children.find((c) => c.type === ':'); + let typeNode = colon?.nextNamedSibling ?? null; + while (typeNode?.type === 'type_modifiers') typeNode = typeNode.nextNamedSibling; + if (typeNode) return typeNode.text; + const initializer = node.namedChildren.find( + (child) => + child.id !== declarator.id && + child.type !== 'binding_pattern_kind' && + child.type !== 'modifiers', + ); + return initializer ? (inferredInitializerType(initializer) ?? 'unknown') : 'unknown'; +} + +function propertyNameNode(node: Parser.SyntaxNode): Parser.SyntaxNode | null { + if (node.type === 'class_parameter') { + return node.children.find((c) => c.type === 'simple_identifier') ?? null; + } + const decl = node.children.find((c) => c.type === 'variable_declaration'); + if (decl) { + return decl.children.find((c) => c.type === 'simple_identifier') ?? null; + } + return node.children.find((c) => c.type === 'simple_identifier') ?? null; +} + +function accessorMetadata( + prop: Parser.SyntaxNode, + propertyVisibility: SyntheticVisibility, + imports: KotlinImportIndex, +): { + getterVisibility: SyntheticVisibility; + setterVisibility: SyntheticVisibility; + skipGetter: boolean; + skipSetter: boolean; +} { + let getter = propertyVisibility; + let setter = propertyVisibility; + let skipGetter = false; + let skipSetter = false; + const propertyModifiers = prop.children.find((c) => c.type === 'modifiers'); + for (const annotation of propertyModifiers?.namedChildren ?? []) { + if ( + annotation.type !== 'annotation' || + !isKotlinJvmAnnotation(annotation, 'JvmName', imports) + ) { + continue; + } + const target = annotation.children.find((c) => c.type === 'use_site_target')?.text; + if (target === 'get:') skipGetter = true; + if (target === 'set:') skipSetter = true; + } + const apply = (node: Parser.SyntaxNode): void => { + const modifiers = node.children.find((c) => c.type === 'modifiers'); + if (!modifiers) return; + if (node.type === 'getter') getter = kotlinVisibility(modifiers); + if (node.type === 'setter') setter = kotlinVisibility(modifiers); + if ( + modifiers.namedChildren.some((annotation) => + isKotlinJvmAnnotation(annotation, 'JvmName', imports), + ) + ) { + if (node.type === 'getter') skipGetter = true; + if (node.type === 'setter') skipSetter = true; + } + }; + for (const child of prop.children) { + if (child.type === 'getter' || child.type === 'setter') apply(child); + } + let sib: Parser.SyntaxNode | null = prop.nextNamedSibling; + while (sib && (sib.type === 'getter' || sib.type === 'setter')) { + apply(sib); + sib = sib.nextNamedSibling; + } + return { + getterVisibility: getter, + setterVisibility: setter, + skipGetter, + skipSetter, + }; +} + +function hasKotlinAccessorBody(prop: Parser.SyntaxNode, kind: 'getter' | 'setter'): boolean { + const hasBody = (node: Parser.SyntaxNode): boolean => + node.type === kind && node.children.some((child) => child.type === 'function_body'); + if (prop.children.some(hasBody)) return true; + let sib: Parser.SyntaxNode | null = prop.nextNamedSibling; + while (sib && (sib.type === 'getter' || sib.type === 'setter')) { + if (hasBody(sib)) return true; + sib = sib.nextNamedSibling; + } + return false; +} + +function functionName(node: Parser.SyntaxNode): string | undefined { + return node.children.find((c) => c.type === 'simple_identifier')?.text; +} + +function functionArity(node: Parser.SyntaxNode): number { + const params = node.children.find((c) => c.type === 'function_value_parameters'); + let arity = + node.childForFieldName('receiver') !== null || + node.namedChildren.some((child) => child.type === 'receiver_type') + ? 1 + : 0; + const modifiers = node.children.find((child) => child.type === 'modifiers'); + if ( + modifiers?.namedChildren.some( + (child) => child.type === 'function_modifier' && child.text === 'suspend', + ) + ) { + arity += 1; + } + for (const child of params?.namedChildren ?? []) { + if (child.type === 'parameter' || child.type === 'parameter_with_optional_type') arity += 1; + } + return arity; +} + +function collectExistingMethods(...bodies: Array): ExistingMethodIndex { + const index = createExistingMethodIndex('exact'); + for (const body of bodies) { + if (!body) continue; + for (const child of body.children) { + if (child.type !== 'function_declaration') continue; + const name = functionName(child); + if (!name) continue; + rememberExistingMethod(index, name, functionArity(child)); + } + } + return index; +} + +function toKtProperty(child: Parser.SyntaxNode, imports: KotlinImportIndex): KtProperty | null { + const isVar = isVarBinding(child); + if (isVar === null) return null; + if (hasJvmField(child, imports) || hasConst(child)) return null; + const nameNode = propertyNameNode(child); + if (!nameNode) return null; + const mods = child.children.find((c) => c.type === 'modifiers'); + const visibility = kotlinVisibility(mods); + const accessor = accessorMetadata(child, visibility, imports); + return { + name: nameNode.text, + type: propertyTypeText(child), + isVar, + skipGetter: accessor.skipGetter, + skipSetter: accessor.skipSetter, + getterVisibility: accessor.getterVisibility, + setterVisibility: accessor.setterVisibility, + startLine: child.startPosition.row + 1, + endLine: child.endPosition.row + 1, + propertyNode: child, + declaratorNode: nameNode, + }; +} + +function collectTypedProperties( + parent: Parser.SyntaxNode | null, + type: 'class_parameter' | 'property_declaration', + imports: KotlinImportIndex, +): KtProperty[] { + if (!parent) return []; + const out: KtProperty[] = []; + for (const child of parent.namedChildren) { + if (child.type !== type) continue; + const prop = toKtProperty(child, imports); + if (prop) out.push(prop); + } + return out; +} + +function findKtClasses(root: Parser.SyntaxNode, imports: KotlinImportIndex): KtClass[] { + const classes: KtClass[] = []; + const graphOwnerNode = (node: Parser.SyntaxNode): Parser.SyntaxNode => { + if (node.type !== 'companion_object') return node; + if (jvmTypeSimpleName(node)) return node; + let current = node.parent; + while (current && !KOTLIN_TYPE_DECLS.has(current.type)) current = current.parent; + return current ?? node; + }; + const walk = (node: Parser.SyntaxNode): void => { + if (KOTLIN_TYPE_DECLS.has(node.type)) { + const ownerNode = graphOwnerNode(node); + const name = jvmTypeSimpleName(ownerNode) ?? ''; + const ctor = node.children.find((c) => c.type === 'primary_constructor') ?? null; + const body = node.children.find((c) => c.type === 'class_body') ?? null; + if (name) { + const properties = [ + ...collectTypedProperties(ctor, 'class_parameter', imports), + ...collectTypedProperties(body, 'property_declaration', imports), + ]; + if (properties.length > 0) { + const ownerBody = + ownerNode.id === node.id + ? null + : (ownerNode.children.find((child) => child.type === 'class_body') ?? null); + classes.push({ + node: ownerNode, + name, + isStatic: node.type === 'companion_object', + isInterface: node.children.some((child) => child.type === 'interface'), + wasHoisted: ownerNode.id !== node.id, + properties, + existingMethods: collectExistingMethods(body, ownerBody), + }); + } + } + if (body) { + for (const child of body.namedChildren) { + if (KOTLIN_TYPE_DECLS.has(child.type)) walk(child); + } + } + return; + } + for (const child of node.namedChildren) walk(child); + }; + walk(root); + return classes; +} + +function planAccessors(cls: KtClass): PlannedJvmAccessor[] { + const planned: PlannedJvmAccessor[] = []; + for (const prop of cls.properties) { + const gName = kotlinGetterName(prop.name); + if (!prop.skipGetter && !hasExistingMethod(cls.existingMethods, gName, 0)) { + planned.push({ + kind: 'getter', + name: gName, + returnType: prop.type, + parameterTypes: [], + visibility: prop.getterVisibility, + isStatic: cls.isStatic, + isAbstract: cls.isInterface && !hasKotlinAccessorBody(prop.propertyNode, 'getter'), + startLine: prop.startLine, + endLine: prop.endLine, + declaratorNode: prop.declaratorNode, + }); + } + if (prop.isVar && !prop.skipSetter) { + const sName = kotlinSetterName(prop.name); + if (!hasExistingMethod(cls.existingMethods, sName, 1)) { + planned.push({ + kind: 'setter', + name: sName, + returnType: 'void', + parameterTypes: [prop.type], + visibility: prop.setterVisibility, + isStatic: cls.isStatic, + isAbstract: cls.isInterface && !hasKotlinAccessorBody(prop.propertyNode, 'setter'), + startLine: prop.startLine, + endLine: prop.endLine, + declaratorNode: prop.declaratorNode, + }); + } + } + } + return planned; +} + +function planKotlinAccessorOwners(rootNode: Parser.SyntaxNode): PlannedJvmAccessorOwner[] { + const owners: PlannedJvmAccessorOwner[] = []; + const imports = collectKotlinImports(rootNode); + for (const cls of findKtClasses(rootNode, imports)) { + const accessors = planAccessors(cls); + const existingIndex = cls.wasHoisted + ? owners.findIndex((owner) => owner.node.id === cls.node.id) + : -1; + const existing = existingIndex >= 0 ? owners[existingIndex] : undefined; + if (existing) { + owners[existingIndex] = { + ...existing, + accessors: [...existing.accessors, ...accessors], + }; + } else { + owners.push({ node: cls.node, name: cls.name, accessors }); + } + } + return owners; +} + +const lombokAccessorSynthesis = createJvmAccessorSynthesis({ + language: 'kotlin', + synthetic: 'kotlin-jvm', + planOwners: planKotlinAccessorOwners, +}); + +export function synthesizeLombokAccessors( + tree: Parser.Tree, + filePath: string, + classOwnersById: ReadonlyMap, +): SyntheticAccessorResult { + return lombokAccessorSynthesis.synthesize(tree, filePath, classOwnersById); +} + +export function synthesizeLombokAccessorCaptures(rootNode: Parser.SyntaxNode): CaptureMatch[] { + return lombokAccessorSynthesis.captures(rootNode); +} diff --git a/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts index 0fffeb60a..983f181b8 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts @@ -27,6 +27,8 @@ import { clearKotlinPackageFacts } from './package-facts.js'; import { attachKotlinSpringDiMetadata } from './spring-di.js'; import { attachKotlinSpringConditionalMetadata } from './spring-conditionals.js'; import { attachKotlinSpringNonHttpHandlerMetadata } from './spring-non-http-handlers.js'; +import { attachKotlinSpringConfigBindings } from './spring-config-bindings.js'; +import { attachKotlinSpringDynamicLookup } from './spring-dynamic-lookup.js'; /** * Kotlin scope resolver for RFC #909 Ring 3. @@ -148,6 +150,8 @@ export const kotlinScopeResolver: ScopeResolver = { attachKotlinSpringConditionalMetadata(graph, parsedFiles, nodeLookup, indexes); attachKotlinSpringDiMetadata(graph, parsedFiles, nodeLookup, indexes); attachKotlinSpringNonHttpHandlerMetadata(graph, parsedFiles, nodeLookup, indexes); + attachKotlinSpringDynamicLookup(graph, parsedFiles, nodeLookup, indexes); + attachKotlinSpringConfigBindings(graph, parsedFiles, nodeLookup, indexes); }, }; diff --git a/gitnexus/src/core/ingestion/languages/kotlin/spring-actuator.ts b/gitnexus/src/core/ingestion/languages/kotlin/spring-actuator.ts new file mode 100644 index 000000000..5acb118b8 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/kotlin/spring-actuator.ts @@ -0,0 +1,229 @@ +import path from 'node:path'; +import type { GraphNode, ParsedImport } from 'gitnexus-shared'; +import type { + DefinitionPropertiesContext, + RuntimeCallableIdentity, + RuntimeSymbolStrategy, +} from '../../language-provider.js'; +import type { SyntaxNode } from '../../utils/ast-helpers.js'; + +const RUNTIME_OWNER_ALIASES = 'runtimeOwnerAliases'; +const RUNTIME_CALLABLE_ALIASES = 'runtimeCallableAliases'; +const KOTLIN_SUSPEND = 'kotlinSuspend'; +const fileFacadeMetadataCache = new WeakMap< + SyntaxNode, + { readonly packageName: string; readonly customFacade: string | undefined } +>(); + +function rootNode(node: SyntaxNode): SyntaxNode { + let current = node; + while (current.parent) current = current.parent; + return current; +} + +function packageName(root: SyntaxNode): string { + const header = root.namedChildren.find((child) => child.type === 'package_header'); + return header?.text.replace(/^package\s+/, '').trim() ?? ''; +} + +function qualify(packageNameValue: string, simpleName: string): string { + return packageNameValue.length === 0 ? simpleName : `${packageNameValue}.${simpleName}`; +} + +function standardFacadeName(filePath: string): string { + const stem = path.basename(filePath).replace(/\.(?:kt|kts)$/i, ''); + return `${stem.charAt(0).toUpperCase()}${stem.slice(1)}Kt`; +} + +function jvmNameIdentifiers( + imports: readonly ParsedImport[], + allowUnqualified: boolean, +): readonly string[] { + const names = new Set(['kotlin.jvm.JvmName']); + if (allowUnqualified) names.add('JvmName'); + for (const parsedImport of imports) { + if (parsedImport.kind !== 'named' && parsedImport.kind !== 'alias') continue; + if (parsedImport.importedName !== 'JvmName') continue; + const target = parsedImport.targetRaw.replace(/\\/g, '/'); + if (target === 'kotlin.jvm' || target === 'kotlin.jvm.JvmName') { + names.add(parsedImport.localName); + } + } + return [...names]; +} + +function annotationJvmName( + source: string, + target = '', + imports: readonly ParsedImport[] = [], + allowUnqualified = true, +): string | undefined { + const names = jvmNameIdentifiers(imports, allowUnqualified); + if (names.length === 0) return undefined; + const escapedTarget = target.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); + const prefix = target.length === 0 ? '' : `${escapedTarget}:`; + const namePattern = [...names] + .map((name) => name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')) + .join('|'); + return new RegExp(`@${prefix}(?:${namePattern})\\s*\\(\\s*["']([^"']+)["']\\s*\\)`).exec( + source, + )?.[1]; +} + +function fileFacadeMetadata( + root: SyntaxNode, + imports: readonly ParsedImport[], + allowUnqualified: boolean, +): { + readonly packageName: string; + readonly customFacade: string | undefined; +} { + const cached = fileFacadeMetadataCache.get(root); + if (cached !== undefined) return cached; + const metadata = { + packageName: packageName(root), + customFacade: annotationJvmName(root.text, 'file', imports, allowUnqualified), + }; + fileFacadeMetadataCache.set(root, metadata); + return metadata; +} + +function hasEnclosingType(node: SyntaxNode): boolean { + let current = node.parent; + while (current) { + if ( + current.type === 'class_declaration' || + current.type === 'object_declaration' || + current.type === 'companion_object' + ) { + return true; + } + current = current.parent; + } + return false; +} + +/** Transient graph metadata used only by the same analysis run's runtime import. */ +export function extractKotlinRuntimeSymbolProperties( + context: DefinitionPropertiesContext, +): Readonly> | undefined { + const properties: Record = {}; + const source = context.definitionNode.text; + const root = rootNode(context.definitionNode); + const allowUnqualifiedJvmName = !/\bannotation\s+class\s+JvmName\b/.test(root.text); + + if ( + (context.nodeLabel === 'Function' || context.nodeLabel === 'Method') && + context.definitionNode.type === 'function_declaration' + ) { + if (/\bsuspend\b/.test(source.slice(0, source.indexOf('fun') + 3))) { + properties[KOTLIN_SUSPEND] = true; + } + const callableJvmName = annotationJvmName( + source, + '', + context.parsedImports, + allowUnqualifiedJvmName, + ); + if (callableJvmName !== undefined) { + properties[RUNTIME_CALLABLE_ALIASES] = [callableJvmName]; + } + if (!hasEnclosingType(context.definitionNode)) { + const facade = fileFacadeMetadata(root, context.parsedImports, allowUnqualifiedJvmName); + properties[RUNTIME_OWNER_ALIASES] = [ + qualify(facade.packageName, facade.customFacade ?? standardFacadeName(context.filePath)), + ]; + } + } else if (context.nodeLabel === 'Property') { + const getterJvmName = annotationJvmName( + source, + 'get', + context.parsedImports, + allowUnqualifiedJvmName, + ); + if (getterJvmName !== undefined) { + properties[RUNTIME_CALLABLE_ALIASES] = [getterJvmName]; + } + if (!hasEnclosingType(context.definitionNode)) { + const facade = fileFacadeMetadata(root, context.parsedImports, allowUnqualifiedJvmName); + properties[RUNTIME_OWNER_ALIASES] = [ + qualify(facade.packageName, facade.customFacade ?? standardFacadeName(context.filePath)), + ]; + } + } + + return Object.keys(properties).length === 0 ? undefined : properties; +} + +function stringArrayProperty(node: GraphNode, property: string): readonly string[] { + const value = node.properties[property]; + return Array.isArray(value) + ? value.filter((item): item is string => typeof item === 'string') + : []; +} + +function callableNames(node: GraphNode): readonly string[] { + return [String(node.properties.name), ...stringArrayProperty(node, RUNTIME_CALLABLE_ALIASES)]; +} + +function propertyGetterNames(node: GraphNode): readonly string[] { + const name = String(node.properties.name); + const capitalized = `${name.charAt(0).toUpperCase()}${name.slice(1)}`; + return [ + name.startsWith('is') && name.length > 2 && /[A-Z]/.test(name.charAt(2)) + ? name + : `get${capitalized}`, + ...stringArrayProperty(node, RUNTIME_CALLABLE_ALIASES), + ]; +} + +function sourceCallableName(runtimeName: string): string { + return runtimeName.endsWith('$default') ? runtimeName.slice(0, -'$default'.length) : runtimeName; +} + +function matchesKotlinCallable(node: GraphNode, runtime: RuntimeCallableIdentity): boolean { + // Kotlin property declarations and their synthesized JVM accessor Methods + // coexist in the graph. Bind runtime getters to the source Property so the + // synthetic accessor cannot turn an otherwise exact match into ambiguity. + if (node.properties.synthetic === 'kotlin-jvm') return false; + + const runtimeName = sourceCallableName(runtime.name); + if (node.label === 'Property') { + const names = propertyGetterNames(node); + if (!names.includes(runtime.name) && !names.includes(runtimeName)) return false; + } else if (!callableNames(node).includes(runtimeName)) { + return false; + } + + const parameterCount = node.properties.parameterCount; + const descriptorTypes = runtime.descriptorParameterTypes; + if ( + typeof parameterCount !== 'number' || + descriptorTypes === undefined || + runtime.name.endsWith('$default') + ) { + return true; + } + if (parameterCount === descriptorTypes.length) return true; + return ( + node.properties[KOTLIN_SUSPEND] === true && + parameterCount + 1 === descriptorTypes.length && + descriptorTypes.at(-1) === 'kotlin/coroutines/Continuation' + ); +} + +export const kotlinRuntimeSymbolStrategy: RuntimeSymbolStrategy = { + callableOwnerAliases(node, owner) { + const aliases = [...stringArrayProperty(node, RUNTIME_OWNER_ALIASES)]; + const ownerName = owner?.properties.qualifiedName; + if (typeof ownerName === 'string') { + aliases.push(ownerName); + if (node.properties.isStatic === true && !ownerName.endsWith('.Companion')) { + aliases.push(`${ownerName}.Companion`); + } + } + return aliases; + }, + + matchesCallable: matchesKotlinCallable, +}; diff --git a/gitnexus/src/core/ingestion/languages/kotlin/spring-config-bindings.ts b/gitnexus/src/core/ingestion/languages/kotlin/spring-config-bindings.ts new file mode 100644 index 000000000..91aa3f9c0 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/kotlin/spring-config-bindings.ts @@ -0,0 +1,467 @@ +import type { KnowledgeGraph } from '../../../graph/types.js'; +import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js'; +import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; +import { makeScopeId, type ParsedFile, type ScopeId } from 'gitnexus-shared'; +import { + bindSpringConfigConsumers, + type SpringConfigConsumer, +} from '../../frameworks/spring/config-bindings.js'; +import { createSpringAnnotationNameResolver } from '../../frameworks/spring/bean-candidates.js'; +import { + parseSpringAnnotationArguments, + parseStaticStringLiteral, +} from '../../frameworks/spring/annotation-arguments.js'; +import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js'; +import { nodeToCapture, type SyntaxNode } from '../../utils/ast-helpers.js'; +import { getKotlinParser } from './query.js'; +import { getKotlinSpringConfigConsumerFacts } from './capture-side-channel.js'; +import { isKotlinPackageSiblingVisibilityIncomplete } from './package-siblings.js'; + +const VALUE_ANNOTATION = 'org.springframework.beans.factory.annotation.Value'; +const CONFIGURATION_PROPERTIES_ANNOTATION = + 'org.springframework.boot.context.properties.ConfigurationProperties'; + +const VALUE_SIMPLE = 'Value'; +const CONFIGURATION_PROPERTIES_SIMPLE = 'ConfigurationProperties'; +const SKIP_USE_SITES = new Set(['get', 'property', 'file']); +const BIND_USE_SITES = new Set(['field', 'set', 'param']); +const OWNER_TYPES = new Set(['class_declaration', 'object_declaration', 'companion_object']); +const INTERPOLATION_TYPES = new Set([ + 'interpolated_identifier', + 'interpolated_expression', + 'interpolation_expression_start', +]); +const STRING_LITERAL_TYPES = new Set(['string_literal', 'character_literal']); + +export interface KotlinSpringConfigConsumerFact { + readonly consumer: SpringConfigConsumer; + readonly annotationName: string; + readonly classScopeId: ScopeId; +} + +interface KotlinAnnotation { + readonly name: string; + readonly node: SyntaxNode; + readonly useSiteTarget?: string; +} + +interface KotlinImports { + readonly exact: ReadonlyMap; + readonly wildcard: ReadonlySet; + readonly localTypes: ReadonlyMap; +} + +function firstDescendantOfType(node: SyntaxNode, type: string): SyntaxNode | undefined { + const stack = [...node.namedChildren].reverse(); + while (stack.length > 0) { + const current = stack.pop(); + if (current === undefined) continue; + if (current.type === type) return current; + for (let index = current.namedChildren.length - 1; index >= 0; index--) { + const child = current.namedChildren[index]; + if (child !== undefined) stack.push(child); + } + } + return undefined; +} + +function ownerName(declaration: SyntaxNode): string | undefined { + if (declaration.type === 'companion_object') { + const named = declaration.namedChildren.find((child) => child.type === 'type_identifier'); + return named?.text.trim() || 'Companion'; + } + return ( + declaration.namedChildren.find((child) => child.type === 'type_identifier')?.text.trim() ?? + declaration.namedChildren.find((child) => child.type === 'simple_identifier')?.text.trim() + ); +} + +function enclosingOwner(node: SyntaxNode): SyntaxNode | undefined { + let current = node.parent; + while (current !== null) { + if (OWNER_TYPES.has(current.type)) return current; + current = current.parent; + } + return undefined; +} + +function classScopeId(filePath: string, declaration: SyntaxNode): ScopeId { + return makeScopeId({ + filePath, + range: nodeToCapture('@scope.class', declaration).range, + kind: 'Class', + }); +} + +function collectKotlinImports(root: SyntaxNode): KotlinImports { + const exact = new Map(); + const wildcard = new Set(); + const localTypes = new Map(); + + for (const header of root.descendantsOfType('import_header')) { + const text = header.text.replace(/^import\s+/, '').trim(); + const aliasMatch = text.match(/^([\w.]+)\s+as\s+(\w+)\s*$/); + if (aliasMatch !== null) { + exact.set(aliasMatch[2], aliasMatch[1]); + continue; + } + if (text.endsWith('.*')) wildcard.add(text.slice(0, -2)); + else { + const simple = text.slice(text.lastIndexOf('.') + 1); + if (simple.length > 0) exact.set(simple, text); + } + } + + for (const type of ['class_declaration', 'object_declaration']) { + for (const declaration of root.descendantsOfType(type)) { + const name = ownerName(declaration); + if (name) { + const declarations = localTypes.get(name) ?? []; + declarations.push(declaration); + localTypes.set(name, declarations); + } + } + } + return { exact, wildcard, localTypes }; +} + +function annotationFromNode(annotation: SyntaxNode): KotlinAnnotation | null { + const nameNode = + firstDescendantOfType(annotation, 'user_type') ?? + firstDescendantOfType(annotation, 'type_identifier') ?? + firstDescendantOfType(annotation, 'simple_identifier'); + if (nameNode === undefined) return null; + const useSiteTarget = annotation.namedChildren + .find((child) => child.type === 'use_site_target') + ?.text.replace(/:\s*$/, '') + .trim(); + return { + name: nameNode.text.trim(), + node: annotation, + ...(useSiteTarget === undefined || useSiteTarget.length === 0 ? {} : { useSiteTarget }), + }; +} + +function annotationsOn(node: SyntaxNode): KotlinAnnotation[] { + const annotations: KotlinAnnotation[] = []; + for (const child of node.namedChildren) { + if (child.type === 'annotation') { + const fact = annotationFromNode(child); + if (fact !== null) annotations.push(fact); + continue; + } + if (child.type !== 'modifiers' && child.type !== 'parameter_modifiers') continue; + for (const nested of child.namedChildren) { + if (nested.type !== 'annotation') continue; + const fact = annotationFromNode(nested); + if (fact !== null) annotations.push(fact); + } + } + return annotations; +} + +function simpleName(rawName: string): string { + const parts = rawName.split('.'); + return parts[parts.length - 1] ?? rawName; +} + +function importedAs( + imports: KotlinImports, + simple: string, + fqn: string, + wildcardPackage: string, +): boolean { + if (imports.exact.get(simple) === fqn) return true; + // An explicit import wins over a star import in Kotlin, so a conflicting + // binding for the same simple name rules the Spring annotation out even when + // its package is wildcard-imported. + return imports.exact.get(simple) === undefined && imports.wildcard.has(wildcardPackage); +} + +function hasVisibleLocalType( + imports: KotlinImports, + simple: string, + annotation: SyntaxNode, +): boolean { + for (const declaration of imports.localTypes.get(simple) ?? []) { + const declarationOwner = enclosingOwner(declaration); + if (declarationOwner === undefined) return true; + let current: SyntaxNode | null = annotation; + while (current !== null) { + if (current.id === declarationOwner.id) return true; + current = current.parent; + } + } + return false; +} + +const SIMPLE_CONFIG_ANNOTATIONS = [ + { + simple: VALUE_SIMPLE, + kind: 'value', + fqn: VALUE_ANNOTATION, + wildcardPackage: 'org.springframework.beans.factory.annotation', + }, + { + simple: CONFIGURATION_PROPERTIES_SIMPLE, + kind: 'configuration-properties', + fqn: CONFIGURATION_PROPERTIES_ANNOTATION, + wildcardPackage: 'org.springframework.boot.context.properties', + }, +] as const; + +function configAnnotationKind( + annotation: KotlinAnnotation, + imports: KotlinImports, +): 'value' | 'configuration-properties' | null { + const rawName = annotation.name; + if (rawName === VALUE_ANNOTATION) return 'value'; + if (rawName === CONFIGURATION_PROPERTIES_ANNOTATION) return 'configuration-properties'; + const simple = simpleName(rawName); + const aliased = imports.exact.get(simple); + if (aliased === VALUE_ANNOTATION) return 'value'; + if (aliased === CONFIGURATION_PROPERTIES_ANNOTATION) return 'configuration-properties'; + for (const candidate of SIMPLE_CONFIG_ANNOTATIONS) { + if (simple !== candidate.simple) continue; + if (hasVisibleLocalType(imports, simple, annotation.node) && !imports.exact.has(simple)) { + return null; + } + return importedAs(imports, simple, candidate.fqn, candidate.wildcardPackage) + ? candidate.kind + : null; + } + return null; +} + +function hasInterpolation(annotation: SyntaxNode): boolean { + const stack: SyntaxNode[] = [...annotation.namedChildren]; + while (stack.length > 0) { + const current = stack.pop(); + if (current === undefined) continue; + if (INTERPOLATION_TYPES.has(current.type)) return true; + stack.push(...current.namedChildren); + } + return false; +} + +function decodeKotlinStringLiteral(literal: string): string | null { + const raw = literal.startsWith('"""') && literal.endsWith('"""'); + const delimiterLength = raw ? 3 : 1; + if (literal.length < delimiterLength * 2) return null; + const body = literal.slice(delimiterLength, -delimiterLength); + if (!raw && /(? + String.fromCharCode(Number.parseInt(hex, 16)), + ) + .replace(/\\(["'\\$btnfr])/g, (_match, escaped: string) => { + const controls: Record = { + b: '\b', + t: '\t', + n: '\n', + f: '\f', + r: '\r', + $: '$', + }; + return controls[escaped] ?? escaped; + }); +} + +function kotlinStringLiterals(annotation: SyntaxNode): string[] { + const literals: string[] = []; + const stack: SyntaxNode[] = [...annotation.namedChildren]; + while (stack.length > 0) { + const current = stack.pop(); + if (current === undefined) continue; + if (STRING_LITERAL_TYPES.has(current.type)) { + const decoded = decodeKotlinStringLiteral(current.text); + if (decoded !== null) literals.push(decoded); + continue; + } + stack.push(...current.namedChildren); + } + return literals; +} + +function parseValuePlaceholderKeys(annotation: SyntaxNode): string[] { + if (hasInterpolation(annotation)) return []; + const keys = new Set(); + for (const literal of kotlinStringLiterals(annotation)) { + for (const match of literal.matchAll(/\$\{([^{}]+)\}/g)) { + const key = match[1].split(':', 1)[0].trim(); + if (/^[A-Za-z0-9_.-]+$/.test(key)) keys.add(key); + } + } + return [...keys]; +} + +function parseConfigurationPropertiesPrefix(annotation: SyntaxNode): string | null { + if (hasInterpolation(annotation)) return null; + const argumentsList = parseSpringAnnotationArguments(annotation.text); + if (argumentsList !== null) { + const named = argumentsList.filter( + (argument) => argument.name === 'prefix' || argument.name === 'value', + ); + const positional = argumentsList.filter((argument) => argument.name === undefined); + const chosen = named.length === 1 ? named[0] : named.length === 0 ? positional[0] : undefined; + if (chosen !== undefined) { + const decoded = parseStaticStringLiteral(chosen.value); + if (decoded === null) return null; + const prefix = decoded.replace(/^\.+|\.+$/g, ''); + if (/^[A-Za-z0-9_.-]+$/.test(prefix)) return prefix; + return null; + } + if (argumentsList.length > 0) return null; + } + const literals = kotlinStringLiterals(annotation); + if (literals.length !== 1) return null; + const prefix = literals[0].trim().replace(/^\.+|\.+$/g, ''); + return /^[A-Za-z0-9_.-]+$/.test(prefix) ? prefix : null; +} + +function allowedUseSite(useSiteTarget: string | undefined): boolean { + if (useSiteTarget === undefined) return true; + if (SKIP_USE_SITES.has(useSiteTarget)) return false; + return BIND_USE_SITES.has(useSiteTarget); +} + +function hasBindingPattern(parameter: SyntaxNode): boolean { + return parameter.namedChildren.some((child) => child.type === 'binding_pattern_kind'); +} + +function propertyName(node: SyntaxNode): string | undefined { + if (node.type === 'class_parameter') { + return node.namedChildren.find((child) => child.type === 'simple_identifier')?.text.trim(); + } + const variable = node.namedChildren.find((child) => child.type === 'variable_declaration'); + return variable?.namedChildren.find((child) => child.type === 'simple_identifier')?.text.trim(); +} + +function underFileAnnotation(node: SyntaxNode): boolean { + let current: SyntaxNode | null = node; + while (current !== null) { + if (current.type === 'file_annotation') return true; + current = current.parent; + } + return false; +} + +function pushValueFacts( + facts: KotlinSpringConfigConsumerFact[], + member: SyntaxNode, + filePath: string, + imports: KotlinImports, +): void { + if (underFileAnnotation(member)) return; + const owner = enclosingOwner(member); + if (owner === undefined) return; + const fieldName = propertyName(member); + if (fieldName === undefined) return; + for (const annotation of annotationsOn(member)) { + if (!allowedUseSite(annotation.useSiteTarget)) continue; + if (configAnnotationKind(annotation, imports) !== 'value') continue; + const keys = parseValuePlaceholderKeys(annotation.node); + if (keys.length === 0) continue; + facts.push({ + consumer: { + kind: 'value', + fieldName, + line: member.startPosition.row + 1, + keys, + }, + annotationName: annotation.name, + classScopeId: classScopeId(filePath, owner), + }); + } +} + +/** Collect config facts from the Kotlin parser's existing AST (no reparse). */ +export function captureKotlinSpringConfigConsumerFacts( + root: SyntaxNode, + filePath: string, +): KotlinSpringConfigConsumerFact[] { + const imports = collectKotlinImports(root); + const facts: KotlinSpringConfigConsumerFact[] = []; + + for (const property of root.descendantsOfType('property_declaration')) { + pushValueFacts(facts, property, filePath, imports); + } + + for (const parameter of root.descendantsOfType('class_parameter')) { + if (!hasBindingPattern(parameter)) continue; + pushValueFacts(facts, parameter, filePath, imports); + } + + for (const type of ['class_declaration', 'object_declaration']) { + for (const declaration of root.descendantsOfType(type)) { + const className = ownerName(declaration); + if (className === undefined) continue; + for (const annotation of annotationsOn(declaration)) { + if (configAnnotationKind(annotation, imports) !== 'configuration-properties') { + continue; + } + const prefix = parseConfigurationPropertiesPrefix(annotation.node); + if (prefix === null) continue; + facts.push({ + consumer: { + kind: 'configuration-properties', + className, + line: declaration.startPosition.row + 1, + prefix, + }, + annotationName: annotation.name, + classScopeId: classScopeId(filePath, declaration), + }); + } + } + } + return facts; +} + +/** Parse Kotlin consumers for focused unit tests; production reuses the worker AST. */ +export function extractKotlinSpringConfigConsumers(source: string): SpringConfigConsumer[] { + const tree = parseSourceSafe(getKotlinParser(), source); + return captureKotlinSpringConfigConsumerFacts(tree.rootNode, '').map( + (fact) => fact.consumer, + ); +} + +export function extractKotlinSpringConfigConsumerFacts( + source: string, +): KotlinSpringConfigConsumerFact[] { + const tree = parseSourceSafe(getKotlinParser(), source); + return captureKotlinSpringConfigConsumerFacts(tree.rootNode, ''); +} + +/** Kotlin ScopeResolver post-resolution hook for Spring configuration consumers. */ +export function attachKotlinSpringConfigBindings( + graph: KnowledgeGraph, + parsedFiles: readonly ParsedFile[], + _nodeLookup: GraphNodeLookup, + indexes: ScopeResolutionIndexes, +): void { + const resolveAnnotation = createSpringAnnotationNameResolver(indexes); + const recognizedAnnotations = new Set([VALUE_ANNOTATION, CONFIGURATION_PROPERTIES_ANNOTATION]); + const batches: Array<{ filePath: string; consumers: SpringConfigConsumer[] }> = []; + for (const parsed of parsedFiles) { + const consumers: SpringConfigConsumer[] = []; + for (const fact of getKotlinSpringConfigConsumerFacts(parsed.filePath)) { + const classScope = indexes.scopeTree.getScope(fact.classScopeId); + if (classScope === undefined || classScope.kind !== 'Class') continue; + const expectedAnnotation = + fact.consumer.kind === 'value' ? VALUE_ANNOTATION : CONFIGURATION_PROPERTIES_ANNOTATION; + const enclosingScope = fact.consumer.kind === 'value' ? classScope.id : classScope.parent; + const resolved = resolveAnnotation( + fact.annotationName, + parsed, + enclosingScope, + recognizedAnnotations, + isKotlinPackageSiblingVisibilityIncomplete(parsed.filePath), + ); + if (resolved === expectedAnnotation) consumers.push(fact.consumer); + } + if (consumers.length > 0) batches.push({ filePath: parsed.filePath, consumers }); + } + bindSpringConfigConsumers(graph, batches); +} diff --git a/gitnexus/src/core/ingestion/languages/kotlin/spring-di.ts b/gitnexus/src/core/ingestion/languages/kotlin/spring-di.ts index acb38360e..af0ba2cec 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/spring-di.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/spring-di.ts @@ -1,5 +1,9 @@ import { makeScopeId } from 'gitnexus-shared'; import { parseSpringInjectionType } from '../../di-extractors/spring.js'; +import { + normalizeSpringFactText, + type SpringArgumentFact, +} from '../../frameworks/spring/argument-facts.js'; import { createSpringDiMetadataAttacher, hasSpringDiRelevantAnnotation, @@ -13,13 +17,29 @@ import { hasSpringBeanFactorySyntax, type SpringBeanFactoryMethodFact, } from '../../frameworks/spring/bean-factories.js'; -import { nodeToCapture, type SyntaxNode } from '../../utils/ast-helpers.js'; +import { hasRecoveredSyntax, nodeToCapture, type SyntaxNode } from '../../utils/ast-helpers.js'; import { getKotlinSpringDiFacts } from './capture-side-channel.js'; import { isKotlinPackageSiblingVisibilityIncomplete } from './package-siblings.js'; export interface KotlinAnnotationSyntaxFact extends SpringDiAnnotationFact { readonly useSiteTarget?: string; readonly line: number; + /** Present only for callers that opt in via `kotlinSpringAnnotationFacts`. */ + readonly args?: readonly SpringArgumentFact[]; +} + +/** + * Options for `kotlinSpringAnnotationFacts`. + * + * The STRUCTURED arguments are opt-in because DI captures every annotated + * constructor parameter, property, and function in the repository, and none of + * its consumers reads them. Note what this does and does not save: every fact + * already carries `text`, the annotation's full source, so the argument TEXT + * crosses the worker boundary either way. What the opt-in avoids is a second, + * parsed copy of that same text on facts that would never look at it. + */ +export interface KotlinSpringAnnotationFactOptions { + readonly includeArguments?: boolean; } export type KotlinSpringDependencyFact = SpringDiDependencyFact; @@ -53,36 +73,128 @@ function firstDescendantOfType(node: SyntaxNode, type: string): SyntaxNode | und return undefined; } -function annotationFact(annotation: SyntaxNode): KotlinAnnotationSyntaxFact | null { +const KOTLIN_COMMENT_NODE_TYPES = new Set(['line_comment', 'multiline_comment']); + +/** + * Kotlin writes annotation arguments and call arguments with the same + * `value_arguments` node, so one reader serves `@KafkaListener(topics = [...])` + * and `kafkaTemplate.send(topic, payload)`. + * + * A named argument keeps its key; everything else — positional values, spreads, + * collection literals, and interpolated strings — is kept as raw text, because + * evaluating it would be resolution. + * + * Returns `null` for a list tree-sitter had to recover, and the callers decide + * what that means: a producer call drops the whole fact, since it has no state + * for "published somewhere unreadable", while an annotation reports no + * arguments and collapses into the marker form. Both answers say "nothing here + * to resolve", which is true; a fabricated value would send a consumer + * somewhere real and wrong. + * + * The check lives HERE, not only in the callers. This function is exported and + * already has a caller in another module, so a guard that every future caller + * has to remember is the same fragility this change set exists to remove — + * `null` makes the decision unavoidable at the type level. Per-argument + * re-checks are still pointless: `hasError` propagates from any argument up to + * the list, so a branch behind this one could never fire. + * + * A named argument is identified by the `=` TOKEN, and the two-child shape is + * only a corroborating detail. Today nothing well formed reaches two children + * without an `=`: an annotated positional argument such as + * `@Suppress("UNCHECKED_CAST") "orders"` arrives as ONE `prefix_expression`, not + * as two children, so the token test is currently redundant. It is kept as the + * leading condition anyway, because the failure it prevents is asymmetric — + * dropping it would let any future two-child positional shape be reported under + * an argument key the source never wrote, which is the failure mode this whole + * change set is about. + */ +export function kotlinValueArgumentFacts(valueArguments: SyntaxNode): SpringArgumentFact[] | null { + if (hasRecoveredSyntax(valueArguments)) return null; + const args: SpringArgumentFact[] = []; + for (const argument of valueArguments.namedChildren) { + if (argument.type !== 'value_argument') continue; + const parts = argument.namedChildren.filter( + (child) => !KOTLIN_COMMENT_NODE_TYPES.has(child.type), + ); + const named = argument.children.some((child) => child.type === '='); + const name = parts[0]; + const value = parts[1]; + if (named && parts.length === 2 && name !== undefined && value !== undefined) { + args.push({ name: name.text.trim(), text: normalizeSpringFactText(value.text) }); + continue; + } + args.push({ text: normalizeSpringFactText(argument.text) }); + } + return args; +} + +/** + * Arguments of one annotation, or `undefined` when it was written without an + * argument list (`@Scheduled`); `@Scheduled()` yields `[]` instead. + * + * Only the annotation's FIRST `user_type` / `constructor_invocation` child is + * read, which is the same element `annotationFact` names. That matters for the + * multi-annotation form `@field:[Alpha Beta("x")]`, where naively taking the + * first constructor invocation would hand Beta's arguments to Alpha. + * + * An argument list that did not parse also yields `undefined`, collapsing into + * the marker-annotation case on purpose: both say there is nothing readable to + * resolve, while the recovered tree would offer values nobody wrote. + */ +function kotlinAnnotationArgumentFacts(annotation: SyntaxNode): SpringArgumentFact[] | undefined { + const named = annotation.namedChildren.find( + (child) => child.type === 'user_type' || child.type === 'constructor_invocation', + ); + if (named === undefined || named.type !== 'constructor_invocation') return undefined; + const valueArguments = named.namedChildren.find((child) => child.type === 'value_arguments'); + if (valueArguments === undefined) return undefined; + // `null` here means recovered syntax; an annotation answers that by reporting + // no arguments at all, which is the marker-annotation form. + return kotlinValueArgumentFacts(valueArguments) ?? undefined; +} + +function annotationFact( + annotation: SyntaxNode, + options: KotlinSpringAnnotationFactOptions, +): KotlinAnnotationSyntaxFact | null { const nameNode = firstDescendantOfType(annotation, 'user_type'); if (nameNode === undefined) return null; const useSiteTarget = annotation.namedChildren .find((child) => child.type === 'use_site_target') ?.text.replace(/:\s*$/, '') .trim(); + const args = + options.includeArguments === true ? kotlinAnnotationArgumentFacts(annotation) : undefined; return { name: nameNode.text.trim(), text: annotation.text.trim(), line: annotation.startPosition.row + 1, ...(useSiteTarget === undefined || useSiteTarget.length === 0 ? {} : { useSiteTarget }), + ...(args === undefined ? {} : { args }), }; } -function annotationsFromModifierContainer(node: SyntaxNode): KotlinAnnotationSyntaxFact[] { +function annotationsFromModifierContainer( + node: SyntaxNode, + options: KotlinSpringAnnotationFactOptions = {}, +): KotlinAnnotationSyntaxFact[] { const facts: KotlinAnnotationSyntaxFact[] = []; for (const child of node.namedChildren) { if (child.type !== 'annotation') continue; - const fact = annotationFact(child); + const fact = annotationFact(child, options); if (fact !== null) facts.push(fact); } return facts; } -export function kotlinSpringAnnotationFacts(node: SyntaxNode): KotlinAnnotationSyntaxFact[] { +export function kotlinSpringAnnotationFacts( + node: SyntaxNode, + options: KotlinSpringAnnotationFactOptions = {}, +): KotlinAnnotationSyntaxFact[] { const facts: KotlinAnnotationSyntaxFact[] = []; for (const child of node.namedChildren) { if (child.type !== 'modifiers' && child.type !== 'parameter_modifiers') continue; - facts.push(...annotationsFromModifierContainer(child)); + facts.push(...annotationsFromModifierContainer(child, options)); } return facts; } diff --git a/gitnexus/src/core/ingestion/languages/kotlin/spring-dynamic-lookup.ts b/gitnexus/src/core/ingestion/languages/kotlin/spring-dynamic-lookup.ts new file mode 100644 index 000000000..184048e07 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/kotlin/spring-dynamic-lookup.ts @@ -0,0 +1,90 @@ +import { makeScopeId } from 'gitnexus-shared'; +import { + createSpringDynamicLookupMetadataAttacher, + springDynamicLookupCardinality, + type SpringDynamicLookupFact, +} from '../../frameworks/spring/dynamic-lookups.js'; +import { + findAncestorBeforeBoundary, + nodeToCapture, + type SyntaxNode, +} from '../../utils/ast-helpers.js'; +import { getKotlinSpringDynamicLookupFacts } from './capture-side-channel.js'; + +// Kotlin emits graph callables for functions and secondary constructors. +// `init {}` / primary-constructor bodies have no independent callable node, so +// attributing their lookups to the enclosing Class would violate graph semantics. +const CALLABLE_NODE_TYPES = new Set(['function_declaration', 'secondary_constructor']); +const NO_CALLABLE_BOUNDARIES = new Set(); +const KOTLIN_CLASS_LITERAL = + /^([A-Za-z_$][A-Za-z0-9_$]*(?:\.[A-Za-z_$][A-Za-z0-9_$]*)*)::class(?:\.java)?$/; + +function navigationParts(node: SyntaxNode): { receiverName: string; methodName: string } | null { + if (node.type !== 'navigation_expression') return null; + const text = node.text.trim(); + const separator = text.lastIndexOf('.'); + if (separator <= 0 || separator === text.length - 1) return null; + return { + receiverName: text.slice(0, separator), + methodName: text.slice(separator + 1), + }; +} + +function singleClassLiteralArgument(node: SyntaxNode): string | null { + const suffix = node.namedChildren.find((child) => child.type === 'call_suffix'); + const argumentsNode = suffix?.namedChildren.find((child) => child.type === 'value_arguments'); + if (argumentsNode === undefined) return null; + const argumentsWithoutComments = argumentsNode.namedChildren.filter( + (child) => child.type !== 'line_comment' && child.type !== 'multiline_comment', + ); + if (argumentsWithoutComments.length !== 1) return null; + const value = argumentsWithoutComments[0]; + if (value?.type !== 'value_argument' || value.namedChildCount !== 1) return null; + return value.namedChild(0)?.text.trim().match(KOTLIN_CLASS_LITERAL)?.[1] ?? null; +} + +/** Capture real Kotlin calls using `Type::class` or `Type::class.java`. */ +export function captureKotlinSpringDynamicLookupFact( + node: SyntaxNode, + filePath: string, +): SpringDynamicLookupFact | null { + if (node.type !== 'call_expression') return null; + const callee = node.namedChildren.find((child) => child.type === 'navigation_expression'); + if (callee === undefined) return null; + const parts = navigationParts(callee); + if (parts === null) return null; + if (springDynamicLookupCardinality(parts.receiverName, parts.methodName) === null) return null; + const targetTypeName = singleClassLiteralArgument(node); + if (targetTypeName === null) return null; + + const owner = findAncestorBeforeBoundary(node, CALLABLE_NODE_TYPES, NO_CALLABLE_BOUNDARIES); + if (owner === null) return null; + const ownerCapture = nodeToCapture('@spring-dynamic-lookup.owner', owner); + return { + ownerScopeId: makeScopeId({ + filePath, + range: ownerCapture.range, + kind: 'Function', + }), + ownerRange: ownerCapture.range, + receiverName: parts.receiverName, + methodName: parts.methodName, + targetTypeName, + }; +} + +/** Standalone extractor for focused tests; production reuses scope-query call nodes. */ +export function captureKotlinSpringDynamicLookupFacts( + rootNode: SyntaxNode, + filePath: string, +): SpringDynamicLookupFact[] { + return rootNode + .descendantsOfType('call_expression') + .map((node) => captureKotlinSpringDynamicLookupFact(node, filePath)) + .filter((fact): fact is SpringDynamicLookupFact => fact !== null); +} + +/** Attach Kotlin lookup facts for later resolution by the shared DI phase. */ +export const attachKotlinSpringDynamicLookup = createSpringDynamicLookupMetadataAttacher({ + getFacts: getKotlinSpringDynamicLookupFacts, +}); diff --git a/gitnexus/src/core/ingestion/languages/kotlin/spring-message-producers.ts b/gitnexus/src/core/ingestion/languages/kotlin/spring-message-producers.ts new file mode 100644 index 000000000..7f32709c8 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/kotlin/spring-message-producers.ts @@ -0,0 +1,132 @@ +import { makeScopeId } from 'gitnexus-shared'; +import { normalizeSpringFactText } from '../../frameworks/spring/argument-facts.js'; +import { + isSpringMessageProducerMethod, + springMessageProducerTemplateOf, + type SpringMessageProducerFact, +} from '../../frameworks/spring/message-producers.js'; +import { + findAncestorBeforeBoundary, + nodeToCapture, + type SyntaxNode, +} from '../../utils/ast-helpers.js'; +import { kotlinValueArgumentFacts } from './spring-di.js'; + +// Kotlin emits graph callables for functions and secondary constructors. +// `init {}` / primary-constructor bodies have no independent callable node, so +// attributing their publishes to the enclosing Class would violate graph +// semantics. +const CALLABLE_NODE_TYPES = new Set(['function_declaration', 'secondary_constructor']); +/** + * A class body ends the search, so the rule above holds at every depth. + * + * Without it the walk passes THROUGH the body of a class or object declared + * inside a function, and the publish in that body's property initializer — which + * likewise has no callable of its own — is attributed to the enclosing function + * instead of being dropped the way its top-level twin is. + */ +const TYPE_BODY_BOUNDARIES = new Set(['class_body', 'enum_class_body']); + +/** + * Strip null-assertion operators from a receiver. + * + * `?.` carries its marker on the navigation suffix, which the structural split + * already discards, but `!!` wraps the receiver in a `postfix_expression` whose + * text ends in the operator — enough to make `kafkaTemplate!!` fail the + * receiver-name check and lose a publish. Unwrapping is limited to `!!` + * because `counter++` produces the same node shape and is not a receiver name. + */ +function withoutNullAssertions(receiver: SyntaxNode): SyntaxNode { + let current = receiver; + while (current.type === 'postfix_expression') { + const operand = current.namedChildren[0]; + if (operand === undefined) return current; + const onlyNullAssertions = current.children.every( + (child) => child.id === operand.id || child.type === '!!', + ); + if (!onlyNullAssertions) return current; + current = operand; + } + return current; +} + +/** + * Split `receiver.method` structurally rather than by text. + * + * Text splitting would leave the safe-call marker on the receiver + * (`kafkaTemplate?` for `kafkaTemplate?.send(...)`). + */ +function navigationParts(callee: SyntaxNode): { receiverName: string; methodName: string } | null { + if (callee.type !== 'navigation_expression') return null; + const suffix = callee.namedChildren.find((child) => child.type === 'navigation_suffix'); + const receiver = callee.namedChildren.find((child) => child.type !== 'navigation_suffix'); + if (suffix === undefined || receiver === undefined) return null; + const methodName = suffix.namedChildren + .find((child) => child.type === 'simple_identifier') + ?.text.trim(); + if (methodName === undefined) return null; + return { + receiverName: normalizeSpringFactText(withoutNullAssertions(receiver).text), + methodName, + }; +} + +/** + * Capture one messaging-template publish from a Kotlin call already surfaced by + * the scope query, without resolving the destination it names. + * + * The destination argument may be a literal, a reference to a constant that + * lives in another file, or a `${...}` placeholder resolved from configuration; + * all three are recorded as written and left to a later phase. + * + * A call whose argument list did not parse yields NO fact, for the reason given + * on the Java side: error recovery guesses argument boundaries, and this fact + * has no way to say "published somewhere unreadable". + */ +export function captureKotlinSpringMessageProducerFact( + node: SyntaxNode, + filePath: string, +): SpringMessageProducerFact | null { + if (node.type !== 'call_expression') return null; + const callee = node.namedChildren[0]; + if (callee === undefined) return null; + const parts = navigationParts(callee); + if (parts === null || !isSpringMessageProducerMethod(parts.methodName)) return null; + const template = springMessageProducerTemplateOf(parts.receiverName, parts.methodName); + if (template === null) return null; + + const callSuffix = node.namedChildren.find((child) => child.type === 'call_suffix'); + // A trailing-lambda call (`send { ... }`) has no argument list at all, which + // is a different fact from an empty one (`send()`). + const valueArguments = callSuffix?.namedChildren.find( + (child) => child.type === 'value_arguments', + ); + // `null` from the reader means tree-sitter had to recover the list. A publish + // fact exists to carry a destination and has no state for "published + // somewhere unreadable", so the whole fact is withheld rather than reported + // with arguments the source never wrote. + const args = valueArguments === undefined ? undefined : kotlinValueArgumentFacts(valueArguments); + if (args === null) return null; + const owner = findAncestorBeforeBoundary(node, CALLABLE_NODE_TYPES, TYPE_BODY_BOUNDARIES); + if (owner === null) return null; + const ownerCapture = nodeToCapture('@spring-message-producer.owner', owner); + return { + ownerScopeId: makeScopeId({ filePath, range: ownerCapture.range, kind: 'Function' }), + ownerRange: ownerCapture.range, + template, + receiverName: parts.receiverName, + methodName: parts.methodName, + ...(args === undefined ? {} : { args }), + }; +} + +/** Standalone extractor for focused tests; production reuses scope-query call nodes. */ +export function captureKotlinSpringMessageProducerFacts( + rootNode: SyntaxNode, + filePath: string, +): SpringMessageProducerFact[] { + return rootNode + .descendantsOfType('call_expression') + .map((node) => captureKotlinSpringMessageProducerFact(node, filePath)) + .filter((fact): fact is SpringMessageProducerFact => fact !== null); +} diff --git a/gitnexus/src/core/ingestion/languages/kotlin/spring-non-http-handlers.ts b/gitnexus/src/core/ingestion/languages/kotlin/spring-non-http-handlers.ts index b46132073..a7b21ffd4 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/spring-non-http-handlers.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/spring-non-http-handlers.ts @@ -1,6 +1,7 @@ import { makeScopeId } from 'gitnexus-shared'; import { createSpringNonHttpHandlerMetadataAttacher, + hasSpringNonHttpHandlerRelevantAnnotation, type SpringNonHttpHandlerAnnotationFact, type SpringNonHttpHandlerFact, } from '../../frameworks/spring/non-http-handlers.js'; @@ -12,10 +13,68 @@ import { kotlinSpringAnnotationFacts } from './spring-di.js'; export type KotlinSpringNonHttpHandlerFact = SpringNonHttpHandlerFact; +/** + * Local names that reach a handler annotation only through an import alias. + * + * `import ...event.EventListener as SpringEvent` makes `@SpringEvent` a handler + * annotation whose simple name matches nothing, which is why the CALLABLE + * capture below has no name prefilter. The alias is not a mystery at capture + * time, though: the import header states both the local name and the FQN it + * stands for, so the same relevance predicate that Java uses on the annotation + * name can be applied to the IMPORTED name and the answer carried back to the + * alias. That recovers a name-based decision without discarding aliases. + * + * Only aliases are collected. A plain or wildcard import leaves the annotation + * written under its own simple name, which the direct check already sees. + */ +function aliasedHandlerAnnotationNames(classNode: SyntaxNode): ReadonlySet { + let root: SyntaxNode = classNode; + while (root.parent !== null) root = root.parent; + + const headers: SyntaxNode[] = []; + for (const child of root.namedChildren) { + if (child.type === 'import_header') headers.push(child); + else if (child.type === 'import_list') { + for (const header of child.namedChildren) { + if (header.type === 'import_header') headers.push(header); + } + } + } + + const aliases = new Set(); + for (const header of headers) { + const alias = header.namedChildren + .find((child) => child.type === 'import_alias') + ?.namedChildren.find((child) => child.type === 'type_identifier') + ?.text.trim(); + if (alias === undefined || alias.length === 0) continue; + const imported = header.namedChildren.find((child) => child.type === 'identifier')?.text.trim(); + if (imported === undefined || imported.length === 0) continue; + if (hasSpringNonHttpHandlerRelevantAnnotation([{ name: imported }])) aliases.add(alias); + } + return aliases; +} + /** * Capture annotated callables conservatively. A simple-name prefilter would * discard Kotlin aliases (for example, `EventListener as SpringEvent`) before * the post-import resolver can map the local name back to its annotation FQN. + * + * That conservatism applies to the CALLABLE — every annotated function still + * produces a fact, whatever its annotations are named. It does NOT have to + * apply to the arguments: reading them unconditionally charged every + * `@Transactional` and `@Deprecated` in a repository for data no consumer + * reads, and unlike the callable itself an argument list can be fetched on + * evidence. Arguments are therefore read in a second pass, for callables that + * either carry a handler annotation under its own name or use a local name this + * file aliased to one — the same two-pass shape as Java, with the alias set + * standing in for the name prefilter Kotlin cannot use. + * + * Measured on 200 annotated NON-handler functions in one file: the side-channel + * payload was 41069 bytes before arguments existed, 78797 with them read + * unconditionally, and 41069 again with this pass — byte for byte what it cost + * before the feature. The 200-handler equivalent pays 58649, which is the + * argument text the consumer asked for. */ export function captureKotlinSpringNonHttpHandlerFacts( classNode: SyntaxNode, @@ -26,9 +85,23 @@ export function captureKotlinSpringNonHttpHandlerFacts( (child) => child.type === 'class_body' || child.type === 'enum_class_body', ); if (body === undefined) return facts; + // Read the import headers at most once per class, and only when some callable + // actually fails the direct name check. + let aliasedHandlerNames: ReadonlySet | undefined; for (const member of body.namedChildren) { if (member.type !== 'function_declaration') continue; - const annotations = kotlinSpringAnnotationFacts(member); + const named = kotlinSpringAnnotationFacts(member); + if (named.length === 0) continue; + let readArguments = hasSpringNonHttpHandlerRelevantAnnotation(named); + if (!readArguments) { + aliasedHandlerNames ??= aliasedHandlerAnnotationNames(classNode); + readArguments = named.some( + (annotation) => aliasedHandlerNames?.has(annotation.name) === true, + ); + } + const annotations = readArguments + ? kotlinSpringAnnotationFacts(member, { includeArguments: true }) + : named; if (annotations.length === 0) continue; const ownerRange = nodeToCapture('@spring-non-http-handler.owner', member).range; facts.push({ @@ -40,6 +113,7 @@ export function captureKotlinSpringNonHttpHandlerFacts( ...(annotation.useSiteTarget === undefined ? {} : { useSiteTarget: annotation.useSiteTarget }), + ...(annotation.args === undefined ? {} : { args: annotation.args }), })), }); } diff --git a/gitnexus/src/core/ingestion/languages/python.ts b/gitnexus/src/core/ingestion/languages/python.ts index ca40a4566..ce8d0fa90 100644 --- a/gitnexus/src/core/ingestion/languages/python.ts +++ b/gitnexus/src/core/ingestion/languages/python.ts @@ -45,6 +45,7 @@ import { import { extractDjangoRoutes } from '../route-extractors/django.js'; import { discoverDjangoRootUrls } from '../route-extractors/django-root-discovery.js'; import { extractPythonModuleConstants } from '../route-extractors/python-const-resolver.js'; +import { pythonDecoratorRouteHandlerName } from '../route-extractors/python-decorator-handler.js'; const BUILT_INS: ReadonlySet = new Set([ 'print', @@ -143,6 +144,7 @@ export const pythonProvider = defineLanguage({ discoverDjangoRootUrls(files, contentMap, reader), extractRoutes: (tree, filePath, reader, parser) => parser ? extractDjangoRoutes(tree, filePath, parser, reader) : [], + decoratorRouteHandlerName: pythonDecoratorRouteHandlerName, labelOverride: pythonFunctionDefinitionLabel, // ── RFC #909 Ring 3: scope-based resolution hooks (RFC §5) ────────── diff --git a/gitnexus/src/core/ingestion/model/field-registry.ts b/gitnexus/src/core/ingestion/model/field-registry.ts index c45fb9cb5..ce9df4982 100644 --- a/gitnexus/src/core/ingestion/model/field-registry.ts +++ b/gitnexus/src/core/ingestion/model/field-registry.ts @@ -2,9 +2,9 @@ * Field Registry * * Owner-scoped field/property index extracted from SymbolTable. - * Stores Property / Variable / Const / Static symbols keyed by - * `ownerNodeId\0fieldName` for O(1) lookup. Supports multiple defs - * under the same (owner, name) — e.g. legacy Property plus a + * Stores Property / Variable / Const / Static symbols in a nested + * `Map>` for O(1) lookup. Supports + * multiple defs under the same (owner, name) — e.g. legacy Property plus a * scope-resolution Variable reconciliation entry. */ @@ -49,13 +49,13 @@ export interface MutableFieldRegistry extends FieldRegistry { // --------------------------------------------------------------------------- export const createFieldRegistry = (): MutableFieldRegistry => { - const fieldByOwner = new Map(); + const fieldByOwner = new Map>(); const lookupAllByOwner = ( ownerNodeId: string, fieldName: string, ): readonly SymbolDefinition[] => { - return fieldByOwner.get(`${ownerNodeId}\0${fieldName}`) ?? EMPTY; + return fieldByOwner.get(ownerNodeId)?.get(fieldName) ?? EMPTY; }; const lookupFieldByOwner = ( @@ -67,12 +67,16 @@ export const createFieldRegistry = (): MutableFieldRegistry => { }; const register = (ownerNodeId: string, fieldName: string, def: SymbolDefinition): void => { - const key = `${ownerNodeId}\0${fieldName}`; - const existing = fieldByOwner.get(key); + let byName = fieldByOwner.get(ownerNodeId); + if (!byName) { + byName = new Map(); + fieldByOwner.set(ownerNodeId, byName); + } + const existing = byName.get(fieldName); if (existing) { existing.push(def); } else { - fieldByOwner.set(key, [def]); + byName.set(fieldName, [def]); } }; diff --git a/gitnexus/src/core/ingestion/model/method-registry.ts b/gitnexus/src/core/ingestion/model/method-registry.ts index 75f9834c2..f64a1d8c7 100644 --- a/gitnexus/src/core/ingestion/model/method-registry.ts +++ b/gitnexus/src/core/ingestion/model/method-registry.ts @@ -2,9 +2,9 @@ * Method Registry * * Owner-scoped method index extracted from SymbolTable. - * Stores Method/Constructor/Function-with-ownerId symbols keyed by - * `ownerNodeId\0methodName` for O(1) lookup. Supports overloads - * (array values) and arity-based filtering. + * Stores Method/Constructor/Function-with-ownerId symbols in a nested + * `Map>` for O(1) lookup. Supports + * overloads (array values) and arity-based filtering. */ import type { SymbolDefinition } from 'gitnexus-shared'; @@ -92,7 +92,7 @@ export interface MutableMethodRegistry extends MethodRegistry { // --------------------------------------------------------------------------- export const createMethodRegistry = (): MutableMethodRegistry => { - const methodByOwner = new Map(); + const methodByOwner = new Map>(); // Secondary flat-by-name index. Values are the SAME SymbolDefinition // references stored under `methodByOwner` — no copy, just a second key. // Populated in lockstep by `register()` and emptied by `clear()`. @@ -102,12 +102,15 @@ export const createMethodRegistry = (): MutableMethodRegistry => { // dedup fast-path. Monotonic: never unset except on `clear()`. let hasFunctionMethodsFlag = false; + const ownerDefs = (ownerNodeId: string, methodName: string): SymbolDefinition[] | undefined => + methodByOwner.get(ownerNodeId)?.get(methodName); + const lookupMethodByOwner = ( ownerNodeId: string, methodName: string, argCount?: number, ): SymbolDefinition | undefined => { - const defs = methodByOwner.get(`${ownerNodeId}\0${methodName}`); + const defs = ownerDefs(ownerNodeId, methodName); if (!defs || defs.length === 0) return undefined; // Arity narrowing: when an argCount is provided and there are multiple @@ -176,16 +179,20 @@ export const createMethodRegistry = (): MutableMethodRegistry => { ownerNodeId: string, methodName: string, ): readonly SymbolDefinition[] => { - return methodByOwner.get(`${ownerNodeId}\0${methodName}`) ?? EMPTY; + return ownerDefs(ownerNodeId, methodName) ?? EMPTY; }; const register = (ownerNodeId: string, methodName: string, def: SymbolDefinition): void => { - const key = `${ownerNodeId}\0${methodName}`; - const existing = methodByOwner.get(key); + let owned = methodByOwner.get(ownerNodeId); + if (!owned) { + owned = new Map(); + methodByOwner.set(ownerNodeId, owned); + } + const existing = owned.get(methodName); if (existing) { existing.push(def); } else { - methodByOwner.set(key, [def]); + owned.set(methodName, [def]); } const byName = methodsByName.get(methodName); if (byName) { diff --git a/gitnexus/src/core/ingestion/model/registration-table.ts b/gitnexus/src/core/ingestion/model/registration-table.ts index 2de59e966..e9e85eeb0 100644 --- a/gitnexus/src/core/ingestion/model/registration-table.ts +++ b/gitnexus/src/core/ingestion/model/registration-table.ts @@ -182,6 +182,9 @@ const LABEL_BEHAVIOR = { Section: 'inert', Route: 'inert', Tool: 'inert', + // A broker address, not a symbol: nothing in any language resolves a name to + // it, so it stays out of the dispatch and callable indexes like `Route`. + Destination: 'inert', // Taint/PDG substrate (issue #2080) — a control-flow node, never a // symbol-resolution target. Inert: file index only, no owner scope. BasicBlock: 'inert', diff --git a/gitnexus/src/core/ingestion/model/type-registry.ts b/gitnexus/src/core/ingestion/model/type-registry.ts index 4dc87e924..135e61dfd 100644 --- a/gitnexus/src/core/ingestion/model/type-registry.ts +++ b/gitnexus/src/core/ingestion/model/type-registry.ts @@ -70,7 +70,7 @@ export const createTypeRegistry = (): MutableTypeRegistry => { const classByName = new Map(); const classByQualifiedName = new Map(); const implByName = new Map(); - const nestedByOwner = new Map(); + const nestedByOwner = new Map>(); const lookupClassByName = (name: string): SymbolDefinition[] => { return classByName.get(name) ?? []; @@ -88,7 +88,7 @@ export const createTypeRegistry = (): MutableTypeRegistry => { ownerNodeId: string, simpleName: string, ): readonly SymbolDefinition[] => { - return nestedByOwner.get(`${ownerNodeId}\0${simpleName}`) ?? EMPTY; + return nestedByOwner.get(ownerNodeId)?.get(simpleName) ?? EMPTY; }; const registerClass = (name: string, qualifiedName: string, def: SymbolDefinition): void => { @@ -121,12 +121,16 @@ export const createTypeRegistry = (): MutableTypeRegistry => { simpleName: string, def: SymbolDefinition, ): void => { - const key = `${ownerNodeId}\0${simpleName}`; - const existing = nestedByOwner.get(key); + let byName = nestedByOwner.get(ownerNodeId); + if (!byName) { + byName = new Map(); + nestedByOwner.set(ownerNodeId, byName); + } + const existing = byName.get(simpleName); if (existing) { existing.push(def); } else { - nestedByOwner.set(key, [def]); + byName.set(simpleName, [def]); } }; diff --git a/gitnexus/src/core/ingestion/pipeline-phases/di.ts b/gitnexus/src/core/ingestion/pipeline-phases/di.ts index 12ee5cb3e..b8bf69bd8 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/di.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/di.ts @@ -102,6 +102,10 @@ function providerCandidates( return recognized.length > 0 ? recognized : all; } +function isConcreteTypeNode(node: GraphNode | undefined): boolean { + return node?.label === 'Class' || node?.label === 'Record' || node?.label === 'Enum'; +} + export const diPhase: PipelinePhase = { name: 'di', deps: ['mro'], @@ -160,21 +164,27 @@ export const diPhase: PipelinePhase = { }; } - const interfaceToImplementers = new Map>(); + const directSubtypes = new Map>(); const directSupertypes = new Map>(); for (const rel of ctx.graph.iterRelationshipsByType('IMPLEMENTS')) { - const set = interfaceToImplementers.get(rel.targetId) ?? new Set(); - set.add(rel.sourceId); - interfaceToImplementers.set(rel.targetId, set); + const subtypes = directSubtypes.get(rel.targetId) ?? new Set(); + subtypes.add(rel.sourceId); + directSubtypes.set(rel.targetId, subtypes); const supertypes = directSupertypes.get(rel.sourceId) ?? new Set(); supertypes.add(rel.targetId); directSupertypes.set(rel.sourceId, supertypes); } for (const rel of ctx.graph.iterRelationshipsByType('EXTENDS')) { + const subtypes = directSubtypes.get(rel.targetId) ?? new Set(); + subtypes.add(rel.sourceId); + directSubtypes.set(rel.targetId, subtypes); const supertypes = directSupertypes.get(rel.sourceId) ?? new Set(); supertypes.add(rel.targetId); directSupertypes.set(rel.sourceId, supertypes); } + const orderedDirectSubtypes = new Map( + [...directSubtypes].map(([typeId, subtypes]) => [typeId, [...subtypes].sort().reverse()]), + ); const memberToClass = new Map(); for (const relationType of ['HAS_PROPERTY', 'HAS_METHOD'] as const) { @@ -187,16 +197,45 @@ export const diPhase: PipelinePhase = { const interfacesByLanguage = new Map(); const classesByLanguage = new Map(); ctx.graph.forEachNode((node) => { - if (node.label !== 'Class' && node.label !== 'Interface') return; + const concreteType = isConcreteTypeNode(node); + if (!concreteType && node.label !== 'Interface') return; const language = node.properties.language; if (typeof language !== 'string' || !candidateLanguages.has(language)) return; - const indexes = node.label === 'Class' ? classesByLanguage : interfacesByLanguage; + const indexes = concreteType ? classesByLanguage : interfacesByLanguage; const index = indexes.get(language) ?? emptyNameIndex(); addIndexedName(index, node); indexes.set(language, index); - if (node.label === 'Class') providerNodes.set(node.id, node); + if (concreteType) providerNodes.set(node.id, node); }); + const concreteSubtypesByRoot = new Map>(); + const concreteSubtypes = (rootTypeId: string, language: string): ReadonlySet => { + const cacheKey = `${language}\0${rootTypeId}`; + const cached = concreteSubtypesByRoot.get(cacheKey); + if (cached !== undefined) return cached; + + const concrete = new Set(); + const queue = [rootTypeId]; + const visited = new Set(); + while (queue.length > 0) { + const typeId = queue.pop(); + if (typeId === undefined || visited.has(typeId)) continue; + visited.add(typeId); + const typeNode = ctx.graph.getNode(typeId); + if ( + typeNode !== undefined && + isConcreteTypeNode(typeNode) && + typeNode.properties.language === language + ) { + concrete.add(typeId); + } + const children = orderedDirectSubtypes.get(typeId) ?? []; + queue.push(...children); + } + concreteSubtypesByRoot.set(cacheKey, concrete); + return concrete; + }; + // A declaration returning a concrete class is assignable to every class or // interface that type extends/implements. Expand once per language+type and // register the declaration under those ancestor names. This keeps named @@ -300,11 +339,15 @@ export const diPhase: PipelinePhase = { continue; } - const structural = new Set(); - if (typeof classEntry === 'string') structural.add(classEntry); - if (typeof interfaceEntry === 'string') { - for (const id of interfaceToImplementers.get(interfaceEntry) ?? []) structural.add(id); - } + const rootTypeId = + typeof classEntry === 'string' + ? classEntry + : typeof interfaceEntry === 'string' + ? interfaceEntry + : undefined; + const structural = new Set( + rootTypeId === undefined ? [] : concreteSubtypes(rootTypeId, candidate.language), + ); for (const id of providedTypes.get(candidate.language)?.get(candidate.targetTypeName) ?? []) { structural.add(id); } diff --git a/gitnexus/src/core/ingestion/pipeline-phases/index.ts b/gitnexus/src/core/ingestion/pipeline-phases/index.ts index 359d80a65..99b7cc253 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/index.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/index.ts @@ -21,6 +21,7 @@ export { type ScopeResolutionOutput, } from '../scope-resolution/pipeline/phase.js'; export { springConfigPhase, type SpringConfigOutput } from './spring-config.js'; +export { springDestinationsPhase, type SpringDestinationsOutput } from './spring-destinations.js'; export { springAutoConfigurationPhase, type SpringAutoConfigurationOutput, diff --git a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts index b6ca75ea0..f9cbbd47b 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts @@ -36,7 +36,7 @@ import { getDurableParsedFileDir, loadDurableParsedFileIndex, prepareDurableParsedFileChunk, - restoreDurableParsedFileShard, + durableChunkHasShards, } from '../../../storage/parsedfile-store.js'; import type { ParseWorkerResult } from '../workers/parse-worker.js'; import { DEFAULT_PDG_MAX_FUNCTION_LINES } from '../cfg/collect.js'; @@ -89,10 +89,9 @@ import type { ExtractedRouterModuleAlias, } from '../route-extractors/fastapi-router-bindings.js'; import { normalizeExtractedRoutePath } from '../route-extractors/route-path.js'; -import { - resolveOperands, - type ModuleConstants, -} from '../route-extractors/python-const-resolver.js'; +import { resolveOperands } from '../route-extractors/python-const-resolver.js'; +import type { ModuleConstants } from '../route-extractors/constant-resolver.js'; +import { prepareRouteConstantsByProvider } from '../language-provider.js'; import { resolveInheritedSpringRoutes, type SharedSpringType, @@ -463,6 +462,9 @@ export async function runChunkedParseAndResolve( * cache analyze run can skip the dominant `extractParsedFile` cost * (otherwise ~58s on a 1000-file repo). */ parsedFiles: import('gitnexus-shared').ParsedFile[]; + /** Repo-wide harvested constants, already prepared per provider. See + * `ParseOutput.moduleConstants` for why this leaves the parse phase. */ + moduleConstants: ReadonlyMap; scopeExtractionFailures: string[]; /** Files excluded because their non-standalone language parser was unavailable. */ unavailableScopeLanguageFiles: number; @@ -739,15 +741,15 @@ export async function runChunkedParseAndResolve( // a sibling of the run-scoped store, NOT cleared per run. Workers write a // shard per chunk hash; on a warm parse-cache hit we restore the chunk's // shards into the run-scoped store so scope-resolution streams them without - // re-parsing. `durableHitKeys` is the prior run's index, version-gated by - // PARSE_CACHE_VERSION (a mismatch ⇒ empty ⇒ every chunk re-dispatches, which - // repopulates the durable store — never the main-thread extract fallback). + // re-parsing. `durableHitEntries` is the prior run's path-coverage index, + // version-gated by PARSE_CACHE_VERSION (a mismatch ⇒ empty ⇒ every chunk + // re-dispatches, which repopulates the durable store). const durableParsedFileDir = parsedFileStorePath !== undefined ? getDurableParsedFileDir(parsedFileStorePath) : undefined; - const durableHitKeys = + const durableHitEntries = durableParsedFileDir !== undefined ? await loadDurableParsedFileIndex(durableParsedFileDir, PARSE_CACHE_VERSION) - : new Set(); + : new Map>(); let chunkCacheHits = 0; let chunkCacheMisses = 0; let reparsedFileCount = 0; @@ -825,11 +827,14 @@ export async function runChunkedParseAndResolve( } if (chunkWorkerData.parsedFiles?.length) { if (parsedFileStorePath) { - await persistParsedFileChunk( + const wrote = await persistParsedFileChunk( parsedFileStorePath, `chunk-${chunkIdx}`, chunkWorkerData.parsedFiles, ); + if (!wrote) { + for (const item of chunkWorkerData.parsedFiles) allParsedFiles.push(item); + } } else { for (const item of chunkWorkerData.parsedFiles) allParsedFiles.push(item); } @@ -1019,8 +1024,16 @@ export async function runChunkedParseAndResolve( // store was introduced, or a pruned/version-stale shard — fall through to // a worker re-dispatch to repopulate them. NEVER let scope-resolution // re-extract on the main thread (the #1983 OOM the durable store closes). + const durableExpectedPaths = + chunkHash === null ? undefined : durableHitEntries.get(chunkHash); const durableHit = - chunkHash !== null && durableParsedFileDir !== undefined && durableHitKeys.has(chunkHash); + cachedRaw !== undefined && + cachedRaw.length > 0 && + chunkHash !== null && + durableParsedFileDir !== undefined && + parsedFileStorePath !== undefined && + durableExpectedPaths !== undefined && + (await durableChunkHasShards(parsedFileStorePath, chunkHash, durableExpectedPaths)); if (cachedRaw && cachedRaw.length > 0 && (durableHit || parsedFileStorePath === undefined)) { // Cache hit: replay cached worker output. Finalize any parked worker @@ -1053,22 +1066,8 @@ export async function runChunkedParseAndResolve( nodesCreated: graph.nodeCount, }, }); - // Restore the chunk's durable ParsedFile shards into the run-scoped - // store so scope-resolution finds full coverage with ZERO main-thread - // re-parse. A verbatim byte copy — byte-identical to a cold run. - if (durableHit && durableParsedFileDir && parsedFileStorePath && chunkHash) { - const restored = await restoreDurableParsedFileShard( - durableParsedFileDir, - parsedFileStorePath, - chunkHash, - ); - if (restored === 0) { - logger.warn( - `parsedfile-cache: durable shards missing for cached chunk ` + - `${chunkHash.slice(0, 8)} — scope-resolution will re-extract these files`, - ); - } - } + // The durable gate already snapshotted warm `.v8` shards into the + // run-scoped store for scope resolution. await applyChunkResults(chunkWorkerData, chunkIdx, chunkFiles, chunkStartMs); } else { // Cache miss: dispatch to workers, capture the raw results, store @@ -1271,11 +1270,27 @@ export async function runChunkedParseAndResolve( // carries `routePathExpr`/`routePathOperands` and an empty `routePath`; we fold // the operands against the repo-wide, file-path-keyed constant map. On failure // we DROP the route (KTD5 skip floor) rather than emit a phantom `POST /`. + // + // Built (and prepared) UNCONDITIONALLY when anything was harvested, because + // the map is also handed to downstream phases on `ParseOutput.moduleConstants` + // — `springDestinations` folds broker-address constants against exactly the + // same table. Preparation runs exactly once, here, on one map, before either + // consumer folds. Deferring it into each consumer instead would need + // `prepareRouteConstants` to be safe to call twice — it materializes deferred + // wildcard bindings IN PLACE — or would leave whichever consumer ran first + // folding against unprepared constants. Neither is worth the coupling; the + // cost here is one pass over the harvested constants of a repo that has some. + const repoConstants = new Map(); + for (const { filePath, constants } of allModuleConstants) { + repoConstants.set(filePath, constants); + } + if (repoConstants.size > 0) { + // Let each language prepare only its own constants slice before folding. + // This is where deferred wildcard bindings can be materialized once per + // provider without naming a language in the shared parse phase. + prepareRouteConstantsByProvider(repoConstants, getProviderForFile); + } if (allDecoratorRoutes.some((dr) => dr.routePathExpr !== undefined)) { - const repoConstants = new Map(); - for (const { filePath, constants } of allModuleConstants) { - repoConstants.set(filePath, constants); - } const resolvedRoutes: ExtractedDecoratorRoute[] = []; let skipped = 0; for (const dr of allDecoratorRoutes) { @@ -1601,6 +1616,10 @@ export async function runChunkedParseAndResolve( // cache: when the file's ParsedFile is here, scope-resolution skips its own // `extractParsedFile` call. parsedFiles: allParsedFiles, + // Repo-wide, file-path-keyed constants, already through each provider's + // `prepareRouteConstants` hook. Empty when no provider harvests constants + // for the languages in this repo. + moduleConstants: repoConstants, scopeExtractionFailures: [...scopeExtractionFailures].sort(), unavailableScopeLanguageFiles, }; diff --git a/gitnexus/src/core/ingestion/pipeline-phases/parse.ts b/gitnexus/src/core/ingestion/pipeline-phases/parse.ts index 8e151ed73..b47ceafdf 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/parse.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/parse.ts @@ -31,6 +31,7 @@ import type { } from '../workers/parse-worker.js'; import { runChunkedParseAndResolve } from './parse-impl.js'; import type { MutableSemanticModel } from '../model/index.js'; +import type { ModuleConstants } from '../route-extractors/constant-resolver.js'; export interface ParseOutput { /** @@ -82,6 +83,22 @@ export interface ParseOutput { * costing ~58s on a 1000-file repo). */ readonly parsedFiles: readonly ParsedFile[]; + /** + * Repo-wide string constants harvested by the providers that declare + * `extractModuleConstants`, keyed by file path and already through each + * provider's `prepareRouteConstants` hook. + * + * Exposed so a later phase can fold a constant reference the same way the + * decorator-route pass does — `springDestinations` resolves a broker address + * written as `Topics.ORDERS` against this table. It is a snapshot: parse does + * not mutate it after returning, and consumers must not either, because the + * preparation that made it foldable has already run. + * + * Empty when no provider in this repo harvests constants. A downstream fold + * against an empty table simply fails to resolve, which is a recorded refusal + * rather than a wrong answer. + */ + readonly moduleConstants: ReadonlyMap; /** Files whose scope extraction failed while legacy parsing continued. */ readonly scopeExtractionFailures: readonly string[]; /** Files omitted because their non-standalone language parser was unavailable. */ diff --git a/gitnexus/src/core/ingestion/pipeline-phases/processes.ts b/gitnexus/src/core/ingestion/pipeline-phases/processes.ts index 24117d532..e34e79d35 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/processes.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/processes.ts @@ -4,7 +4,8 @@ * Detects execution flows (processes) and creates Process nodes + * STEP_IN_PROCESS edges. Also links Route/Tool nodes to processes. * - * @deps communities, routes, tools, pruneLocalSymbols, structure, parse + * @deps communities, routes, tools, springAutoConfiguration, + * pruneLocalSymbols, structure, parse * @reads graph (all nodes and relationships), communityResult, routeRegistry, * toolDefs, parse's allFetchCalls + allORMQueries (R3-6 sink sites) * @writes graph (Process nodes, STEP_IN_PROCESS edges, ENTRY_POINT_OF edges) @@ -52,7 +53,15 @@ export const processesPhase: PipelinePhase = { // sinks rather than failing the phase. `pruneLocalSymbols` is declared // explicitly so process extraction always reads the trimmed graph even if a // future option drops the intervening `mro`/`communities` phases. - deps: ['communities', 'routes', 'tools', 'pruneLocalSymbols', 'structure', 'parse'], + deps: [ + 'communities', + 'routes', + 'tools', + 'springAutoConfiguration', + 'pruneLocalSymbols', + 'structure', + 'parse', + ], async execute( ctx: PipelineContext, @@ -212,8 +221,38 @@ export const processesPhase: PipelinePhase = { }); }); + // The static registry is finalized before Spring runtime enrichment. Merge + // runtime-confirmed Route nodes from the graph after the explicit + // springAutoConfiguration dependency has completed, so Actuator-only + // mappings participate in the same process-linking path. + const processRouteRegistry = new Map(routeRegistry); + ctx.graph.forEachNode((node) => { + if ( + node.label !== 'Route' || + node.properties.runtimeSource !== 'spring-actuator' || + node.properties.runtimeConfirmed !== true + ) { + return; + } + const url = typeof node.properties.name === 'string' ? node.properties.name : undefined; + const filePath = + typeof node.properties.filePath === 'string' ? node.properties.filePath : undefined; + const method = + typeof node.properties.method === 'string' ? node.properties.method : undefined; + if (url === undefined || filePath === undefined) return; + const key = routeNodeKey(method, url); + if (!processRouteRegistry.has(key)) { + processRouteRegistry.set(key, { + filePath, + source: 'spring-actuator-runtime', + url, + ...(method === undefined ? {} : { method }), + }); + } + }); + // Link Route and Tool nodes to Processes - if (routeRegistry.size > 0 || toolDefs.length > 0) { + if (processRouteRegistry.size > 0 || toolDefs.length > 0) { // Two-tier route lookup, mirroring the tool tables 10 lines below. // Routes whose handler resolved key by `handlerSymbolId` (read from // the Route node's graph properties — routes.ts stamps it there) and @@ -228,7 +267,7 @@ export const processesPhase: PipelinePhase = { // routes phase stamps on the Route node was never consulted. const routesByHandlerId = new Map(); const routesWithoutHandlerByFile = new Map(); - for (const [, entry] of routeRegistry) { + for (const [, entry] of processRouteRegistry) { // Push the Route node identity (`routeNodeKey`), not the bare URL, so the // ENTRY_POINT_OF edge targets the same node id the routes phase created // (#2289: a same-URL GET/POST pair is two distinct Route nodes). diff --git a/gitnexus/src/core/ingestion/pipeline-phases/routes.ts b/gitnexus/src/core/ingestion/pipeline-phases/routes.ts index e93fa033c..42bf639d2 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/routes.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/routes.ts @@ -336,32 +336,31 @@ export const routesPhase: PipelinePhase = { let handlerContents: Map | undefined; if (routeRegistry.size > 0) { - const handlerPathFor = (routeKey: string, entry: RouteEntry): string => { - if (entry.source !== DATA_ROUTE_TABLE_SOURCE) return entry.filePath; - const handlerSymbolId = routeHandlerSymbols.get(routeKey); - const resolvedPath = handlerSymbolId - ? ctx.graph.getNode(handlerSymbolId)?.properties.filePath - : undefined; - return typeof resolvedPath === 'string' ? resolvedPath : entry.filePath; - }; - const handlerPaths = [...routeRegistry].map(([key, entry]) => handlerPathFor(key, entry)); - handlerContents = await readFileContents(ctx.repoPath, handlerPaths); + // Resolve once so content attribution, the route stamp, and the edge use + // the same live graph node. Pre-seeded routes never own handler symbols. + const routes = [...routeRegistry].map(([routeKey, entry]) => { + const id = preSeededKeys.has(routeKey) ? undefined : routeHandlerSymbols.get(routeKey); + const node = id ? ctx.graph.getNode(id) : undefined; + const handlerSymbol = id && node ? { id, node } : undefined; + const resolvedPath = + entry.source === DATA_ROUTE_TABLE_SOURCE + ? handlerSymbol?.node.properties.filePath + : undefined; + const handlerPath = typeof resolvedPath === 'string' ? resolvedPath : entry.filePath; + return { routeKey, entry, handlerSymbol, handlerPath }; + }); + handlerContents = await readFileContents( + ctx.repoPath, + routes.map(({ handlerPath }) => handlerPath), + ); - for (const [routeKey, entry] of routeRegistry) { + for (const { routeKey, entry, handlerSymbol, handlerPath } of routes) { const { source: routeSource, method: routeMethod, url } = entry; - const handlerPath = handlerPathFor(routeKey, entry); const content = handlerContents.get(handlerPath); - // A pre-seeded route can never legitimately appear in - // `routeHandlerSymbols`, so a key that does is a route that LOST (#3049). - const handlerSymbolId = preSeededKeys.has(routeKey) - ? undefined - : routeHandlerSymbols.get(routeKey); + const handlerSymbolId = handlerSymbol?.id; const analysisContent = entry.source === DATA_ROUTE_TABLE_SOURCE && content - ? handlerSymbolContent( - content, - handlerSymbolId ? ctx.graph.getNode(handlerSymbolId) : undefined, - ) + ? handlerSymbolContent(content, handlerSymbol?.node) : content; const { responseKeys, errorKeys } = analysisContent @@ -397,6 +396,19 @@ export const routesPhase: PipelinePhase = { confidence: 1.0, reason: routeSource, }); + + // Keep the file edge for existing extractor queries; add the live + // definition edge for explicit handler-level traversal. + if (handlerSymbolId) { + ctx.graph.addRelationship({ + id: generateId('HANDLES_ROUTE', `${handlerSymbolId}->${routeNodeId}`), + sourceId: handlerSymbolId, + targetId: routeNodeId, + type: 'HANDLES_ROUTE', + confidence: 1.0, + reason: routeSource, + }); + } } if (isDev) { diff --git a/gitnexus/src/core/ingestion/pipeline-phases/runner.ts b/gitnexus/src/core/ingestion/pipeline-phases/runner.ts index 0bfc45bd4..da8e4dd8f 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/runner.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/runner.ts @@ -16,23 +16,36 @@ import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js'; import { isDev } from '../utils/env.js'; import { logger } from '../../logger.js'; + +function assertUniquePhaseNames(phases: readonly PipelinePhase[]): void { + const seen = new Set(); + for (const phase of phases) { + if (seen.has(phase.name)) { + throw new Error(`Duplicate phase name: '${phase.name}'`); + } + seen.add(phase.name); + } +} + /** * Validate that the phases form a valid dependency graph (no cycles, all deps present). * Returns phases in topological execution order. + * + * `satisfied` names phases whose results are already available (a deferred + * follow-up run over the same context, #3016). Their edges are dropped rather + * than validated, because they are resolved by definition. */ -function topologicalSort(phases: readonly PipelinePhase[]): PipelinePhase[] { - const phaseMap = new Map(); - for (const phase of phases) { - if (phaseMap.has(phase.name)) { - throw new Error(`Duplicate phase name: '${phase.name}'`); - } - phaseMap.set(phase.name, phase); - } +function topologicalSort( + phases: readonly PipelinePhase[], + satisfied: ReadonlySet = new Set(), +): PipelinePhase[] { + assertUniquePhaseNames(phases); + const phaseMap = new Map(phases.map((p) => [p.name, p])); // Validate all deps exist for (const phase of phases) { for (const dep of phase.deps) { - if (!phaseMap.has(dep)) { + if (!phaseMap.has(dep) && !satisfied.has(dep)) { throw new Error(`Phase '${phase.name}' depends on '${dep}', which is not registered`); } } @@ -43,8 +56,9 @@ function topologicalSort(phases: readonly PipelinePhase[]): PipelinePhase[] { const reverseDeps = new Map(); for (const phase of phases) { - inDegree.set(phase.name, phase.deps.length); - for (const dep of phase.deps) { + const pendingDeps = phase.deps.filter((dep) => !satisfied.has(dep)); + inDegree.set(phase.name, pendingDeps.length); + for (const dep of pendingDeps) { let rev = reverseDeps.get(dep); if (!rev) { rev = []; @@ -143,15 +157,30 @@ function findCyclePath( * * @param phases All phases to execute (order doesn't matter — sorted internally) * @param ctx Shared pipeline context + * @param seed Results of phases that already ran against this same context, + * available to `phases` as dependencies (#3016 deferred derived + * phases). Included in the returned map. * @returns Map of phase name → PhaseResult (all completed phases) */ export async function runPipeline( phases: readonly PipelinePhase[], ctx: PipelineContext, + seed?: ReadonlyMap>, ): Promise>> { + // A seeded phase has already run against this context; re-running it would + // apply its graph writes a second time. "Already ran" is the whole meaning of + // the seed, so honour it here rather than making every caller pre-filter. + const satisfied = new Set(seed?.keys() ?? []); let sorted: PipelinePhase[]; try { - sorted = topologicalSort(phases); + // Duplicate names must be rejected on the caller-supplied list *before* + // seed-filtering. Filtering first would drop a seeded duplicate and let + // `topologicalSort` see a unique name (#3102). + assertUniquePhaseNames(phases); + sorted = topologicalSort( + phases.filter((p) => !satisfied.has(p.name)), + satisfied, + ); } catch (err) { // Emit a terminal 'error' progress event for graph-validation failures // (cycle detected, duplicate phase, missing dep) so CLI/MCP consumers see @@ -171,7 +200,7 @@ export async function runPipeline( } throw err; } - const results = new Map>(); + const results = new Map>(seed); for (const phase of sorted) { const start = Date.now(); diff --git a/gitnexus/src/core/ingestion/pipeline-phases/scan.ts b/gitnexus/src/core/ingestion/pipeline-phases/scan.ts index f8629ea0d..f41c4f84e 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/scan.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/scan.ts @@ -12,6 +12,8 @@ import type { PipelinePhase, PipelineContext } from './types.js'; import { walkRepositoryPaths } from '../filesystem-walker.js'; +import fs from 'node:fs/promises'; +import path from 'node:path'; export interface ScanOutput { scannedFiles: { path: string; size: number }[]; @@ -19,6 +21,104 @@ export interface ScanOutput { totalFiles: number; } +const SPRING_ACTUATOR_ENDPOINT_FILES = new Set([ + 'mappings.json', + 'beans.json', + 'conditions.json', + 'configprops.json', + 'env.json', +]); + +/** + * Runtime snapshots are external analysis inputs, not repository source. When + * a configured input lives below the repository root, exclude it before any + * downstream phase reads file contents. This is especially important for + * Actuator env/configprops payloads: their values must never become File-node + * content or enter FTS merely because the snapshot directory is in the repo. + */ +type CompiledActuatorExclusion = + | { kind: 'repo-root-endpoints' } + | { kind: 'dir'; resolved: string }; + +async function canonicalPath(filePath: string): Promise { + return fs.realpath(filePath).catch(() => path.resolve(filePath)); +} + +async function compileActuatorExclusions( + repoPath: string, + inputPaths: readonly string[], +): Promise<{ repo: string; exclusions: CompiledActuatorExclusion[] }> { + const repo = await canonicalPath(repoPath); + const lexicalRepo = path.resolve(repoPath); + const compiled: CompiledActuatorExclusion[] = []; + const seen = new Set(); + for (const inputPath of inputPaths) { + const lexicalInput = path.resolve(repoPath, inputPath); + const lexicalRelative = path.relative(lexicalRepo, lexicalInput); + const canonicalInput = await canonicalPath(lexicalInput); + const candidateInputs = new Set([canonicalInput]); + if ( + lexicalRelative === '' || + (lexicalRelative !== '..' && + !lexicalRelative.startsWith(`..${path.sep}`) && + !path.isAbsolute(lexicalRelative)) + ) { + // Preserve the configured in-repo alias as an exclusion even when its + // real target is outside the repository. + candidateInputs.add(path.resolve(repo, lexicalRelative)); + } + for (const input of candidateInputs) { + const inputRelativeToRepo = path.relative(repo, input); + if (inputRelativeToRepo === '') { + if (seen.has('')) continue; + seen.add(''); + compiled.push({ kind: 'repo-root-endpoints' }); + continue; + } + if ( + inputRelativeToRepo === '..' || + inputRelativeToRepo.startsWith(`..${path.sep}`) || + path.isAbsolute(inputRelativeToRepo) + ) { + continue; + } + if (seen.has(input)) continue; + seen.add(input); + compiled.push({ kind: 'dir', resolved: input }); + } + } + return { repo, exclusions: compiled }; +} + +function matchesActuatorExclusion( + canonicalRepoPath: string, + filePath: string, + exclusions: readonly CompiledActuatorExclusion[], +): boolean { + if (exclusions.length === 0) return false; + const candidate = path.resolve(canonicalRepoPath, filePath); + for (const exclusion of exclusions) { + if (exclusion.kind === 'repo-root-endpoints') { + const candidateRelativeToRepo = path.relative(canonicalRepoPath, candidate); + if ( + path.dirname(candidateRelativeToRepo) === '.' && + SPRING_ACTUATOR_ENDPOINT_FILES.has(path.basename(candidateRelativeToRepo).toLowerCase()) + ) { + return true; + } + continue; + } + const relative = path.relative(exclusion.resolved, candidate); + if ( + relative === '' || + (relative !== '..' && !relative.startsWith(`..${path.sep}`) && !path.isAbsolute(relative)) + ) { + return true; + } + } + return false; +} + export const scanPhase: PipelinePhase = { name: 'scan', deps: [], @@ -30,15 +130,25 @@ export const scanPhase: PipelinePhase = { message: 'Scanning repository...', }); + const { repo: canonicalRepoPath, exclusions: actuatorExclusions } = + await compileActuatorExclusions(ctx.repoPath, [ + ...(ctx.options?.springActuatorPath === undefined ? [] : [ctx.options.springActuatorPath]), + ...(ctx.options?.springActuatorScanExclusions ?? []), + ]); let scannedFiles; try { scannedFiles = await walkRepositoryPaths(ctx.repoPath, (current, total, filePath) => { const scanProgress = Math.round((current / total) * 15); + const isRuntimeInput = matchesActuatorExclusion( + canonicalRepoPath, + filePath, + actuatorExclusions, + ); ctx.onProgress({ phase: 'extracting', percent: scanProgress, message: 'Scanning repository...', - detail: filePath, + ...(isRuntimeInput ? {} : { detail: filePath }), stats: { filesProcessed: current, totalFiles: total, @@ -56,6 +166,12 @@ export const scanPhase: PipelinePhase = { throw err; } + if (actuatorExclusions.length > 0) { + scannedFiles = scannedFiles.filter( + (file) => !matchesActuatorExclusion(canonicalRepoPath, file.path, actuatorExclusions), + ); + } + const totalFiles = scannedFiles.length; const allPaths = scannedFiles.map((f) => f.path); diff --git a/gitnexus/src/core/ingestion/pipeline-phases/spring-auto-configuration.ts b/gitnexus/src/core/ingestion/pipeline-phases/spring-auto-configuration.ts index a578bfa0d..ba0bad613 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/spring-auto-configuration.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/spring-auto-configuration.ts @@ -21,6 +21,11 @@ import { SPRING_AUTO_CONFIGURATION_IMPORT_REASON, SPRING_AUTO_CONFIGURATION_SYNTHETIC_DESCRIPTION, } from '../frameworks/spring/auto-configuration.js'; +import { + importSpringActuatorRuntime, + MAX_RUNTIME_RECORDS, + type SpringActuatorImportStats, +} from '../frameworks/spring/actuator-runtime.js'; import { isDev } from '../utils/env.js'; import type { StructureOutput } from './structure.js'; import type { PipelineContext, PipelinePhase, PhaseResult } from './types.js'; @@ -74,6 +79,7 @@ export interface SpringAutoConfigurationOutput { readonly metadataFiles: number; readonly autoConfigurations: number; readonly ambiguousAutoConfigurations: number; + readonly actuatorRuntime?: SpringActuatorImportStats; } export function classifySpringAutoConfigurationMetadata( @@ -305,10 +311,25 @@ export const springAutoConfigurationPhase: PipelinePhase 0) { + logger.warn( + `Spring Actuator runtime import reached the ${MAX_RUNTIME_RECORDS.toLocaleString('en-US')}-record limit for: ${actuatorRuntime.truncatedEndpoints.join(', ')}. Runtime evidence is incomplete.`, + ); + } + return { metadataFiles, autoConfigurations, ambiguousAutoConfigurations: ambiguousQualifiedNames.size, + ...(actuatorRuntime === undefined ? {} : { actuatorRuntime }), }; }, }; diff --git a/gitnexus/src/core/ingestion/pipeline-phases/spring-destinations.ts b/gitnexus/src/core/ingestion/pipeline-phases/spring-destinations.ts new file mode 100644 index 000000000..0106eaa98 --- /dev/null +++ b/gitnexus/src/core/ingestion/pipeline-phases/spring-destinations.ts @@ -0,0 +1,565 @@ +/** + * Phase: springDestinations + * + * Materializes Spring async messaging as graph structure: a `Destination` node + * per broker address, `CONSUMES_FROM` from every `@KafkaListener`-family + * handler, and `PUBLISHES_TO` from every messaging-template publish. The + * inbound and outbound facts are captured during parse and survive the parse + * cache; until now nothing read them. + * + * Shaped after `Route` + `HANDLES_ROUTE` in `routes.ts` — a framework overlay + * node keyed by what it names, with the callable pointing at it, down to the + * detail that the key pairs the address with the one dimension that can make + * two same-named things different places: the method for a Route, the broker + * here. + * + * ── THE KEYING RULE, WHICH IS THE POINT OF THE PHASE ───────────────────── + * + * A `Destination` connects two services precisely because both sides mint the + * SAME node id from the SAME address on the SAME broker. That is the whole + * value, and it is also the whole hazard: an address that could not be resolved + * must never be allowed to key a node. + * + * resolved id = generateId('Destination', `
`) `address` present + * unresolved id = generateId('Destination', ) `address` ABSENT + * + * Two unrelated services that each merely write `@KafkaListener(topics = + * "${app.topic}")` have said nothing whatever about each other. Keyed on the + * placeholder text they would land on one node and READ AS CONNECTED, in a + * report, as a fact. A missing edge is visible as a gap; a false one is not. + * + * A status property would not have prevented this, and neither would a second + * label: the id is what merges the nodes, and both sides would still compute + * the same id. Only the KEY prevents it, so an unresolved destination is keyed + * by its source LOCATION — a value no second file can produce. + * + * The same rule governs the `address` PROPERTY, which is the join key a + * cross-repository pass would match on. It is written only when resolved. An + * absent property cannot match another absent property, so the structural + * guarantee survives being read back out of the database. The unresolved + * spelling is kept in `name`, for a human reading the node. + * + * The BROKER is part of the connecting key rather than a reason to withdraw + * one — see {@link destinationNodeKey}, which owns that argument and the + * evidence for it. Two brokers claiming one address is therefore an ordinary + * two-node situation here, exactly like `GET /x` and `POST /x`, and this phase + * needs no vocabulary for it: nothing is being taken away, so there is nothing + * to diagnose. + * + * `name` must never be used to join two destinations, and nothing does — but + * the stronger claim that nothing reads it at all would be false. `Destination` + * is in `VALID_NODE_LABELS`, and `mcp/local/local-backend.ts` resolves a + * symbol with an unlabeled `WHERE n.name = $symName`, so a destination can be + * returned by name like any other node. That is a lookup, not a join: it + * matches a caller-supplied string against one node, never one destination + * against another, so it cannot manufacture the connection this phase exists to + * prevent. + * + * @deps parse, scopeResolution, springConfig + * @reads Spring messaging capture facts, Method/Function nodes, Property nodes + * @writes Destination nodes; CONSUMES_FROM / PUBLISHES_TO / USES edges + */ + +import type { GraphNode, Range } from 'gitnexus-shared'; +import { generateId } from '../../../lib/utils.js'; +import { logger } from '../../logger.js'; +import type { KnowledgeGraph } from '../../graph/types.js'; +import { SPRING_CONFIG_DESCRIPTION } from '../frameworks/spring/config-bindings.js'; +import { + parseSpringStringLiteral, + resolveSpringDestination, + selectConsumerDestinationArguments, + selectProducerDestinationArguments, + type SpringDestinationCandidate, + type SpringDestinationRefusal, + type SpringDestinationResolution, + type SpringDestinationSelection, +} from '../frameworks/spring/destinations.js'; +import { destinationNodeKey } from '../destination-key.js'; +import { getProviderForFile } from '../languages/index.js'; +import { isDev } from '../utils/env.js'; +import type { ModuleConstants } from '../route-extractors/constant-resolver.js'; +import type { ParseOutput } from './parse.js'; +import type { PipelineContext, PipelinePhase, PhaseResult } from './types.js'; +import { getPhaseOutput } from './types.js'; + +export interface SpringDestinationsOutput { + /** Destination nodes keyed by `(broker, address)`, and which therefore + * connect to every other site that named the same address on the same + * broker. */ + readonly resolvedDestinations: number; + /** Destination nodes keyed by source location, and therefore unable to + * connect: the address did not resolve. */ + readonly unresolvedDestinations: number; + /** CONSUMES_FROM + PUBLISHES_TO edges emitted. */ + readonly edges: number; + /** + * Every refusal, counted by reason. This is the phase's real measure: the + * feature is judged on the unresolved FRACTION, and a silent skip would hide + * exactly the number that says whether it works. + */ + readonly refusalsByReason: Readonly>; + /** Destination -> Property provenance edges for `${key}` placeholders. */ + readonly configKeyLinks: number; +} + +/** + * Exact-range index of callable nodes, mirroring the bridge in + * `non-http-handlers.ts`. + * + * A duplicate range maps to `null` rather than to one of its nodes: two + * callables sharing a span means the index cannot say which one publishes, and + * attributing the publish to an arbitrary one of them is the failure this + * phase is least able to detect afterwards. The File-level edge below is the + * fallback, so a `null` here costs precision, not the fact. + */ +function callableOwnersByRange(graph: KnowledgeGraph): ReadonlyMap { + const owners = new Map(); + for (const node of graph.iterNodes()) { + if ( + (node.label !== 'Method' && node.label !== 'Function') || + typeof node.properties.filePath !== 'string' + ) { + continue; + } + const key = `${node.properties.filePath}\0${node.properties.startLine}\0${node.properties.endLine}`; + owners.set(key, owners.has(key) ? null : node); + } + return owners; +} + +/** Capture ranges are 1-based; graph nodes carry 0-based lines. */ +function ownerKey(filePath: string, range: Range): string { + return `${filePath}\0${range.startLine - 1}\0${range.endLine - 1}`; +} + +/** + * Spring configuration `Property` nodes grouped by KEY. + * + * Deliberately a multimap. `spring-config.ts` keys a Property node per FILE + * (`spring-config::`), so one key declared in `application.yml` and + * again in `application-prod.yml` is TWO nodes. Linking to only the first would + * silently pin a destination to an arbitrary profile. An EMPTY match set is + * normal, not an error — a key may be supplied by an environment variable or a + * config server and never appear in a checked-in file at all. + */ +function springConfigPropertiesByKey(graph: KnowledgeGraph): ReadonlyMap { + const byKey = new Map(); + for (const node of graph.iterNodes()) { + if (node.label !== 'Property') continue; + const description = node.properties.description; + if (typeof description !== 'string' || !description.startsWith(SPRING_CONFIG_DESCRIPTION)) { + continue; + } + const key = node.properties.name; + if (typeof key !== 'string' || key === '') continue; + const existing = byKey.get(key); + if (existing === undefined) byKey.set(key, [node.id]); + else existing.push(node.id); + } + return byKey; +} + +/** + * Fold a constant reference against the harvested repo constants, using the + * owning provider's own fold when it declares one. + * + * Only providers that declare `extractModuleConstants` contribute to the table, + * so a language that harvests nothing simply resolves nothing here and the + * cascade records `unresolved-constant`. That is a countable gap, not a wrong + * answer. + */ +function makeConstantResolver( + filePath: string, + repo: ReadonlyMap, +): ((name: string) => string | null) | undefined { + if (repo.size === 0) return undefined; + const fold = getProviderForFile(filePath)?.foldRoutePathOperands; + if (fold === undefined) return undefined; + return (name: string): string | null => fold(filePath, [{ kind: 'ref', name }], repo); +} + +interface DestinationSite { + readonly filePath: string; + /** Owner callable's capture range, when the fact carried one. */ + readonly ownerRange?: Range; + /** Owner callable's scope id. Unique per callable even when the range is + * absent, which is the only reason an unresolved key is site-unique on the + * handler side — see {@link destinationNodeId}. */ + readonly ownerScopeId: string; + readonly candidate: SpringDestinationCandidate; + readonly resolution: SpringDestinationResolution; +} + +/** + * Identity for a destination node. + * + * The connecting key is `(broker, address)` and nothing else — not the file, + * not the site. That is what lets a publisher in one module and a subscriber in + * another meet on one node, which is the entire point, and it is minted by the + * framework-neutral {@link destinationNodeKey} so a non-Spring producer can mint + * the same identity without importing anything Spring. + * + * The broker is IN that key rather than a reason to withhold one. The argument + * for it, including the inferred-broker objection and why the previous rule was + * worse, lives on `destinationNodeKey` — one copy, next to the code that + * decides it. + * + * The site key has to identify the site EXACTLY. It carries the file path, so + * no second file can ever produce it — that is the cross-repository guarantee, + * and nothing added below can weaken it. Everything else in the key is there to + * keep two sites inside ONE file apart, which is the same false identity at a + * smaller scale: + * + * - the owner SCOPE ID, because two callables can start on the same line and + * because `ownerRange` is optional on a handler fact — keyed on the line + * alone, a file whose handlers carried no range collapsed every consumer in + * it onto line 0; + * - the full owner RANGE, which separates two sites the scope id cannot (a + * scope id is stable, but two callables that share one are still two + * callables); + * - the raw TEXT plus argument and element position, because two publishes to + * two different placeholders inside one method share the owner entirely. + * + * The residual: two publishes with identical text at identical argument + * positions inside ONE callable share a node. They are indistinguishable to + * this phase — a producer fact carries the owner's range, not the call's — and + * merging two publishes of the same unreadable address from one method is the + * one collapse that asserts nothing false about anybody. + */ +function destinationNodeId(site: DestinationSite): string { + if (site.resolution.kind === 'resolved') { + return generateId( + 'Destination', + destinationNodeKey(site.candidate.broker, site.resolution.address), + ); + } + const { candidate, ownerRange } = site; + const position = + ownerRange === undefined + ? 'no-range' + : `${ownerRange.startLine}:${ownerRange.startCol}:${ownerRange.endLine}:${ownerRange.endCol}`; + return generateId( + 'Destination', + [ + site.filePath, + site.ownerScopeId, + position, + candidate.role, + candidate.source, + candidate.argIndex, + candidate.elementIndex, + candidate.rawText, + ].join(':'), + ); +} + +/** + * Display spelling for a destination's `name`. + * + * A resolved destination's `name` is its address, bare. An unresolved one keeps + * the source text — but unquoted, so the two are spelled the same way. Keeping + * the quotes on one and not the other made the same value read differently + * depending on whether it resolved, for no gain to the human the property + * exists for. + */ +function destinationDisplayName(rawText: string): string { + return parseSpringStringLiteral(rawText) ?? rawText.trim(); +} + +function edgeReason(candidate: SpringDestinationCandidate): string { + const argument = candidate.argName ?? `arg${candidate.argIndex}`; + const element = `${argument}[${candidate.elementIndex}]`; + const exchange = candidate.exchange === undefined ? '' : ` exchange=${candidate.exchange}`; + return `spring-${candidate.source}:${element}${exchange}`; +} + +export const springDestinationsPhase: PipelinePhase = { + name: 'springDestinations', + // `parse` supplies the file list and the harvested constants; `scopeResolution` + // must have run so the Method/Function nodes exist AND so each provider's + // `applyCaptureSideChannel` has restored the messaging facts onto the main + // thread; `springConfig` must have run so the Property nodes a `${key}` + // placeholder links to are already in the graph. + deps: ['parse', 'scopeResolution', 'springConfig'], + + async execute( + ctx: PipelineContext, + deps: ReadonlyMap>, + ): Promise { + // `allPaths`, NOT `parsedFiles`. On any run with a storage path — which is + // every run of the CLI — worker-produced ParsedFiles are flushed to a disk + // store and `ParseOutput.parsedFiles` comes back EMPTY, with + // scope-resolution streaming them back per language. Iterating it therefore + // found nothing in production while every in-process test passed, because a + // direct pipeline call has no storage path and keeps them in memory. + // + // The fact stores are keyed by file path and are populated by the same + // streaming pass, so the path list is the right cursor for them anyway: it + // does not care how the ParsedFile got to scope resolution. + const { allPaths, moduleConstants } = getPhaseOutput(deps, 'parse'); + const refusalsByReason: Record = {}; + const countRefusal = (reason: SpringDestinationRefusal): void => { + refusalsByReason[reason] = (refusalsByReason[reason] ?? 0) + 1; + }; + + // ── Gather: facts → candidates → resolutions ────────────────────────── + const sites: DestinationSite[] = []; + for (const filePath of allPaths) { + const provider = getProviderForFile(filePath); + const facts = provider?.getSpringMessagingFacts?.(filePath); + if (facts === undefined) continue; + if (facts.handlers.length === 0 && facts.producers.length === 0) continue; + + // Built lazily and reused for the whole file: the fold state behind it is + // per-call, but resolving the provider and checking the table is not free + // and every candidate in the file wants the same closure. + const constant = makeConstantResolver(filePath, moduleConstants); + // The owning language's capability, not its name. In an interpolating + // language `"orders-$env"` is a runtime template rather than an address + // and `"${app.topic}"` is a template rather than a Spring placeholder; + // the resolver needs to know which regime it is in, and this phase must + // not learn which language that is (AGENTS.md — shared ingestion code + // plugs language behaviour in through provider hooks). + const interpolatesStringLiterals = provider?.interpolatesStringLiterals === true; + const record = ( + selection: SpringDestinationSelection, + ownerScopeId: string, + ownerRange: Range | undefined, + ): void => { + for (const refusal of selection.refusals) countRefusal(refusal.reason); + for (const candidate of selection.candidates) { + const resolution = resolveSpringDestination(candidate, { + constant, + interpolatesStringLiterals, + }); + if (resolution.kind === 'unresolved') countRefusal(resolution.reason); + sites.push({ + filePath, + ...(ownerRange === undefined ? {} : { ownerRange }), + ownerScopeId, + candidate, + resolution, + }); + } + }; + + for (const handler of facts.handlers) { + for (const annotation of handler.annotations) { + // A Kotlin use-site target describes a generated property element, not + // the callable, so its arguments are not this handler's. + if (annotation.useSiteTarget !== undefined) continue; + const selection = selectConsumerDestinationArguments(annotation.name, annotation.args); + if (selection === null) continue; + record(selection, String(handler.ownerScopeId), handler.ownerRange); + } + } + for (const producer of facts.producers) { + record( + selectProducerDestinationArguments(producer), + String(producer.ownerScopeId), + producer.ownerRange, + ); + } + } + if (sites.length === 0) { + return { + resolvedDestinations: 0, + unresolvedDestinations: 0, + edges: 0, + refusalsByReason, + configKeyLinks: 0, + }; + } + + // ── Emit ────────────────────────────────────────────────────────────── + // + // A single pass. There used to be a preliminary one that looked for an + // address claimed by two brokers so the emit could withdraw it from both — + // the disagreement had to be known before the first node was minted, + // because re-keying a node after it has grown an edge is the kind of work + // that is easy to get half-right. With the broker in the key there is + // nothing to decide up front: each site's identity is a function of that + // site alone, so no other site can change it and no lookahead is needed. + const owners = callableOwnersByRange(ctx.graph); + const configProperties = springConfigPropertiesByKey(ctx.graph); + const linkedConfigKeys = new Set(); + let resolvedDestinations = 0; + let unresolvedDestinations = 0; + let edges = 0; + let configKeyLinks = 0; + + for (const site of sites) { + const { candidate, resolution } = site; + // The one predicate the rest of the loop is written against: may this + // site's node be keyed by its address, and therefore meet another site on + // it? Exactly when the address resolved. Otherwise the node gets its + // location-based key, no `address` property, and its own file. + const connects = resolution.kind === 'resolved'; + const nodeId = destinationNodeId(site); + const isNew = ctx.graph.getNode(nodeId) === undefined; + if (isNew) { + if (connects) resolvedDestinations += 1; + else unresolvedDestinations += 1; + ctx.graph.addNode({ + id: nodeId, + label: 'Destination', + properties: { + // For a resolved address this equals `address`. For an unresolved + // one it is the UNRESOLVED SPELLING, unquoted, kept so a human + // reading the node sees what the source actually said. Either way + // it is kept out of `address` unless the node connects, so nothing + // joins on it. + // + // Note that `name` is the ADDRESS, not the node key: two nodes on + // one spelling over two brokers share a `name` and differ by id. + // That is deliberate — `name` is for a human, and telling them the + // topic is called `kafka orders` would be a lie. + name: + resolution.kind === 'resolved' + ? resolution.address + : destinationDisplayName(candidate.rawText), + // A CONNECTING destination carries NO location, and that is load + // bearing rather than cosmetic. + // + // It is shared by every site that names the address, so no single + // file identifies it — but more importantly, the incremental + // writeback deletes by location: `deleteNodesForFiles` issues + // `MATCH (n:) WHERE n.filePath IN [...] DETACH DELETE n` for + // every changed file. Stamping the first-seen file here would make + // a shared destination collateral damage whenever THAT file + // changed, and DETACH DELETE would take its edges from every OTHER + // file with it. Those files are not in the write set, so their + // edges would never be rebuilt: a publisher and a subscriber that + // genuinely agree on an address would silently stop being + // connected, depending on which of them the indexer happened to + // walk first. Omitting the property makes the `IN` predicate unable + // to match, so the node survives the writeback and every referrer + // keeps its edge. + // + // The cost used to be the opposite error — a destination whose + // last referrer was deleted lingered as an edgeless orphan until a + // full rebuild — and that is no longer paid. Because the per-file + // predicate can neither remove such a node nor admit a newly + // introduced one, the whole layer is instead delete-alled + // (`deleteAllDestinations`) and re-included graph-wide + // (`isGraphWideNode`) on every incremental writeback. Both halves + // move together: the delete without the re-include drops the layer, + // and the re-include without the delete duplicates every edge. + // + // A NON-CONNECTING destination is the opposite case — it belongs to + // exactly one site, its id already says so, and it SHOULD be + // deleted and re-created with its file. + // `''`, not absent: `NodeProperties.filePath` is required, and the + // empty string is the established spelling for a node with no file + // (`pipeline-phases/communities.ts` does the same). It is equally + // unmatchable by the `IN` predicate and the CSV writes it as an + // empty field, which COPY loads as NULL. + ...(connects + ? { filePath: '' } + : { + filePath: site.filePath, + ...(site.ownerRange === undefined + ? {} + : { + startLine: site.ownerRange.startLine - 1, + endLine: site.ownerRange.endLine - 1, + }), + }), + // `address` IS THE JOIN KEY and is written ONLY when the node + // connects. See the module header: an absent property cannot match + // another absent property, so the structural guarantee survives + // being read back out of the database. + // + // `resolution` always says how the node got here — the provenance + // of a real address, or the named refusal that stopped one. Every + // value in the column now comes from the resolver's own closed + // vocabulary, because the phase no longer has a verdict of its own + // to record: nothing is withdrawn here. + ...(resolution.kind === 'resolved' + ? { address: resolution.address, resolution: resolution.via } + : { resolution: resolution.reason }), + ...(resolution.kind === 'unresolved' && resolution.configKey !== undefined + ? { configKey: resolution.configKey } + : {}), + // The `${key:default}` default text. Kept because the source wrote + // it and throwing it away would make an overridable default + // indistinguishable from a bare `${key}`; NOT an address and never + // part of the id, because configuration can override it and this + // graph cannot see whether it did. + ...(resolution.kind === 'unresolved' && resolution.configDefault !== undefined + ? { configDefault: resolution.configDefault } + : {}), + broker: candidate.broker, + }, + }); + } + + // Link a placeholder's KEY to the configuration entries that could supply + // it — `${key}` and `${key:default}` alike, since a default changes + // nothing about where the real value comes from. This is PROVENANCE, not + // resolution: the node stays unresolved and keeps its location-based id + // even when Property nodes are found, because the VALUE is still not in + // the graph and letting a Property sighting upgrade the node would + // reintroduce the false connection the keying rule exists to prevent. + // `${}` names no key at all and reports none, so nothing is ever looked + // up under the empty string. + if (resolution.kind === 'unresolved' && resolution.configKey !== undefined) { + for (const propertyId of configProperties.get(resolution.configKey) ?? []) { + const linkId = `${nodeId}->${propertyId}`; + if (linkedConfigKeys.has(linkId)) continue; + linkedConfigKeys.add(linkId); + ctx.graph.addRelationship({ + id: generateId('USES', linkId), + sourceId: nodeId, + targetId: propertyId, + type: 'USES', + confidence: 1.0, + reason: `spring-destination:config-key:${resolution.configKey}`, + }); + configKeyLinks += 1; + } + } + + // One edge per address. An array-valued `topics` really does subscribe to + // several places, and each gets its own edge rather than a group node, so + // "who reads from `a`" stays one hop. `reason` carries which argument and + // which element it came from. + const type = candidate.role === 'consumer' ? 'CONSUMES_FROM' : 'PUBLISHES_TO'; + const owner = + site.ownerRange === undefined + ? undefined + : (owners.get(ownerKey(site.filePath, site.ownerRange)) ?? undefined); + // Prefer the callable; fall back to its File when the owner is unknown or + // ambiguous. Unlike `routes.ts` this does NOT emit both — there is no + // legacy File-level consumer to keep working here, and a second edge per + // publish would double the async surface of the graph for no query. + const sourceId = owner?.id ?? generateId('File', site.filePath); + if (owner === undefined && ctx.graph.getNode(sourceId) === undefined) continue; + const reason = edgeReason(candidate); + ctx.graph.addRelationship({ + id: generateId(type, `${sourceId}->${nodeId}:${reason}`), + sourceId, + targetId: nodeId, + type, + confidence: 1.0, + reason, + }); + edges += 1; + } + + if (isDev) { + logger.info( + `📮 Spring destinations: ${resolvedDestinations} resolved, ${unresolvedDestinations} unresolved, ${edges} edges`, + ); + } + + return { + resolvedDestinations, + unresolvedDestinations, + edges, + refusalsByReason, + configKeyLinks, + }; + }, +}; diff --git a/gitnexus/src/core/ingestion/pipeline.ts b/gitnexus/src/core/ingestion/pipeline.ts index 050822e87..00892d07c 100644 --- a/gitnexus/src/core/ingestion/pipeline.ts +++ b/gitnexus/src/core/ingestion/pipeline.ts @@ -35,6 +35,7 @@ import { springConfigPhase, springAutoConfigurationPhase, springAopPhase, + springDestinationsPhase, springAopInheritancePhase, pruneLocalSymbolsPhase, taintSummariesPhase, @@ -46,6 +47,7 @@ import { PhaseRegistry, type ScopeResolutionOutput, type PipelinePhase, + type PipelineContext, type CommunitiesOutput, type ProcessesOutput, } from './pipeline-phases/index.js'; @@ -58,6 +60,20 @@ export interface PipelineOptions { * to retain those nodes under `skipGraphPhases`. */ skipGraphPhases?: boolean; + /** + * Skip only Leiden community detection and process/flow extraction (#3016). + * MRO/DI still run. Used on warm incremental analyze so persisted + * Community/Process rows can be kept instead of wipe+rewrite. + */ + skipDerivedGraphPhases?: boolean; + /** + * Explicit local Spring Boot Actuator snapshot input. Accepts a directory + * containing endpoint-named JSON files or a JSON bundle keyed by endpoint. + * Undefined keeps runtime enrichment completely disabled. + */ + springActuatorPath?: string; + /** Repo-relative Actuator inputs retained only for a cleanup scan. */ + springActuatorScanExclusions?: readonly string[]; /** Per-advice Spring AOP candidate inspection cap. `0` disables this cap. */ springAopMaxCandidateInspectionsPerAdvice?: number; /** Aggregate Spring AOP candidate inspection cap for one analysis. `0` disables this cap. */ @@ -272,7 +288,8 @@ export interface PipelineOptions { * Phase dependency graph: * * scan → structure → [springConfig, markdown, cobol] → parse → [routes, tools, orm] - * → crossFile → scopeResolution → [springAutoConfiguration, springAop] → pruneLocalSymbols + * → crossFile → scopeResolution → [springAutoConfiguration, springAop, + * springDestinations] → pruneLocalSymbols * → mro → springAopInheritance → di → communities → processes * * To add a new phase: create a file in pipeline-phases/, export the phase @@ -301,6 +318,11 @@ export function buildPhaseList(options?: PipelineOptions): PipelinePhase[] { .register(scopeResolutionPhase) .register(springAutoConfigurationPhase) .register(springAopPhase) + // Async messaging overlay. Must follow scopeResolution twice over: the + // owner Method/Function nodes have to exist, and each provider's + // `applyCaptureSideChannel` has to have restored the messaging facts onto + // the main thread. It also reads the Property nodes springConfig emits. + .register(springDestinationsPhase) .register(pruneLocalSymbolsPhase) // M4 (#2084): interprocedural taint fixpoint — the first real opt-in // pdg-gated phase. Off ⇒ absent ⇒ byte-identical graph. No always-on @@ -310,8 +332,12 @@ export function buildPhaseList(options?: PipelineOptions): PipelinePhase[] { .register(mroPhase, { enabledWhen: (o) => !o.skipGraphPhases }) .register(springAopInheritancePhase, { enabledWhen: (o) => !o.skipGraphPhases }) .register(diPhase, { enabledWhen: (o) => !o.skipGraphPhases }) - .register(communitiesPhase, { enabledWhen: (o) => !o.skipGraphPhases }) - .register(processesPhase, { enabledWhen: (o) => !o.skipGraphPhases }) + .register(communitiesPhase, { + enabledWhen: (o) => !o.skipGraphPhases && o.skipDerivedGraphPhases !== true, + }) + .register(processesPhase, { + enabledWhen: (o) => !o.skipGraphPhases && o.skipDerivedGraphPhases !== true, + }) // Normalize a missing options object once here so phase predicates above // take a required PipelineOptions and need no `?.` guard (#2080 review S1). .build(options ?? {}) @@ -351,18 +377,19 @@ export const runPipelineFromRepo = async ( } const phases = buildPhaseList(options); + const ctx: PipelineContext = { + repoPath, + graph: graphEmitSink ?? graph, + onProgress, + options, + pipelineStart, + graphEmit: graphEmitSink, + }; let graphEmitManifest: GraphEmitManifest | undefined; let results; try { - results = await runPipeline(phases, { - repoPath, - graph: graphEmitSink ?? graph, - onProgress, - options, - pipelineStart, - graphEmit: graphEmitSink, - }); + results = await runPipeline(phases, ctx); graphEmitManifest = graphEmitSink?.finalize(); } finally { // Release per-pair fds when the pipeline threw before finalize ran. @@ -412,7 +439,7 @@ export const runPipelineFromRepo = async ( }, }); - return { + const result: PipelineResult = { // The RAW graph, deliberately — NOT `graphEmitSink`. Phases above received // the sink so their reads are complete, but `loadGraphToLbug` feeds this to // `streamAllCSVsToDisk`, and the sink's complete iterator would then emit @@ -434,4 +461,39 @@ export const runPipelineFromRepo = async ( pdgEmitManifest, propertyInference, }; + + // #3016: hand back a way to run the derived phases `skipDerivedGraphPhases` + // held back. Which phases those are is answered by re-asking the registry + // with only that flag cleared — the one form of the question that stays + // correct when a different predicate (`skipGraphPhases`) also disables them, + // since then they are absent for a reason a deferred run cannot fix and the + // filter yields nothing. The sink guard mirrors the `graph` note above: a + // streaming run is a full rebuild, which never sets the skip flag, so an + // active sink here means the two got combined by mistake — and deferred + // phases writing into a finalized sink would emit past its manifest. + const deferredDerivedPhases = + options?.skipDerivedGraphPhases === true && graphEmitSink === undefined + ? buildPhaseList({ ...options, skipDerivedGraphPhases: false }).filter( + (p) => (p.name === 'communities' || p.name === 'processes') && !results.has(p.name), + ) + : []; + + if (deferredDerivedPhases.length > 0) { + result.runDeferredDerivedPhases = async () => { + const derived = await runPipeline(deferredDerivedPhases, ctx, results); + // Presence-checked for the same reason as the block above: a phase the + // registry filtered out is absent, and `getPhaseOutput` throws on absent. + if (derived.has('communities')) { + result.communityResult = getPhaseOutput( + derived, + 'communities', + ).communityResult; + } + if (derived.has('processes')) { + result.processResult = getPhaseOutput(derived, 'processes').processResult; + } + }; + } + + return result; }; diff --git a/gitnexus/src/core/ingestion/route-extractors/constant-resolver.ts b/gitnexus/src/core/ingestion/route-extractors/constant-resolver.ts index 9b70b458c..8605f5a02 100644 --- a/gitnexus/src/core/ingestion/route-extractors/constant-resolver.ts +++ b/gitnexus/src/core/ingestion/route-extractors/constant-resolver.ts @@ -66,6 +66,30 @@ export interface ModuleConstants { readonly literals: Map; readonly exprs: Map; readonly imports: Map; + /** + * On-demand (wildcard) import specifiers whose bound member names could not + * be enumerated at extract time — Java `import static a.b.C.*;`, Python + * `from m import *`. The agnostic fold never reads this (it has no way to + * enumerate a target module's exports); a language binding materializes the + * promised bindings from a repo-wide map after extraction — see + * `expandJavaWildcardStaticImports` in the Java binding — so they resolve + * through the plain `imports` path with no special cases in the fold. + */ + readonly wildcardImports?: readonly string[]; +} + +const NO_UNFOLDABLE_DECLARATIONS: ReadonlySet = new Set(); + +/** + * Declaration keys a language extractor found but could not fold. Java and + * Kotlin both use this metadata to keep lower-priority imports from replacing + * a real local declaration; other producers simply return the empty set. + */ +export function unfoldableDeclarationsOf(mc: ModuleConstants | undefined): ReadonlySet { + const declarations = ( + mc as (ModuleConstants & { readonly unfoldableDeclarations?: unknown }) | undefined + )?.unfoldableDeclarations; + return declarations instanceof Set ? declarations : NO_UNFOLDABLE_DECLARATIONS; } /** Repo-wide map: unique file key (e.g. `app/constants.py`) → that file's diff --git a/gitnexus/src/core/ingestion/route-extractors/java-const-resolver.ts b/gitnexus/src/core/ingestion/route-extractors/java-const-resolver.ts index 0ca5e71e2..95f9f4710 100644 --- a/gitnexus/src/core/ingestion/route-extractors/java-const-resolver.ts +++ b/gitnexus/src/core/ingestion/route-extractors/java-const-resolver.ts @@ -49,6 +49,7 @@ import { type ModuleConstants, type Operand, type RepoConstants, + unfoldableDeclarationsOf, } from './constant-resolver.js'; export type { @@ -58,6 +59,11 @@ export type { RepoConstants, } from './constant-resolver.js'; +export interface JavaModuleConstants extends ModuleConstants { + /** Declaration keys whose initializer exists but cannot be folded. */ + readonly unfoldableDeclarations: ReadonlySet; +} + /** * Cheap content gate: can this Java file DEFINE a string constant that a route * annotation might reference? @@ -140,10 +146,16 @@ export const resolveJavaImport: ImportResolver = (_importingFileKey, moduleSpec, const asPath = moduleSpec.replace(/\./g, '/'); const classFile = `${asPath}.java`; + // Compare in POSIX space: on Windows the repo keys can carry backslash + // separators, which would otherwise never match a '/'-joined class file + // (observed as 675 calls with zero hits on a backslash-keyed repo). + const toPosix = (p: string): string => p.replace(/\\/g, '/'); + // Exact package-path suffix match, unique or nothing. let hit: string | null = null; for (const key of repoKeys) { - if (key === classFile || key.endsWith(`/${classFile}`)) { + const posixKey = toPosix(key); + if (posixKey === classFile || posixKey.endsWith(`/${classFile}`)) { if (hit !== null) return null; // 2+ modules carry this FQN — unresolvable hit = key; } @@ -264,29 +276,53 @@ export function parseJavaConstOperands( * Last-wins in source order; a non-foldable rebind (`X = compute()`) drops X * to unresolvable rather than keeping a stale literal. */ -export function extractJavaModuleConstants(tree: Parser.Tree): ModuleConstants { +export function extractJavaModuleConstants(tree: Parser.Tree): JavaModuleConstants { const literals = new Map(); const exprs = new Map(); const imports = new Map(); + const unfoldableDeclarations = new Set(); + + // On-demand static imports (`import static a.b.C.*`) — expanded post-map + // by expandJavaWildcardStaticImports below. + const wildcardImports: string[] = []; // Pass 1: imports (both shapes). const walkImports = (node: Parser.SyntaxNode): void => { if (node.type === 'import_declaration') { // import a.b.C; | import static a.b.C; | import static a.b.C.F; + // import static a.b.C.*; — asterisk is a sibling of scoped_identifier + // (tree-sitter-java), not the last path segment. Same detection as + // import-decomposer.ts (`static-wildcard`). const isStatic = node.children.some((c) => c.type === 'static' && c.text === 'static'); - const scoped = node.children.find((c) => c.type === 'scoped_identifier'); + const isWildcard = node.children.some((c) => c.type === 'asterisk'); + const scoped = + node.children.find((c) => c.type === 'scoped_identifier') ?? + node.children.find((c) => c.type === 'identifier'); if (scoped) { const text = scoped.text; - const lastDot = text.lastIndexOf('.'); - const fqn = text.slice(0, lastDot); - const name = text.slice(lastDot + 1); - if (isStatic) { - // import static a.b.C.F → local F from module a.b.C, original F. - imports.set(name, { module: fqn, originalName: name }); + if (isStatic && isWildcard) { + // Class FQN only — members are materialized post-map. + if (text.length > 0 && !wildcardImports.includes(text)) { + wildcardImports.push(text); + } } else { - // import a.b.C → module IS the class FQN; originalName is the class - // simple name. resolveJavaImport maps `a.b.C` → `a/b/C.java`. - imports.set(name, { module: text, originalName: name }); + const lastDot = text.lastIndexOf('.'); + const fqn = text.slice(0, lastDot); + const name = text.slice(lastDot + 1); + if (isStatic) { + // Preserve the declaring type in the target lookup. Constants from + // multiple types share one file-level map, so a bare `F` could + // otherwise resolve to a sibling type's flattened field. + const declaringClass = fqn.slice(fqn.lastIndexOf('.') + 1); + imports.set(name, { + module: fqn, + originalName: `${declaringClass}.${name}`, + }); + } else { + // import a.b.C → module IS the class FQN; originalName is the class + // simple name. resolveJavaImport maps `a.b.C` → `a/b/C.java`. + imports.set(name, { module: text, originalName: name }); + } } } } @@ -343,6 +379,7 @@ export function extractJavaModuleConstants(tree: Parser.Tree): ModuleConstants { if (operands === null) { literals.delete(name); exprs.delete(name); + unfoldableDeclarations.add(name); // …and the static IMPORT of the same simple name. A local // `static final String` shadows `import static a.b.C.PATH` inside // that class (JLS 6.4.1), so the correct answer for a non-foldable @@ -363,9 +400,12 @@ export function extractJavaModuleConstants(tree: Parser.Tree): ModuleConstants { if (qname) { literals.delete(qname); exprs.delete(qname); + unfoldableDeclarations.add(qname); } continue; } + unfoldableDeclarations.delete(name); + if (qname) unfoldableDeclarations.delete(qname); const literalValue = operands.length === 1 && operands[0].kind === 'literal' ? (operands[0] as { value: string }).value @@ -436,7 +476,13 @@ export function extractJavaModuleConstants(tree: Parser.Tree): ModuleConstants { }; walkTypes(tree.rootNode, false); - return { literals, exprs, imports: imports as Map }; + return { + literals, + exprs, + imports: imports as Map, + wildcardImports, + unfoldableDeclarations, + }; } /** @@ -578,6 +624,7 @@ function computeJavaFold( if (literal !== undefined) return literal; const expr = mc.exprs.get(name); if (expr !== undefined) return foldOperands(fileKey, expr, state, depth + 1); + if (unfoldableDeclarationsOf(mc).has(name)) return null; const imp = mc.imports.get(name); if (imp !== undefined) { const targetFile = resolveJavaImport(fileKey, imp.module, constantKeys); @@ -628,3 +675,117 @@ export function foldJavaOperands( const out = foldOperands(fileKey, operands, newFoldState(repo), 0); return out === '' ? null : out; } + +/** + * Constant-defining file keys used by Java import resolution. + * + * Build once per repo pass. Recomputing this set for every wildcard-importing + * controller makes expansion quadratic in controller count. + */ +export function buildJavaConstantKeys(repo: RepoConstants): ReadonlySet { + const repoKeys = new Set(); + for (const [key, target] of repo) { + if (target.literals.size > 0 || target.exprs.size > 0) repoKeys.add(key); + } + return repoKeys; +} + +/** Direct static members owned by `classSimple`, excluding nested-type members. */ +function directJavaMembers(target: ModuleConstants, classSimple: string): Set { + const members = new Set(); + const prefix = `${classSimple}.`; + for (const map of [target.literals, target.exprs]) { + for (const key of map.keys()) { + if (!key.startsWith(prefix)) continue; + const member = key.slice(prefix.length); + if (member.length > 0 && !member.includes('.')) members.add(member); + } + } + return members; +} + +export interface JavaConstantIndex { + readonly keys: ReadonlySet; + /** Every resolvable path suffix as a dotted module name; null means ambiguous. */ + readonly byModule: ReadonlyMap; + /** Direct members by constant-defining file, built once for all importers. */ + readonly membersByFile: ReadonlyMap>; +} + +/** + * Build all Java import suffixes once, turning repeated wildcard target lookup + * from O(importers × constant files) into O(path segments + importers). + */ +export function buildJavaConstantIndex(repo: RepoConstants): JavaConstantIndex { + const keys = buildJavaConstantKeys(repo); + const byModule = new Map(); + const membersByFile = new Map>(); + for (const key of keys) { + const normalized = key.replace(/\\/g, '/').replace(/^\.\//, ''); + if (!normalized.endsWith('.java')) continue; + const segments = normalized.slice(0, -'.java'.length).split('/'); + const classSimple = segments[segments.length - 1]; + const constants = repo.get(key); + if (constants) membersByFile.set(key, directJavaMembers(constants, classSimple)); + for (let start = 0; start < segments.length; start++) { + const moduleName = segments.slice(start).join('.'); + const existing = byModule.get(moduleName); + if (existing === undefined) byModule.set(moduleName, key); + else if (existing !== key) byModule.set(moduleName, null); + } + } + return { keys, byModule, membersByFile }; +} + +export function expandJavaWildcardStaticImports( + mc: ModuleConstants, + _fileKey: string, + repo: RepoConstants, + index: JavaConstantIndex = buildJavaConstantIndex(repo), +): ModuleConstants { + const wildcards = mc.wildcardImports; + if (!wildcards || wildcards.length === 0) return mc; + // Resolve targets against constant-DEFINING files only. Ingestion's harvest + // also admits import-only files; measuring uniqueness over every key made + // a duplicate empty FQN floor ingestion to skip while group still folded + // (#2980 R4). + const explicitImports = new Set(mc.imports.keys()); + const pending = new Map(); + for (const fqn of wildcards) { + const targetKey = index.byModule.get(fqn) ?? null; + if (targetKey === null) continue; + const classSimple = fqn.slice(fqn.lastIndexOf('.') + 1); + const members = index.membersByFile.get(targetKey); + if (!members) continue; + for (const name of members) { + // Same-file declarations and explicit imports have higher precedence + // than on-demand imports. An unfoldable declaration must remain a skip, + // not be resurrected from a wildcard target. + if ( + mc.literals.has(name) || + mc.exprs.has(name) || + unfoldableDeclarationsOf(mc).has(name) || + explicitImports.has(name) + ) { + continue; + } + const binding = { module: fqn, originalName: `${classSimple}.${name}` }; + const previous = pending.get(name); + if (previous === undefined) pending.set(name, binding); + else if (previous !== null && previous.module !== fqn) pending.set(name, null); + } + } + for (const [name, binding] of pending) { + if (binding !== null) mc.imports.set(name, binding); + } + return mc; +} + +/** Prepare every Java constants entry with one shared suffix index. */ +export function prepareJavaRouteConstants(repo: RepoConstants): JavaConstantIndex { + const index = buildJavaConstantIndex(repo); + for (const [fileKey, mc] of repo) { + expandJavaWildcardStaticImports(mc, fileKey, repo, index); + } + return index; +} diff --git a/gitnexus/src/core/ingestion/route-extractors/kotlin-const-resolver.ts b/gitnexus/src/core/ingestion/route-extractors/kotlin-const-resolver.ts index ceb1f92e1..311612e2d 100644 --- a/gitnexus/src/core/ingestion/route-extractors/kotlin-const-resolver.ts +++ b/gitnexus/src/core/ingestion/route-extractors/kotlin-const-resolver.ts @@ -95,6 +95,7 @@ * @PostMapping(ApiPaths.ORDERS) // qualified * @PostMapping(com.example.app.api.ApiPaths.ORDERS) // FQN-qualified * @PostMapping(ORDERS) // single-name import + * @PostMapping(ORDERS) after import com.example.api.* // package-star import * @PostMapping(ApiPaths.BASE + "/orders") // inline concat * * Which ANNOTATIONS count as routes is a separate question this module has no @@ -122,22 +123,18 @@ * * WHERE THIS IS WIRED. Java reaches its binding from BOTH layers: the group * extractor (`group/extractors/http-patterns/java.ts`) and the ingestion - * provider (`languages/java.ts`, via `extractModuleConstants` + - * `foldRoutePathOperands`). Kotlin is wired into the GROUP layer only, because - * the ingestion fold in `pipeline-phases/parse-impl.ts` runs exclusively over - * `decoratorRoutes` — and `languages/kotlin.ts` declares no - * `extractDecoratorRoutes`, since the ingestion Spring extractor (`spring.ts`) - * is bound to `tree-sitter-java` and its node types. Declaring the constant - * hooks on the Kotlin provider today would harvest a map on every Kotlin file - * that nothing consumes. An ingestion-side Kotlin route extractor is the - * prerequisite; when it lands, this binding is what its provider hooks should - * point at, and no change here is needed. + * provider (`languages/java.ts`). Kotlin now does the same: the group extractor + * (`group/extractors/http-patterns/kotlin.ts`) plus `languages/kotlin.ts` + * (`extractDecoratorRoutes`, `extractModuleConstants`, `foldRoutePathOperands`). + * The dedicated ingestion walker is `route-extractors/kotlin-spring.ts`; it + * does not reuse Java `spring.ts`. */ import type Parser from 'tree-sitter'; import { unquoteSpringLiteral } from './spring-shared.js'; import { MAX_FOLD_LENGTH, + unfoldableDeclarationsOf, type ImportBinding, type ModuleConstants, type Operand, @@ -150,6 +147,7 @@ export type { Operand, RepoConstants, } from './constant-resolver.js'; +export { unfoldableDeclarationsOf } from './constant-resolver.js'; /** * What {@link extractKotlinModuleConstants} returns: the agnostic @@ -177,6 +175,8 @@ export interface KotlinModuleConstants extends ModuleConstants { readonly packageName: string; /** Declaration keys whose initializer cannot be folded. */ readonly unfoldableDeclarations: ReadonlySet; + /** Top-level properties and types that shadow lower-priority star imports. */ + readonly topLevelDeclarations: ReadonlySet; } /** @@ -189,15 +189,12 @@ function declaredPackageOf(mc: ModuleConstants | undefined): string | null { return typeof declared === 'string' ? declared : null; } -const NO_UNFOLDABLE_DECLARATIONS: ReadonlySet = new Set(); +const NO_TOP_LEVEL_DECLARATIONS: ReadonlySet = new Set(); -/** - * Kotlin declaration keys known to exist but not fold, or an empty set when - * `mc` came from another language binding. - */ -export function unfoldableDeclarationsOf(mc: ModuleConstants | undefined): ReadonlySet { - const declarations = (mc as KotlinModuleConstants | undefined)?.unfoldableDeclarations; - return declarations instanceof Set ? declarations : NO_UNFOLDABLE_DECLARATIONS; +/** Kotlin top-level names known to shadow package-star imports. */ +function topLevelDeclarationsOf(mc: ModuleConstants | undefined): ReadonlySet { + const declarations = (mc as KotlinModuleConstants | undefined)?.topLevelDeclarations; + return declarations instanceof Set ? declarations : NO_TOP_LEVEL_DECLARATIONS; } /** Source extensions a Kotlin declaration can live in. */ @@ -443,6 +440,16 @@ export interface KotlinConstantIndex { /** Does this file contribute declarations to Kotlin import ambiguity? */ function contributesKotlinConstants(mc: ModuleConstants): boolean { + return ( + mc.literals.size > 0 || + mc.exprs.size > 0 || + unfoldableDeclarationsOf(mc).size > 0 || + topLevelDeclarationsOf(mc).size > 0 + ); +} + +/** Foldable or explicitly unfoldable constants that must live in index projections. */ +function hasIndexedConstants(mc: ModuleConstants): boolean { return mc.literals.size > 0 || mc.exprs.size > 0 || unfoldableDeclarationsOf(mc).size > 0; } @@ -459,6 +466,7 @@ function topLevelDeclarationNames(mc: ModuleConstants): Set { const dot = key.indexOf('.'); names.add(dot < 0 ? key : key.slice(0, dot)); } + for (const name of topLevelDeclarationsOf(mc)) names.add(name); return names; } @@ -509,6 +517,65 @@ export function buildKotlinConstantIndex(repo: RepoConstants): KotlinConstantInd return { repo, constantKeys, byPackage: mutablePackages, byFqn }; } +/** Read-only one-entry overlay without copying the repo-wide constant map. */ +class KotlinConstantOverlay implements ReadonlyMap { + readonly [Symbol.toStringTag] = 'KotlinConstantOverlay'; + + constructor( + private readonly base: RepoConstants, + private readonly overlayKey: string, + private readonly overlayValue: ModuleConstants, + ) {} + + get size(): number { + return this.base.size + (this.base.has(this.overlayKey) ? 0 : 1); + } + + get(key: string): ModuleConstants | undefined { + return key === this.overlayKey ? this.overlayValue : this.base.get(key); + } + + has(key: string): boolean { + return key === this.overlayKey || this.base.has(key); + } + + *entries(): MapIterator<[string, ModuleConstants]> { + let replaced = false; + for (const [key, value] of this.base) { + if (key === this.overlayKey) { + replaced = true; + yield [key, this.overlayValue]; + } else { + yield [key, value]; + } + } + if (!replaced) yield [this.overlayKey, this.overlayValue]; + } + + *keys(): MapIterator { + for (const [key] of this.entries()) yield key; + } + + *values(): MapIterator { + for (const [, value] of this.entries()) yield value; + } + + [Symbol.iterator](): MapIterator<[string, ModuleConstants]> { + return this.entries(); + } + + forEach( + callbackfn: ( + value: ModuleConstants, + key: string, + map: ReadonlyMap, + ) => void, + thisArg?: unknown, + ): void { + for (const [key, value] of this.entries()) callbackfn.call(thisArg, value, key, this); + } +} + /** * Add one scan-time file without rebuilding the base index when it only imports * constants. A newly discovered declaration is rare and rebuilds once for that @@ -519,10 +586,15 @@ export function overlayKotlinConstantIndex( fileKey: string, mc: ModuleConstants, ): KotlinConstantIndex { + // Same-file shadows are read from `mc` itself. Rebuild only when this key's + // constant projections would change; a controller with a class name but no + // foldable constants stays on the overlay so scan stays linear. + const existing = index.repo.get(fileKey); + if (!hasIndexedConstants(mc) && (!existing || !hasIndexedConstants(existing))) { + return { ...index, repo: new KotlinConstantOverlay(index.repo, fileKey, mc) }; + } const repo = new Map(index.repo); - const replacing = repo.has(fileKey); repo.set(fileKey, mc); - if (!replacing && !contributesKotlinConstants(mc)) return { ...index, repo }; return buildKotlinConstantIndex(repo); } @@ -858,9 +930,9 @@ function declaredPackage(root: Parser.SyntaxNode): string { } /** - * Extract the declared package, file-level string constants and import bindings - * of one parsed Kotlin file into the {@link KotlinModuleConstants} shape the - * resolver consumes. + * Extract the declared package, file-level string constants, named imports and + * package-star import scopes of one parsed Kotlin file into the + * {@link KotlinModuleConstants} shape the resolver consumes. * * Constants come from the three carriers Kotlin allows a caller to reach without * an instance: file top level, `object` members, and `companion object` members. @@ -907,21 +979,25 @@ export function extractKotlinModuleConstants(tree: Parser.Tree): KotlinModuleCon const literals = new Map(); const exprs = new Map(); const imports = new Map(); + const wildcardImports: string[] = []; const unfoldableDeclarations = new Set(); + const topLevelDeclarations = new Set(); // Pass 1: imports. const walkImports = (node: Parser.SyntaxNode): void => { if (node.type === 'import_header') { - // `import a.b.*` binds no single name — nothing to key the fold on, and - // guessing which package member a bare reference came from is exactly the - // wrong answer. Skipped, so such a reference floors to skip. const isWildcard = node.children.some((c) => c.type === 'wildcard_import'); const identifier = node.children.find((c) => c.type === 'identifier'); - if (!isWildcard && identifier) { + if (identifier) { const segments = identifier.namedChildren .filter((c) => c.type === 'simple_identifier') .map((c) => unquoteKotlinIdentifier(c.text)); - if (segments.length >= 2) { + if (isWildcard) { + // Package star (`pkg.*`) or classifier star (`Type.*`). Resolution + // decides which reading the specifier actually names. + const scope = segments.join('.'); + if (scope.length > 0 && !wildcardImports.includes(scope)) wildcardImports.push(scope); + } else if (segments.length >= 2) { const spec = segments.join('.'); const originalName = segments[segments.length - 1]; const aliasNode = node.children @@ -999,6 +1075,24 @@ export function extractKotlinModuleConstants(tree: Parser.Tree): KotlinModuleCon const withScope = (scope: string | null, scopes: readonly string[]): readonly string[] => scope === null || scopes[0] === scope ? scopes : [scope, ...scopes]; + // Package-star imports have lower priority than declarations in this file, + // including declarations the string-constant extractor intentionally does + // not harvest (a `var`, plain class, or unfoldable property). Record their + // names separately so a star import cannot turn one of those shadows into a + // false route constant. + for (const child of tree.rootNode.children ?? []) { + if (child.type === 'property_declaration') { + const declaration = child.children.find((c) => c.type === 'variable_declaration'); + const name = declaration?.namedChildren.find((c) => c.type === 'simple_identifier'); + if (name) topLevelDeclarations.add(unquoteKotlinIdentifier(name.text)); + continue; + } + if (child.type === 'object_declaration' || child.type === 'class_declaration') { + const name = child.children.find((c) => c.type === 'type_identifier'); + if (name) topLevelDeclarations.add(unquoteKotlinIdentifier(name.text)); + } + } + const walkDeclarations = ( node: Parser.SyntaxNode, enclosingType: string | null, @@ -1103,8 +1197,10 @@ export function extractKotlinModuleConstants(tree: Parser.Tree): KotlinModuleCon literals, exprs, imports, + wildcardImports, packageName: declaredPackage(tree.rootNode), unfoldableDeclarations, + topLevelDeclarations, }; } @@ -1211,6 +1307,87 @@ function resolveImportedName( return resolveWithState(owner.fileKey, `${owner.localName}.${imp.originalName}`, state, depth); } +/** + * Resolve one name contributed by Kotlin star imports. + * + * Star imports have lower priority than local declarations and explicit + * imports; callers enforce that ordering before reaching this helper. A name + * must identify one declaration across every imported scope. Two stars + * exporting the same name, a duplicated FQN, or a package and a classifier + * that disagree on the target, floor to null rather than guessing. + * + * A specifier is tried as a package (`import pkg.*`) and as a classifier + * (`import Type.*` — object, class, or companion members). Kotlin allows both. + */ +function resolveKotlinWildcardImportTarget( + mc: ModuleConstants, + name: string, + index: KotlinConstantIndex, +): KotlinImportTarget | null { + const scopes = mc.wildcardImports; + if (!scopes || scopes.length === 0) return null; + + let resolved: KotlinImportTarget | null = null; + for (const rawScope of scopes) { + const scope = unquoteKotlinDottedName(rawScope); + const candidates: KotlinImportTarget[] = []; + + const bucket = index.byPackage.get(scope); + if (bucket?.declarers.has(name)) { + const fileKey = bucket.declarers.get(name); + if (fileKey === null || fileKey === undefined) return null; + candidates.push({ fileKey, localName: name }); + } + + const owner = index.byFqn.get(scope); + if (owner === null) return null; + if (owner !== undefined) { + const localName = `${owner.localName}.${name}`; + const target = index.repo.get(owner.fileKey); + if ( + target && + (target.literals.has(localName) || + target.exprs.has(localName) || + unfoldableDeclarationsOf(target).has(localName)) + ) { + candidates.push({ fileKey: owner.fileKey, localName }); + } + } + + for (const candidate of candidates) { + if ( + resolved !== null && + (resolved.fileKey !== candidate.fileKey || resolved.localName !== candidate.localName) + ) { + return null; + } + resolved = candidate; + } + } + return resolved; +} + +/** + * Resolve a top-level name visible from the importing file's own package. + * `undefined` means the package does not declare the name; `null` means it is + * ambiguous and must floor rather than fall through to a star import. + */ +function resolveKotlinSamePackageTarget( + fileKey: string, + name: string, + index: KotlinConstantIndex, +): KotlinImportTarget | null | undefined { + const packageName = declaredPackageOf(index.repo.get(fileKey)); + if (packageName === null) return undefined; + const bucket = index.byPackage.get(packageName); + if (!bucket?.declarers.has(name)) return undefined; + const declaringFile = bucket.declarers.get(name); + if (declaringFile === null || declaringFile === undefined) return null; + // Same-file literals, expressions, and shadows are handled before this step. + if (declaringFile === fileKey) return undefined; + return { fileKey: declaringFile, localName: name }; +} + function computeKotlinFold( fileKey: string, name: string, @@ -1218,6 +1395,8 @@ function computeKotlinFold( depth: number, ): string | null { const { repo } = state.index; + const mc = repo.get(fileKey); + if (!mc) return null; // Qualified reference (`ApiPaths.ORDERS`): constants and imports are keyed by // their IN-FILE name, so a dotted name never hits directly. Split head.tail, // resolve the head through the importing file's type import, then look the @@ -1239,6 +1418,28 @@ function computeKotlinFold( // uses the declaring type's real name. return resolveWithState(target.fileKey, `${target.localName}.${tail}`, state, depth + 1); } + // A same-file top-level declaration outranks every star import. + if (!topLevelDeclarationsOf(mc).has(head)) { + const samePackageTarget = resolveKotlinSamePackageTarget(fileKey, head, state.index); + if (samePackageTarget === null) return null; + if (samePackageTarget !== undefined) { + return resolveWithState( + samePackageTarget.fileKey, + `${samePackageTarget.localName}.${tail}`, + state, + depth + 1, + ); + } + const wildcardTarget = resolveKotlinWildcardImportTarget(mc, head, state.index); + if (wildcardTarget !== null) { + return resolveWithState( + wildcardTarget.fileKey, + `${wildcardTarget.localName}.${tail}`, + state, + depth + 1, + ); + } + } // Un-imported qualified name (FQN form `com.example.app.api.ApiPaths.ORDERS`): // try the longest dotted prefix that resolves to a file. const parts = name.split('.'); @@ -1261,14 +1462,29 @@ function computeKotlinFold( // an operand may itself be a QUALIFIED reference (`X = ApiPaths.Y + "/tail"`) // and the core only knows bare names: it would look `ApiPaths.Y` up in maps // keyed by simple name, miss, and floor the whole chain to null. - const mc = repo.get(fileKey); - if (!mc) return null; const literal = mc.literals.get(name); if (literal !== undefined) return literal; const expr = mc.exprs.get(name); if (expr !== undefined) return foldOperands(fileKey, expr, state, depth + 1); const imp = mc.imports.get(name); if (imp !== undefined) return resolveImportedName(fileKey, imp, state, depth + 1); + // Any local declaration still shadows a lower-priority package-star import, + // including a var/plain type that is absent from the constant maps. + if (topLevelDeclarationsOf(mc).has(name) || unfoldableDeclarationsOf(mc).has(name)) return null; + const samePackageTarget = resolveKotlinSamePackageTarget(fileKey, name, state.index); + if (samePackageTarget === null) return null; + if (samePackageTarget !== undefined) { + return resolveWithState( + samePackageTarget.fileKey, + samePackageTarget.localName, + state, + depth + 1, + ); + } + const wildcardTarget = resolveKotlinWildcardImportTarget(mc, name, state.index); + if (wildcardTarget !== null) { + return resolveWithState(wildcardTarget.fileKey, wildcardTarget.localName, state, depth + 1); + } return null; } diff --git a/gitnexus/src/core/ingestion/route-extractors/kotlin-spring.ts b/gitnexus/src/core/ingestion/route-extractors/kotlin-spring.ts new file mode 100644 index 000000000..1ac21e2f9 --- /dev/null +++ b/gitnexus/src/core/ingestion/route-extractors/kotlin-spring.ts @@ -0,0 +1,326 @@ +/** + * Kotlin Spring MVC route annotations for the ingestion pipeline (#3130). + * + * This module only walks a caller-provided tree. In particular, it does not + * import or load tree-sitter-kotlin at module initialization, so platforms + * without the optional grammar can still load the language-provider registry. + */ +import type Parser from 'tree-sitter'; +import type { ExtractedDecoratorRoute } from '../workers/parse-worker.js'; +import { + intersectSpringHttpMethods, + springAnnotationHttpMethods, + unquoteSpringLiteral, +} from './spring-shared.js'; +import { + extractKotlinModuleConstants, + parseKotlinConstOperands, + unfoldableDeclarationsOf, + unquoteKotlinIdentifier, + type KotlinModuleConstants, + type Operand, +} from './kotlin-const-resolver.js'; + +/** Direct declaration annotations in source order. */ +function declarationAnnotations(node: Parser.SyntaxNode): Parser.SyntaxNode[] { + const modifiers = node.namedChildren.find((child) => child.type === 'modifiers'); + return modifiers?.namedChildren.filter((child) => child.type === 'annotation') ?? []; +} + +/** `@Foo`, `@Foo(...)`, or `@a.b.Foo(...)` → `Foo`. */ +function annotationName(annotation: Parser.SyntaxNode): string | null { + const constructor = annotation.namedChildren.find( + (child) => child.type === 'constructor_invocation', + ); + const userType = + annotation.namedChildren.find((child) => child.type === 'user_type') ?? + constructor?.namedChildren.find((child) => child.type === 'user_type'); + const identifiers = + userType?.namedChildren.filter((child) => child.type === 'type_identifier') ?? []; + const identifier = identifiers.at(-1); + return identifier ? unquoteKotlinIdentifier(identifier.text) : null; +} + +function annotationArguments(annotation: Parser.SyntaxNode): Parser.SyntaxNode[] { + const constructor = annotation.namedChildren.find( + (child) => child.type === 'constructor_invocation', + ); + const values = constructor?.namedChildren.find((child) => child.type === 'value_arguments'); + return values?.namedChildren.filter((child) => child.type === 'value_argument') ?? []; +} + +interface AnnotationArgument { + readonly name?: string; + readonly expression: Parser.SyntaxNode; +} + +/** + * Kotlin represents positional and named annotation arguments with the same + * `value_argument` node. A direct `=` token distinguishes the named form. + */ +function readAnnotationArgument(argument: Parser.SyntaxNode): AnnotationArgument | null { + const named = argument.children.some((child) => child.type === '='); + if (!named) { + const expression = argument.namedChild(0); + return expression ? { expression } : null; + } + const key = argument.namedChild(0); + const expression = argument.namedChild(1); + if (key?.type !== 'simple_identifier' || !expression) return null; + return { name: unquoteKotlinIdentifier(key.text), expression }; +} + +function routeArguments(annotation: Parser.SyntaxNode): AnnotationArgument[] | null { + const out: AnnotationArgument[] = []; + for (const argument of annotationArguments(annotation)) { + const parsed = readAnnotationArgument(argument); + if (!parsed) return null; + if (parsed.name === undefined || parsed.name === 'path' || parsed.name === 'value') { + out.push(parsed); + } + } + return out; +} + +/** A fully static Kotlin string literal: no `$name` or `${expr}` children. */ +function isPlainStringLiteral(node: Parser.SyntaxNode): boolean { + return ( + node.type === 'string_literal' && + node.namedChildren.every((child) => child.type === 'string_content') + ); +} + +/** + * Empty `[]` / `arrayOf()` is Spring "no path", not an unresolvable prefix. + * tree-sitter-kotlin may put a zero-width recovery child inside `[]`. + */ +function isEmptyKotlinPathCollection(node: Parser.SyntaxNode): boolean { + if (node.type === 'collection_literal') { + return node.namedChildren.every((child) => child.text.length === 0); + } + if (node.type !== 'call_expression') return false; + const callee = node.namedChild(0); + if (callee?.type !== 'simple_identifier' || unquoteKotlinIdentifier(callee.text) !== 'arrayOf') { + return false; + } + const suffix = node.namedChildren.find((child) => child.type === 'call_suffix'); + const args = suffix?.namedChildren.find((child) => child.type === 'value_arguments'); + if (!args) return true; + return args.namedChildren.every((child) => child.type !== 'value_argument'); +} + +function isKotlinInterface(node: Parser.SyntaxNode): boolean { + return node.children.some((child) => child.type === 'interface'); +} + +function isAbstractOrSealedClass(node: Parser.SyntaxNode): boolean { + const modifiers = node.namedChildren.find((child) => child.type === 'modifiers'); + return ( + modifiers?.namedChildren.some((child) => { + if (child.type !== 'inheritance_modifier' && child.type !== 'class_modifier') { + return false; + } + const text = child.text.trim(); + return text === 'abstract' || text === 'sealed'; + }) === true + ); +} + +function directFunctions(node: Parser.SyntaxNode): Parser.SyntaxNode[] { + const body = node.namedChildren.find((child) => child.type === 'class_body'); + return body?.namedChildren.filter((child) => child.type === 'function_declaration') ?? []; +} + +function functionName(node: Parser.SyntaxNode): string | null { + const field = node.childForFieldName('name'); + const identifier = + field?.type === 'simple_identifier' + ? field + : node.namedChildren.find((child) => child.type === 'simple_identifier'); + return identifier ? unquoteKotlinIdentifier(identifier.text) : null; +} + +/** + * `springAnnotationHttpMethods` parses Java `{A, B}` collections. + * Translate only Kotlin `method = [A, B]` before delegating. + */ +function kotlinSpringHttpMethods(name: string, annotation: Parser.SyntaxNode): readonly string[] { + if (name !== 'RequestMapping') return springAnnotationHttpMethods(name, annotation.text); + const normalized = annotation.text.replace( + /(\bmethod\s*=\s*)\[([^\]]*)\]/gs, + (_match, assignment: string, values: string) => `${assignment}{${values}}`, + ); + return springAnnotationHttpMethods(name, normalized); +} + +function typeName(node: Parser.SyntaxNode): string | null { + const identifier = node.children.find((child) => child.type === 'type_identifier'); + return identifier ? unquoteKotlinIdentifier(identifier.text) : null; +} + +/** + * Qualified enclosing type paths, innermost first. Kotlin companion members + * are keyed through their enclosing class, so companion_object itself adds no + * segment. + */ +function enclosingTypeNames(node: Parser.SyntaxNode): string[] { + const simpleNames: string[] = []; + for (let current: Parser.SyntaxNode | null = node.parent; current; current = current.parent) { + if (current.type !== 'class_declaration' && current.type !== 'object_declaration') continue; + const name = typeName(current); + if (name) simpleNames.push(name); + } + return simpleNames.map((_, index) => simpleNames.slice(index).reverse().join('.')); +} + +function declarationExists(constants: KotlinModuleConstants, name: string): boolean { + return ( + constants.literals.has(name) || + constants.exprs.has(name) || + unfoldableDeclarationsOf(constants).has(name) + ); +} + +/** + * The provider fold hook has a language-neutral three-argument signature and + * cannot receive a Kotlin reference site's enclosing type chain. Qualify only + * names whose owner is proven by this same tree; unresolved/imported names are + * left untouched for the repo-wide resolver. + */ +function qualifySameFileOperands( + operands: readonly Operand[], + functionNode: Parser.SyntaxNode, + constants: KotlinModuleConstants, +): Operand[] { + const enclosingTypes = enclosingTypeNames(functionNode); + return operands.map((operand) => { + if (operand.kind === 'literal') return operand; + for (const owner of enclosingTypes) { + const qualified = `${owner}.${operand.name}`; + if (declarationExists(constants, qualified)) { + return { kind: 'ref', name: qualified }; + } + } + return operand; + }); +} + +interface ClassMapping { + readonly prefix: string; + readonly methods: readonly string[]; +} + +/** + * Read the one optional class-level RequestMapping. A present route member must + * be exactly one plain string literal, an empty `[]`/`arrayOf()` (no prefix), + * or absent; constants, interpolation, non-empty collections, duplicate + * mappings, and dynamic expressions fail closed for the whole class. + */ +function classMapping(annotations: readonly Parser.SyntaxNode[]): ClassMapping | null { + const mappings = annotations.filter( + (annotation) => annotationName(annotation) === 'RequestMapping', + ); + if (mappings.length === 0) return { prefix: '', methods: ['*'] }; + if (mappings.length !== 1) return null; + + const mapping = mappings[0]; + const paths = routeArguments(mapping); + if (paths === null || paths.length > 1) return null; + + let prefix = ''; + if (paths.length === 1) { + const path = paths[0].expression; + if (isEmptyKotlinPathCollection(path)) { + prefix = ''; + } else if (isPlainStringLiteral(path)) { + const literal = unquoteSpringLiteral(path.text); + if (literal === null) return null; + prefix = literal; + } else { + return null; + } + } + + const methods = kotlinSpringHttpMethods('RequestMapping', mapping); + return methods.length === 0 ? null : { prefix, methods }; +} + +/** + * Extract direct Spring handler methods from concrete Kotlin RestControllers. + */ +export function extractKotlinSpringRoutes( + tree: Parser.Tree, + filePath: string, + lineOffset = 0, +): ExtractedDecoratorRoute[] { + const routes: ExtractedDecoratorRoute[] = []; + let moduleConstants: KotlinModuleConstants | undefined; + + for (const classNode of tree.rootNode.descendantsOfType('class_declaration')) { + if (isKotlinInterface(classNode) || isAbstractOrSealedClass(classNode)) continue; + const annotations = declarationAnnotations(classNode); + const annotationNames = annotations.map(annotationName); + if (!annotationNames.includes('RestController')) continue; + if (annotationNames.includes('FeignClient')) continue; + + const ownerMapping = classMapping(annotations); + if (ownerMapping === null) continue; + + for (const functionNode of directFunctions(classNode)) { + const handlerName = functionName(functionNode); + if (!handlerName) continue; + + for (const annotation of declarationAnnotations(functionNode)) { + const decoratorName = annotationName(annotation); + if (!decoratorName) continue; + + const methodMethods = kotlinSpringHttpMethods(decoratorName, annotation); + const methods = intersectSpringHttpMethods(ownerMapping.methods, methodMethods); + if (methods.length === 0) continue; + + const paths = routeArguments(annotation); + if (paths === null || paths.length > 1) continue; + + let routePath = ''; + let routePathExpr: string | undefined; + let routePathOperands: Operand[] | undefined; + if (paths.length === 1) { + const expression = paths[0].expression; + if (isEmptyKotlinPathCollection(expression)) { + routePath = ''; + } else if (isPlainStringLiteral(expression)) { + const literal = unquoteSpringLiteral(expression.text); + if (literal === null) continue; + routePath = literal; + } else { + const operands = parseKotlinConstOperands(expression); + if (operands === null) continue; + moduleConstants ??= extractKotlinModuleConstants(tree); + routePathExpr = expression.text; + routePathOperands = qualifySameFileOperands(operands, functionNode, moduleConstants); + } + } + + for (const httpMethod of methods) { + routes.push({ + filePath, + routePath, + httpMethod, + decoratorName, + lineNumber: annotation.startPosition.row + lineOffset, + ...(ownerMapping.prefix ? { prefix: ownerMapping.prefix } : {}), + handlerName, + ...(routePathExpr === undefined + ? {} + : { + routePathExpr, + routePathOperands, + }), + }); + } + } + } + } + + return routes; +} diff --git a/gitnexus/src/core/ingestion/route-extractors/python-decorator-handler.ts b/gitnexus/src/core/ingestion/route-extractors/python-decorator-handler.ts new file mode 100644 index 000000000..fca06c857 --- /dev/null +++ b/gitnexus/src/core/ingestion/route-extractors/python-decorator-handler.ts @@ -0,0 +1,19 @@ +/** + * Return the function name attached to a Python decorator's immediate + * `decorated_definition`; reject every other shape rather than climbing. + */ + +import type { SyntaxNode } from '../utils/ast-helpers.js'; + +export function pythonDecoratorRouteHandlerName(decoratorNode: SyntaxNode): string | undefined { + const decorated = decoratorNode.parent; + if (decorated === null || decorated.type !== 'decorated_definition') return undefined; + + // `async def` is still a `function_definition` in tree-sitter-python (the + // `async` keyword is an anonymous child), so async handlers need no branch. + const definition = decorated.childForFieldName('definition'); + if (!definition || definition.type !== 'function_definition') return undefined; + + const name = definition.childForFieldName('name')?.text; + return name !== undefined && name.length > 0 ? name : undefined; +} diff --git a/gitnexus/src/core/ingestion/scope-extractor.ts b/gitnexus/src/core/ingestion/scope-extractor.ts index 39b8667ae..fde3c7b26 100644 --- a/gitnexus/src/core/ingestion/scope-extractor.ts +++ b/gitnexus/src/core/ingestion/scope-extractor.ts @@ -1775,6 +1775,7 @@ const KNOWN_SUB_TAGS: ReadonlySet = new Set([ '@scope.lexical-names', '@declaration.name', '@declaration.qualified_name', + '@declaration.is-synthetic', '@import.name', '@import.source', '@import.alias', diff --git a/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/ids.ts b/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/ids.ts index d0f99de15..55becdcb6 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/ids.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/graph-bridge/ids.ts @@ -36,6 +36,32 @@ import { import { templateConstraintsIdTag } from '../../utils/template-arguments.js'; import { parameterShapeIdTag } from '../../utils/method-props.js'; import { definitionIdPosition } from '../utils/definition-id.js'; + +const defGraphIdMemoByLookup = new WeakMap>(); + +const isResolveDefGraphIdMemoEnabled = (): boolean => { + const raw = process.env.GITNEXUS_RESOLVE_DEF_GRAPH_ID_MEMO; + if (raw === undefined || raw.trim() === '') return true; + const value = raw.trim().toLowerCase(); + return value !== '0' && value !== 'false' && value !== 'off' && value !== 'no'; +}; + +const defGraphIdMemoKey = ( + filePath: string, + def: { + nodeId?: string; + qualifiedName?: string; + type?: NodeLabel; + parameterTypes?: readonly string[]; + parameterTypeClasses?: readonly ParameterTypeClass[]; + parameterCount?: number; + templateArguments?: readonly string[]; + templateConstraints?: unknown; + namespacePrefix?: string; + }, +): string => + `${filePath}\0${def.nodeId ?? ''}\0${def.type ?? ''}\0${def.qualifiedName ?? ''}\0${def.parameterCount ?? ''}\0${(def.parameterTypes ?? []).join(',')}\0${(def.parameterTypeClasses ?? []).join(',')}\0${def.namespacePrefix ?? ''}\0${(def.templateArguments ?? []).join(',')}\0${templateConstraintsIdTag(def.templateConstraints)}`; + /** * Labels that may legitimately ANCHOR a CALLS/ACCESSES edge as the * source ("caller"). A Variable / Property can be the TARGET of an @@ -230,6 +256,38 @@ export function resolveDefGraphId( namespacePrefix?: string; }, nodeLookup: GraphNodeLookup, +): string | undefined { + if (!isResolveDefGraphIdMemoEnabled()) { + return resolveDefGraphIdUncached(filePath, def, nodeLookup); + } + const qn = def.qualifiedName; + if (qn === undefined || qn.length === 0) return undefined; + let bucket = defGraphIdMemoByLookup.get(nodeLookup); + if (bucket === undefined) { + bucket = new Map(); + defGraphIdMemoByLookup.set(nodeLookup, bucket); + } + const key = defGraphIdMemoKey(filePath, def); + if (bucket.has(key)) return bucket.get(key); + const resolved = resolveDefGraphIdUncached(filePath, def, nodeLookup); + bucket.set(key, resolved); + return resolved; +} + +function resolveDefGraphIdUncached( + filePath: string, + def: { + nodeId?: string; + qualifiedName?: string; + type?: NodeLabel; + parameterTypes?: readonly string[]; + parameterTypeClasses?: readonly ParameterTypeClass[]; + parameterCount?: number; + templateArguments?: readonly string[]; + templateConstraints?: unknown; + namespacePrefix?: string; + }, + nodeLookup: GraphNodeLookup, ): string | undefined { const qn = def.qualifiedName; if (qn === undefined || qn.length === 0) return undefined; diff --git a/gitnexus/src/core/ingestion/scope-resolution/passes/callable-value-flow.ts b/gitnexus/src/core/ingestion/scope-resolution/passes/callable-value-flow.ts index 6a8e1ba83..4ce65c63a 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/passes/callable-value-flow.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/passes/callable-value-flow.ts @@ -82,6 +82,15 @@ export interface EmitCallableValueFlowInput { readonly isCallableValueTarget?: (def: SymbolDefinition) => boolean; readonly hasFileLocalCallableLinkage?: (def: SymbolDefinition) => boolean; readonly onWarn?: (warning: CallableValueFlowWarning) => void; + /** When set, skip a second `collectDeferredIndirectSites` walk. */ + readonly deferredIndirectSites?: ReadonlySet; + /** When set, skip a second `referenceSites` signature walk. */ + readonly callSignaturesBySite?: ReadonlyMap; +} + +export interface DeferredIndirectCollection { + readonly sites: ReadonlySet; + readonly callSignaturesBySite: ReadonlyMap; } /** Position key shared with the existing free/reference skip-set contract. */ @@ -100,7 +109,16 @@ export function collectDeferredIndirectSites( parsedFiles: readonly ParsedFile[], scopes?: ScopeResolutionIndexes, ): ReadonlySet { + return collectDeferredIndirectCollection(parsedFiles, scopes).sites; +} + +/** One `referenceSites` walk for deferred keys and call-signature evidence. */ +export function collectDeferredIndirectCollection( + parsedFiles: readonly ParsedFile[], + scopes?: ScopeResolutionIndexes, +): DeferredIndirectCollection { const out = new Set(); + const callSignaturesBySite = new Map(); const flowCells = new Set(); if (scopes !== undefined) { for (const parsed of parsedFiles) { @@ -113,11 +131,23 @@ export function collectDeferredIndirectSites( } } for (const parsed of parsedFiles) { - const canonical = new Set( - parsed.referenceSites - .filter((site) => site.kind === 'call') - .map((site) => callableFlowSiteKey(parsed.filePath, site.atRange)), - ); + const canonical = new Set(); + for (const site of parsed.referenceSites) { + if (site.kind !== 'call') continue; + const key = callableFlowSiteKey(parsed.filePath, site.atRange); + canonical.add(key); + const signature: CallableFlowExpectedSignature = { + ...(site.arity !== undefined ? { parameterCount: site.arity } : {}), + ...(site.argumentTypes !== undefined ? { parameterTypes: site.argumentTypes } : {}), + ...(site.argumentTypeClasses !== undefined + ? { parameterTypeClasses: site.argumentTypeClasses } + : {}), + }; + const previous = callSignaturesBySite.get(key); + if (previous === undefined || signatureEvidence(signature) > signatureEvidence(previous)) { + callSignaturesBySite.set(key, signature); + } + } for (const site of parsed.callableFlowSites ?? []) { if (site.kind !== 'invoke') continue; const key = callableFlowSiteKey(parsed.filePath, site.callSite); @@ -132,7 +162,7 @@ export function collectDeferredIndirectSites( } } } - return out; + return { sites: out, callSignaturesBySite }; } function flowCellOperand(site: CallableFlowSite): CallableFlowOperand | undefined { @@ -156,7 +186,8 @@ function flowCellOperand(site: CallableFlowSite): CallableFlowOperand | undefine export function emitCallableValueFlow(input: EmitCallableValueFlowInput): CallableValueFlowResult { const facts: FileFact[] = []; const invokes: FileInvoke[] = []; - const canonicalInvokeKeys = collectDeferredIndirectSites(input.parsedFiles, input.scopes); + const canonicalInvokeKeys = + input.deferredIndirectSites ?? collectDeferredIndirectSites(input.parsedFiles, input.scopes); let unmatchedInvokes = 0; for (const parsed of input.parsedFiles) { for (const site of parsed.callableFlowSites ?? []) { @@ -485,7 +516,7 @@ export function emitCallableValueFlow(input: EmitCallableValueFlowInput): Callab targetIndexes, aliasesByTargetId, ); - const callSignaturesBySite = indexCallSignatures(input.parsedFiles); + const callSignaturesBySite = input.callSignaturesBySite ?? indexCallSignatures(input.parsedFiles); const dynamicCallees = new Map>(); const dynamicOverflow = new Set(); const dynamicTargetHistory = new Map>(); diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts index 689780d40..f39aa4dd0 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts @@ -90,7 +90,7 @@ import { import { emitImportEdges } from '../graph-bridge/imports-to-edges.js'; import { callableFlowSiteKey, - collectDeferredIndirectSites, + collectDeferredIndirectCollection, emitCallableValueFlow, } from '../passes/callable-value-flow.js'; import type { ScopeResolver, UndecidedSatisfaction } from '../contract/scope-resolver.js'; @@ -988,7 +988,8 @@ export function runScopeResolution( // ── Phase 4: emit graph edges (LOAD-BEARING ORDER — see I1) ──────────── input.onProgress?.('linking symbols', files.length, files.length); const handledSites = new Set(preEmittedInheritanceSites); - const deferredIndirectSites = collectDeferredIndirectSites(emitParsedFiles, indexes); + const deferredIndirectCollection = collectDeferredIndirectCollection(emitParsedFiles, indexes); + const deferredIndirectSites = deferredIndirectCollection.sites; const callableArgumentSites = new Set(); if (input.pdg !== true && deferredIndirectSites.size > 0) { for (const parsed of emitParsedFiles) { @@ -1237,6 +1238,8 @@ export function runScopeResolution( collapseByCallerTarget: provider.collapseMemberCallsByCallerTarget === true, isCallableValueTarget: provider.isCallableValueTarget, hasFileLocalCallableLinkage: provider.hasFileLocalCallableLinkage, + deferredIndirectSites, + callSignaturesBySite: deferredIndirectCollection.callSignaturesBySite, onWarn: (warning) => logger.warn( warning, diff --git a/gitnexus/src/core/ingestion/utils/ast-helpers.ts b/gitnexus/src/core/ingestion/utils/ast-helpers.ts index 1a88c9a15..2349b1426 100644 --- a/gitnexus/src/core/ingestion/utils/ast-helpers.ts +++ b/gitnexus/src/core/ingestion/utils/ast-helpers.ts @@ -462,6 +462,25 @@ export function walkNamedTree(node: SyntaxNode, cb: (node: SyntaxNode) => void): } } +/** + * True when a node is, or contains, tree-sitter error recovery. + * + * After a syntax error the parser keeps going by guessing node boundaries, so + * the surviving tree stays WELL FORMED while describing text that was never + * written that way: an unterminated argument list can absorb the source of the + * next declaration into an `ERROR` child, and an assignment with no right-hand + * side gets a `MISSING` value node whose text is invented. A capture that reads + * such a subtree emits facts that look ordinary and are false, which is worse + * than emitting nothing — so callers that record source text verbatim should + * check this first and fail closed. + * + * `hasError` covers the subtree; `isMissing` is checked as well because a node + * inserted by recovery is the one case where the node itself carries the flag. + */ +export function hasRecoveredSyntax(node: SyntaxNode): boolean { + return node.hasError || node.isMissing; +} + /** Return the first matching ancestor unless a boundary ancestor is reached first. */ export function findAncestorBeforeBoundary( node: SyntaxNode, diff --git a/gitnexus/src/core/ingestion/utils/test-file-path.ts b/gitnexus/src/core/ingestion/utils/test-file-path.ts new file mode 100644 index 000000000..ceee59c56 --- /dev/null +++ b/gitnexus/src/core/ingestion/utils/test-file-path.ts @@ -0,0 +1,87 @@ +/** + * Test-file path classification — the single source of truth. + * + * WHY THIS MODULE EXISTS + * + * Two independent copies of this predicate existed and had drifted apart: + * + * - `core/ingestion/entry-point-scoring.ts` `isTestFile` — excludes test + * files from process entry-point detection. + * - `mcp/local/local-backend.ts` `isTestFilePath` — backs the + * `includeTests` flag on `impact` / `trace` / `context`. + * + * They answered "is this a test file?" differently, so the same path could be a + * test in one code path and not the other. The MCP copy recognized no C#, Java, + * or Swift test convention at all, meaning `includeTests: false` silently failed + * to filter them; the scoring copy missed `/conftest.` (it already matched + * `/test/`, so `/test/fixtures/` was never a scoring gap). + * + * The duplication was not gratuitous: `entry-point-scoring.ts` imports the + * language-provider registry, and #2802 deliberately cut that closure out of MCP + * server startup. Importing it back into `local-backend.ts` would reintroduce + * that cost. So the shared predicate lives here instead, with NO imports — pure + * string matching — and both callers delegate to it. + * + * Keep it dependency-free. Anything imported here lands in MCP startup. + */ + +/** + * Lowercase forward-slash path substrings. Directory needles include a leading + * slash so they match path components after the caller slash-prefixes relative + * paths. `/test/` already covers Maven `src/test` and `/test/fixtures/`; + * `/tests/` covers Laravel `tests/Feature` and `/tests/fixtures/`. + */ +const TEST_PATH_SUBSTRINGS: readonly string[] = [ + '.test.', + '.spec.', + '__tests__/', + '__mocks__/', + '/test/', + '/tests/', + '/testing/', + '/spec/', + '/test_', + '/conftest.', + '/uitests/', + '.tests/', + '.test/', + '.integrationtests/', + '.unittests/', + '/testproject/', +]; + +/** Case-insensitive suffixes that already include a delimiter (`_test.py`, not `test.py`). */ +const TEST_PATH_DELIMITED_SUFFIXES: readonly string[] = [ + '_test.py', + '_test.go', + '_spec.rb', + '_test.rb', +]; + +/** + * Case-sensitive `Test`/`Tests`/`Spec` suffixes. Lowercasing first would also + * match production names such as `Contest.swift` and `Latest.php`. + */ +const TEST_PATH_CASED_SUFFIXES: readonly string[] = [ + 'Tests.swift', + 'Test.swift', + 'Tests.cs', + 'Test.cs', + 'Test.php', + 'Spec.php', +]; + +/** Absent / empty paths are not test paths. */ +export function isTestFilePath(filePath: string | null | undefined): boolean { + if (!filePath) return false; + const slashed = filePath.replace(/\\/g, '/'); + const prefixed = slashed.startsWith('/') ? slashed : `/${slashed}`; + const lower = prefixed.toLowerCase(); + if (TEST_PATH_SUBSTRINGS.some((needle) => lower.includes(needle))) return true; + if (TEST_PATH_DELIMITED_SUFFIXES.some((suffix) => lower.endsWith(suffix))) return true; + // Xcode `{Product}UITests` targets. Slash-anchored `/uitests/` does not match + // `MyAppUITests`; an unanchored `uitests/` substring also matches `fruitests`. + if (prefixed.split('/').some((seg) => seg.endsWith('UITests'))) return true; + const basename = prefixed.slice(prefixed.lastIndexOf('/') + 1); + return TEST_PATH_CASED_SUFFIXES.some((suffix) => basename.endsWith(suffix)); +} diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index 29c94e8ee..3fd98a3d4 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -1476,6 +1476,7 @@ import { type ModuleConstants, type Operand, } from '../route-extractors/python-const-resolver.js'; +import { unfoldableDeclarationsOf } from '../route-extractors/constant-resolver.js'; /** * Report a non-fatal worker issue to the pool over IPC so a caught error is not @@ -1578,6 +1579,11 @@ const processFileGroup = ( } const provider = getProvider(language); + // Owner map for provider.synthesizeStructureMembers: type-declaration AST + // node id → graph node id for classes THIS file's capture loop materialized. + // Keyed by in-memory AST identity (never persisted); filled below. + const classOwnersByNodeId = new Map(); + // #2687: ONE pass over `matches` yields both suppression sets — the // definition-name claims by rank (callable > Property > value), so the dedup // below cannot depend on tree-sitter's match order, and the concrete-typedef @@ -1790,12 +1796,14 @@ const processFileGroup = ( const httpMethod = ['GET', 'POST', 'PUT', 'DELETE', 'PATCH'].includes(method) ? method : 'GET'; + const handlerName = provider.decoratorRouteHandlerName?.(decoratorNode); const base = { filePath: file.path, httpMethod, decoratorName, lineNumber: decoratorNode.startPosition.row + lineOffset, ...(decoratorReceiver ? { decoratorReceiver } : {}), + ...(handlerName ? { handlerName } : {}), }; if (decoratorArgStr) { // String-literal path (the fast path, unchanged). Empty-string @@ -2900,6 +2908,7 @@ const processFileGroup = ( { nodeLabel, nodeName, + filePath: file.path, definitionNode, parsedImports: parsedFile?.parsedImports ?? [], isExported, @@ -2987,6 +2996,17 @@ const processFileGroup = ( : {}), }); + // Class-like definitions register their AST node id → graph node id for + // provider.synthesizeStructureMembers. The definition node is the same + // type-declaration AST node that the provider-specific planner receives. + if ( + isClassLikeLabel && + definitionNode && + provider.classExtractor?.isTypeDeclaration(definitionNode) + ) { + classOwnersByNodeId.set(definitionNode.id, nodeId); + } + // Object-literal callables remain file definitions as well as members of // their exported binding. Class members still use HAS_METHOD alone. const isTopLevelObjectCallable = @@ -3070,7 +3090,17 @@ const processFileGroup = ( // without booting a worker. if (provider.extractModuleConstants && shouldHarvestModuleConstants(provider, parseContent)) { const constants = provider.extractModuleConstants(tree); - if (constants.literals.size > 0 || constants.exprs.size > 0 || constants.imports.size > 0) { + const topLevelDeclarations = ( + constants as ModuleConstants & { readonly topLevelDeclarations?: unknown } + ).topLevelDeclarations; + if ( + constants.literals.size > 0 || + constants.exprs.size > 0 || + constants.imports.size > 0 || + (constants.wildcardImports?.length ?? 0) > 0 || + unfoldableDeclarationsOf(constants).size > 0 || + (topLevelDeclarations instanceof Set && topLevelDeclarations.size > 0) + ) { (result.moduleConstants ??= []).push({ filePath: file.path, constants }); } } @@ -3092,6 +3122,19 @@ const processFileGroup = ( if (springTypes.length > 0) (result.springTypes ??= []).push(...springTypes); } + if (provider.synthesizeStructureMembers) { + const synthetic = provider.synthesizeStructureMembers(tree, file.path, classOwnersByNodeId); + for (const node of synthetic.nodes) { + result.nodes.push(node as ParsedNode); + } + for (const sym of synthetic.symbols) { + result.symbols.push(sym as ParsedSymbol); + } + for (const rel of synthetic.relationships) { + result.relationships.push(rel as ParsedRelationship); + } + } + // Vue: emit CALLS edges for components used in