Merge remote-tracking branch 'upstream/main' into codex/xaml-support-3202

# Conflicts:
#	gitnexus/src/core/run-analyze.ts
This commit is contained in:
azizur100389 2026-09-20 03:34:26 +01:00
commit 248db190ef
126 changed files with 10686 additions and 1272 deletions

View file

@ -19,8 +19,8 @@ runs:
# Browsers are installed explicitly by e2e. Typecheck only needs types.
PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD: '1'
# Compile shared with the web package's TypeScript 5. Do not npm-ci
# gitnexus-shared (TypeScript 7 optional-platform install, ~7 minutes).
# Compile shared with the web package's TypeScript 7. Do not npm-ci
# gitnexus-shared (a second optional-platform install, ~7 minutes).
- name: Build gitnexus-shared
# node + lib/tsc.js — same on Windows/macOS/Linux. Do not use .bin/tsc
# (tsc.cmd on Windows; execFileSync cannot launch .cmd without a shell).

View file

@ -93,6 +93,9 @@ updates:
# The extension version is a separate upstream constant, not derivable
# from the core version (see vendor/lbug-fts/manifest.json).
- dependency-name: '@ladybugdb/core'
- dependency-name: typescript
update-types:
- version-update:semver-major
# gitnexus-web (thin frontend client).
- package-ecosystem: npm
@ -111,6 +114,10 @@ updates:
labels:
- dependencies
- frontend
ignore:
- dependency-name: typescript
update-types:
- version-update:semver-major
# Shared types package.
- package-ecosystem: npm
@ -128,3 +135,9 @@ updates:
include: scope
labels:
- dependencies
ignore:
# Keep shared on the same TypeScript major as CLI/web. A shared-only
# major bump reintroduces a compiler split this repo unified.
- dependency-name: typescript
update-types:
- version-update:semver-major

View file

@ -618,6 +618,17 @@ jobs:
run: node --import tsx bench/rust-cargo-targets/measure.mjs --check
working-directory: gitnexus
- name: Swift Package.swift import-resolve guards (#2964, #2931)
if: ${{ !cancelled() }}
# Build-free: same baseline approach as parse-dispatch-rounds —
# exact declared/SDK/undeclared floors plus a fingerprint, then
# ratio timing only (resolveSwiftImportTarget, swiftPackageStrategy,
# parseSwiftPackageManifest 4n/n). Pins declaration-only resolve,
# https:// factory survival, and #2931 segment-boundary membership.
# See bench/swift-package-imports/measure.mjs.
run: node --import tsx bench/swift-package-imports/measure.mjs --check
working-directory: gitnexus
- name: MCP tools/list countRepos vs listRepos guards (#3259, #3184)
if: ${{ !cancelled() }}
# Build-free: exact registry cardinality + tool-roster + schema-flag

View file

@ -48,7 +48,7 @@ jobs:
persist-credentials: false
- name: Initialize CodeQL
uses: github/codeql-action/init@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9
uses: github/codeql-action/init@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0
with:
languages: ${{ matrix.language }}
queries: security-and-quality
@ -87,6 +87,6 @@ jobs:
- '.github/scripts/fetch-lbug-fts-artifacts.mjs'
- name: Perform CodeQL Analysis
uses: github/codeql-action/analyze@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9
uses: github/codeql-action/analyze@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0
with:
category: '/language:${{ matrix.language }}'

View file

@ -330,24 +330,20 @@ jobs:
set -euo pipefail
# The benchmark's task bindings sandbox-copy node_modules from the
# monorepo root as well as gitnexus-shared and gitnexus (see the
# sandbox_copy entries in tasks.scenarios.yaml). The two steps below
# install the subpackage trees; the root tree needs its own install
# or capture_task_dependency_binding aborts at task binding on the
# missing root node_modules.
# sandbox_dependencies entries in tasks.scenarios.yaml). gitnexus
# npm ci plus build compiles shared through parent lib/tsc.js; do
# not npm ci gitnexus-shared (second TypeScript 7 optional install).
npm ci
- name: Build pinned shared runtime
run: |
set -euo pipefail
npm ci
npm run build
working-directory: gitnexus-shared
- name: Install and build pinned GitNexus runtime
run: |
set -euo pipefail
npm ci
npm run build
# Task YAML still mounts gitnexus-shared/node_modules. Keep an empty
# directory so capture_task_dependency_binding does not abort, without
# installing TypeScript 7 inside shared.
mkdir -p ../gitnexus-shared/node_modules
working-directory: gitnexus
- name: Point the benchmark task repo at the checkout

View file

@ -53,6 +53,6 @@ jobs:
retention-days: 5
- name: Upload to Security tab
uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9
uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0
with:
sarif_file: results.sarif

View file

@ -76,7 +76,7 @@ jobs:
exit-code: '0'
- name: Upload to Security tab
uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9
uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0
with:
sarif_file: trivy-${{ matrix.image.name }}.sarif
category: trivy-${{ matrix.image.name }}

View file

@ -76,7 +76,7 @@ jobs:
continue-on-error: true
- name: Upload SARIF
uses: github/codeql-action/upload-sarif@cdf488f595d80d6e07e03d4674febd5ab45fa938 # v4.37.9
uses: github/codeql-action/upload-sarif@b96794f015dfd88f77b49b1c93e0fa7110f94c63 # v4.38.0
with:
sarif_file: zizmor.sarif
category: zizmor

View file

@ -16,14 +16,19 @@ This project uses the [PolyForm Noncommercial License 1.0.0](https://polyformpro
**Prerequisites:** Node.js — `gitnexus/` requires `^22.18.0 || >=24.11.0` and `gitnexus-web/` requires `^20.19.0 || >=22.12.0` (enforced via the `engines` field in each package). Use `nvm install` to match the local version.
1. Clone the repository.
2. **Shared package:** `cd gitnexus-shared && npm install && npm run build`
3. **CLI / MCP package:** `cd ../gitnexus && npm install && npm run build`
4. **Web UI (if needed):** `cd ../gitnexus-web && npm install`
5. Run tests as described in [TESTING.md](TESTING.md).
2. **CLI / MCP package:** `cd gitnexus && npm install && npm run build`
`prepare` / `scripts/build.js` compiles `gitnexus-shared` with
`node …/typescript/lib/tsc.js` from this package. Do not `npm install` or
`npm ci` inside `gitnexus-shared/` — that is a second TypeScript 7
optional-platform install and is not what `setup-gitnexus` does.
3. **Web UI (if needed):** `cd gitnexus-web && npm install`
If you skipped step 2, compile shared with the web compiler first:
`cd gitnexus-shared && node ../gitnexus-web/node_modules/typescript/lib/tsc.js`
4. Run tests as described in [TESTING.md](TESTING.md).
The CLI build imports `gitnexus-shared`, so a fresh clone must install and build
the shared package before running `npm install` in `gitnexus/`. This is the same
order used by the repository's `setup-gitnexus` CI action.
The CLI build imports `gitnexus-shared`, so `gitnexus-shared/dist` must exist
before `gitnexus` typecheck. That emit uses a parent package's TypeScript 7
`lib/tsc.js`, matching `setup-gitnexus`, `setup-gitnexus-web`, and Vercel.
### Containerized development (optional)
@ -70,7 +75,7 @@ Commits within a PR may use any style — only the **merged PR title** shows up
## Before you open a PR
- [ ] Tests pass for the packages you touched (`gitnexus` and/or `gitnexus-web`).
- [ ] Typecheck passes: `npx tsc --noEmit` in `gitnexus/` and `npx tsc -b --noEmit` in `gitnexus-web/`.
- [ ] Typecheck passes: `npx tsc --noEmit` in `gitnexus/` and `npx tsc -b --noEmit` in `gitnexus-web/`. Those commands use TypeScript 7. The web app tsconfig lists `lib` `DOM`/`DOM.Iterable` and `jsx: react-jsx` so React/JSX typecheck on 7; Vite/`vitest` keep `@vitejs/plugin-react` with the automatic JSX runtime. Build `gitnexus-shared/dist` first with a parent `lib/tsc.js` (a `gitnexus` install/`npm run build` does this). Repo ESLint stays syntax-only on a TypeScript 5.x peer until typescript-eslint supports 7.
- [ ] No secrets, tokens, or machine-specific paths committed.
- [ ] Documentation updated if behavior or public CLI/MCP contract changes.
- [ ] Every new `GITNEXUS_*` environment variable has a row in the **Environment variables** table in [README.md](README.md) — variable, default, effect, and when to tune it.

6
DoD.md
View file

@ -113,7 +113,7 @@ Run the commands relevant to the touched area. If something cannot be run in the
### 4.1 Build ordering
- [ ] `gitnexus-shared/` dist is built before consuming packages are typechecked or tested (CI uses the `setup-gitnexus` action for this — local runs must match).
- [ ] `gitnexus-shared/` dist is built before consuming packages are typechecked or tested (CI uses the `setup-gitnexus` action, which compiles shared with the parent TypeScript 7 `lib/tsc.js` — local runs must match).
### 4.2 If `gitnexus/` changed
@ -123,13 +123,13 @@ Run the commands relevant to the touched area. If something cannot be run in the
### 4.3 If `gitnexus-web/` changed
- [ ] `cd gitnexus-web && npx tsc -b --noEmit`
- [ ] `cd gitnexus-web && npx tsc -b --noEmit` (TypeScript 7 typechecks React/JSX: `jsx: react-jsx`, `lib` includes `DOM`)
- [ ] `cd gitnexus-web && npm test`
- [ ] `cd gitnexus-web && npm run test:e2e` when browser flows or user-facing UI behavior changed
### 4.4 If `gitnexus-shared/` changed
- [ ] Shared package builds cleanly (`npm run build` in `gitnexus-shared/`)
- [ ] Shared package builds cleanly from a parent TypeScript 7 shim after that parent is installed (`cd gitnexus-shared && node ../gitnexus/node_modules/typescript/lib/tsc.js`, or `node ../gitnexus-web/node_modules/typescript/lib/tsc.js` after a web install). Do not `npm install` / `npm ci` in `gitnexus-shared/` for this check.
- [ ] Dependent packages still typecheck and test after the shared change — verify both CLI and web consumers together
### 4.5 If CI workflows or release pipelines changed

View file

@ -407,7 +407,8 @@ backoff. Invalid `.gitnexusrc` or ignore-file reloads pause ordinary refreshes
until the control file is fixed. Stop the watcher with Ctrl+C.
Watch mode accepts `--debounce`, `--workers`, `--worker-timeout`,
`--max-file-size`, `--branch`, `--pdg`, `--skip-fts`, `--name`, `--allow-duplicate-name`, and
`--max-file-size`, `--max-processes`, `--max-process-branching`,
`--max-process-trace-depth`, `--max-entry-point-candidates`, `--branch`, `--pdg`, `--skip-fts`, `--name`, `--allow-duplicate-name`, and
`--verbose`. Explicit one-shot options such as `--force`, `--repair-fts`,
embedding flags, `--skills`, `--self-commit`, `--index-only`, and `--skip-git`
are rejected. Unsupported defaults from `.gitnexusrc` are ignored with a
@ -453,6 +454,8 @@ gitnexus analyze --verbose # Log skipped files when parsers are unavailabl
gitnexus analyze --worker-timeout 60 # Increase worker idle timeout for slow parses
gitnexus analyze --workers <n> # Parse worker pool size (>=1; default: cores-1, capped at 16,
# auto-sized to the repo). 0 is rejected — there is no sequential mode.
gitnexus analyze --max-processes <n> # Process-detection process cap (replaces dynamic max(20, round(symbols/10)))
gitnexus analyze --max-entry-point-candidates <n> # Ranked entry-point pool (default 200; raise when the warning names it)
gitnexus analyze --spring-actuator ./actuator # Enrich with local Spring Boot Actuator JSON snapshots
gitnexus analyze --asyncapi-spec ./docs/asyncapi # Resolve broker addresses from AsyncAPI 3.x documents
gitnexus analyze --wal-checkpoint-threshold 67108864 # LadybugDB WAL auto-checkpoint threshold in bytes
@ -576,15 +579,15 @@ Notes:
- The default branch is resolved as: `--default-branch` > `.gitnexusrc` `defaultBranch`/`branch` > auto-detected `origin/HEAD` > `main`.
- `skipContextFiles` / `skipAiContext` are aliases for `skipAgentsMd` — they skip the `AGENTS.md` / `CLAUDE.md` block only. They do **not** imply `skipSkills`. `indexOnly` is the stronger option that skips all file injection.
- Supported keys: `defaultBranch` (`branch`), `skipAgentsMd` (`skipContextFiles`, `skipAiContext`), `skipSkills`, `indexOnly`, `stats`/`noStats`, `embeddings`, `dropEmbeddings`, `name`, `allowDuplicateName`, `maxFileSize`, `workerTimeout`, `walCheckpointThreshold`, `workers`, `springActuator`, `embeddingThreads`, `embeddingBatchSize`, `embeddingSubBatchSize`, `embeddingDevice`.
- The file is JSON only. Unknown keys and invalid values fail fast with an actionable error before analysis starts.
- Supported keys: `defaultBranch` (`branch`), `skipAgentsMd` (`skipContextFiles`, `skipAiContext`), `skipSkills`, `indexOnly`, `stats`/`noStats`, `embeddings`, `dropEmbeddings`, `name`, `allowDuplicateName`, `maxFileSize`, `workerTimeout`, `walCheckpointThreshold`, `workers`, `maxProcesses`, `maxProcessBranching`, `maxProcessTraceDepth`, `maxEntryPointCandidates`, `springActuator`, `embeddingThreads`, `embeddingBatchSize`, `embeddingSubBatchSize`, `embeddingDevice`.
- The file is JSON only. Unknown keys and wrong JSON types fail fast with an actionable error before analysis starts. Process-detection knobs (`maxProcesses`, `maxProcessBranching`, `maxProcessTraceDepth`, `maxEntryPointCandidates`) that are not a positive integer warn and fall through to env, then the built-in default.
</details>
<details>
<summary><strong>Environment variables</strong></summary>
Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max-file-size`, `--verbose`). Use the env-var form when you'd otherwise repeat the same flag every run, or when invoking GitNexus from a long-running host (MCP server, eval-server, CI shell) that already manages its own environment. CLI flags take precedence over env vars; env vars take precedence over built-in defaults.
Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max-file-size`, `--verbose`). Use the env-var form when you'd otherwise repeat the same flag every run, or when invoking GitNexus from a long-running host (MCP server, eval-server, CI shell) that already manages its own environment. CLI flags take precedence over `.gitnexusrc`, which takes precedence over env vars, which take precedence over built-in defaults.
| Variable | Default | Effect | Tune when… |
| ----------------------------------------------- | ---------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
@ -601,6 +604,10 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max
| `GITNEXUS_PROFILE_DEFERRED_SLOW_MS` | `3000` (verbose) / `5000` | Per-file threshold in ms above which `processCallsFromExtracted` emits a `slow file …` log line. Parsed via `Number()`: accepts integers (`5000`), scientific notation (`2.5e3`), decimals (`.5`), and hex (`0x10`). Non-finite or non-positive values fall back to the default. | Hunting a few outlier files dominating the deferred call-resolution stage; lower to surface more, raise to focus only on the worst. |
| `PROF_LBUG_LOAD` | unset | When `1`, emits one `[lbug-load prof]` summary line per `loadGraphToLbug` call breaking the graph-DB persistence wall into stages (`csv-emit` / `copy-nodes` / `copy-rels` / `fallback` / `total`) plus node & edge counts. Zero-cost when unset. | Attributing large-repo analyze wall time across CSV generation vs. LadybugDB `COPY` (issue #2203) — the analyze "emit" timing is the scope-resolution bucket, not this DB-write path. |
| `GITNEXUS_MAX_FILE_SIZE` | `512` (KB) | Walker skip threshold in KB. Hard cap is `32768` (tree-sitter buffer ceiling). Equivalent to `--max-file-size <kb>`. | Indexing repos with intentionally-large source files (generated parsers, vendored bundles) that should still be parsed. |
| `GITNEXUS_MAX_PROCESSES` | dynamic (`max(20, round(symbols/10))`) | Analyze-time process-detection process cap. Equivalent to `--max-processes <n>` / `.gitnexusrc` `maxProcesses`. Explicit values replace the dynamic formula (not a multiplier). `0` is invalid, not unlimited. Changing this re-detects flows on the next analyze without `--force`. Distinct from query-time `IMPACT_MAX_CHUNKS`. | `[processes] … whole flows are MISSING` names `--max-processes` after entry points were never traced or flows were dropped. Tracing does not start the next entry once collected traces already reach `maxProcesses * 2`; a started entry can still emit every trace that entry produces. |
| `GITNEXUS_MAX_PROCESS_BRANCHING` | `4` | Analyze-time per-node branching cap during flow tracing. Equivalent to `--max-process-branching <n>`. Shape-only: raising it shortens fewer traces; it does not restore whole missing flows. | A flow is present but `calleesDropped` is high at debug. |
| `GITNEXUS_MAX_PROCESS_TRACE_DEPTH` | `10` | Analyze-time DFS depth cap during flow tracing. Equivalent to `--max-process-trace-depth <n>`. Shape-only. | A reported flow is shorter than the code path (`tracesDepthCapped` at debug). |
| `GITNEXUS_MAX_ENTRY_POINT_CANDIDATES` | `200` | Ranked entry-point candidate pool. Equivalent to `--max-entry-point-candidates <n>`. Raising `--max-processes` alone does not clear `entryPointCandidatesDropped`. Doubling the current cap is the usual first raise; setting it to the full remaining candidate count can exhaust CPU and memory. | The `[processes]` warning reports candidate entry points that never ranked in. |
| `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS` | `30000` | Worker idle timeout in milliseconds before retry/fallback. Equivalent to `--worker-timeout <seconds>` × 1000. | Slow-parsing files (large minified JS, deeply-nested TS types) that legitimately need more than 30s. |
| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget in milliseconds for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. | Slow or heavily loaded hosts where a full pool cold-starting concurrently needs more than 5s, and analyze aborts with "did not report ready within 5000ms". |
| `GITNEXUS_FTS_STEMMER` | `porter` | Stemmer used when rebuilding BM25/FTS indexes. Use `none` for CJK-heavy repositories, or a language stemmer such as `german`, `french`, or `spanish` for matching repository comments. Re-run `gitnexus analyze --repair-fts` after changing it. | Keyword search quality is poor for non-English comments or identifiers under English stemming. |
@ -904,9 +911,10 @@ The web UI uses the same indexing pipeline as the CLI but runs entirely in WebAs
```bash
git clone https://github.com/abhigyanpatwari/gitnexus.git
cd gitnexus/gitnexus-shared && npm install && npm run build
cd ../gitnexus-web && npm install
npm run dev
cd gitnexus/gitnexus-web && npm install
# Compile sibling gitnexus-shared with this package's TypeScript 7 (do not npm ci shared).
cd ../gitnexus-shared && node ../gitnexus-web/node_modules/typescript/lib/tsc.js
cd ../gitnexus-web && npm run dev
# Then in another terminal, start the backend the frontend connects to:
npx gitnexus@latest serve
```

View file

@ -39,6 +39,8 @@ From `gitnexus-web/`:
### Before opening a PR
```bash
# gitnexus-shared/dist must exist first. `npm install` / `npm run build` in
# gitnexus/ compiles it via parent `lib/tsc.js` (do not npm ci gitnexus-shared).
cd gitnexus && npx tsc --noEmit && npm test
cd ../gitnexus-web && npx tsc -b --noEmit && npm test
```

14
eval/uv.lock generated
View file

@ -170,15 +170,15 @@ wheels = [
[[package]]
name = "anyio"
version = "4.12.1"
version = "4.14.2"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "idna" },
{ name = "typing-extensions", marker = "python_full_version < '3.13'" },
]
sdist = { url = "https://files.pythonhosted.org/packages/96/f0/5eb65b2bb0d09ac6776f2eb54adee6abe8228ea05b20a5ad0e4945de8aac/anyio-4.12.1.tar.gz", hash = "sha256:41cfcc3a4c85d3f05c932da7c26d0201ac36f72abd4435ba90d0464a3ffed703", size = 228685, upload-time = "2026-01-06T11:45:21.246Z" }
sdist = { url = "https://files.pythonhosted.org/packages/61/cc/a381afa6efea9f496eff839d4a6a1aed3bfafc7b3ab4b0d1b243a12573dd/anyio-4.14.2.tar.gz", hash = "sha256:cfa139f3ed1a23ee8f88a145ddb5ac7605b8bbfd8592baacd7ce3d8bb4313c7f", size = 260176, upload-time = "2026-07-12T20:29:07.082Z" }
wheels = [
{ url = "https://files.pythonhosted.org/packages/38/0e/27be9fdef66e72d64c0cdc3cc2823101b80585f8119b5c112c2e8f5f7dab/anyio-4.12.1-py3-none-any.whl", hash = "sha256:d405828884fc140aa80a3c667b8beed277f1dfedec42ba031bd6ac3db606ab6c", size = 113592, upload-time = "2026-01-06T11:45:19.497Z" },
{ url = "https://files.pythonhosted.org/packages/da/35/f2287558c17e29fafc8ef3daf819bb9834061cfa43bff8014f7df7f63bdc/anyio-4.14.2-py3-none-any.whl", hash = "sha256:9f505dda5ac9f0c8309b5e8bd445a8c2bf7246f3ce950121e45ea15bc41d1494", size = 125813, upload-time = "2026-07-12T20:29:05.763Z" },
]
[[package]]
@ -2513,11 +2513,11 @@ wheels = [
[[package]]
name = "pygments"
version = "2.19.2"
version = "2.20.0"
source = { registry = "https://pypi.org/simple" }
sdist = { url = "https://files.pythonhosted.org/packages/b0/77/a5b8c569bf593b0140bde72ea885a803b82086995367bf2037de0159d924/pygments-2.19.2.tar.gz", hash = "sha256:636cb2477cec7f8952536970bc533bc43743542f70392ae026374600add5b887", size = 4968631, upload-time = "2025-06-21T13:39:12.283Z" }
sdist = { url = "https://files.pythonhosted.org/packages/c3/b2/bc9c9196916376152d655522fdcebac55e66de6603a76a02bca1b6414f6c/pygments-2.20.0.tar.gz", hash = "sha256:6757cd03768053ff99f3039c1a36d6c0aa0b263438fcab17520b30a303a82b5f", size = 4955991, upload-time = "2026-03-29T13:29:33.898Z" }
wheels = [
{ url = "https://files.pythonhosted.org/packages/c7/21/705964c7812476f378728bdf590ca4b771ec72385c533964653c68e86bdc/pygments-2.19.2-py3-none-any.whl", hash = "sha256:86540386c03d588bb81d44bc3928634ff26449851e99741617ecb9037ee5ec0b", size = 1225217, upload-time = "2025-06-21T13:39:07.939Z" },
{ url = "https://files.pythonhosted.org/packages/f4/7e/a72dd26f3b0f4f2bf1dd8923c85f7ceb43172af56d63c7383eb62b332364/pygments-2.20.0-py3-none-any.whl", hash = "sha256:81a9e26dd42fd28a23a2d169d86d7ac03b46e2f8b59ed4698fb4785f946d0176", size = 1231151, upload-time = "2026-03-29T13:29:30.038Z" },
]
[[package]]
@ -2574,7 +2574,7 @@ name = "pyroscope-io"
version = "0.8.16"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "cffi", marker = "sys_platform != 'win32'" },
{ name = "cffi" },
]
wheels = [
{ url = "https://files.pythonhosted.org/packages/a8/50/607b38b120ba8adad954119ba512c53590c793f0cf7f009ba6549e4e1d77/pyroscope_io-0.8.16-py2.py3-none-macosx_11_0_arm64.whl", hash = "sha256:e07edcfd59f5bdce42948b92c9b118c824edbd551730305f095a6b9af401a9e8", size = 3138869, upload-time = "2026-01-22T06:23:24.664Z" },

View file

@ -98,6 +98,7 @@ export type { ResolveTypeRefContext } from './scope-resolution/resolve-type-ref.
// ScopeExtractor output contracts (RFC §3.2 Phase 1; Ring 2 PKG #919)
export type { ParsedFile } from './scope-resolution/parsed-file.js';
export type { CallResultAssignmentSite } from './scope-resolution/call-result-assignment-site.js';
export type {
ReferenceSite,
ReferenceKind,

View file

@ -0,0 +1,14 @@
import type { Range, ScopeId } from './types.js';
/**
* Compact extraction-time identity for `lhs = callee()`.
*
* The call-site range uses the same call-expression anchor as reference
* resolution, allowing downstream passes to join this fact to the exact
* resolved callee id without relying on a possibly polluted type binding.
*/
export interface CallResultAssignmentSite {
readonly callSite: Range;
readonly inScope: ScopeId;
readonly lhs: string;
}

View file

@ -56,6 +56,7 @@ import type { ParsedImport } from './types.js';
import type { SymbolDefinition } from './symbol-definition.js';
import type { ReferenceSite } from './reference-site.js';
import type { CallableFlowSite } from './callable-flow-site.js';
import type { CallResultAssignmentSite } from './call-result-assignment-site.js';
export interface ParsedFile {
readonly filePath: string;
@ -81,6 +82,8 @@ export interface ParsedFile {
* syntax remain source-compatible; consumers normalize absence to `[]`.
*/
readonly callableFlowSites?: readonly CallableFlowSite[];
/** Exact call-expression → assigned local identity for return-type replay. */
readonly callResultAssignmentSites?: readonly CallResultAssignmentSite[];
/**
* Opaque, language-private serialization of capture-time side-channel
* state that a provider's `emitScopeCaptures` populates into module-level

File diff suppressed because it is too large Load diff

View file

@ -20,10 +20,10 @@
"dependencies": {
"@langchain/anthropic": "^1.5.8",
"@langchain/core": "^1.2.8",
"@langchain/google-genai": "^2.2.0",
"@langchain/google-genai": "^2.3.1",
"@langchain/langgraph": "^1.4.14",
"@langchain/ollama": "^1.3.0",
"@langchain/openai": "^1.5.3",
"@langchain/openai": "^1.5.13",
"@sigma/edge-curve": "^3.1.0",
"@tailwindcss/vite": "^4.3.3",
"axios": "^1.20.0",
@ -40,12 +40,12 @@
"i18next-browser-languagedetector": "^8.2.1",
"langchain": "^1.5.11",
"lru-cache": "^11.5.2",
"lucide-react": "^1.31.0",
"lucide-react": "^1.44.0",
"mermaid": "^11.17.2",
"mnemonist": "^0.40.4",
"pandemonium": "^2.4.0",
"react": "^19.2.5",
"react-dom": "^19.2.8",
"react": "^19.3.0",
"react-dom": "^19.3.0",
"react-i18next": "^17.0.13",
"react-markdown": "^10.1.0",
"react-syntax-highlighter": "^16.1.1",
@ -57,23 +57,23 @@
"zod": "^4.5.4"
},
"devDependencies": {
"@babel/types": "^8.0.4",
"@playwright/test": "^1.62.1",
"@testing-library/jest-dom": "^7.0.0",
"@babel/types": "^8.0.5",
"@playwright/test": "^1.63.0",
"@testing-library/jest-dom": "^7.0.1",
"@testing-library/react": "^16.3.3",
"@testing-library/user-event": "^14.6.7",
"@types/dompurify": "^3.2.0",
"@types/node": "^26.0.1",
"@types/react": "^19.2.18",
"@types/react-dom": "^19.2.4",
"@types/node": "^26.5.1",
"@types/react": "^19.3.0",
"@types/react-dom": "^19.3.0",
"@types/react-syntax-highlighter": "^15.5.13",
"@vercel/node": "^5.10.2",
"@vitejs/plugin-react": "^6.1.1",
"@vitest/coverage-v8": "^4.1.11",
"jsdom": "^30.0.1",
"tree-sitter-wasms": "^0.1.13",
"typescript": "^5.4.5",
"vite": "^8.1.5",
"typescript": "^7.0.2",
"vite": "^8.3.0",
"vitest": "^4.1.10",
"wait-on": "^9.1.0"
},

View file

@ -0,0 +1,16 @@
import { readFileSync } from 'node:fs';
import path from 'node:path';
import { fileURLToPath } from 'node:url';
import { describe, expect, it } from 'vitest';
const webRoot = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
describe('TypeScript 7 + React frontend toolchain', () => {
it('keeps Vite and Vitest on the automatic JSX runtime that matches react-jsx', () => {
for (const configFile of ['vite.config.ts', 'vitest.config.ts'] as const) {
const source = readFileSync(path.join(webRoot, configFile), 'utf8');
expect(source).toContain('@vitejs/plugin-react');
expect(source).toContain("jsxRuntime: 'automatic'");
}
});
});

View file

@ -1,26 +1,29 @@
{
"compilerOptions": {
"target": "ESNext",
"lib": ["ESNext", "DOM", "DOM.Iterable"],
"useDefineForClassFields": true,
"module": "ESNext",
"moduleResolution": "Bundler",
"moduleDetection": "force",
"allowImportingTsExtensions": true,
"resolveJsonModule": true,
"isolatedModules": true,
"noEmit": true,
"jsx": "react-jsx",
"jsxImportSource": "react",
"strict": true,
"esModuleInterop": true,
"allowSyntheticDefaultImports": true,
"forceConsistentCasingInFileNames": true,
"skipLibCheck": true,
"baseUrl": "./",
"noUncheckedSideEffectImports": true,
"rootDir": "./src",
"paths": {
"@/*": ["./src/*"],
"@shared/*": ["../shared/*"]
"@/*": ["./src/*"]
},
"types": ["vite/client"]
},
"include": ["src", "../shared"],
"include": ["src"],
"exclude": ["src/**/*.test.ts", "src/**/*.test.tsx"]
}

View file

@ -2,11 +2,15 @@
"compilerOptions": {
"composite": true,
"skipLibCheck": true,
"target": "ESNext",
"lib": ["ESNext"],
"module": "ESNext",
"moduleResolution": "Bundler",
"moduleDetection": "force",
"allowSyntheticDefaultImports": true,
"esModuleInterop": true,
"noEmit": true,
"rootDir": ".",
"types": ["node"]
},
"include": ["vite.config.ts"]

View file

@ -1,3 +1,3 @@
{
"installCommand": "cd ../gitnexus-shared && npm install && npm run build && cd ../gitnexus-web && npm ci --include=dev"
"installCommand": "npm ci --include=dev && cd ../gitnexus-shared && node ../gitnexus-web/node_modules/typescript/lib/tsc.js"
}

View file

@ -8,7 +8,7 @@ const _require = createRequire(import.meta.url);
const gitnexusPkg = _require('../gitnexus/package.json');
export default defineConfig({
plugins: [react(), tailwindcss()],
plugins: [react({ jsxRuntime: 'automatic' }), tailwindcss()],
define: {
__REQUIRED_NODE_VERSION__: JSON.stringify(gitnexusPkg.engines.node.replace(/[>=^~\s]/g, '')),
},

View file

@ -7,7 +7,7 @@ const _require = createRequire(import.meta.url);
const gitnexusPkg = _require('../gitnexus/package.json');
export default defineConfig({
plugins: [react()],
plugins: [react({ jsxRuntime: 'automatic' })],
define: {
__REQUIRED_NODE_VERSION__: JSON.stringify(gitnexusPkg.engines.node.replace(/[>=^~\s]/g, '')),
},

View file

@ -245,6 +245,8 @@ gitnexus analyze --skip-agents-md # Preserve custom AGENTS.md/CLAUDE.md gitnexu
gitnexus analyze --skip-skills # Skip installing standard .claude/skills/gitnexus-* skill files
gitnexus analyze --skip-git # Index folders that are not Git repositories
gitnexus analyze --workers <n> # Parse worker pool size (>=1; default: cores-1, capped at 16)
gitnexus analyze --max-processes <n> # Process-detection process cap (replaces dynamic max(20, round(symbols/10)))
gitnexus analyze --max-entry-point-candidates <n> # Ranked entry-point pool (default 200; raise when the warning names it)
gitnexus analyze --spring-actuator ./actuator # Enrich with local Spring Boot Actuator JSON snapshots
gitnexus analyze --verbose # Log skipped files when parsers are unavailable
gitnexus analyze --max-file-size 1024 # Skip files larger than N KB (default: 512, cap: 32768)
@ -298,7 +300,8 @@ installation. Run a one-shot `gitnexus analyze` when those generated files need
updating. Stop watch mode with Ctrl+C.
Watch mode accepts `--debounce`, `--workers`, `--worker-timeout`,
`--max-file-size`, `--branch`, `--pdg`, `--name`, `--allow-duplicate-name`, and
`--max-file-size`, `--max-processes`, `--max-process-branching`,
`--max-process-trace-depth`, `--max-entry-point-candidates`, `--branch`, `--pdg`, `--name`, `--allow-duplicate-name`, and
`--verbose`. Explicit one-shot options such as `--force`, `--repair-fts`,
embedding flags, `--skills`, `--default-branch`, `--skip-agents-md`,
`--skip-skills`, `--no-stats`, `--self-commit`, `--index-only`, and `--skip-git`
@ -764,6 +767,22 @@ npx gitnexus analyze
Values above **32768 KB (32 MB)** are clamped to the tree-sitter parser ceiling; invalid values fall back to the 512 KB default with a one-time warning. When an override is active, `analyze` prints the effective threshold in its startup banner (e.g. `GITNEXUS_MAX_FILE_SIZE: effective threshold 2048KB (default 512KB)`).
### Process detection reports missing flows
On a large repository, `analyze` may warn that `[processes] … whole flows are MISSING`. That means ranked entry points or completed flows were sampled away by the analyze-time detection budget — not that the code path is absent, and not the query-time `IMPACT_MAX_CHUNKS` cap.
Defaults stay in place when nothing is set: dynamic `maxProcesses = max(20, round(non-File symbols / 10))`, branching `4`, trace depth `10`, entry-point candidate pool `200`. Raise a knob only when the warning names it:
```bash
# Usual first move when entryPointCandidatesDropped is the loud counter
npx gitnexus analyze --max-entry-point-candidates 400
# When ranked entry points were never traced, or flows were dropped at maxProcesses
npx gitnexus analyze --max-processes 80
```
Equivalent `.gitnexusrc` keys: `maxProcesses`, `maxProcessBranching`, `maxProcessTraceDepth`, `maxEntryPointCandidates`. Equivalent env vars: `GITNEXUS_MAX_PROCESSES`, `GITNEXUS_MAX_PROCESS_BRANCHING`, `GITNEXUS_MAX_PROCESS_TRACE_DEPTH`, `GITNEXUS_MAX_ENTRY_POINT_CANDIDATES`. Precedence is CLI > `.gitnexusrc` > env > default. `0` is invalid, not unlimited. Changing these knobs re-runs process detection on the next `analyze` without `--force`. Raising them increases CPU and memory; this is not a heap-OOM fix.
### Analyze reports a worker timeout
Worker parse timeouts are recoverable. GitNexus retries stalled worker jobs with backoff, splits large jobs to isolate slow files, and quarantines a file that repeatedly crashes its worker (respawning the slot so the pool keeps going). If a large repository needs more time per worker job, use either:

View file

@ -659,6 +659,11 @@
"path_segments": 13,
"probe": "ExternalPkg0"
},
"context": {
"target": "App",
"with_context": "Sources/App/Lib.swift,Sources/Models/User.swift",
"without_context": "Sources/App/Lib.swift"
},
"_measured": {
"collide_ms": 0.819,
"collide_scaling_ratio": 3.454,

View file

@ -627,7 +627,7 @@ const HEAP_BUDGETED = [
* their hooks declare three or four parameters — so their numbers stay exactly
* where they were.
*/
const CONTEXT_LANGS = ['php', 'java', 'kotlin', 'python'];
const CONTEXT_LANGS = ['php', 'java', 'kotlin', 'python', 'swift'];
/**
* Needs `node --expose-gc` to force collection for a clean delta; without it
@ -1869,7 +1869,7 @@ function resolveOne(lang, from, target, pass) {
if (lang === 'swift') {
return resolveSwiftImportTarget(
{ kind: 'namespace', localName: 'X', importedName: 'X', targetRaw: target },
{ fromFile: from, allFilePaths },
{ fromFile: from, allFilePaths, parsedFiles: pass.parsedFiles },
);
}
if (lang === 'rust') return resolveRustImportTarget(target, from, allFilePaths, undefined);
@ -2326,6 +2326,26 @@ const CONTEXT_PROBE = {
probeFile('app/main.py', [['Function', 'app.main.run']]),
],
},
/**
* `import App` where App `@_exported import`s Models. With parsedFiles the
* re-export closure unions Models' files; without it the adapter returns
* only App's own file. Two distinct non-null answers, so a dropped
* `parsedFiles` cannot look like a miss.
*/
swift: {
from: 'Sources/Client/Main.swift',
target: 'App',
parsedFiles: [
{
...probeFile('Sources/App/Lib.swift', [['Class', 'App.Lib']]),
parsedImports: [
{ kind: 'reexport', localName: 'Models', importedName: 'Models', targetRaw: 'Models' },
],
},
probeFile('Sources/Models/User.swift', [['Class', 'Models.User']]),
probeFile('Sources/Client/Main.swift', [['Class', 'Client.Main']]),
],
},
};
/** Resolve the probe twice through `resolveOne` — once with the pass's parsed
@ -3065,7 +3085,7 @@ expectNoOrphanKeys(
// against a claim in a comment. `run.ts` passes the fifth argument to every
// provider; which ones can OBSERVE it is decided by how many parameters each
// hook declares, and that is a number the registry can be asked for. Today
// exactly four answer 5 (php, java, kotlin, python) and the other thirteen answer 3 or 4 —
// exactly five answer 5 (php, java, kotlin, python, swift) and the other twelve answer 3 or 4 —
// which is why thirteen arms can ignore this whole question and their numbers
// did not move when it was fixed.
//

View file

@ -117,7 +117,7 @@
"_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior fc81941b0a921074fa80dc448284de9a23bd07358ddc84d4894797cc08c3fe83 -> 1c8c9c4b54036fa24c2a81e39ea530e938645c856d369075e5f437da78218c57."
},
"swift": {
"fingerprint": "adef9284feaecd39cb490aebce83876e15b9150c7a04b00a396feb78b7e1e0a9",
"fingerprint": "decf74c01af0c7f403e203b19c2bd92dbf95b4562f6da2b0ac5117ef872204da",
"scaling_budget": 1.5,
"_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 5f923c6604d825d12b249f31c155b0f4d13a8379d532e5dde64a0f9b15cf4725 -> 7687ee2466e16020a12440a03fbda53e63aa05f94b4481f6133c09867a0d560d; scaling 1.042 < 1.5.",
"_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Swift function-value callable flow facts with invocation-result suppression. Prior 180ac68e780bdf6f9089d53f51cbb9a66aed3e7774631cc3fcbaae5020213998 -> 5f923c6604d825d12b249f31c155b0f4d13a8379d532e5dde64a0f9b15cf4725; scaling 1.043 < 1.5.",
@ -125,7 +125,13 @@
"_rebaselined_2522_review_fixes": "PR #2522 review fixes: assignment target:/result: fields join the shared fallback. Prior 7687ee2466e16020a12440a03fbda53e63aa05f94b4481f6133c09867a0d560d -> 115c5da807e36bb12fdeba28e44f2b6484ef322ff26c19fa0f191febaf774248; scaling ratio re-verified within budget.",
"_rebaselined_receiver_chain_2747": "#2747 receiver-chain rollout: call matches whose receiver is itself an expression now carry `@reference.receiver-chain`, a compact encoding of the receiver's structure, so resolution types it by folding instead of re-parsing receiver source text. Capture GROUP counts are unchanged \u2014 the tag is added to existing call matches, never a new match \u2014 so this is digest drift only. Prior 115c5da807e36bb12fdeba28e44f2b6484ef322ff26c19fa0f191febaf774248 -> a6fca5f052ae5ec635b56051e28a168c864a988b2221a3279ddd69807378ba0b.",
"_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior a6fca5f052ae5ec635b56051e28a168c864a988b2221a3279ddd69807378ba0b -> 2f04ae960123cf50138a49fabdc5a146c2963170cecf5755c552b23c9055a9e7.",
"_rebaselined_inferred_field_receiver_2807": "#2807: optional property annotations (`var a: Outer?`) now emit a type binding. The prior pattern required the `user_type` to be a DIRECT child of the annotation, so an `optional_type` wrapper meant an optional field was never typed at all and its receiver could not resolve. ADDS @type-binding.annotation captures on the optional form only; no capture is removed. Prior 2f04ae960123cf50138a49fabdc5a146c2963170cecf5755c552b23c9055a9e7 -> adef9284feaecd39cb490aebce83876e15b9150c7a04b00a396feb78b7e1e0a9; scaling 1.023 < 1.5."
"_rebaselined_inferred_field_receiver_2807": "#2807: optional property annotations (`var a: Outer?`) now emit a type binding. The prior pattern required the `user_type` to be a DIRECT child of the annotation, so an `optional_type` wrapper meant an optional field was never typed at all and its receiver could not resolve. ADDS @type-binding.annotation captures on the optional form only; no capture is removed. Prior 2f04ae960123cf50138a49fabdc5a146c2963170cecf5755c552b23c9055a9e7 -> adef9284feaecd39cb490aebce83876e15b9150c7a04b00a396feb78b7e1e0a9; scaling 1.023 < 1.5.",
"_rebaselined_nested_constructor_3262": "#3262: the existing nested-constructor fixture now uses the qualified legal form `extension Outer.Container`, covering full extension-owner recovery. Fixture-only capture drift; no capture logic changed. Prior adef9284feaecd39cb490aebce83876e15b9150c7a04b00a396feb78b7e1e0a9 -> 3ac80b64776969effb75c8d32aa0bc38fc6a41c6c115a33829c502559fa404f9; scaling 1.025 < 1.5.",
"_rebaselined_3309_call_result_assignment": "#3309: the Swift protocol-extension regression fixture adds one file with an untyped local initialized from a helper call, plus a compound-field receiver call. The capture query is unchanged; this is fixture-corpus growth only. capture_groups_fp 1235 -> 1272 and fixture_count 62 -> 64; synthetic scale counts remain 5012/16012. Prior adef9284feaecd39cb490aebce83876e15b9150c7a04b00a396feb78b7e1e0a9 -> 4d32535d454cd79086f0fcaec3b75de855c1314e73630321460e569f19aa484f; measured scaling 1.020 < 1.5.",
"_rebaselined_3309_exact_return_replay": "#3309 follow-up: Swift callable declarations now emit exact return-type captures, and untyped local initializers emit call-result assignment captures through direct, try, await, and try-await forms. This is intentional capture-set growth covered by exact-callable collision and wrapper regressions. Prior 4d32535d454cd79086f0fcaec3b75de855c1314e73630321460e569f19aa484f -> 724553c91e8ebcf872d7302b14105d59476c230f9b623acdacc0b58e39912329; hosted benchmark scaling remained within the existing 1.5 budget.",
"_rebaselined_3308_merge_main": "Merge origin/main into #3308: combine #3262 fixture-owner recovery with #3309 protocol-extension capture growth. Prior 3ac80b64776969effb75c8d32aa0bc38fc6a41c6c115a33829c502559fa404f9 + 724553c91e8ebcf872d7302b14105d59476c230f9b623acdacc0b58e39912329 -> b0c89f8de1a4ce30409182e79c1c14cbc49a509d18d853c0d7c5f22f1f66be06; capture_groups_fp 1297, fixture_count 67.",
"_rebaselined_3308_public_extension": "#3308 review: the nested-constructor fixture now uses `public extension Outer.Container` so owner recovery is gated on modifier-prefixed headers. Fixture-only digest drift. Prior b0c89f8de1a4ce30409182e79c1c14cbc49a509d18d853c0d7c5f22f1f66be06 -> a6d61e3749c9c02b9a54611e42d03db02f8fabfded3d7ce53a1b7229d3d38d59; capture_groups_fp 1297, fixture_count 67.",
"_rebaselined_3308_public_enclosing_types": "#3308 review: `Outer` / `Container` / `Entry` in the nested-constructor fixture are now `public` so the `public extension` is valid Swift. Fixture-only digest drift (Types.swift capture text). Prior a6d61e3749c9c02b9a54611e42d03db02f8fabfded3d7ce53a1b7229d3d38d59 -> decf74c01af0c7f403e203b19c2bd92dbf95b4562f6da2b0ac5117ef872204da; capture_groups_fp 1297."
},
"dart": {
"fingerprint": "3a8ddabbeb1cba47a4757451d4f79d726ca230fd15e860772b11526fbb1c6687",

View file

@ -0,0 +1,52 @@
{
"_what": "Baselines for bench/swift-package-imports/measure.mjs --check. Guards Swift Package.swift declaration resolve after #2964 / #2931: declared-target hits, SDK and undeclared-folder misses, empty declaredTargets fail-closed, @_exported union, #2931 nested-prefix membership, first-wins grouping, and https:// factory survival. Same approach as bench/parse-dispatch-rounds/baselines.json — exact floors and a fingerprint first; the only timing arms are ratios.",
"_triage": "READ THIS BEFORE RE-RUNNING. targets, files, imports, declared_resolved, sdk_external, undeclared_external, empty_declared_external, reexport_extra, nested_repeat_resolved, first_wins, parse_targets, parse_binary_skipped, url_comment_targets, parse_complete and layout_fingerprint are DETERMINISTIC: a re-run never changes them, and none may be re-baselined to make CI green. resolve_scaling_ratio, strategy_scaling_ratio and parse_scaling_ratio are the only timing arms; runner contention dominates them, so re-run on an idle machine before investigating and read the reported `reps` first. If exactly one arm fails and it is a timing arm, suspect the machine.",
"targets": 16,
"files": 260,
"imports": 12288,
"_shape_note": "THE FLOOR. Without these three, every arm below is a ceiling over nothing. declared_resolved only asserts something while the corpus still walks many files and imports. Shrink it to one happy-path module and declared_resolved still reads 32 and still passes, asserting a property the corpus no longer has. Files-per-target is 16 and does not scale — 4x grows target count so a hit returns a constant-size file list. Scaling files-per-target instead makes the return-list copy look quadratic even when the index is linear (the same trap bench/import-target documents for Swift collide_scaling).",
"declared_resolved": 32,
"sdk_external": 32,
"undeclared_external": 16,
"empty_declared_external": 1,
"reexport_extra": 1,
"nested_repeat_resolved": 1,
"first_wins": 1,
"_membership_note": "Exact resolve/grouping counts on the unique query set (one File0 per target). declared_resolved=32 is two declared spellings (Mod{t+1} and Mod{t+1}.Model) across 16 targets. sdk_external=32 is Foundation+UIKit. undeclared_external=16 is the in-repo CoreUI folder that Package.swift does not declare — the hole bench/import-target's Swift arm cannot see. empty_declared_external=1 pins an empty declaredTargets map. reexport_extra=1 pins import Mod0 including Mod1 after @_exported. nested_repeat_resolved=1 pins vendor/Sources/Mod0/Sources/Mod0/Nested.swift (#2931). first_wins=1 pins a two-prefix file into Mod0 only.",
"parse_targets": 3,
"parse_binary_skipped": 1,
"url_comment_targets": 1,
"parse_complete": 1,
"_parse_note": "Exact parse floors on a fixed manifest, not the scaling corpus. parse_targets=3 is Models+App+AppTests. parse_binary_skipped=1 means Lib/Gen/CFoo never entered the map. url_comment_targets=1 pins https:// on the same line as a later .target. parse_complete=1 pins both manifests readable.",
"layout_fingerprint": "2b220baa1df63778489949f9596879ba1c89a49e62c57bd9d2336fd8d74c17bd",
"_layout_fingerprint_note": "sha256 over sorted from|target->files rows on the unique query set. A change here is a BEHAVIOUR change — the resolver returned a different target set. Explain it, never re-baseline it alone.",
"resolve_scaling_budget": 1.9,
"strategy_scaling_budget": 1.6,
"parse_scaling_budget": 1.9,
"_resolve_scaling_note": "(t_4n / t_n) / 4 for resolveSwiftImportTarget over the declared corpus; ~1.0 is linear. A RATIO rather than a millisecond ceiling, deliberately: wall-clock is runner-speed-dependent, and this repo has already been bitten by a fixed ms budget. A per-import allFilePaths scan scores ~4. Budget is 1.9, i.e. 1.51x the measured maximum 1.261 — this file's siblings use ~1.5x on ratios. min-of-15 estimator.",
"_strategy_scaling_note": "(t_4n / t_n) / 4 for swiftPackageStrategy (the import-config WeakMap index). Measured maximum 1.030; budget 1.6 is 1.55x, matching parse-dispatch-rounds' pack_scaling_budget.",
"_parse_scaling_note": "(t_4n / t_n) / 4 for parseSwiftPackageManifest over 256 vs 1024 factories. Measured maximum 1.276; budget 1.9 is 1.49x. The small cell is still sub-millisecond, so this budget is looser than strategy on purpose.",
"_measured": {
"resolve_scaling_ratio": 1.261,
"resolve_scaling_ratio_samples": [1.261, 1.201, 1.182, 1.241, 1.212],
"strategy_scaling_ratio": 1.03,
"strategy_scaling_ratio_samples": [1.019, 0.994, 0.97, 1.015, 1.03],
"parse_scaling_ratio": 1.276,
"parse_scaling_ratio_samples": [1.25, 1.276, 1.215, 1.224, 1.227],
"resolve_ms": 3.5,
"resolve_ms_4x": 17.16,
"strategy_ms": 0.67,
"strategy_ms_4x": 2.65,
"parse_ms": 0.41,
"parse_ms_4x": 2.01,
"reps": 15
},
"_measured_note": "Maxima (and sample lists) over 5 consecutive local runs using the min-of-15 estimator from parse-dispatch-rounds. Milliseconds are diagnostic context only — nothing gates on them."
}

View file

@ -0,0 +1,475 @@
/**
* Build-free bench for Swift Package.swift declaration resolve (#2964 / #2931).
*
* WHY THIS EXISTS. `bench/import-target`'s Swift arm never threads a
* `resolutionConfig`. It times the no-manifest directory-segment index
* (`getSwiftModuleIndex`) and cannot see a revert that starts scanning
* `allFilePaths` per `import`, drops declaration-only resolve, treats
* `https://` as a comment, or matches a target dir inside another path
* segment (#2931). Graph output on a tiny fixture is identical either way.
* This file is the same shape as `bench/parse-dispatch-rounds`: exact floors
* first, ratio timing only, never a millisecond ceiling.
*
* ARMS:
*
* - `targets` / `files` / `imports` — EXACT, and they are the FLOOR.
* `declared_resolved` only asserts something while the corpus still has
* many files and imports to walk. Shrink it to one happy-path module and
* the resolve arm still passes, gating a property the corpus no longer has.
*
* - `declared_resolved` / `sdk_external` / `undeclared_external` /
* `empty_declared_external` / `reexport_extra` / `nested_repeat_resolved` /
* `first_wins` — EXACT. Declaration map hits stay hits. Foundation / UIKit
* and an in-repo `CoreUI` folder that is NOT in Package.swift stay
* external. An empty `declaredTargets` map fails every name closed. Import
* of a module that `@_exported import`s another unions that module's files.
* A `#2931` nested `Sources/Mod0` path still belongs to Mod0. Grouping
* assigns a two-prefix file to the FIRST target only.
*
* - `parse_targets` / `parse_binary_skipped` / `url_comment_targets` /
* `parse_complete` — EXACT. Source factories are kept; binary/plugin
* factories are not modules; `https://` on the same line does not hide a
* later `.target`.
*
* - `layout_fingerprint` — EXACT. sha256 over sorted `from|target->files`
* rows on the unique query set. Catches a target-set change that leaves
* the counts intact.
*
* - `resolve_scaling_ratio` / `strategy_scaling_ratio` / `parse_scaling_ratio`
* — the only timing arms, RATIOS not millisecond ceilings.
* `(t_4n / t_n) / 4` divides the machine out; ~1.0 is linear.
* Files-per-target is FIXED while target count (and therefore files and
* imports) grow 4x, so a hit returns a constant-size file list. A correct
* once-per-pass index is then linear in imports; a per-import scan of
* allFilePaths scores ~4. Parse scales factory count 4x on a fixed-shape
* manifest.
*
* Usage:
* node --import tsx bench/swift-package-imports/measure.mjs
* node --import tsx bench/swift-package-imports/measure.mjs --check
*/
import { createHash } from 'node:crypto';
import { readFileSync } from 'node:fs';
import { performance } from 'node:perf_hooks';
import { parseSwiftPackageManifest } from '../../src/core/ingestion/language-config.ts';
import { swiftPackageStrategy } from '../../src/core/ingestion/import-resolvers/configs/swift.ts';
import { resolveSwiftImportTarget } from '../../src/core/ingestion/languages/swift/import-target.ts';
import { groupSwiftFilesBySpmTarget } from '../../src/core/ingestion/languages/swift/target-grouping.ts';
const baselines = JSON.parse(readFileSync(new URL('./baselines.json', import.meta.url), 'utf8'));
const REPS = 15;
const TARGETS_SMALL = 16;
const FILES_PER_TARGET = 16;
const QUERY_REPEATS = 8;
const PARSE_TARGETS = 256;
const SCALE = 4;
function queryKinds(targetCount) {
return [
(t) => ({ raw: `Mod${(t + 1) % targetCount}`, expect: 'declared' }),
() => ({ raw: 'Foundation', expect: 'sdk' }),
() => ({ raw: 'CoreUI', expect: 'undeclared' }),
(t) => ({ raw: `Mod${(t + 1) % targetCount}.Model`, expect: 'declared' }),
() => ({ raw: 'UIKit', expect: 'sdk' }),
() => ({ raw: 'Ghost', expect: 'miss' }),
];
}
function parsedImport(targetRaw) {
return { kind: 'namespace', localName: targetRaw, importedName: targetRaw, targetRaw };
}
function stubFile(filePath, parsedImports = []) {
return {
filePath,
moduleScope: `module:${filePath}`,
scopes: [],
parsedImports,
localDefs: [],
referenceSites: [],
};
}
function targetMap(targetCount) {
const targets = new Map();
for (let t = 0; t < targetCount; t++) targets.set(`Mod${t}`, `Sources/Mod${t}`);
return targets;
}
function declaredConfig(targets) {
return { origin: 'package.swift', targets, declaredTargets: targets };
}
function buildCorpus(targetCount) {
const targets = targetMap(targetCount);
const files = [];
const parsedFiles = [];
for (let t = 0; t < targetCount; t++) {
for (let i = 0; i < FILES_PER_TARGET; i++) {
const filePath = `Sources/Mod${t}/File${i}.swift`;
files.push(filePath);
const imports =
t === 0 && i === 0
? [{ kind: 'reexport', localName: 'Mod1', importedName: 'Mod1', targetRaw: 'Mod1' }]
: [];
parsedFiles.push(stubFile(filePath, imports));
}
}
const foundation = 'Sources/Foundation/Thing.swift';
const coreUi = 'Sources/CoreUI/View.swift';
const nested = 'vendor/Sources/Mod0/Sources/Mod0/Nested.swift';
const clash = 'Sources/Mod0/Vendor/Sources/Mod1/Clash.swift';
for (const extra of [foundation, coreUi, nested, clash]) {
files.push(extra);
parsedFiles.push(stubFile(extra));
}
const kinds = queryKinds(targetCount);
const queries = [];
for (let t = 0; t < targetCount; t++) {
for (let i = 0; i < FILES_PER_TARGET; i++) {
const from = `Sources/Mod${t}/File${i}.swift`;
for (let r = 0; r < QUERY_REPEATS; r++) {
for (const kind of kinds) {
const { raw, expect } = kind(t);
queries.push({ from, raw, expect, unique: r === 0 && i === 0 });
}
}
}
}
return {
files,
parsedFiles,
queries,
targets,
config: declaredConfig(targets),
extras: { foundation, coreUi, nested, clash },
targetCount,
};
}
function outcomeKey(from, raw, result) {
if (result == null) return `${from}|${raw}-><null>`;
const list = typeof result === 'string' ? [result] : [...result];
return `${from}|${raw}->${list.sort().join(',')}`;
}
function resultFiles(result) {
if (result == null) return [];
return typeof result === 'string' ? [result] : [...result];
}
function resolveOne(query, allFilePaths, config, parsedFiles) {
return resolveSwiftImportTarget(parsedImport(query.raw), {
fromFile: query.from,
allFilePaths,
resolutionConfig: config,
parsedFiles,
});
}
function strategyCtx(files, config) {
return {
allFilePaths: new Set(files),
allFileList: files,
normalizedFileList: files.map((p) => p.replace(/\\/g, '/')),
index: {},
resolveCache: new Map(),
configs: {
tsconfigPaths: null,
goModule: null,
composerConfig: null,
swiftPackageConfig: config,
csharpConfigs: [],
},
};
}
function resolvePass(corpus) {
const allFilePaths = new Set(corpus.files);
let resolved = 0;
for (const query of corpus.queries) {
if (resolveOne(query, allFilePaths, corpus.config, corpus.parsedFiles) != null) resolved++;
}
return resolved;
}
function strategyPass(corpus) {
const ctx = strategyCtx(corpus.files, corpus.config);
let resolved = 0;
for (const query of corpus.queries) {
if (swiftPackageStrategy(query.raw, query.from, ctx) != null) resolved++;
}
return resolved;
}
function manifestFor(targetCount) {
const rows = [];
for (let i = 0; i < targetCount; i++) {
rows.push(
` .target(name: "Mod${i}", dependencies: [.product(name: "X", package: "https://example.com/x")]),`,
);
if (i % 8 === 0) {
rows.push(
` .binaryTarget(name: "Bin${i}", url: "https://example.com/b${i}.xcframework"),`,
);
}
}
return `let package = Package(\n name: "Demo",\n targets: [\n${rows.join('\n')}\n ]\n)\n`;
}
function fastest(fn, reps) {
fn();
let best = Infinity;
for (let r = 0; r < reps; r++) {
const t0 = performance.now();
fn();
best = Math.min(best, performance.now() - t0);
}
return best;
}
function correctness(corpus) {
const allFilePaths = new Set(corpus.files);
const records = [];
let declaredResolved = 0;
let sdkExternal = 0;
let undeclaredExternal = 0;
const unique = corpus.queries.filter((q) => q.unique);
for (const query of unique) {
const result = resolveOne(query, allFilePaths, corpus.config, corpus.parsedFiles);
records.push(outcomeKey(query.from, query.raw, result));
if (query.expect === 'declared' && result != null) declaredResolved++;
if (query.expect === 'sdk' && result == null) sdkExternal++;
if (query.expect === 'undeclared' && result == null) undeclaredExternal++;
}
const fromOther = 'Sources/Mod2/File0.swift';
const reexport = resultFiles(
resolveOne({ from: fromOther, raw: 'Mod0' }, allFilePaths, corpus.config, corpus.parsedFiles),
);
const reexportExtra =
reexport.includes('Sources/Mod0/File0.swift') && reexport.includes('Sources/Mod1/File0.swift')
? 1
: 0;
const nestedRepeatResolved = reexport.includes(corpus.extras.nested) ? 1 : 0;
const empty = resolveOne(
{ from: fromOther, raw: 'Mod0' },
allFilePaths,
{ origin: 'package.swift', targets: corpus.targets, declaredTargets: new Map() },
corpus.parsedFiles,
);
const emptyDeclaredExternal = empty == null ? 1 : 0;
const groups = groupSwiftFilesBySpmTarget(corpus.files, (p) => p, corpus.targets);
const firstWins =
groups.get('Mod0')?.includes(corpus.extras.clash) === true &&
groups.get('Mod1')?.includes(corpus.extras.clash) !== true
? 1
: 0;
const parseFixed = parseSwiftPackageManifest(`
let package = Package(
name: "Demo",
dependencies: [.package(url: "https://example.com/foo.git", from: "1.0.0")],
targets: [
.target(name: "Models"),
.target(name: "App"),
.binaryTarget(name: "Lib", url: "https://example.com/Lib.xcframework", checksum: "abc"),
.testTarget(name: "AppTests"),
.plugin(name: "Gen"),
.systemLibrary(name: "CFoo"),
]
)
`);
const urlComment = parseSwiftPackageManifest(
'let package = Package(name: "Demo", dependencies: [.package(url: "https://example.com/foo.git", from: "1.0.0")], targets: [.target(name: "T")])',
);
return {
declaredResolved,
sdkExternal,
undeclaredExternal,
emptyDeclaredExternal,
reexportExtra,
nestedRepeatResolved,
firstWins,
parseTargets: parseFixed.targets.size,
parseBinarySkipped:
parseFixed.targets.has('Lib') ||
parseFixed.targets.has('Gen') ||
parseFixed.targets.has('CFoo')
? 0
: 1,
urlCommentTargets: urlComment.targets.get('T') === 'Sources/T' ? 1 : 0,
parseComplete: parseFixed.complete && urlComment.complete ? 1 : 0,
fingerprint: createHash('sha256').update(records.sort().join('\n')).digest('hex'),
};
}
const small = buildCorpus(TARGETS_SMALL);
const large = buildCorpus(TARGETS_SMALL * SCALE);
const shape = correctness(small);
const smallResolveMs = fastest(() => resolvePass(small), REPS);
const largeResolveMs = fastest(() => resolvePass(large), REPS);
const resolveScaling = largeResolveMs / smallResolveMs / SCALE;
const smallStrategyMs = fastest(() => strategyPass(small), REPS);
const largeStrategyMs = fastest(() => strategyPass(large), REPS);
const strategyScaling = largeStrategyMs / smallStrategyMs / SCALE;
const parseSmallSrc = manifestFor(PARSE_TARGETS);
const parseLargeSrc = manifestFor(PARSE_TARGETS * SCALE);
const parseSmall = parseSwiftPackageManifest(parseSmallSrc);
const parseLarge = parseSwiftPackageManifest(parseLargeSrc);
const smallParseMs = fastest(() => parseSwiftPackageManifest(parseSmallSrc), REPS);
const largeParseMs = fastest(() => parseSwiftPackageManifest(parseLargeSrc), REPS);
const parseScaling = largeParseMs / smallParseMs / SCALE;
const files = small.files.length;
const imports = small.queries.length;
console.log(`targets : ${small.targetCount} (expect ${baselines.targets})`);
console.log(`files : ${files} (expect ${baselines.files})`);
console.log(`imports : ${imports} (expect ${baselines.imports})`);
console.log(
`declared_resolved : ${shape.declaredResolved} (expect ${baselines.declared_resolved})`,
);
console.log(`sdk_external : ${shape.sdkExternal} (expect ${baselines.sdk_external})`);
console.log(
`undeclared_external : ${shape.undeclaredExternal} (expect ${baselines.undeclared_external})`,
);
console.log(
`empty_declared_external: ${shape.emptyDeclaredExternal} (expect ${baselines.empty_declared_external})`,
);
console.log(
`reexport_extra : ${shape.reexportExtra} (expect ${baselines.reexport_extra})`,
);
console.log(
`nested_repeat_resolved : ${shape.nestedRepeatResolved} (expect ${baselines.nested_repeat_resolved})`,
);
console.log(`first_wins : ${shape.firstWins} (expect ${baselines.first_wins})`);
console.log(`parse_targets : ${shape.parseTargets} (expect ${baselines.parse_targets})`);
console.log(
`parse_binary_skipped : ${shape.parseBinarySkipped} (expect ${baselines.parse_binary_skipped})`,
);
console.log(
`url_comment_targets : ${shape.urlCommentTargets} (expect ${baselines.url_comment_targets})`,
);
console.log(
`parse_complete : ${shape.parseComplete} (expect ${baselines.parse_complete})`,
);
console.log(`layout_fingerprint : ${shape.fingerprint}`);
console.log(
`resolve_scaling_ratio : ${resolveScaling.toFixed(3)} (budget <= ${baselines.resolve_scaling_budget}; ~1.0 is linear)`,
);
console.log(
`strategy_scaling_ratio : ${strategyScaling.toFixed(3)} (budget <= ${baselines.strategy_scaling_budget}; ~1.0 is linear)`,
);
console.log(
`parse_scaling_ratio : ${parseScaling.toFixed(3)} (budget <= ${baselines.parse_scaling_budget}; ~1.0 is linear)`,
);
console.log(
`reps : ${REPS} resolve ${smallResolveMs.toFixed(2)}ms / 4x ${largeResolveMs.toFixed(2)}ms strategy ${smallStrategyMs.toFixed(2)}ms / 4x ${largeStrategyMs.toFixed(2)}ms parse ${smallParseMs.toFixed(2)}ms / 4x ${largeParseMs.toFixed(2)}ms`,
);
console.log(
`parse_scale_shape : ${parseSmall.targets.size} -> ${parseLarge.targets.size} targets (expect ${PARSE_TARGETS} -> ${PARSE_TARGETS * SCALE})`,
);
if (process.argv.includes('--check')) {
let failed = false;
if (shape.fingerprint !== baselines.layout_fingerprint) {
failed = true;
console.error(
`\nFAIL layout_fingerprint: ${shape.fingerprint}\n` +
` expected ${baselines.layout_fingerprint}\n` +
` Unique-query target set moved. Explain it; do not re-baseline alone.`,
);
}
const exact = [
['targets', small.targetCount],
['files', files],
['imports', imports],
['declared_resolved', shape.declaredResolved],
['sdk_external', shape.sdkExternal],
['undeclared_external', shape.undeclaredExternal],
['empty_declared_external', shape.emptyDeclaredExternal],
['reexport_extra', shape.reexportExtra],
['nested_repeat_resolved', shape.nestedRepeatResolved],
['first_wins', shape.firstWins],
['parse_targets', shape.parseTargets],
['parse_binary_skipped', shape.parseBinarySkipped],
['url_comment_targets', shape.urlCommentTargets],
['parse_complete', shape.parseComplete],
];
for (const [name, value] of exact) {
if (value !== baselines[name]) {
failed = true;
console.error(`\nFAIL ${name}: ${value}, expected exactly ${baselines[name]}.`);
}
}
if (
small.targetCount !== baselines.targets ||
files !== baselines.files ||
imports !== baselines.imports
) {
failed = true;
console.error(
`\nFAIL shape: the corpus must stay large enough that declared_resolved still measures a walk.`,
);
}
if (
parseSmall.targets.size !== PARSE_TARGETS ||
parseLarge.targets.size !== PARSE_TARGETS * SCALE
) {
failed = true;
console.error(
`\nFAIL parse scale shape: ${parseSmall.targets.size} -> ${parseLarge.targets.size}, ` +
`expected ${PARSE_TARGETS} -> ${PARSE_TARGETS * SCALE}.`,
);
}
if (resolveScaling > baselines.resolve_scaling_budget) {
failed = true;
console.error(
`\nFAIL resolve_scaling_ratio: ${resolveScaling.toFixed(3)} exceeds ` +
`${baselines.resolve_scaling_budget} (~1.0 is linear).\n` +
` resolveSwiftImportTarget grew superlinearly in file+import count — a\n` +
` per-import scan of allFilePaths scores ~4 here. Re-run on an idle\n` +
` machine before investigating, and check \`reps\` first.`,
);
}
if (strategyScaling > baselines.strategy_scaling_budget) {
failed = true;
console.error(
`\nFAIL strategy_scaling_ratio: ${strategyScaling.toFixed(3)} exceeds ` +
`${baselines.strategy_scaling_budget} (~1.0 is linear).\n` +
` swiftPackageStrategy grew superlinearly — usually the WeakMap target\n` +
` index falling back to O(imports × files). Re-run idle; check \`reps\`.`,
);
}
if (parseScaling > baselines.parse_scaling_budget) {
failed = true;
console.error(
`\nFAIL parse_scaling_ratio: ${parseScaling.toFixed(3)} exceeds ` +
`${baselines.parse_scaling_budget} (~1.0 is linear).\n` +
` parseSwiftPackageManifest grew superlinearly in factory count.\n` +
` Re-run on an idle machine before investigating, and check \`reps\` first.`,
);
}
if (failed) process.exit(1);
console.log('\nOK — within budget.');
}

View file

@ -68,7 +68,7 @@
"@vitest/coverage-v8": "^4.0.18",
"gitnexus-shared": "file:../gitnexus-shared",
"tsx": "^4.0.0",
"typescript": "^5.4.5",
"typescript": "^7.0.2",
"vitest": "^4.0.18"
},
"engines": {
@ -1331,6 +1331,346 @@
"@types/node": "*"
}
},
"node_modules/@typescript/typescript-aix-ppc64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-aix-ppc64/-/typescript-aix-ppc64-7.0.2.tgz",
"integrity": "sha512-MTKKkWB7p/0E9xi1d1tHtZ5PiLkGEMIq88pK2CubZjOsLtYTLqhgIgi6zepFa+9GHZ6h05NMCkQxGKiPXMxXtQ==",
"cpu": [
"ppc64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"aix"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-darwin-arm64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-darwin-arm64/-/typescript-darwin-arm64-7.0.2.tgz",
"integrity": "sha512-gowzar9MwS/aRWp6f3a4KUqzRjAZjOsmGNCM6LcTgXum+dBfgsBVMN+AgvOCCbguXyick6LJhpBszxMebJ8syA==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"darwin"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-darwin-x64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-darwin-x64/-/typescript-darwin-x64-7.0.2.tgz",
"integrity": "sha512-SZ9xZInqApNlNGc9s0W1VSsktYSOe9cFqNOIqmN1Gs8SmkjKZYFt017G4VwPxASInODuAdbTW7sXiFUf893RgA==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"darwin"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-freebsd-arm64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-freebsd-arm64/-/typescript-freebsd-arm64-7.0.2.tgz",
"integrity": "sha512-W5NH4y/J0plIIS5b2xvTEkU7JFxyqdMAOgf+Ilhl0vHQXKO5dZoxd+C/jEtq56c4F3wk71RB4BMRQ2XdI+bwYQ==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"freebsd"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-freebsd-x64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-freebsd-x64/-/typescript-freebsd-x64-7.0.2.tgz",
"integrity": "sha512-UMGDx5sTpzNw3WiPebH7l90IWfJggEd+egHt/q6p7/Cm3zqoV7VxkGXt+3DxPIw8CcmvAB0j3sVVfbhX+M4Tpw==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"freebsd"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-linux-arm": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-arm/-/typescript-linux-arm-7.0.2.tgz",
"integrity": "sha512-gffT3xPz9sR7j/YJExkyPntrI0P2EP9XbOyWzth2/Gs0RstK+90RBcO0ncXoXy/beYll1SXw846Nf2zdnEz0QQ==",
"cpu": [
"arm"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-linux-arm64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-arm64/-/typescript-linux-arm64-7.0.2.tgz",
"integrity": "sha512-Qh4eU4/y3yDjnfjjyPYihMj5/ODIlmt+Bzu17OI+fiSRDW57QmU5SiN63exPRNJPKUzcc1INa1NXdrJ+MqHjUQ==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-linux-loong64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-loong64/-/typescript-linux-loong64-7.0.2.tgz",
"integrity": "sha512-uEHck9i8hoAzXPiYRib1O7miOnz23SxIeVl6F4LXox+qov1K35jHcEW6VHKvZI+pyvl7fZEP4MCU5LYvIq1GuQ==",
"cpu": [
"loong64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-linux-mips64el": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-mips64el/-/typescript-linux-mips64el-7.0.2.tgz",
"integrity": "sha512-R4KvAMnE43W5Qeqb0Ly56O3mWMWIAgsMyz36DCaycd5nbg/9kzm0liw3JocfRqyJY0KPmzFjbswozXyW0DnIYA==",
"cpu": [
"mips64el"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-linux-ppc64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-ppc64/-/typescript-linux-ppc64-7.0.2.tgz",
"integrity": "sha512-DORx5b3sd/4S7eayxm4FQv+A7CrkUIGRaHiwI8oiHTAI1fAPWhF4J0vAlkC8biAlHSVVwxMQ3tjZ2/DVbnQiiA==",
"cpu": [
"ppc64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-linux-riscv64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-riscv64/-/typescript-linux-riscv64-7.0.2.tgz",
"integrity": "sha512-wf0jqEDOjrPRnKwYRyyJDRo11KMbvMFrU+q4zqKyChODBzvlkbhNQfKvLxQCcwTpdDaXSHZTVuh0JoCrKCUMHQ==",
"cpu": [
"riscv64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-linux-s390x": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-s390x/-/typescript-linux-s390x-7.0.2.tgz",
"integrity": "sha512-IkwJc3L7yhytWd/ewjyxNDfOmswCm9GWMJT/ue/dU4aZNbwZeYAetq42VyLmsmSjvoX7z74X6ZaYCtzAr0EuGw==",
"cpu": [
"s390x"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-linux-x64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-linux-x64/-/typescript-linux-x64-7.0.2.tgz",
"integrity": "sha512-EYdf2cNg7rgCWJnxCdJ+F3V39O8ihb37eHAu1LK8oAFizgTQbPOK7zHHXbPt8rX24COqODXeI3sIf0fCXG7H/A==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-netbsd-arm64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-netbsd-arm64/-/typescript-netbsd-arm64-7.0.2.tgz",
"integrity": "sha512-+polYF4MF04aPpO5FTkHran9yUQDSXqy5GiSDKpsll5jy3l3+g9QLhpf39T+ePtefhXLOGrLl0QIjkQP6VnelA==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"netbsd"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-netbsd-x64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-netbsd-x64/-/typescript-netbsd-x64-7.0.2.tgz",
"integrity": "sha512-8YIT0EHM/3dq10ZOVF/A7pc/YSMtbcecct4rWtexrnSCHOPcpC2KTLXfTCR6vDpnSiY12heNb1GiN/wu+T/FyA==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"netbsd"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-openbsd-arm64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-openbsd-arm64/-/typescript-openbsd-arm64-7.0.2.tgz",
"integrity": "sha512-APT8+ClYnuYm1u9+kgGXoMj2VzWzcymwh2gNSQVySHfkRDGOTVkoWLjCmOQSaO+PoqQ57B0flRp9SA+7GnnkzQ==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"openbsd"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-openbsd-x64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-openbsd-x64/-/typescript-openbsd-x64-7.0.2.tgz",
"integrity": "sha512-yX7s+Q0Dln0Dt9tEzZsAjXXR/+ytBM7AlglaqyeMPxQszJ1JhlJdZ6jLA+IzldHtflX81em7lDao1xXu+aRRkg==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"openbsd"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-sunos-x64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-sunos-x64/-/typescript-sunos-x64-7.0.2.tgz",
"integrity": "sha512-dLJDGaLZ1D4HPQn62u1n8mBDkJREwMsAkCdkwd4Ieqw+x3TUyTsqY0YiBCtE6H6OzzgGk3iuZ3vFWRS+E8/d1g==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"sunos"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-win32-arm64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-win32-arm64/-/typescript-win32-arm64-7.0.2.tgz",
"integrity": "sha512-Gyl1Vy6OsWesLzmq+EP0Fb7b4Nid5232AvcA2SFcdYreldpNtYFFofPjnt62y9hQy7VTaZp65ICJjuAQRaVcIQ==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"win32"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@typescript/typescript-win32-x64": {
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/@typescript/typescript-win32-x64/-/typescript-win32-x64-7.0.2.tgz",
"integrity": "sha512-0BQ3HkAHHlKLSp1qRvf3SUhGpGsDuhB/jgFw75guyqbxJqEaS0Cw/VFO8i2nHglJUzQCRtMMR/IBAKE3ETMC4g==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"win32"
],
"engines": {
"node": ">=16.20.0"
}
},
"node_modules/@vitest/coverage-v8": {
"version": "4.1.11",
"resolved": "https://registry.npmjs.org/@vitest/coverage-v8/-/coverage-v8-4.1.11.tgz",
@ -2908,9 +3248,9 @@
"license": "MIT"
},
"node_modules/js-yaml": {
"version": "5.4.1",
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-5.4.1.tgz",
"integrity": "sha512-28R/k+NAjeuf7+CKlTxWZVExJGwVVLwY06DgEnOMz2gEpfNkDcD7QvyiVPT0xy0XXhU8vHsd4Ot42OOPdJG7dQ==",
"version": "5.4.2",
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-5.4.2.tgz",
"integrity": "sha512-m+aqu+LwO1O6sIopafj8HUVl5aawITwZQe/yHpMCKjaWBaA/d07B/QdMb3529REftiU+RMMHL3Vlsw3hON7vWg==",
"funding": [
{
"type": "github",
@ -3809,9 +4149,9 @@
"license": "MIT"
},
"node_modules/proxy-addr": {
"version": "2.0.7",
"resolved": "https://registry.npmjs.org/proxy-addr/-/proxy-addr-2.0.7.tgz",
"integrity": "sha512-llQsMLSUDUPT44jdrU/O37qlnifitDP+ZwrmmZcoSKyLKvtZxpyV0n2/bD/N4tBAAZ/gJEdZU7KMraoK1+XYAg==",
"version": "2.0.8",
"resolved": "https://registry.npmjs.org/proxy-addr/-/proxy-addr-2.0.8.tgz",
"integrity": "sha512-5nnx0yGyVUcY6t9RnWcARWtwT9F1D8O9rt08htPvnd49W1IgZtmLkhu9WfMzQj1cFxjHIO6connUNVW5k7AVyQ==",
"license": "MIT",
"dependencies": {
"forwarded": "0.2.0",
@ -3819,6 +4159,10 @@
},
"engines": {
"node": ">= 0.10"
},
"funding": {
"type": "opencollective",
"url": "https://opencollective.com/express"
}
},
"node_modules/pump": {
@ -4671,17 +5015,38 @@
}
},
"node_modules/typescript": {
"version": "5.9.3",
"resolved": "https://registry.npmjs.org/typescript/-/typescript-5.9.3.tgz",
"integrity": "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw==",
"version": "7.0.2",
"resolved": "https://registry.npmjs.org/typescript/-/typescript-7.0.2.tgz",
"integrity": "sha512-8FYau96o3NKOhbjKi/qNvG/W5jhzxkbdm5sj9AbZ/5T5sWqn3hJgLfGx27sRKZWTvyzCP8dLRBTf5tBTSRVUNA==",
"dev": true,
"license": "Apache-2.0",
"bin": {
"tsc": "bin/tsc",
"tsserver": "bin/tsserver"
"tsc": "bin/tsc"
},
"engines": {
"node": ">=14.17"
"node": ">=16.20.0"
},
"optionalDependencies": {
"@typescript/typescript-aix-ppc64": "7.0.2",
"@typescript/typescript-darwin-arm64": "7.0.2",
"@typescript/typescript-darwin-x64": "7.0.2",
"@typescript/typescript-freebsd-arm64": "7.0.2",
"@typescript/typescript-freebsd-x64": "7.0.2",
"@typescript/typescript-linux-arm": "7.0.2",
"@typescript/typescript-linux-arm64": "7.0.2",
"@typescript/typescript-linux-loong64": "7.0.2",
"@typescript/typescript-linux-mips64el": "7.0.2",
"@typescript/typescript-linux-ppc64": "7.0.2",
"@typescript/typescript-linux-riscv64": "7.0.2",
"@typescript/typescript-linux-s390x": "7.0.2",
"@typescript/typescript-linux-x64": "7.0.2",
"@typescript/typescript-netbsd-arm64": "7.0.2",
"@typescript/typescript-netbsd-x64": "7.0.2",
"@typescript/typescript-openbsd-arm64": "7.0.2",
"@typescript/typescript-openbsd-x64": "7.0.2",
"@typescript/typescript-sunos-x64": "7.0.2",
"@typescript/typescript-win32-arm64": "7.0.2",
"@typescript/typescript-win32-x64": "7.0.2"
}
},
"node_modules/typical": {

View file

@ -129,7 +129,7 @@
"@vitest/coverage-v8": "^4.0.18",
"gitnexus-shared": "file:../gitnexus-shared",
"tsx": "^4.0.0",
"typescript": "^5.4.5",
"typescript": "^7.0.2",
"vitest": "^4.0.18"
},
"overrides": {

View file

@ -54,8 +54,8 @@ if (!fs.existsSync(SHARED_ROOT)) {
// Launch tsc as `node typescript/lib/tsc.js` on every OS. The `.bin/tsc` /
// `tsc.cmd` shims are Windows-only wrappers; `execFileSync` cannot spawn a
// `.cmd` without a shell, and a separate `npm ci` in gitnexus-shared pulls
// TypeScript 7 optional platform packages (7+ minutes in CI).
// `.cmd` without a shell, and a separate `npm ci` in gitnexus-shared would
// pull a second TypeScript 7 optional-platform install (7+ minutes in CI).
const tscJs = path.join(ROOT, 'node_modules', 'typescript', 'lib', 'tsc.js');
if (!fs.existsSync(tscJs)) {
console.error(

View file

@ -33,8 +33,10 @@ import path from 'node:path';
import { readRepoControlFile } from '../config/repo-control-file.js';
import {
InvalidBranchError,
sanitizeDetectedBranch,
validateBranchName as validateBranchNameCore,
} from '../core/git-ref.js';
export { sanitizeDetectedBranch };
import type { AnalyzeOptions } from './analyze-options.js';
export const GITNEXUS_RC_FILENAME = '.gitnexusrc';
@ -100,6 +102,10 @@ const KEY_SPECS: Record<string, KeySpec> = {
workerTimeout: { target: 'workerTimeout', kind: 'numeric-string' },
walCheckpointThreshold: { target: 'walCheckpointThreshold', kind: 'numeric-string' },
workers: { target: 'workers', kind: 'numeric-string' },
maxProcesses: { target: 'maxProcesses', kind: 'numeric-string' },
maxProcessBranching: { target: 'maxProcessBranching', kind: 'numeric-string' },
maxProcessTraceDepth: { target: 'maxProcessTraceDepth', kind: 'numeric-string' },
maxEntryPointCandidates: { target: 'maxEntryPointCandidates', kind: 'numeric-string' },
embeddingThreads: { target: 'embeddingThreads', kind: 'numeric-string' },
embeddingBatchSize: { target: 'embeddingBatchSize', kind: 'numeric-string' },
embeddingSubBatchSize: { target: 'embeddingSubBatchSize', kind: 'numeric-string' },
@ -172,20 +178,6 @@ export function validateBranchName(value: string, source: string): string {
}
}
/**
* Best-effort validation for an auto-detected branch (from git). Never throws —
* returns `undefined` for anything unusable so the resolver falls back to the
* next precedence tier.
*/
export function sanitizeDetectedBranch(value: string | null | undefined): string | undefined {
if (!value) return undefined;
try {
return validateBranchName(value, 'detected branch');
} catch {
return undefined;
}
}
const normalizeValue = (kind: ValueKind, value: unknown, key: string): unknown => {
const source = `${GITNEXUS_RC_FILENAME} "${key}"`;
switch (kind) {

View file

@ -115,6 +115,14 @@ export interface AnalyzeOptions {
walCheckpointThreshold?: string;
/** Parse worker pool size (>=1); 0 is rejected (no sequential mode). */
workers?: string;
/** Process-detection process cap. Positive integer string; `0` is invalid. */
maxProcesses?: string;
/** Process-detection per-node branching cap. Positive integer string. */
maxProcessBranching?: string;
/** Process-detection DFS depth cap. Positive integer string. */
maxProcessTraceDepth?: string;
/** Ranked entry-point candidate pool. Positive integer string. */
maxEntryPointCandidates?: string;
embeddingThreads?: string;
embeddingBatchSize?: string;
embeddingSubBatchSize?: string;

View file

@ -22,6 +22,10 @@ import {
import type { AnalyzeOptions } from './analyze-options.js';
import { ensureHeap } from './analyze.js';
import { cliError, cliInfo, cliWarn } from './cli-message.js';
import {
formatInvalidProcessDetectionOverride,
parseProcessDetectionBudgetStrings,
} from '../core/ingestion/process-detection-budget.js';
import {
WATCH_FULL_REFRESH_PATH,
WatchRefreshQueue,
@ -164,6 +168,17 @@ export async function resolveWatchOptions(
const workerPoolSize = positiveInteger(merged.workers, '--workers');
const workerTimeoutSeconds = positiveInteger(merged.workerTimeout, 'workerTimeout');
const maxFileSize = positiveInteger(merged.maxFileSize, 'maxFileSize', MAX_FILE_SIZE_KB);
const processDetection = parseProcessDetectionBudgetStrings(
{
maxProcesses: merged.maxProcesses,
maxProcessBranching: merged.maxProcessBranching,
maxProcessTraceDepth: merged.maxProcessTraceDepth,
maxEntryPointCandidates: merged.maxEntryPointCandidates,
},
(flag, raw) => {
cliWarn(formatInvalidProcessDetectionOverride(flag, raw));
},
);
setEnvironment(
'GITNEXUS_MAX_FILE_SIZE',
@ -183,6 +198,10 @@ export async function resolveWatchOptions(
registryName: merged.name,
allowDuplicateName: merged.allowDuplicateName,
workerPoolSize,
maxProcesses: processDetection.maxProcesses,
maxProcessBranching: processDetection.maxProcessBranching,
maxProcessTraceDepth: processDetection.maxProcessTraceDepth,
maxEntryPointCandidates: processDetection.maxEntryPointCandidates,
fetchWrappers: merged.fetchWrappers,
skipAgentsMd: true,
skipSkills: true,

View file

@ -54,6 +54,12 @@ import type { AnalyzeOptions } from './analyze-options.js';
import { runFullAnalysis } from '../core/run-analyze.js';
import { getRuntimeFingerprint } from '../core/platform/capabilities.js';
import { getMaxFileSizeBannerMessage } from '../core/ingestion/utils/max-file-size.js';
import {
formatInvalidProcessDetectionOverride,
formatProcessDetectionBudgetBanner,
parseProcessDetectionBudgetStrings,
resolveProcessDetectionBudget,
} from '../core/ingestion/process-detection-budget.js';
import { warnMissingOptionalGrammars, getOptionalGrammarExtensions } from './optional-grammars.js';
import { glob } from 'glob';
import fs from 'fs/promises';
@ -947,6 +953,18 @@ const analyzeCommandImpl = async (
workerPoolSize = parsedWorkers;
}
const processDetectionFromFlags = parseProcessDetectionBudgetStrings(
{
maxProcesses: options.maxProcesses,
maxProcessBranching: options.maxProcessBranching,
maxProcessTraceDepth: options.maxProcessTraceDepth,
maxEntryPointCandidates: options.maxEntryPointCandidates,
},
(flag, raw) => {
cliWarn(` ${formatInvalidProcessDetectionOverride(flag, raw)}\n`);
},
);
// Parse `--embeddings [limit]`: `true` → default cap, string → numeric cap
// (0 disables the cap entirely). Validated up here so failures match the
// sibling-validation pattern (exit before bar.start() — otherwise
@ -1236,6 +1254,12 @@ const analyzeCommandImpl = async (
if (maxFileSizeBanner) {
console.log(`${maxFileSizeBanner}\n`);
}
const processDetectionBanner = formatProcessDetectionBudgetBanner(
resolveProcessDetectionBudget(processDetectionFromFlags),
);
if (processDetectionBanner) {
console.log(`${processDetectionBanner}\n`);
}
// ── CLI progress bar setup ─────────────────────────────────────────
const barOptions: cliProgress.Options & { terminal?: CliProgressTerminal } = {
@ -1376,6 +1400,10 @@ const analyzeCommandImpl = async (
// GITNEXUS_WORKER_POOL_SIZE env mutation. `undefined` defers to the
// env / auto-formula fallback inside the pipeline.
workerPoolSize,
maxProcesses: processDetectionFromFlags.maxProcesses,
maxProcessBranching: processDetectionFromFlags.maxProcessBranching,
maxProcessTraceDepth: processDetectionFromFlags.maxProcessTraceDepth,
maxEntryPointCandidates: processDetectionFromFlags.maxEntryPointCandidates,
// Extra fetch-wrapper names from `.gitnexusrc` (#1589/#1852 residual);
// forwarded to the routes phase consumer scan.
fetchWrappers: options.fetchWrappers,

View file

@ -72,6 +72,10 @@ const OPTION_DESCRIPTION_KEYS = {
'analyze|--worker-timeout <seconds>': 'help.option.analyze.workerTimeout',
'analyze|--wal-checkpoint-threshold <bytes>': 'help.option.analyze.walCheckpointThreshold',
'analyze|--workers <n>': 'help.option.analyze.workers',
'analyze|--max-processes <n>': 'help.option.analyze.maxProcesses',
'analyze|--max-process-branching <n>': 'help.option.analyze.maxProcessBranching',
'analyze|--max-process-trace-depth <n>': 'help.option.analyze.maxProcessTraceDepth',
'analyze|--max-entry-point-candidates <n>': 'help.option.analyze.maxEntryPointCandidates',
'analyze|--embedding-threads <n>': 'help.option.analyze.embeddingThreads',
'analyze|--embedding-batch-size <n>': 'help.option.analyze.embeddingBatchSize',
'analyze|--embedding-sub-batch-size <n>': 'help.option.analyze.embeddingSubBatchSize',

View file

@ -254,6 +254,14 @@ export const en = {
'LadybugDB WAL auto-checkpoint threshold in bytes during analyze (integer >= -1; default: 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB).',
'help.option.analyze.workers':
'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.',
'help.option.analyze.maxProcesses':
'Process-detection process cap (positive integer). Replaces the dynamic max(20, round(symbols/10)) formula. Default: dynamic.',
'help.option.analyze.maxProcessBranching':
'Process-detection per-node branching cap (positive integer). Default: 4.',
'help.option.analyze.maxProcessTraceDepth':
'Process-detection DFS depth cap (positive integer). Default: 10.',
'help.option.analyze.maxEntryPointCandidates':
'Ranked entry-point candidate pool (positive integer). Default: 200. Raise when the warning names this knob; doubling is the usual first raise.',
'help.option.analyze.embeddingThreads': 'Limit local ONNX embedding CPU threads',
'help.option.analyze.embeddingBatchSize': 'Number of nodes per embedding batch',
'help.option.analyze.embeddingSubBatchSize': 'Number of chunks per embedding model call',
@ -358,5 +366,5 @@ export const en = {
'help.identityCache.environment':
'\nAnalyzer identity cache:\n GITNEXUS_ANALYZER_IDENTITY_CACHE_DIR=/absolute/protected/dir\n Operator-trusted persistent cache for warm cross-process status. The directory must pre-exist, be outside the GitNexus package/build roots, and contain no symlink or junction components. Defaults remain fail-closed on platforms without POSIX ownership APIs.',
'help.analyze.environment':
'\nEnvironment variables:\n GITNEXUS_NO_GITIGNORE=1 Skip .gitignore parsing (still reads .gitnexusignore)\n GITNEXUS_MAX_FILE_SIZE=N Override large-file skip threshold (KB). Default 512, max 32768.\n GITNEXUS_STORAGE_PATH=/absolute/index Complete external index directory. Preserves the existing configuration semantics and overrides GITNEXUS_STORAGE_ROOT when both are set.\n GITNEXUS_STORAGE_ROOT=/absolute/root External index root; each repository uses an isolated <repo-basename>-<canonical-path-hash>/ slot.\n GITNEXUS_CONTENT_RETENTION=full Source-text retention profile: full, symbol, or none. Default full.\n GITNEXUS_ANALYZER_IDENTITY_CACHE_DIR=/absolute/protected/dir Operator-trusted persistent analyzer identity cache; must pre-exist, be outside package/build roots, and contain no symlink/junction components.\n GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS=N Worker idle timeout in milliseconds. Default 30000.\n GITNEXUS_WAL_CHECKPOINT_THRESHOLD=N LadybugDB WAL auto-checkpoint threshold in bytes (default 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB).\n GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES=N Worker job byte budget. Default 8388608.\n GITNEXUS_WORKER_POOL_SIZE=N Parse worker count override. Default cores-1 capped at 16.\n GITNEXUS_PARSE_CHUNK_CONCURRENCY=N Concurrent in-flight parse chunks. Default 2.\n GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT=N Max replacement spawns per slot before drop. Default 3.\n GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS=N Total retry wall-time per job. Default 5x sub-batch timeout.\n GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD=N Per-slot deaths to trip circuit breaker. Default max(3, poolSize).\n GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS=N Max wait at pool shutdown for a retired worker still inside native code (terminated at its next safe point instead of aborting the process). Default 30000.\n GITNEXUS_CPP_CAPTURE_BUDGET_MS=N Per-file wall-clock budget for C++ capture extraction; on breach the file keeps partial captures with a warning. Default 20000.\n GITNEXUS_EMBEDDING_THREADS=N Limit local ONNX CPU threads for --embeddings.\n GITNEXUS_EMBEDDING_RETRY_TIMEOUTS=1 Retry per-attempt HTTP embedding timeouts through GITNEXUS_EMBEDDING_MAX_ATTEMPTS (default off; timeouts stay terminal).\n GITNEXUS_SEMANTIC_EXACT_SCAN_LIMIT=N Max embedding chunks for exact-scan fallback. Default 10000.\n GITNEXUS_VECTOR_MAX_DISTANCE=N Max accepted semantic/vector cosine distance (0 < N <= 2; higher values clamp to 2). Default 0.6 for MCP, 0.5 elsewhere.\n\nFlags override the corresponding env vars when both are provided.\n\nTip: `.gitnexusignore` supports `.gitignore`-style negation. Add e.g.\n `!__tests__/` to index a directory that is auto-filtered by default (#771).',
'\nEnvironment variables:\n GITNEXUS_NO_GITIGNORE=1 Skip .gitignore parsing (still reads .gitnexusignore)\n GITNEXUS_MAX_FILE_SIZE=N Override large-file skip threshold (KB). Default 512, max 32768.\n GITNEXUS_STORAGE_PATH=/absolute/index Complete external index directory. Preserves the existing configuration semantics and overrides GITNEXUS_STORAGE_ROOT when both are set.\n GITNEXUS_STORAGE_ROOT=/absolute/root External index root; each repository uses an isolated <repo-basename>-<canonical-path-hash>/ slot.\n GITNEXUS_CONTENT_RETENTION=full Source-text retention profile: full, symbol, or none. Default full.\n GITNEXUS_ANALYZER_IDENTITY_CACHE_DIR=/absolute/protected/dir Operator-trusted persistent analyzer identity cache; must pre-exist, be outside package/build roots, and contain no symlink/junction components.\n GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS=N Worker idle timeout in milliseconds. Default 30000.\n GITNEXUS_WAL_CHECKPOINT_THRESHOLD=N LadybugDB WAL auto-checkpoint threshold in bytes (default 67108864 = 64 MiB; -1 keeps Ladybug stock ~16 MiB).\n GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES=N Worker job byte budget. Default 8388608.\n GITNEXUS_WORKER_POOL_SIZE=N Parse worker count override. Default cores-1 capped at 16.\n GITNEXUS_PARSE_CHUNK_CONCURRENCY=N Concurrent in-flight parse chunks. Default 2.\n GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT=N Max replacement spawns per slot before drop. Default 3.\n GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS=N Total retry wall-time per job. Default 5x sub-batch timeout.\n GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD=N Per-slot deaths to trip circuit breaker. Default max(3, poolSize).\n GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS=N Max wait at pool shutdown for a retired worker still inside native code (terminated at its next safe point instead of aborting the process). Default 30000.\n GITNEXUS_CPP_CAPTURE_BUDGET_MS=N Per-file wall-clock budget for C++ capture extraction; on breach the file keeps partial captures with a warning. Default 20000.\n GITNEXUS_EMBEDDING_THREADS=N Limit local ONNX CPU threads for --embeddings.\n GITNEXUS_EMBEDDING_RETRY_TIMEOUTS=1 Retry per-attempt HTTP embedding timeouts through GITNEXUS_EMBEDDING_MAX_ATTEMPTS (default off; timeouts stay terminal).\n GITNEXUS_SEMANTIC_EXACT_SCAN_LIMIT=N Max embedding chunks for exact-scan fallback. Default 10000.\n GITNEXUS_VECTOR_MAX_DISTANCE=N Max accepted semantic/vector cosine distance (0 < N <= 2; higher values clamp to 2). Default 0.6 for MCP, 0.5 elsewhere.\n GITNEXUS_MAX_PROCESSES=N Process-detection process cap (positive integer). Replaces the dynamic max(20, round(symbols/10)) formula. Distinct from query-time IMPACT_MAX_CHUNKS.\n GITNEXUS_MAX_PROCESS_BRANCHING=N Process-detection per-node branching cap. Default 4.\n GITNEXUS_MAX_PROCESS_TRACE_DEPTH=N Process-detection DFS depth cap. Default 10.\n GITNEXUS_MAX_ENTRY_POINT_CANDIDATES=N Ranked entry-point candidate pool. Default 200. Raise when the warning names this knob; doubling is the usual first raise.\n\nCLI flags take precedence over `.gitnexusrc`, which takes precedence over env vars, which take precedence over built-in defaults.\n\nTip: `.gitnexusignore` supports `.gitignore`-style negation. Add e.g.\n `!__tests__/` to index a directory that is auto-filtered by default (#771).',
} as const;

View file

@ -235,6 +235,12 @@ export const zhCN = {
'analyze 期间 LadybugDB WAL 自动 checkpoint 阈值(字节,整数 >= -1;默认:67108864 = 64 MiB;-1 保持 Ladybug 默认约 16 MiB)。',
'help.option.analyze.workers':
'解析 worker 池大小(>=1)。默认:cores-1,最多 16,按仓库规模自适应。',
'help.option.analyze.maxProcesses':
'流程检测的流程数量上限(正整数)。覆盖动态的 max(20, round(symbols/10)) 公式。默认:动态。',
'help.option.analyze.maxProcessBranching': '流程检测的单节点分支上限(正整数)。默认:4。',
'help.option.analyze.maxProcessTraceDepth': '流程检测的 DFS 深度上限(正整数)。默认:10。',
'help.option.analyze.maxEntryPointCandidates':
'排序后的入口点候选池(正整数)。默认:200。仅在警告点名该上限时提高;那时通常先翻倍。',
'help.option.analyze.embeddingThreads': '限制本地 ONNX 嵌入 CPU 线程数',
'help.option.analyze.embeddingBatchSize': '每个嵌入批次的节点数',
'help.option.analyze.embeddingSubBatchSize': '每次嵌入模型调用的分块数',
@ -331,5 +337,5 @@ export const zhCN = {
'help.identityCache.environment':
'\n分析器身份缓存:\n GITNEXUS_ANALYZER_IDENTITY_CACHE_DIR=/absolute/protected/dir\n 由操作员明确信任的持久缓存,用于跨进程快速查询状态。目录必须预先存在、位于 GitNexus 包/构建根目录之外,且路径中不得包含符号链接或 junction。缺少 POSIX 所有权 API 的平台默认保持故障关闭。',
'help.analyze.environment':
'\n环境变量:\n GITNEXUS_NO_GITIGNORE=1 跳过 .gitignore 解析(仍读取 .gitnexusignore)\n GITNEXUS_MAX_FILE_SIZE=N 覆盖大文件跳过阈值(KB)。默认 512,最大 32768。\n GITNEXUS_STORAGE_PATH=/absolute/index 完整外部索引目录。保留既有配置语义;与 GITNEXUS_STORAGE_ROOT 同时设置时优先使用。\n GITNEXUS_STORAGE_ROOT=/absolute/root 外部索引根目录;每个仓库使用独立的 <仓库名>-<规范路径哈希>/ 子目录。\n GITNEXUS_CONTENT_RETENTION=full 源码文本保留策略:full、symbol 或 none。默认 full。\n GITNEXUS_ANALYZER_IDENTITY_CACHE_DIR=/absolute/protected/dir 由操作员明确信任的持久分析器身份缓存;目录必须预先存在、位于包/构建根目录之外,且路径中不得包含符号链接或 junction。\n GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS=N Worker 空闲超时(毫秒)。默认 30000。\n GITNEXUS_WAL_CHECKPOINT_THRESHOLD=N LadybugDB WAL 自动 checkpoint 阈值(字节,默认 67108864 = 64 MiB;-1 保持 Ladybug 默认约 16 MiB)。\n GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES=N Worker 作业字节预算。默认 8388608。\n GITNEXUS_WORKER_POOL_SIZE=N 解析 worker 数量覆盖值。默认 cores-1,最多 16。\n GITNEXUS_PARSE_CHUNK_CONCURRENCY=N 并发进行中的解析分块数。默认 2。\n GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT=N 每个 slot 丢弃前允许的最大替换进程数。默认 3。\n GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS=N 每个作业的总重试墙钟时间。默认 5 倍子批次超时。\n GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD=N 每个 slot 触发熔断的死亡次数。默认 max(3, poolSize)。\n GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS=N 线程池关闭时等待仍在原生代码中的已退役 worker 的最长时间(到达安全点后再终止,避免进程级 abort)。默认 30000。\n GITNEXUS_CPP_CAPTURE_BUDGET_MS=N C++ 捕获提取的每文件墙钟预算;超出后该文件保留部分捕获并输出警告。默认 20000。\n GITNEXUS_EMBEDDING_THREADS=N 限制 --embeddings 的本地 ONNX CPU 线程数。\n GITNEXUS_EMBEDDING_RETRY_TIMEOUTS=1 将单次 HTTP 嵌入超时纳入 GITNEXUS_EMBEDDING_MAX_ATTEMPTS 重试(默认关闭,超时仍为终止错误)。\n GITNEXUS_SEMANTIC_EXACT_SCAN_LIMIT=N exact-scan 回退的最大嵌入分块数。默认 10000。\n GITNEXUS_VECTOR_MAX_DISTANCE=N 语义/向量搜索接受的最大余弦距离(0 < N <= 2;超出则钳制为 2)。MCP 默认 0.6,其他路径默认 0.5。\n\n当参数和对应环境变量同时提供时,参数优先。\n\n提示:`.gitnexusignore` 支持 `.gitignore` 风格的取反。比如添加\n `!__tests__/` 可以索引默认自动过滤的目录(#771)。',
'\n环境变量:\n GITNEXUS_NO_GITIGNORE=1 跳过 .gitignore 解析(仍读取 .gitnexusignore)\n GITNEXUS_MAX_FILE_SIZE=N 覆盖大文件跳过阈值(KB)。默认 512,最大 32768。\n GITNEXUS_STORAGE_PATH=/absolute/index 完整外部索引目录。保留既有配置语义;与 GITNEXUS_STORAGE_ROOT 同时设置时优先使用。\n GITNEXUS_STORAGE_ROOT=/absolute/root 外部索引根目录;每个仓库使用独立的 <仓库名>-<规范路径哈希>/ 子目录。\n GITNEXUS_CONTENT_RETENTION=full 源码文本保留策略:full、symbol 或 none。默认 full。\n GITNEXUS_ANALYZER_IDENTITY_CACHE_DIR=/absolute/protected/dir 由操作员明确信任的持久分析器身份缓存;目录必须预先存在、位于包/构建根目录之外,且路径中不得包含符号链接或 junction。\n GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS=N Worker 空闲超时(毫秒)。默认 30000。\n GITNEXUS_WAL_CHECKPOINT_THRESHOLD=N LadybugDB WAL 自动 checkpoint 阈值(字节,默认 67108864 = 64 MiB;-1 保持 Ladybug 默认约 16 MiB)。\n GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES=N Worker 作业字节预算。默认 8388608。\n GITNEXUS_WORKER_POOL_SIZE=N 解析 worker 数量覆盖值。默认 cores-1,最多 16。\n GITNEXUS_PARSE_CHUNK_CONCURRENCY=N 并发进行中的解析分块数。默认 2。\n GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT=N 每个 slot 丢弃前允许的最大替换进程数。默认 3。\n GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS=N 每个作业的总重试墙钟时间。默认 5 倍子批次超时。\n GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD=N 每个 slot 触发熔断的死亡次数。默认 max(3, poolSize)。\n GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS=N 线程池关闭时等待仍在原生代码中的已退役 worker 的最长时间(到达安全点后再终止,避免进程级 abort)。默认 30000。\n GITNEXUS_CPP_CAPTURE_BUDGET_MS=N C++ 捕获提取的每文件墙钟预算;超出后该文件保留部分捕获并输出警告。默认 20000。\n GITNEXUS_EMBEDDING_THREADS=N 限制 --embeddings 的本地 ONNX CPU 线程数。\n GITNEXUS_EMBEDDING_RETRY_TIMEOUTS=1 将单次 HTTP 嵌入超时纳入 GITNEXUS_EMBEDDING_MAX_ATTEMPTS 重试(默认关闭,超时仍为终止错误)。\n GITNEXUS_SEMANTIC_EXACT_SCAN_LIMIT=N exact-scan 回退的最大嵌入分块数。默认 10000。\n GITNEXUS_VECTOR_MAX_DISTANCE=N 语义/向量搜索接受的最大余弦距离(0 < N <= 2;超出则钳制为 2)。MCP 默认 0.6,其他路径默认 0.5。\n GITNEXUS_MAX_PROCESSES=N 流程检测的流程数量上限(正整数)。覆盖动态的 max(20, round(symbols/10)) 公式。与查询时的 IMPACT_MAX_CHUNKS 无关。\n GITNEXUS_MAX_PROCESS_BRANCHING=N 流程检测的单节点分支上限。默认 4。\n GITNEXUS_MAX_PROCESS_TRACE_DEPTH=N 流程检测的 DFS 深度上限。默认 10。\n GITNEXUS_MAX_ENTRY_POINT_CANDIDATES=N 排序后的入口点候选池。默认 200。仅在警告点名该上限时提高;那时通常先翻倍。\n\nCLI 参数优先于 `.gitnexusrc`,后者优先于环境变量,环境变量优先于内置默认值。\n\n提示:`.gitnexusignore` 支持 `.gitignore` 风格的取反。比如添加\n `!__tests__/` 可以索引默认自动过滤的目录(#771)。',
} satisfies EnglishMessages;

View file

@ -163,6 +163,22 @@ program
'--workers <n>',
'Parse worker pool size (>=1). Default: cores-1 capped at 16, auto-sized to the repo.',
)
.option(
'--max-processes <n>',
'Process-detection process cap (positive integer). Replaces the dynamic max(20, round(symbols/10)) formula. Default: dynamic.',
)
.option(
'--max-process-branching <n>',
'Process-detection per-node branching cap (positive integer). Default: 4.',
)
.option(
'--max-process-trace-depth <n>',
'Process-detection DFS depth cap (positive integer). Default: 10.',
)
.option(
'--max-entry-point-candidates <n>',
'Ranked entry-point candidate pool (positive integer). Default: 200. Raise when the warning names this knob; doubling is the usual first raise.',
)
.option(
'--spring-actuator <path>',
'Import local Spring Boot Actuator JSON snapshots (mappings, beans, conditions, ' +

View file

@ -0,0 +1,483 @@
/**
* Disk-backed restore cache for CodeEmbedding rows (#3306).
*
* `loadCachedEmbeddings` used to `getAll()` the table and `map(Number)` every
* vector into a JS `number[]`. On a large already-indexed repo that single
* structure OOMs the V8 heap during "Caching embeddings..." even when the
* incremental diff is a handful of nodes.
*
* This module keeps metadata in RAM and writes vectors to a temp Float32
* spill. Restore materializes only the rows that Phase 3.5 will re-insert,
* in the existing 200-row batches.
*/
import { closeSync, openSync, readSync, unlinkSync, writeSync } from 'node:fs';
import { AsyncLocalStorage } from 'node:async_hooks';
import { randomBytes } from 'node:crypto';
import os from 'node:os';
import path from 'node:path';
import type { CachedEmbedding } from './types.js';
/** In-RAM vector copies stay below this row count; larger tables use the spill. */
export const DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT = 2048;
const SPILL_MAGIC = 'GNXE';
const SPILL_VERSION = 1;
const SPILL_HEADER_BYTES = 12;
export interface CachedEmbeddingMeta {
nodeId: string;
chunkIndex: number;
startLine: number;
endLine: number;
contentHash?: string;
/** Row order in the spill file (and in `embeddings` when in-memory). */
vectorIndex: number;
}
export interface EmbeddingVectorSpill {
path: string;
dims: number;
rowCount: number;
}
export interface CachedEmbeddingsSnapshot {
embeddingNodeIds: Set<string>;
/** Populated only when the table is at or under the in-memory row limit. */
embeddings: CachedEmbedding[];
rows: CachedEmbeddingMeta[];
spill?: EmbeddingVectorSpill;
}
export interface LoadCachedEmbeddingsOptions {
/**
* Keep full `number[]` vectors in RAM at or below this many rows.
* `0` always spills. Default {@link DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT}
* or `GITNEXUS_EMBEDDING_CACHE_IN_MEMORY_LIMIT`.
*/
inMemoryRowLimit?: number;
/** Directory for the spill file (default `os.tmpdir()`). */
spillDir?: string;
}
export interface CachedEmbeddingsBuilder {
embeddingNodeIds: Set<string>;
rows: CachedEmbeddingMeta[];
/** Float32 vectors kept in RAM until the in-memory row limit is exceeded. */
inMemory: Float32Array[] | null;
inMemoryRowLimit: number;
writer: EmbeddingSpillWriter;
}
export function emptyCachedEmbeddingsSnapshot(): CachedEmbeddingsSnapshot {
return { embeddingNodeIds: new Set(), embeddings: [], rows: [] };
}
export function resolveEmbeddingCacheInMemoryRowLimit(override?: number): number {
if (override !== undefined) {
if (!Number.isFinite(override) || override < 0) {
return DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT;
}
return Math.floor(override);
}
const raw = process.env.GITNEXUS_EMBEDDING_CACHE_IN_MEMORY_LIMIT;
if (raw === undefined || raw === '') return DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT;
const parsed = parseInt(raw, 10);
return Number.isFinite(parsed) && parsed >= 0
? parsed
: DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT;
}
export function normalizeCachedEmbeddings(raw: {
embeddingNodeIds?: Set<string>;
embeddings?: CachedEmbedding[];
rows?: CachedEmbeddingMeta[];
spill?: EmbeddingVectorSpill;
}): CachedEmbeddingsSnapshot {
const embeddings = raw.embeddings ?? [];
const embeddingNodeIds = raw.embeddingNodeIds ?? new Set(embeddings.map((row) => row.nodeId));
const rows =
raw.rows ??
embeddings.map((row, vectorIndex) => ({
nodeId: row.nodeId,
chunkIndex: row.chunkIndex,
startLine: row.startLine,
endLine: row.endLine,
contentHash: row.contentHash,
vectorIndex,
}));
return { embeddingNodeIds, embeddings, rows, spill: raw.spill };
}
export function cacheRowCount(snapshot: CachedEmbeddingsSnapshot): number {
return snapshot.rows.length > 0 ? snapshot.rows.length : snapshot.embeddings.length;
}
export function snapshotEmbeddingDims(snapshot: CachedEmbeddingsSnapshot): number | undefined {
if (snapshot.spill && snapshot.spill.dims > 0) return snapshot.spill.dims;
const dims = snapshot.embeddings[0]?.embedding.length;
return dims && dims > 0 ? dims : undefined;
}
export function coerceEmbeddingToFloat32(embedding: unknown): Float32Array | null {
if (embedding == null) return null;
if (embedding instanceof Float32Array) {
return embedding.length > 0 ? embedding : null;
}
if (ArrayBuffer.isView(embedding) && !(embedding instanceof DataView)) {
const view = embedding as Exclude<ArrayBufferView, DataView> & { length: number };
if (view.length === 0) return null;
return Float32Array.from({ length: view.length }, (_, i) => Number(view[i]));
}
if (
typeof embedding === 'object' &&
typeof (embedding as Iterable<unknown>)[Symbol.iterator] === 'function'
) {
const arr = Array.isArray(embedding)
? (embedding as unknown[])
: Array.from(embedding as Iterable<unknown>);
if (arr.length === 0) return null;
return Float32Array.from(arr, (value) => Number(value));
}
return null;
}
export function float32ToNumberArray(vec: Float32Array): number[] {
const out = new Array<number>(vec.length);
for (let i = 0; i < vec.length; i++) out[i] = vec[i]!;
return out;
}
/**
* `fs.writeSync` can return a short byte count. Loop until the whole buffer
* lands, matching `sync-csv-writer.ts`, so a partial write never advances
* `rowCount` on a truncated vector.
*/
function writeAllSync(fd: number, data: Uint8Array): void {
let offset = 0;
while (offset < data.length) {
const n = writeSync(fd, data, offset, data.length - offset);
if (n <= 0) {
throw new Error(`embedding spill short write: wrote ${n} of ${data.length - offset} bytes`);
}
offset += n;
}
}
function unlinkBestEffort(filePath: string): void {
try {
unlinkSync(filePath);
} catch {
/* ENOENT or already removed */
}
}
const liveSpillPaths = new Set<string>();
const spillScope = new AsyncLocalStorage<Set<string>>();
let spillExitHookInstalled = false;
function trackLiveSpillPath(filePath: string): void {
liveSpillPaths.add(filePath);
spillScope.getStore()?.add(filePath);
if (!spillExitHookInstalled) {
spillExitHookInstalled = true;
process.on('exit', () => {
for (const spillPath of liveSpillPaths) {
unlinkBestEffort(spillPath);
}
});
}
}
function untrackLiveSpillPath(filePath: string): void {
liveSpillPaths.delete(filePath);
}
/** Best-effort unlink of every tracked spill. Safe to call more than once. */
export function discardLiveEmbeddingSpills(): void {
for (const spillPath of [...liveSpillPaths]) {
unlinkBestEffort(spillPath);
liveSpillPaths.delete(spillPath);
}
}
/** Run `fn` so later {@link discardScopedEmbeddingSpills} only unlinks this run. */
export function withEmbeddingSpillScope<T>(fn: () => T): T {
return spillScope.run(new Set(), fn);
}
/** Unlink spills created inside the current {@link withEmbeddingSpillScope}. */
export function discardScopedEmbeddingSpills(): void {
const owned = spillScope.getStore();
if (!owned) return;
for (const spillPath of [...owned]) {
unlinkBestEffort(spillPath);
liveSpillPaths.delete(spillPath);
owned.delete(spillPath);
}
}
export class EmbeddingSpillWriter {
readonly path: string;
dims = 0;
rowCount = 0;
private fd: number | null = null;
private closed = false;
constructor(dir: string) {
this.path = path.join(
dir,
`gitnexus-embed-restore-${process.pid}-${randomBytes(8).toString('hex')}.bin`,
);
}
append(vec: Float32Array): void {
if (this.closed) {
throw new Error('embedding spill writer already closed');
}
if (this.fd === null) {
this.dims = vec.length;
this.fd = openSync(this.path, 'wx', 0o600);
trackLiveSpillPath(this.path);
const header = Buffer.alloc(SPILL_HEADER_BYTES);
header.write(SPILL_MAGIC, 0, 4, 'ascii');
header.writeUInt8(SPILL_VERSION, 4);
header.writeUInt32LE(this.dims, 5);
writeAllSync(this.fd, header);
} else if (vec.length !== this.dims) {
throw new Error(
`embedding dim mismatch while spilling: got ${vec.length}, expected ${this.dims}`,
);
}
writeAllSync(this.fd, Buffer.from(vec.buffer, vec.byteOffset, vec.byteLength));
this.rowCount++;
}
finish(): EmbeddingVectorSpill | undefined {
if (this.fd !== null) {
closeSync(this.fd);
this.fd = null;
}
this.closed = true;
if (this.rowCount === 0) {
this.unlinkQuiet();
return undefined;
}
return { path: this.path, dims: this.dims, rowCount: this.rowCount };
}
abort(): void {
const opened = this.fd !== null;
if (this.fd !== null) {
try {
closeSync(this.fd);
} catch {
/* already closed */
}
this.fd = null;
}
this.closed = true;
if (opened || this.rowCount > 0) {
this.unlinkQuiet();
}
}
private unlinkQuiet(): void {
unlinkBestEffort(this.path);
untrackLiveSpillPath(this.path);
}
}
/** Validates the spill header once and reads vectors without reopening the file. */
export class EmbeddingSpillReader {
private fd: number | null = null;
private readonly bytesPerVec: number;
readonly dims: number;
readonly rowCount: number;
constructor(spill: EmbeddingVectorSpill) {
this.rowCount = spill.rowCount;
const fd = openSync(spill.path, 'r');
try {
const header = Buffer.alloc(SPILL_HEADER_BYTES);
const headerRead = readSync(fd, header, 0, SPILL_HEADER_BYTES, 0);
if (headerRead !== SPILL_HEADER_BYTES || header.toString('ascii', 0, 4) !== SPILL_MAGIC) {
throw new Error(`invalid embedding spill header: ${spill.path}`);
}
if (header.readUInt8(4) !== SPILL_VERSION) {
throw new Error(`unsupported embedding spill version in ${spill.path}`);
}
const dims = header.readUInt32LE(5);
if (dims !== spill.dims) {
throw new Error(`embedding spill dim mismatch: file ${dims}, expected ${spill.dims}`);
}
this.dims = dims;
this.bytesPerVec = dims * 4;
this.fd = fd;
} catch (err) {
closeSync(fd);
throw err;
}
}
read(indices: readonly number[]): Float32Array[] {
if (this.fd === null) {
throw new Error('embedding spill reader already closed');
}
const out: Float32Array[] = [];
for (const index of indices) {
if (!Number.isInteger(index) || index < 0 || index >= this.rowCount) {
throw new Error(`embedding spill index out of range: ${index}`);
}
const offset = SPILL_HEADER_BYTES + index * this.bytesPerVec;
const copy = new Float32Array(this.dims);
const bytes = new Uint8Array(copy.buffer, copy.byteOffset, this.bytesPerVec);
const n = readSync(this.fd, bytes, 0, this.bytesPerVec, offset);
if (n !== this.bytesPerVec) {
throw new Error(`short embedding spill read at index ${index}`);
}
out.push(copy);
}
return out;
}
close(): void {
if (this.fd === null) return;
closeSync(this.fd);
this.fd = null;
}
}
export function readSpillVectors(
spill: EmbeddingVectorSpill,
indices: readonly number[],
): Float32Array[] {
const reader = new EmbeddingSpillReader(spill);
try {
return reader.read(indices);
} finally {
reader.close();
}
}
export function disposeEmbeddingSpill(spill?: EmbeddingVectorSpill): void {
if (!spill?.path) return;
unlinkBestEffort(spill.path);
untrackLiveSpillPath(spill.path);
}
export function createCachedEmbeddingsBuilder(
options?: LoadCachedEmbeddingsOptions,
): CachedEmbeddingsBuilder {
const inMemoryRowLimit = resolveEmbeddingCacheInMemoryRowLimit(options?.inMemoryRowLimit);
return {
embeddingNodeIds: new Set(),
rows: [],
inMemory: inMemoryRowLimit <= 0 ? null : [],
inMemoryRowLimit,
writer: new EmbeddingSpillWriter(options?.spillDir ?? os.tmpdir()),
};
}
export function ingestCachedEmbeddingRow(
builder: CachedEmbeddingsBuilder,
row: Record<string, unknown> | unknown[],
hasContentHash: boolean,
): void {
const rec = row as Record<string, unknown> & unknown[];
const nodeId = String(rec.nodeId ?? rec[0] ?? '');
if (!nodeId) return;
const embedding = rec.embedding ?? rec[4];
const f32 = coerceEmbeddingToFloat32(embedding);
if (!f32) return;
builder.embeddingNodeIds.add(nodeId);
const meta: CachedEmbeddingMeta = {
nodeId,
chunkIndex: Number(rec.chunkIndex ?? rec[1] ?? 0),
startLine: Number(rec.startLine ?? rec[2] ?? 0),
endLine: Number(rec.endLine ?? rec[3] ?? 0),
contentHash: hasContentHash
? ((rec.contentHash ?? rec[5] ?? undefined) as string | undefined)
: undefined,
vectorIndex: builder.rows.length,
};
builder.rows.push(meta);
if (builder.inMemory && builder.rows.length <= builder.inMemoryRowLimit) {
builder.inMemory.push(f32);
return;
}
if (builder.inMemory) {
for (const prior of builder.inMemory) {
builder.writer.append(prior);
}
builder.inMemory = null;
}
builder.writer.append(f32);
}
export function finalizeCachedEmbeddingsSnapshot(
builder: CachedEmbeddingsBuilder,
): CachedEmbeddingsSnapshot {
const inMemory = builder.inMemory;
if (inMemory) {
builder.writer.abort();
return {
embeddingNodeIds: builder.embeddingNodeIds,
embeddings: builder.rows.map((meta, i) => ({
nodeId: meta.nodeId,
chunkIndex: meta.chunkIndex,
startLine: meta.startLine,
endLine: meta.endLine,
contentHash: meta.contentHash,
embedding: float32ToNumberArray(inMemory[i]!),
})),
rows: builder.rows,
};
}
return {
embeddingNodeIds: builder.embeddingNodeIds,
embeddings: [],
rows: builder.rows,
spill: builder.writer.finish(),
};
}
export function abortCachedEmbeddingsBuilder(builder: CachedEmbeddingsBuilder): void {
builder.writer.abort();
}
export function materializeCachedEmbeddings(
snapshot: CachedEmbeddingsSnapshot,
metas: readonly CachedEmbeddingMeta[],
spillReader?: EmbeddingSpillReader,
): CachedEmbedding[] {
if (metas.length === 0) return [];
if (snapshot.spill && snapshot.embeddings.length === 0) {
const indices = metas.map((meta) => meta.vectorIndex);
const vectors = spillReader
? spillReader.read(indices)
: readSpillVectors(snapshot.spill, indices);
return metas.map((meta, i) => ({
nodeId: meta.nodeId,
chunkIndex: meta.chunkIndex,
startLine: meta.startLine,
endLine: meta.endLine,
contentHash: meta.contentHash,
embedding: float32ToNumberArray(vectors[i]!),
}));
}
if (snapshot.embeddings.length === 0) return [];
const byKey = new Map(
snapshot.embeddings.map((row) => [`${row.nodeId}:${row.chunkIndex}`, row] as const),
);
return metas.map((meta) => {
const hit =
byKey.get(`${meta.nodeId}:${meta.chunkIndex}`) ?? snapshot.embeddings[meta.vectorIndex];
if (!hit) {
throw new Error(`missing cached embedding ${meta.nodeId}:${meta.chunkIndex}`);
}
return hit;
});
}

View file

@ -1,10 +1,10 @@
/**
* Git ref-name validation used by both the CLI and the HTTP analyze route.
*
* Lives in `core/` so `server/api.ts` does not import `cli/analyze-config`
* (that import closed a cli → server → cli cycle: `cli/serve.ts` already
* imports `createServer`). The CLI keeps a thin wrapper that rethrows
* {@link InvalidBranchError} as `GitNexusRcError`.
* Lives in `core/` so `server/api.ts` and `run-analyze.ts` do not import
* `cli/analyze-config` (that import closed a cli → server → cli cycle:
* `cli/serve.ts` already imports `createServer`). The CLI keeps a thin
* wrapper that rethrows {@link InvalidBranchError} as `GitNexusRcError`.
*/
/** Git refs longer than this are almost certainly a mistake / injection attempt. */
@ -126,3 +126,49 @@ export function validateBranchName(value: string, source: string): string {
}
return trimmed;
}
/**
* Best-effort validation for an auto-detected branch (from git). Returns the
* trimmed name, or `undefined` for anything unusable so callers fall back to
* the next precedence tier or leave the index unlabeled. Swallows
* {@link InvalidBranchError} only; unexpected errors are rethrown.
*/
export function sanitizeDetectedBranch(value: string | null | undefined): string | undefined {
if (!value) return undefined;
try {
return validateBranchName(value, 'detected branch');
} catch (err) {
if (err instanceof InvalidBranchError) return undefined;
throw err;
}
}
/**
* Render a rejected checkout name for `onLog`. Hidden / bidi / control code
* points become `\uXXXX` so a git-legal U+202E name cannot reverse the
* warning in a terminal. C1 controls (U+0080–U+009F, including NEL U+0085)
* and remaining Unicode whitespace (`/\s/` — NBSP, U+2028/U+2029, ideographic
* space, etc.) are not all in {@link isHiddenOrControl}; escape them the same
* way so the ASCII escape survives `stripControlCharacters` and the warning
* stays one line. ASCII `"` is escaped; other characters (including backticks)
* stay visible.
*/
export function formatRejectedBranchForLog(value: string): string {
const shouldEscapeRejectedBranchChar = (cp: number, ch: string): boolean =>
isHiddenOrControl(cp) || (cp >= 0x80 && cp <= 0x9f) || /\s/.test(ch);
let out = '';
for (const ch of value) {
const cp = ch.codePointAt(0);
if (cp !== undefined && shouldEscapeRejectedBranchChar(cp, ch)) {
out += `\\u${cp.toString(16).padStart(4, '0')}`;
continue;
}
if (ch === '"') {
out += '\\"';
continue;
}
out += ch;
}
return out;
}

View file

@ -13,18 +13,22 @@
* every strategy invocation, per `import-processor`'s build-once
* context). Lookup per import is then O(1).
*
* Behavior is preserved bit-for-bit: a file is attributed to a target
* iff its **forward-slash (backslash-normalized), case-sensitive** path
* starts with `<targetDir>/`, matching the old
* `normalizedFileList[i].startsWith(targetDir + '/')` comparison
* (`normalizedFileList` is only backslash→forward-slash normalized — NOT
* lowercased — so the match is case-sensitive); the returned paths are
* the original-case `allFileList` entries; and the per-target file ORDER
* follows `allFileList`, so the emitted `{ kind: 'files', files }` set and
* ordering are identical to the old scan.
* A file is attributed to a target iff its **forward-slash
* (backslash-normalized), case-sensitive** path matches at a
* **segment boundary**: it starts with `<targetDir>/` or contains
* `/<targetDir>/`. `normalizedFileList` is
* only backslash→forward-slash normalized — NOT lowercased — so the
* match is case-sensitive. The returned paths are the original-case
* `allFileList` entries, and the per-target file ORDER follows
* `allFileList`.
*
* Import-config fans a file to every matching declared target. Grouping
* (`groupSwiftFilesBySpmTarget`) is first-target-wins. That divergence
* is intentional.
*/
import { SupportedLanguages } from 'gitnexus-shared';
import { coerceDeclaredSwiftTargets, swiftDeclaredTargetPrefix } from '../../language-config.js';
import type { ImportResolutionConfig, ImportResolverStrategy, ResolveCtx } from '../types.js';
interface SwiftTargetIndex {
@ -66,10 +70,10 @@ function getSwiftTargetIndex(
// Pre-compute each target's directory prefix once (original case, to
// match the legacy comparison against the forward-slash-normalized,
// case-sensitive file list — see module docstring).
const targetPrefixes: { name: string; prefix: string }[] = [];
const targetDirs: { name: string; prefix: string }[] = [];
const byTarget = new Map<string, string[]>();
for (const [name, dir] of targets) {
targetPrefixes.push({ name, prefix: dir + '/' });
targetDirs.push({ name, prefix: swiftDeclaredTargetPrefix(dir) });
byTarget.set(name, []);
}
@ -82,10 +86,10 @@ function getSwiftTargetIndex(
for (let i = 0; i < ctx.allFileList.length; i++) {
const norm = ctx.normalizedFileList[i];
if (!norm.endsWith('.swift')) continue;
for (const { name, prefix } of targetPrefixes) {
if (norm.startsWith(prefix)) {
byTarget.get(name)!.push(ctx.allFileList[i]);
}
for (const { name, prefix } of targetDirs) {
if (!norm.startsWith(prefix) && !norm.includes(`/${prefix}`)) continue;
const bucket = byTarget.get(name);
if (bucket !== undefined) bucket.push(ctx.allFileList[i]);
}
}
@ -97,19 +101,19 @@ function getSwiftTargetIndex(
/** Swift Package.swift target map resolution strategy. */
export const swiftPackageStrategy: ImportResolverStrategy = (rawImportPath, _filePath, ctx) => {
const swiftPackageConfig = ctx.configs.swiftPackageConfig;
if (swiftPackageConfig) {
// Only the targets map is needed; build the index lazily so repos
// without a Package.swift config pay nothing.
if (swiftPackageConfig.targets.has(rawImportPath)) {
const index = getSwiftTargetIndex(ctx, swiftPackageConfig.targets);
const files = index.byTarget.get(rawImportPath);
if (files !== undefined && files.length > 0) {
// Copy so callers can't mutate the cached index bucket.
return { kind: 'files', files: [...files] };
}
}
if (swiftPackageConfig == null) return null;
const declared = coerceDeclaredSwiftTargets(swiftPackageConfig);
if (declared == null) return null;
const moduleName = rawImportPath.split('.')[0];
if (moduleName === '' || !declared.has(moduleName)) {
return null;
}
return null; // External framework (Foundation, UIKit, etc.)
const index = getSwiftTargetIndex(ctx, declared);
const files = index.byTarget.get(moduleName);
if (files !== undefined && files.length > 0) {
return { kind: 'files', files: [...files] };
}
return null;
};
export const swiftImportConfig: ImportResolutionConfig = {

View file

@ -175,9 +175,55 @@ export function csharpScanToEvidence(scan: CSharpProjectScan): CSharpNamespaceEv
}
/** Swift Package Manager module config */
export type SwiftPackageConfigOrigin = 'package.swift' | 'directories';
export interface SwiftPackageConfig {
/** Map of target name -> source directory path (e.g., "SiuperModel" -> "Package/Sources/SiuperModel") */
targets: Map<string, string>;
/**
* `package.swift` — extracted from a readable Package.swift with no
* completeness hazards. Explicit import resolve may treat this as a
* declaration map (empty means every name is external).
* `directories` — inferred from `Sources/*` (or Package/Sources / src)
* when no usable declaration exists. Grouping uses this; import resolve
* must not.
* Omitted on hand-built test configs: treated as a declaration map so
* existing `{ targets }` fixtures stay valid.
*/
origin?: SwiftPackageConfigOrigin;
/**
* Declaration map when `origin` is `package.swift` (may be empty).
* Grouping uses `targets`, which is this map when it is non-empty and the
* inferred `Sources/*` map when the declaration is empty — so a
* binary-only Package.swift does not collapse every file into `__default__`.
*/
declaredTargets?: Map<string, string>;
}
/**
* Declaration view for explicit import resolve. `origin: 'directories'`
* is grouping-only. A hand-built `{ targets }` with no origin stays a
* declaration so existing fixtures keep working.
*/
export function coerceDeclaredSwiftTargets(
resolutionConfig: unknown,
): ReadonlyMap<string, string> | null {
const config = resolutionConfig as Partial<SwiftPackageConfig> | null | undefined;
if (config == null) return null;
if (config.origin === 'directories') return null;
if (config.declaredTargets instanceof Map) return config.declaredTargets;
if (config.targets instanceof Map) return config.targets;
return null;
}
/** Segment-boundary prefix for a Package.swift `path:`. `"."` / `"./"` is the package root. */
export function swiftDeclaredTargetPrefix(dir: string): string {
let norm = dir.replace(/\\/g, '/');
while (norm.startsWith('./')) {
norm = norm.slice(2);
}
norm = norm.replace(/\/+$/, '');
return norm === '' || norm === '.' ? '' : `${norm}/`;
}
/** Zig package config parsed from build.zig.zon and the root build.zig */
@ -616,12 +662,768 @@ async function collectDeclaredNamespaces(
return structure.incomplete ? 'truncated' : 'ok';
}
export async function loadSwiftPackageConfig(repoRoot: string): Promise<SwiftPackageConfig | null> {
// Swift imports are module-name based (e.g., `import SiuperModel`)
// SPM convention: Sources/<TargetName>/ or Package/Sources/<TargetName>/
// We scan for these directories to build a target map
const targets = new Map<string, string>();
const SWIFT_SOURCE_FACTORY_NAMES = ['target', 'executableTarget', 'testTarget', 'macro'] as const;
const SWIFT_SKIP_FACTORY_NAMES = ['binaryTarget', 'plugin', 'systemLibrary'] as const;
const SWIFT_SKIP_FACTORIES = new Set<string>(SWIFT_SKIP_FACTORY_NAMES);
const SWIFT_FACTORY_RE = new RegExp(
`\\.(${[...SWIFT_SOURCE_FACTORY_NAMES, ...SWIFT_SKIP_FACTORY_NAMES].join('|')})\\s*\\(`,
'g',
);
function extractBalancedParen(source: string, openIndex: number): string | null {
let depth = 0;
let inString: '"' | "'" | null = null;
let escape = false;
let inLineComment = false;
let blockCommentDepth = 0;
for (let i = openIndex; i < source.length; i++) {
const ch = source[i];
const next = source[i + 1];
if (inLineComment) {
if (ch === '\n') inLineComment = false;
continue;
}
if (blockCommentDepth > 0) {
if (ch === '*' && next === '/') {
blockCommentDepth--;
i++;
} else if (ch === '/' && next === '*') {
blockCommentDepth++;
i++;
}
continue;
}
if (inString !== null) {
if (escape) {
escape = false;
continue;
}
if (ch === '\\') {
escape = true;
continue;
}
if (ch === inString) inString = null;
continue;
}
if (ch === '"' || ch === "'") {
inString = ch;
continue;
}
if (ch === '/' && next === '/') {
// `https://` lives inside a string, already excluded above.
inLineComment = true;
i++;
continue;
} else if (ch === '/' && next === '*') {
blockCommentDepth = 1;
i++;
continue;
}
if (ch === '(') depth++;
else if (ch === ')') {
depth--;
if (depth === 0) return source.slice(openIndex + 1, i);
}
}
return null;
}
function isSwiftIdentCont(ch: string | undefined): boolean {
return ch !== undefined && /[A-Za-z0-9_]/.test(ch);
}
function skipSwiftWsAndComments(source: string, start: number): number | null {
let i = start;
while (i < source.length) {
const ch = source[i];
const next = source[i + 1];
if (/\s/.test(ch)) {
i++;
continue;
}
if (ch === '/' && next === '/') {
const nl = source.indexOf('\n', i + 2);
if (nl === -1) return null;
i = nl + 1;
continue;
}
if (ch === '/' && next === '*') {
let depth = 1;
i += 2;
while (i < source.length && depth > 0) {
if (source[i] === '/' && source[i + 1] === '*') {
depth++;
i += 2;
} else if (source[i] === '*' && source[i + 1] === '/') {
depth--;
i += 2;
} else {
i++;
}
}
if (depth !== 0) return null;
continue;
}
return i;
}
return null;
}
/** First `name:` / `path:` string outside comments. Escapes and interpolations are unreadable. */
function readSwiftFactoryField(
block: string,
field: 'name' | 'path',
): { value: string | undefined; keyPresent: boolean } {
let inString: '"' | "'" | null = null;
let escape = false;
let inLineComment = false;
let blockCommentDepth = 0;
for (let i = 0; i < block.length; i++) {
const ch = block[i];
const next = block[i + 1];
if (inLineComment) {
if (ch === '\n') inLineComment = false;
continue;
}
if (blockCommentDepth > 0) {
if (ch === '*' && next === '/') {
blockCommentDepth--;
i++;
} else if (ch === '/' && next === '*') {
blockCommentDepth++;
i++;
}
continue;
}
if (inString !== null) {
if (escape) {
escape = false;
continue;
}
if (ch === '\\') {
escape = true;
continue;
}
if (ch === inString) inString = null;
continue;
}
if (ch === '"' || ch === "'") {
inString = ch;
continue;
}
if (ch === '/' && next === '/') {
inLineComment = true;
i++;
continue;
}
if (ch === '/' && next === '*') {
blockCommentDepth = 1;
i++;
continue;
}
if (!/[A-Za-z_]/.test(ch)) continue;
let j = i + 1;
while (j < block.length && isSwiftIdentCont(block[j])) j++;
if (block.slice(i, j) !== field) {
i = j - 1;
continue;
}
const colonAt = skipSwiftWsAndComments(block, j);
if (colonAt === null || block[colonAt] !== ':') {
i = j - 1;
continue;
}
const valueAt = skipSwiftWsAndComments(block, colonAt + 1);
if (valueAt === null) return { value: undefined, keyPresent: true };
const quote = block[valueAt];
if (quote !== '"' && quote !== "'") return { value: undefined, keyPresent: true };
const parsed = readSwiftSimpleQuotedString(block, valueAt);
if (parsed === null) return { value: undefined, keyPresent: true };
return { value: parsed, keyPresent: true };
}
return { value: undefined, keyPresent: false };
}
/** Quoted literal with no escapes. Any `\` (including `\u{…}` and `\(`) is unreadable. */
function readSwiftSimpleQuotedString(source: string, openIndex: number): string | null {
const quote = source[openIndex];
let i = openIndex + 1;
while (i < source.length) {
const ch = source[i];
if (ch === '\\') return null;
if (ch === quote) return source.slice(openIndex + 1, i);
if (ch === '\n') return null;
i++;
}
return null;
}
function swiftManifestHasCompletenessHazard(source: string): boolean {
let inString: '"' | "'" | null = null;
let escape = false;
let inLineComment = false;
let blockCommentDepth = 0;
let atLineStart = true;
for (let i = 0; i < source.length; i++) {
const ch = source[i];
const next = source[i + 1];
if (inLineComment) {
if (ch === '\n') {
inLineComment = false;
atLineStart = true;
}
continue;
}
if (blockCommentDepth > 0) {
if (ch === '*' && next === '/') {
blockCommentDepth--;
i++;
} else if (ch === '/' && next === '*') {
blockCommentDepth++;
i++;
} else if (ch === '\n') {
atLineStart = true;
}
continue;
}
if (inString !== null) {
if (escape) {
escape = false;
continue;
}
if (ch === '\\') {
escape = true;
continue;
}
if (ch === inString) inString = null;
else if (ch === '\n') atLineStart = true;
continue;
}
if (ch === '"' || ch === "'") {
inString = ch;
atLineStart = false;
continue;
}
if (ch === '/' && next === '/') {
inLineComment = true;
i++;
atLineStart = false;
continue;
}
if (ch === '/' && next === '*') {
blockCommentDepth = 1;
i++;
atLineStart = false;
continue;
}
if (ch === '\n') {
atLineStart = true;
continue;
}
if (atLineStart && /\s/.test(ch)) continue;
if (atLineStart && ch === '#') {
if (source.startsWith('if', i + 1) && !isSwiftIdentCont(source[i + 3])) return true;
if (source.startsWith('elseif', i + 1) && !isSwiftIdentCont(source[i + 7])) return true;
}
atLineStart = false;
}
return false;
}
interface SwiftCommentScan {
i: number;
inString: '"' | "'" | null;
escape: boolean;
inLineComment: boolean;
blockCommentDepth: number;
}
function newSwiftCommentScan(): SwiftCommentScan {
return { i: 0, inString: null, escape: false, inLineComment: false, blockCommentDepth: 0 };
}
/** Resume the comment/string walk up to `upTo`. Matches are left-to-right, so this is O(n) over the file. */
function advanceSwiftCommentScan(source: string, state: SwiftCommentScan, upTo: number): void {
let { i, inString, escape, inLineComment, blockCommentDepth } = state;
for (; i < upTo; i++) {
const ch = source[i];
const next = source[i + 1];
if (inLineComment) {
if (ch === '\n') inLineComment = false;
continue;
}
if (blockCommentDepth > 0) {
if (ch === '*' && next === '/') {
blockCommentDepth--;
i++;
} else if (ch === '/' && next === '*') {
blockCommentDepth++;
i++;
}
continue;
}
if (inString !== null) {
if (escape) {
escape = false;
continue;
}
if (ch === '\\') {
escape = true;
continue;
}
if (ch === inString) inString = null;
continue;
}
if (ch === '"' || ch === "'") {
inString = ch;
continue;
}
if (ch === '/' && next === '/') {
inLineComment = true;
i++;
continue;
}
if (ch === '/' && next === '*') {
blockCommentDepth = 1;
i++;
}
}
state.i = i;
state.inString = inString;
state.escape = escape;
state.inLineComment = inLineComment;
state.blockCommentDepth = blockCommentDepth;
}
function swiftPathIsUnreadable(customPath: string | undefined, hasPathKey: boolean): boolean {
if (!hasPathKey) return false;
return customPath === undefined || customPath === '' || customPath.includes('\\(');
}
/** Heuristic Package.swift scan. Never shells out to `swift package dump-package`. */
export function parseSwiftPackageManifest(source: string): {
targets: Map<string, string>;
complete: boolean;
} {
const targets = new Map<string, string>();
if (swiftManifestHasCompletenessHazard(source)) {
return { targets, complete: false };
}
const packageTargets = inspectSwiftPackageTargets(source);
SWIFT_FACTORY_RE.lastIndex = 0;
let match: RegExpExecArray | null;
let sawUnreadableFactory = false;
const commentScan = newSwiftCommentScan();
let coveredEnd = -1;
while ((match = SWIFT_FACTORY_RE.exec(source)) !== null) {
advanceSwiftCommentScan(source, commentScan, match.index);
if (
commentScan.inLineComment ||
commentScan.blockCommentDepth > 0 ||
commentScan.inString !== null
) {
continue;
}
if (
packageTargets.sawPackage &&
!packageTargets.arraySpans.some(([lo, hi]) => match.index >= lo && match.index <= hi)
) {
continue;
}
if (match.index > 0 && match.index < coveredEnd) continue;
const kind = match[1];
const paren = source.indexOf('(', match.index);
const block = extractBalancedParen(source, paren);
if (block === null) {
sawUnreadableFactory = true;
continue;
}
coveredEnd = Math.max(coveredEnd, paren + 1 + block.length + 1);
if (SWIFT_SKIP_FACTORIES.has(kind)) continue;
const nameField = readSwiftFactoryField(block, 'name');
if (nameField.value === undefined || nameField.value === '') {
sawUnreadableFactory = true;
continue;
}
const name = nameField.value;
const pathField = readSwiftFactoryField(block, 'path');
const customPath = pathField.value;
if (swiftPathIsUnreadable(customPath, pathField.keyPresent)) {
sawUnreadableFactory = true;
continue;
}
const dir = customPath ?? (kind === 'testTarget' ? `Tests/${name}` : `Sources/${name}`);
const existing = targets.get(name);
if (existing === undefined) {
targets.set(name, dir);
} else if (customPath !== undefined && existing === `Sources/${name}`) {
// A later `.target(name:path:)` wins over an earlier same-name
// factory that only implied the default path.
targets.set(name, customPath);
}
}
return {
targets,
complete: !sawUnreadableFactory && !packageTargets.helperBuilt,
};
}
interface SwiftPackageTargetsInspection {
helperBuilt: boolean;
sawPackage: boolean;
arraySpans: Array<[number, number]>;
}
const SWIFT_ALL_FACTORY_NAMES = new Set<string>([
...SWIFT_SOURCE_FACTORY_NAMES,
...SWIFT_SKIP_FACTORY_NAMES,
]);
/** Locate `Package(...)`'s `targets:` argument. Product `targets:` stay nested. */
function inspectSwiftPackageTargets(source: string): SwiftPackageTargetsInspection {
const arraySpans: Array<[number, number]> = [];
let helperBuilt = false;
const seen = { package: false };
const unreadable = forEachSwiftPackageArgs(
source,
(args, argsStart) => {
const found = inspectPackageTargetsArg(args);
if (found.helperBuilt) {
helperBuilt = true;
return true;
}
if (found.arrayStart !== null && found.arrayEnd !== null) {
arraySpans.push([argsStart + found.arrayStart, argsStart + found.arrayEnd]);
}
return false;
},
seen,
);
return { helperBuilt: helperBuilt || unreadable, sawPackage: seen.package, arraySpans };
}
/** Walk `Package(` calls outside comments/strings. Unclosed `Package(` is incomplete. */
function forEachSwiftPackageArgs(
source: string,
visit: (args: string, argsStart: number) => boolean,
seen: { package: boolean },
): boolean {
let inString: '"' | "'" | null = null;
let escape = false;
let inLineComment = false;
let blockCommentDepth = 0;
for (let i = 0; i < source.length; i++) {
const ch = source[i];
const next = source[i + 1];
if (inLineComment) {
if (ch === '\n') inLineComment = false;
continue;
}
if (blockCommentDepth > 0) {
if (ch === '*' && next === '/') {
blockCommentDepth--;
i++;
} else if (ch === '/' && next === '*') {
blockCommentDepth++;
i++;
}
continue;
}
if (inString !== null) {
if (escape) {
escape = false;
continue;
}
if (ch === '\\') {
escape = true;
continue;
}
if (ch === inString) inString = null;
continue;
}
if (ch === '"' || ch === "'") {
inString = ch;
continue;
}
if (ch === '/' && next === '/') {
inLineComment = true;
i++;
continue;
}
if (ch === '/' && next === '*') {
blockCommentDepth = 1;
i++;
continue;
}
if (
!source.startsWith('Package', i) ||
isSwiftIdentCont(source[i + 7]) ||
(i > 0 && isSwiftIdentCont(source[i - 1]))
) {
continue;
}
const parenAt = skipSwiftWsAndComments(source, i + 7);
if (parenAt === null || source[parenAt] !== '(') continue;
seen.package = true;
const args = extractBalancedParen(source, parenAt);
if (args === null) return true;
if (visit(args, parenAt + 1)) return true;
i = parenAt + args.length + 1;
}
return false;
}
function inspectPackageTargetsArg(args: string): {
helperBuilt: boolean;
arrayStart: number | null;
arrayEnd: number | null;
} {
let inString: '"' | "'" | null = null;
let escape = false;
let inLineComment = false;
let blockCommentDepth = 0;
let paren = 0;
for (let i = 0; i < args.length; i++) {
const ch = args[i];
const next = args[i + 1];
if (inLineComment) {
if (ch === '\n') inLineComment = false;
continue;
}
if (blockCommentDepth > 0) {
if (ch === '*' && next === '/') {
blockCommentDepth--;
i++;
} else if (ch === '/' && next === '*') {
blockCommentDepth++;
i++;
}
continue;
}
if (inString !== null) {
if (escape) {
escape = false;
continue;
}
if (ch === '\\') {
escape = true;
continue;
}
if (ch === inString) inString = null;
continue;
}
if (ch === '"' || ch === "'") {
inString = ch;
continue;
}
if (ch === '/' && next === '/') {
inLineComment = true;
i++;
continue;
}
if (ch === '/' && next === '*') {
blockCommentDepth = 1;
i++;
continue;
}
if (ch === '(') {
paren++;
continue;
}
if (ch === ')') {
paren--;
continue;
}
if (paren !== 0) continue;
if (
!args.startsWith('targets', i) ||
isSwiftIdentCont(args[i + 7]) ||
(i > 0 && isSwiftIdentCont(args[i - 1]))
) {
continue;
}
const colonAt = skipSwiftWsAndComments(args, i + 7);
if (colonAt === null || args[colonAt] !== ':') {
i += 6;
continue;
}
return classifyPackageTargetsValue(args, colonAt + 1);
}
return { helperBuilt: false, arrayStart: null, arrayEnd: null };
}
function classifyPackageTargetsValue(
args: string,
afterColon: number,
): {
helperBuilt: boolean;
arrayStart: number | null;
arrayEnd: number | null;
} {
const start = skipSwiftWsAndComments(args, afterColon);
if (start === null) return { helperBuilt: true, arrayStart: null, arrayEnd: null };
if (args[start] === '[') {
const close = matchSwiftSquare(args, start);
if (close === null) return { helperBuilt: true, arrayStart: null, arrayEnd: null };
const next = skipSwiftWsAndComments(args, close + 1);
if (next !== null && args[next] === '+') {
return { helperBuilt: true, arrayStart: start, arrayEnd: close };
}
if (packageTargetsArrayHasComputed(args, start, close)) {
return { helperBuilt: true, arrayStart: start, arrayEnd: close };
}
return { helperBuilt: false, arrayStart: start, arrayEnd: close };
}
return { helperBuilt: true, arrayStart: null, arrayEnd: null };
}
function packageTargetsArrayHasComputed(source: string, open: number, close: number): boolean {
let inString: '"' | "'" | null = null;
let escape = false;
let inLineComment = false;
let blockCommentDepth = 0;
let paren = 0;
let bracket = 0;
for (let i = open; i < close; i++) {
const ch = source[i];
const next = source[i + 1];
if (inLineComment) {
if (ch === '\n') inLineComment = false;
continue;
}
if (blockCommentDepth > 0) {
if (ch === '*' && next === '/') {
blockCommentDepth--;
i++;
} else if (ch === '/' && next === '*') {
blockCommentDepth++;
i++;
}
continue;
}
if (inString !== null) {
if (escape) {
escape = false;
continue;
}
if (ch === '\\') {
escape = true;
continue;
}
if (ch === inString) inString = null;
continue;
}
if (ch === '"' || ch === "'") {
inString = ch;
continue;
}
if (ch === '/' && next === '/') {
inLineComment = true;
i++;
continue;
}
if (ch === '/' && next === '*') {
blockCommentDepth = 1;
i++;
continue;
}
if (ch === '[') {
bracket++;
continue;
}
if (ch === ']') {
bracket--;
continue;
}
if (ch === '(') {
paren++;
continue;
}
if (ch === ')') {
paren--;
continue;
}
if (bracket !== 1 || paren !== 0) continue;
if (ch === ',' || /\s/.test(ch)) continue;
if (ch === '.') {
let j = i + 1;
while (j < close && isSwiftIdentCont(source[j])) j++;
const name = source.slice(i + 1, j);
const after = skipSwiftWsAndComments(source, j);
if (after !== null && source[after] === '(' && SWIFT_ALL_FACTORY_NAMES.has(name)) {
const block = extractBalancedParen(source, after);
if (block === null) return true;
i = after + block.length + 1;
continue;
}
return true;
}
return true;
}
return false;
}
function matchSwiftSquare(source: string, openIndex: number): number | null {
let depth = 0;
let inString: '"' | "'" | null = null;
let escape = false;
let inLineComment = false;
let blockCommentDepth = 0;
for (let i = openIndex; i < source.length; i++) {
const ch = source[i];
const next = source[i + 1];
if (inLineComment) {
if (ch === '\n') inLineComment = false;
continue;
}
if (blockCommentDepth > 0) {
if (ch === '*' && next === '/') {
blockCommentDepth--;
i++;
} else if (ch === '/' && next === '*') {
blockCommentDepth++;
i++;
}
continue;
}
if (inString !== null) {
if (escape) {
escape = false;
continue;
}
if (ch === '\\') {
escape = true;
continue;
}
if (ch === inString) inString = null;
continue;
}
if (ch === '"' || ch === "'") {
inString = ch;
continue;
}
if (ch === '/' && next === '/') {
inLineComment = true;
i++;
continue;
} else if (ch === '/' && next === '*') {
blockCommentDepth = 1;
i++;
continue;
}
if (ch === '[') depth++;
else if (ch === ']') {
depth--;
if (depth === 0) return i;
}
}
return null;
}
async function inferSwiftDirectoryTargets(repoRoot: string): Promise<Map<string, string>> {
const targets = new Map<string, string>();
const sourceDirs = ['Sources', 'Package/Sources', 'src'];
for (const sourceDir of sourceDirs) {
try {
@ -636,12 +1438,41 @@ export async function loadSwiftPackageConfig(repoRoot: string): Promise<SwiftPac
// Directory doesn't exist
}
}
return targets;
}
if (targets.size > 0) {
if (isDev) {
logger.info(`📦 Loaded ${targets.size} Swift package targets`);
export async function loadSwiftPackageConfig(repoRoot: string): Promise<SwiftPackageConfig | null> {
try {
const source = await fs.readFile(path.join(repoRoot, 'Package.swift'), 'utf-8');
const parsed = parseSwiftPackageManifest(source);
if (parsed.complete) {
if (isDev) {
logger.info(`📦 Loaded ${parsed.targets.size} Swift package targets from Package.swift`);
}
if (parsed.targets.size > 0) {
return {
targets: parsed.targets,
origin: 'package.swift',
declaredTargets: parsed.targets,
};
}
const inferred = await inferSwiftDirectoryTargets(repoRoot);
return {
targets: inferred,
origin: 'package.swift',
declaredTargets: parsed.targets,
};
}
return { targets };
} catch {
// Missing or unreadable — fall through to inferred folders.
}
const inferred = await inferSwiftDirectoryTargets(repoRoot);
if (inferred.size > 0) {
if (isDev) {
logger.info(`📦 Inferred ${inferred.size} Swift source folders`);
}
return { targets: inferred, origin: 'directories' };
}
return null;
}

View file

@ -48,6 +48,7 @@ import {
synthesizeSwiftReceiverBinding,
} from './receiver-binding.js';
import { synthesizeSwiftSignatureBindings } from './signature-bindings.js';
import { swiftMethodConfig } from '../../method-extractors/configs/swift.js';
import { getSwiftParser, getSwiftScopeQuery } from './query.js';
import { preprocessSwiftConditionalDirectives } from './conditional-directive-preprocess.js';
import { recordCacheHit, recordCacheMiss } from './cache-stats.js';
@ -198,6 +199,26 @@ export function emitSwiftScopeCaptures(
continue;
}
// The query deliberately recognizes the compact `lhs = call()` shape;
// enforce "untyped lhs" here because tree-sitter queries cannot express
// absence of Swift's sibling type_annotation robustly. A typed declaration
// remains authoritative and must never enter return-type replay.
if (grouped['@call-result-assignment.call'] !== undefined) {
const callNode = nodeIfType(nodeMap['@call-result-assignment.call'], 'call_expression');
let property = callNode?.parent;
while (property?.type === 'await_expression' || property?.type === 'try_expression') {
property = property.parent;
}
if (
property?.type !== 'property_declaration' ||
property.namedChildren.some((child) => child.type === 'type_annotation')
) {
continue;
}
out.push(grouped);
continue;
}
// ── Field accesses: a `navigation_expression` (`obj.field`) is one of
// three things. Drop it when it's a call's callee (`u.save` in
// `u.save()` — the @reference.call.member query already covers that).
@ -314,7 +335,17 @@ export function emitSwiftScopeCaptures(
nodeMap['@declaration.constructor'],
...FUNCTION_NODE_TYPES,
);
if (fnNodeForArity !== null) attachArityMetadata(grouped, fnNodeForArity);
if (fnNodeForArity !== null) {
attachArityMetadata(grouped, fnNodeForArity);
const returnType = swiftMethodConfig.extractReturnType?.(fnNodeForArity);
if (returnType !== undefined && returnType !== '') {
grouped['@declaration.return-type'] = syntheticCapture(
'@declaration.return-type',
fnNodeForArity,
returnType,
);
}
}
// Structural receiver chain for a call whose receiver is itself an
// expression, so resolution can type it by folding over structure
// instead of re-parsing the receiver's source text. Self-gating: a
@ -338,7 +369,17 @@ export function emitSwiftScopeCaptures(
const declTag = FUNCTION_DECL_TAGS.find((t) => grouped[t] !== undefined);
if (declTag !== undefined) {
const fnNode = nodeIfType(nodeMap[declTag], ...FUNCTION_NODE_TYPES);
if (fnNode !== null) attachArityMetadata(grouped, fnNode);
if (fnNode !== null) {
attachArityMetadata(grouped, fnNode);
const returnType = swiftMethodConfig.extractReturnType?.(fnNode);
if (returnType !== undefined && returnType !== '') {
grouped['@declaration.return-type'] = syntheticCapture(
'@declaration.return-type',
fnNode,
returnType,
);
}
}
}
// ── Constructor calls: Swift has no `new`, so `Foo()` is a free call

View file

@ -14,12 +14,15 @@
* Module identity: Swift has no in-source `package X` marker. Module
* membership is the SPM target *subtree* (`Sources/<Target>/…`), threaded
* in via the SPM target map (`resolutionConfig` → `coerceSwiftTargets`)
* and grouped by `groupSwiftFilesBySpmTarget` — replicating legacy
* `groupSwiftFilesByTarget`. With no scanned source dir the map is null
* and grouped by `groupSwiftFilesBySpmTarget`. The helper preserves the
* legacy `groupSwiftFilesByTarget` bucketing contract for ordinary layouts
* while intentionally fixing #2931's repeated-prefix edge case. With no
* target map (no Package.swift and no scanned `Sources/*`) the map is null
* and all files form one `__default__` module (single-Xcode-project
* assumption). Every pair of distinct `.swift` files in the same module
* gets a directed IMPORTS edge in both directions (whole-module
* visibility is symmetric).
* assumption). An inferred folder map must keep sibling folders isolated.
* Every pair of distinct `.swift`
* files in the same module gets a directed IMPORTS edge in both directions
* (whole-module visibility is symmetric).
*
* Node identity + edge construction mirror the generic `emitImportEdges`
* convention (`graph-bridge/imports-to-edges.ts`): `generateId('File', path)`
@ -41,8 +44,6 @@ export function emitSwiftImplicitImportEdges(
_nodeLookup: GraphNodeLookup,
resolutionConfig?: unknown,
): void {
// Group files by SPM target subtree (the module). No-source-dir → all
// files in one `__default__` bucket.
const targets = coerceSwiftTargets(resolutionConfig);
const filesByTarget = groupSwiftFilesBySpmTarget(
parsedFiles,
@ -51,16 +52,15 @@ export function emitSwiftImplicitImportEdges(
);
for (const [, group] of filesByTarget) {
if (group.length < 2) continue; // no siblings to import
if (group.length < 2) continue;
for (const source of group) {
for (const target of group) {
if (source.filePath === target.filePath) continue; // no self-import
const dedupKey = `${source.filePath}->${target.filePath}`;
for (const dest of group) {
if (source.filePath === dest.filePath) continue;
const dedupKey = `${source.filePath}->${dest.filePath}`;
graph.addRelationship({
id: generateId('IMPORTS', dedupKey),
sourceId: generateId('File', source.filePath),
targetId: generateId('File', target.filePath),
targetId: generateId('File', dest.filePath),
type: 'IMPORTS',
confidence: 1.0,
reason: 'swift-scope: implicit module visibility',

View file

@ -4,32 +4,30 @@
* `@import.name` / `@import.testable` that `interpretSwiftImport`
* consumes.
*
* Swift imports are whole-module (no named members), so this is 1:1 —
* one `import` produces exactly one import. The split layer exposes the
* module name and the `@testable` flag without pushing raw-text parsing
* into `interpret.ts`.
* import Foundation → kind=namespace, source=Foundation
* import Foo.Bar → kind=namespace, source=Foo
* import struct Foo.Bar → kind=named, source=Foo, name=Bar
* @testable import MyApp → kind=namespace, source=MyApp, testable=1
* @_exported import Foo → kind=reexport, source=Foo, name=Foo
* @_exported import struct Foo.Bar → kind=reexport, source=Foo, name=Bar
*
* import Foundation → kind=namespace, source=Foundation
* import Foo.Bar → kind=namespace, source=Foo (SPM target),
* name=Foo.Bar (full path, for reference)
* @testable import MyApp → kind=namespace, source=MyApp, testable=1
*
* Verified against tree-sitter-swift 0.7.1:
* (import_declaration
* (modifiers (attribute (user_type (type_identifier))))? ; @testable / @_exported
* (identifier (simple_identifier)+)) ; one per dotted segment
* Import-kind (`struct`/`class`/…) is not a named tree-sitter child
* (hidden `_import_kind` in 0.7.1). Read it from the statement text.
* `@_exported` / `@testable` live on `modifiers`.
*/
import type { Capture, CaptureMatch } from 'gitnexus-shared';
import { nodeToCapture, syntheticCapture, type SyntaxNode } from '../../utils/ast-helpers.js';
const IMPORT_KIND_TOKEN_RE = /^(struct|class|enum|protocol|func|let|var|typealias)\b/;
interface SwiftImportSpec {
/** SPM target name — the first dotted segment (`Foo` in `import Foo.Bar`). */
readonly source: string;
/** Full dotted module path (`Foo.Bar`). */
readonly memberName: string;
readonly fullPath: string;
/** True for `@testable import` (test-scope visibility; resolves identically). */
readonly testable: boolean;
readonly exported: boolean;
readonly importKind: string | null;
readonly atNode: SyntaxNode;
}
@ -42,14 +40,15 @@ export function splitSwiftImport(stmtNode: SyntaxNode): CaptureMatch | null {
function parseSwiftImport(node: SyntaxNode): SwiftImportSpec | null {
let testable = false;
let exported = false;
let identifierNode: SyntaxNode | null = null;
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (child === null) continue;
if (child.type === 'modifiers') {
// Any attribute whose text mentions `testable` flips the flag.
if (/\btestable\b/.test(child.text)) testable = true;
if (swiftModifiersHaveAttribute(child.text, 'testable')) testable = true;
if (swiftModifiersHaveAttribute(child.text, '_exported')) exported = true;
} else if (child.type === 'identifier') {
identifierNode = child;
}
@ -57,36 +56,219 @@ function parseSwiftImport(node: SyntaxNode): SwiftImportSpec | null {
if (identifierNode === null) return null;
// The module path is one or more simple_identifier children, one per
// dotted segment. The SPM target is the FIRST segment.
const importKind = importKindFromClause(node, identifierNode);
const segments: string[] = [];
for (let i = 0; i < identifierNode.namedChildCount; i++) {
const seg = identifierNode.namedChild(i);
if (seg !== null && seg.type === 'simple_identifier') segments.push(seg.text);
}
if (segments.length === 0) {
// Fall back to the raw identifier text (e.g. a grammar shape we didn't
// anticipate). Split on `.` to recover the target segment.
const raw = identifierNode.text.trim();
if (raw === '') return null;
const parts = raw.split('.');
return { source: parts[0], fullPath: raw, testable, atNode: node };
segments.push(...raw.split('.'));
}
return {
source: segments[0],
memberName: segments.length > 1 ? segments[segments.length - 1] : segments[0],
fullPath: segments.join('.'),
testable,
exported,
importKind,
atNode: node,
};
}
function isSwiftIdentCont(ch: string | undefined): boolean {
return ch !== undefined && /[A-Za-z0-9_]/.test(ch);
}
/** First `word` outside comments and strings. Linear scan. */
function indexOfBareWord(text: string, word: string): number {
let inString: '"' | "'" | null = null;
let escape = false;
let inLineComment = false;
let blockCommentDepth = 0;
for (let i = 0; i < text.length; i++) {
const ch = text[i];
const next = text[i + 1];
if (inLineComment) {
if (ch === '\n') inLineComment = false;
continue;
}
if (blockCommentDepth > 0) {
if (ch === '*' && next === '/') {
blockCommentDepth--;
i++;
} else if (ch === '/' && next === '*') {
blockCommentDepth++;
i++;
}
continue;
}
if (inString !== null) {
if (escape) {
escape = false;
continue;
}
if (ch === '\\') {
escape = true;
continue;
}
if (ch === inString) inString = null;
continue;
}
if (ch === '"' || ch === "'") {
inString = ch;
continue;
}
if (ch === '/' && next === '/') {
inLineComment = true;
i++;
continue;
}
if (ch === '/' && next === '*') {
blockCommentDepth = 1;
i++;
continue;
}
if (
text.startsWith(word, i) &&
!isSwiftIdentCont(text[i + word.length]) &&
(i === 0 || !isSwiftIdentCont(text[i - 1]))
) {
return i;
}
}
return -1;
}
/** `@name` token in modifier text — not `_exported` / `testable` inside a string. */
function swiftModifiersHaveAttribute(text: string, name: 'testable' | '_exported'): boolean {
let inString: '"' | "'" | null = null;
let escape = false;
let inLineComment = false;
let blockCommentDepth = 0;
for (let i = 0; i < text.length; i++) {
const ch = text[i];
const next = text[i + 1];
if (inLineComment) {
if (ch === '\n') inLineComment = false;
continue;
}
if (blockCommentDepth > 0) {
if (ch === '*' && next === '/') {
blockCommentDepth--;
i++;
} else if (ch === '/' && next === '*') {
blockCommentDepth++;
i++;
}
continue;
}
if (inString !== null) {
if (escape) {
escape = false;
continue;
}
if (ch === '\\') {
escape = true;
continue;
}
if (ch === inString) inString = null;
continue;
}
if (ch === '"' || ch === "'") {
inString = ch;
continue;
}
if (ch === '/' && next === '/') {
inLineComment = true;
i++;
continue;
}
if (ch === '/' && next === '*') {
blockCommentDepth = 1;
i++;
continue;
}
if (ch !== '@') continue;
let j = i + 1;
while (j < text.length && /\s/.test(text[j])) j++;
if (text.startsWith(name, j) && !isSwiftIdentCont(text[j + name.length])) return true;
}
return false;
}
/** Kind token from the import clause only — skip `@available(..., message: "import struct")`. */
function importKindFromClause(node: SyntaxNode, identifierNode: SyntaxNode): string | null {
const identRel = identifierNode.startIndex - node.startIndex;
const before = identRel >= 0 ? node.text.slice(0, identRel) : node.text;
const importAt = indexOfBareWord(before, 'import');
const clause = importAt === -1 ? before : before.slice(importAt);
return kindAfterImportKeyword(clause);
}
/** After `import`, skip whitespace and comments, then read a kind token. Linear: no nested-quantifier backtracking. */
function kindAfterImportKeyword(clause: string): string | null {
const start = indexOfBareWord(clause, 'import');
if (start === -1) return null;
let i = start + 'import'.length;
let skipped = false;
while (i < clause.length) {
const ch = clause[i];
const next = clause[i + 1];
if (/\s/.test(ch)) {
skipped = true;
i += 1;
continue;
}
if (ch === '/' && next === '/') {
const nl = clause.indexOf('\n', i + 2);
if (nl === -1) return null;
skipped = true;
i = nl + 1;
continue;
}
if (ch === '/' && next === '*') {
let depth = 1;
i += 2;
while (i < clause.length && depth > 0) {
if (clause[i] === '/' && clause[i + 1] === '*') {
depth++;
i += 2;
} else if (clause[i] === '*' && clause[i + 1] === '/') {
depth--;
i += 2;
} else {
i++;
}
}
if (depth !== 0) return null;
skipped = true;
continue;
}
break;
}
if (!skipped) return null;
return IMPORT_KIND_TOKEN_RE.exec(clause.slice(i))?.[1] ?? null;
}
function bindingKind(spec: SwiftImportSpec): 'namespace' | 'named' | 'reexport' {
if (spec.exported) return 'reexport';
if (spec.importKind !== null && spec.fullPath.includes('.')) return 'named';
return 'namespace';
}
function buildImportMatch(stmtNode: SyntaxNode, spec: SwiftImportSpec): CaptureMatch {
const kind = bindingKind(spec);
const nameText = kind === 'namespace' ? spec.fullPath : spec.memberName;
const m: Record<string, Capture> = {
'@import.statement': nodeToCapture('@import.statement', stmtNode),
'@import.kind': syntheticCapture('@import.kind', spec.atNode, 'namespace'),
'@import.kind': syntheticCapture('@import.kind', spec.atNode, kind),
'@import.source': syntheticCapture('@import.source', spec.atNode, spec.source),
'@import.name': syntheticCapture('@import.name', spec.atNode, spec.fullPath),
'@import.name': syntheticCapture('@import.name', spec.atNode, nameText),
};
if (spec.testable) {
m['@import.testable'] = syntheticCapture('@import.testable', spec.atNode, '1');

View file

@ -1,37 +1,27 @@
/**
* `resolveImportTarget` adapter for the Swift `ScopeResolver`.
*
* Swift's `import ModuleName` brings in a whole SPM target / framework
* module. The scope-resolution contract passes only `allFilePaths` (no
* `SwiftPackageConfig`), so we resolve a module name to the `.swift`
* files under a directory segment named after the module — the SPM
* convention `Sources/<Module>/*.swift` (and the common
* `<Module>/*.swift` layout). This needs no manifest parsing.
* A Package.swift declaration map (`origin: 'package.swift'`) resolves
* only declared target names. Otherwise refuse well-known SDK module
* names and fall back to the memoized directory-segment index so local
* folder modules still resolve without a manifest.
*
* Same-module (intra-target) visibility — the bulk of Swift cross-file
* resolution, which needs NO `import` statement — is handled separately
* by `populateSwiftTargetSiblings` (see `target-siblings.ts`). This
* adapter only resolves EXPLICIT `import` statements (cross-module).
*
* Returns all matching files (one ImportEdge per file, like Go's
* package resolver) so every exported symbol in the module materializes
* a binding. Returns `null` for external frameworks (Foundation, UIKit,
* …) that have no in-repo directory.
*
* Performance: the directory→files grouping is memoized on the stable
* `allFilePaths` Set identity (the same Set is threaded to every import
* in a run), so it is built once per run — NOT once per import. Mirrors
* Python's `getPythonFileIndex` WeakMap pattern (PR #1918).
* Same-module visibility without `import` is `populateSwiftTargetSiblings`.
* This adapter only resolves EXPLICIT cross-module `import`s.
*/
import type { ParsedImport, WorkspaceIndex } from 'gitnexus-shared';
import type { ParsedFile, ParsedImport, WorkspaceIndex } from 'gitnexus-shared';
import { perFileSet } from '../../import-resolvers/per-file-set.js';
import { coerceDeclaredSwiftTargets, swiftDeclaredTargetPrefix } from '../../language-config.js';
import { isSwiftSdkModule } from './sdk-modules.js';
export interface SwiftResolveContext {
readonly fromFile: string;
/** `ReadonlySet` so the orchestrator's stable run-level set flows
* straight through to the memoized index key. */
readonly allFilePaths: ReadonlySet<string>;
readonly resolutionConfig?: unknown;
readonly parsedFiles?: readonly ParsedFile[];
}
interface SwiftModuleIndex {
@ -40,16 +30,17 @@ interface SwiftModuleIndex {
readonly byModule: Map<string, string[]>;
}
interface SwiftDeclaredFileIndex {
readonly declared: ReadonlyMap<string, string>;
readonly byName: ReadonlyMap<string, string[]>;
}
const getSwiftModuleIndex = perFileSet((allFilePaths: ReadonlySet<string>): SwiftModuleIndex => {
const byModule = new Map<string, string[]>();
for (const raw of allFilePaths) {
const norm = raw.replace(/\\/g, '/');
if (!norm.endsWith('.swift')) continue;
// Each interior directory segment is a candidate module name. A file
// `Sources/Models/User.swift` is attributed to module `Sources` and
// module `Models`; an `import Models` then resolves to it.
const segments = norm.split('/');
// Drop the filename (last segment); the rest are directory segments.
for (let i = 0; i < segments.length - 1; i++) {
const seg = segments[i];
if (seg === '') continue;
@ -65,12 +56,70 @@ const getSwiftModuleIndex = perFileSet((allFilePaths: ReadonlySet<string>): Swif
return { byModule };
});
export function resolveSwiftImportTarget(
parsedImport: ParsedImport,
workspaceIndex: WorkspaceIndex,
): string | readonly string[] | null {
const SWIFT_DECLARED_INDEX = new WeakMap<ReadonlySet<string>, SwiftDeclaredFileIndex>();
function getDeclaredFilesByName(
allFilePaths: ReadonlySet<string>,
declared: ReadonlyMap<string, string>,
): ReadonlyMap<string, string[]> {
const hit = SWIFT_DECLARED_INDEX.get(allFilePaths);
if (hit !== undefined && hit.declared === declared) return hit.byName;
const dirs = [...declared.entries()].map(([name, dir]) => ({
name,
prefix: swiftDeclaredTargetPrefix(dir),
}));
const byName = new Map<string, string[]>();
for (const { name } of dirs) byName.set(name, []);
for (const raw of allFilePaths) {
const norm = raw.replace(/\\/g, '/');
if (!norm.endsWith('.swift')) continue;
for (const { name, prefix } of dirs) {
if (!norm.startsWith(prefix) && !norm.includes(`/${prefix}`)) continue;
const bucket = byName.get(name);
if (bucket !== undefined) bucket.push(raw);
}
}
const index = { declared, byName };
SWIFT_DECLARED_INDEX.set(allFilePaths, index);
return byName;
}
const getSwiftReexportFlag = perFileSet(
(parsedFiles: readonly ParsedFile[]): { hasReexport: boolean } => {
for (const parsed of parsedFiles) {
for (const imp of parsed.parsedImports) {
if (imp.kind === 'reexport') return { hasReexport: true };
}
}
return { hasReexport: false };
},
);
const getSwiftParsedByPath = perFileSet(
(parsedFiles: readonly ParsedFile[]): ReadonlyMap<string, ParsedFile> => {
const byPath = new Map<string, ParsedFile>();
for (const parsed of parsedFiles) {
byPath.set(parsed.filePath, parsed);
}
return byPath;
},
);
function excludeImporter(files: readonly string[], fromFile: string): string[] {
return files.filter((f) => f !== fromFile);
}
function firstSwiftModuleSegment(targetRaw: string): string | null {
if (targetRaw === '') return null;
const moduleName = targetRaw.split('.')[0];
return moduleName === '' ? null : moduleName;
}
function narrowContext(workspaceIndex: WorkspaceIndex): SwiftResolveContext | null {
const ctx = workspaceIndex as SwiftResolveContext | undefined;
// Duck-type the set (PR #1918 P2: don't `instanceof Set`).
const allFilePaths = (ctx as { allFilePaths?: unknown } | undefined)?.allFilePaths;
if (
ctx === undefined ||
@ -80,18 +129,84 @@ export function resolveSwiftImportTarget(
) {
return null;
}
return ctx;
}
// Swift import target is the SPM module name (first dotted segment).
const targetRaw = parsedImport.targetRaw;
if (targetRaw === null || targetRaw === '') return null;
const moduleName = targetRaw.split('.')[0];
/** Module files only — no @_exported closure. Null means external / unknown. */
function resolveSwiftModuleFiles(moduleName: string, ctx: SwiftResolveContext): string[] | null {
if (moduleName === '') return null;
const declared = coerceDeclaredSwiftTargets(ctx.resolutionConfig);
if (declared !== null) {
if (!declared.has(moduleName)) return null;
const files = getDeclaredFilesByName(ctx.allFilePaths, declared).get(moduleName);
if (files === undefined) return null;
const out = excludeImporter(files, ctx.fromFile);
return out.length > 0 ? out : null;
}
if (isSwiftSdkModule(moduleName)) return null;
const index = getSwiftModuleIndex(ctx.allFilePaths);
const files = index.byModule.get(moduleName);
if (files === undefined || files.length === 0) return null; // external framework
// Exclude the importer itself (a file under `Foo/` importing `Foo`).
const out = files.filter((f) => f !== ctx.fromFile);
if (files === undefined || files.length === 0) return null;
const out = excludeImporter(files, ctx.fromFile);
return out.length > 0 ? out : null;
}
function expandSwiftReexportFiles(seed: readonly string[], ctx: SwiftResolveContext): string[] {
const parsedFiles = ctx.parsedFiles;
if (parsedFiles === undefined || parsedFiles.length === 0) return [...seed];
if (!getSwiftReexportFlag(parsedFiles).hasReexport) return [...seed];
const byPath = getSwiftParsedByPath(parsedFiles);
const seenModules = new Set<string>();
const out = new Set(seed);
const queue = [...seed];
let head = 0;
while (head < queue.length) {
const file = queue[head++];
const parsed = byPath.get(file);
if (parsed === undefined) continue;
for (const imp of parsed.parsedImports) {
if (imp.kind !== 'reexport') continue;
const targetRaw = imp.targetRaw;
if (targetRaw === null) continue;
const moduleName = firstSwiftModuleSegment(targetRaw);
if (moduleName === null || seenModules.has(moduleName)) continue;
// `@_exported import struct Models.User` re-exports User, not Models.
// File-level resolve cannot attribute a member to a file, so skip
// the whole-module enqueue rather than painting every Models file.
if (imp.importedName !== moduleName) continue;
seenModules.add(moduleName);
const more = resolveSwiftModuleFiles(moduleName, ctx);
if (more === null) continue;
for (const next of more) {
if (out.has(next)) continue;
out.add(next);
queue.push(next);
}
}
}
return [...out];
}
export function resolveSwiftImportTarget(
parsedImport: ParsedImport,
workspaceIndex: WorkspaceIndex,
): string | readonly string[] | null {
const ctx = narrowContext(workspaceIndex);
if (ctx === null) return null;
const targetRaw = parsedImport.targetRaw;
if (targetRaw === null) return null;
const moduleName = firstSwiftModuleSegment(targetRaw);
if (moduleName === null) return null;
const files = resolveSwiftModuleFiles(moduleName, ctx);
if (files === null) return null;
const expanded = expandSwiftReexportFiles(files, ctx);
return expanded.length > 0 ? expanded : null;
}

View file

@ -27,7 +27,12 @@
export { emitSwiftScopeCaptures } from './captures.js';
export { getSwiftCaptureCacheStats, resetSwiftCaptureCacheStats } from './cache-stats.js';
export { interpretSwiftImport, interpretSwiftTypeBinding } from './interpret.js';
export {
interpretSwiftImport,
interpretSwiftTypeBinding,
normalizeSwiftTypeName,
stripSwiftTypePreservingDecoration,
} from './interpret.js';
export { swiftMergeBindings } from './merge-bindings.js';
export { swiftArityCompatibility } from './arity.js';
export { resolveSwiftImportTarget, type SwiftResolveContext } from './import-target.js';

View file

@ -19,17 +19,23 @@ export function interpretSwiftImport(captures: CaptureMatch): ParsedImport | nul
const sourceCap = captures['@import.source'];
if (sourceCap === undefined) return null;
// Swift imports are whole-module (wildcard semantics): `import Foundation`
// brings the entire module into scope, no named members. The SPM target
// (first dotted segment) is the resolution target; the full path is kept
// as importedName for reference. `@testable` resolves identically to a
// plain import (same module is visible in test scope).
const source = sourceCap.text;
const fullPath = captures['@import.name']?.text ?? source;
const name = captures['@import.name']?.text ?? source;
const kindText = captures['@import.kind']?.text;
if (kindText === 'reexport' || kindText === 'named') {
return {
kind: kindText,
localName: name,
importedName: name,
targetRaw: source,
};
}
return {
kind: 'namespace',
localName: source,
importedName: fullPath,
importedName: name,
targetRaw: source,
};
}
@ -46,7 +52,7 @@ export function interpretSwiftTypeBinding(captures: CaptureMatch): ParsedTypeBin
// `[User]` → User (array sugar)
// `Array<User>` / `Optional<User>` → User (single-arg generic)
// `Foundation.URL` → URL (qualifier)
const rawType = stripQualifier(stripGeneric(stripArraySugar(stripOptional(typeCap.text.trim()))));
const rawType = normalizeSwiftTypeName(typeCap.text);
let source: TypeRef['source'] = 'parameter-annotation';
if (captures['@type-binding.self'] !== undefined) source = 'self';
@ -58,6 +64,22 @@ export function interpretSwiftTypeBinding(captures: CaptureMatch): ParsedTypeBin
return { boundName: nameCap.text, rawTypeName: rawType, source };
}
export function normalizeSwiftTypeName(text: string): string {
return stripQualifier(stripGeneric(stripArraySugar(stripOptional(text.trim()))));
}
/**
* Type-preserving decoration only — used by `stripTypePreservingDecoration`
* for class lookup and declared-return replay. `User?` / `User!` → `User`.
* Arrays, generics, and nested `Foo.Bar` are left intact so a binding used
* for member lookup does not follow the element type or the trailing ident.
*/
export function stripSwiftTypePreservingDecoration(typeName: string): string | undefined {
const trimmed = typeName.trim();
if (trimmed.endsWith('?') || trimmed.endsWith('!')) return trimmed.slice(0, -1).trim();
return undefined;
}
/** `User?` / `User!` → `User`. */
function stripOptional(text: string): string {
if (text.endsWith('?') || text.endsWith('!')) return text.slice(0, -1).trim();

View file

@ -188,6 +188,32 @@ const SWIFT_SCOPE_QUERY = `
(call_expression
(simple_identifier) @reference.name) @reference.call.free
;; Exact identity for \`let lhs = callee()\`. The call-expression anchor is
;; byte-identical to @reference.call.free / member for downstream position join.
(property_declaration
name: (pattern
bound_identifier: (simple_identifier) @call-result-assignment.lhs)
value: (call_expression) @call-result-assignment.call)
(property_declaration
name: (pattern
bound_identifier: (simple_identifier) @call-result-assignment.lhs)
value: (await_expression
(call_expression) @call-result-assignment.call))
(property_declaration
name: (pattern
bound_identifier: (simple_identifier) @call-result-assignment.lhs)
value: (try_expression
(call_expression) @call-result-assignment.call))
(property_declaration
name: (pattern
bound_identifier: (simple_identifier) @call-result-assignment.lhs)
value: (try_expression
(await_expression
(call_expression) @call-result-assignment.call)))
;; ── References — member / method calls: \`obj.method(...)\` ───────────
;; navigation_expression carries the receiver (target:) and the member
;; (suffix > navigation_suffix > simple_identifier). \`self\` is a

View file

@ -14,7 +14,8 @@
* re-keys an `extension Foo { … }` to a `class_declaration`-style def
* named `Foo`, so its members land on `Foo`'s scope and the shared
* `populateClassOwnedMembers` stamps them with `Foo`'s ownerId — the
* same mechanism C# uses for `partial class`. No separate hoist pass.
* same mechanism C# uses for `partial class`. Cross-file extensions
* that mint no type def are reconciled by `populateWorkspaceOwners`.
* - **Labeled arguments** narrow by ARITY only (count-primary, labels
* soft) — see `arity.ts`. Label-precise dispatch is deferred to the
* type-binding layer.
@ -36,17 +37,18 @@
* Array where Element: Equatable`) are not narrowed — the `Self`
* type of a protocol method resolves to the protocol, not the
* conforming type.
* 2. **Cross-module `import` resolution** is still directory-segment
* based (`import Foo` → files under a `Foo/` dir); explicit imports do
* not yet consult the SPM target map (follow-up, tracked under #1935).
* Same-target visibility (the common case) IS SPM-target-subtree
* accurate — handled by sibling augmentation grouped via
* `groupSwiftFilesBySpmTarget`, not by explicit imports.
* 2. **Cross-module `import` resolution** uses a Package.swift
* declaration map when one is present, otherwise the directory-segment
* index minus well-known SDK module names (#2964). Same-target
* visibility (the common case) is SPM-target-subtree grouping via
* `groupSwiftFilesBySpmTarget`, not explicit imports.
* 3. **Operator / subscript overloads** dispatch by name only.
* 4. **`@_exported import` re-exports** are treated as plain imports.
* 4. **`@_exported import`** is `ParsedImport` `kind: 'reexport'` and
* in-repo modules are closed transitively at resolve time. `public
* import` is not a re-export.
*/
import type { ParsedFile } from 'gitnexus-shared';
import type { ParsedFile, SymbolDefinition } from 'gitnexus-shared';
import { SupportedLanguages } from 'gitnexus-shared';
import { loadSwiftPackageConfig } from '../../language-config.js';
import { buildMro, defaultLinearize } from '../../scope-resolution/passes/mro.js';
@ -66,6 +68,8 @@ import {
mirrorSwiftSiblingTypeBindings,
type SwiftResolveContext,
} from './index.js';
import { stripSwiftTypePreservingDecoration } from './interpret.js';
import { coerceSwiftTargets, groupSwiftFilesBySpmTarget } from './target-grouping.js';
import { swiftIsGlobalNameFallbackPlausible } from './name-fallback-visibility.js';
const ZERO_RANGE = { startLine: 0, startCol: 0, endLine: 0, endCol: 0 } as const;
@ -84,12 +88,18 @@ const swiftScopeResolver: ScopeResolver = {
// `goScopeResolver`'s `loadGoModulePath`.
loadResolutionConfig: (repoPath: string) => loadSwiftPackageConfig(repoPath),
resolveImportTarget: (targetRaw, fromFile, allFilePaths) => {
const ws: SwiftResolveContext = { fromFile, allFilePaths };
resolveImportTarget: (targetRaw, fromFile, allFilePaths, resolutionConfig, context) => {
const ws: SwiftResolveContext = {
fromFile,
allFilePaths,
resolutionConfig,
parsedFiles: context?.parsedFiles,
};
return resolveSwiftImportTarget(
interpretSwiftImport({
'@import.source': { name: '@import.source', text: targetRaw, range: ZERO_RANGE },
}) ?? { kind: 'namespace', localName: targetRaw, importedName: targetRaw, targetRaw },
context?.parsedImport ??
interpretSwiftImport({
'@import.source': { name: '@import.source', text: targetRaw, range: ZERO_RANGE },
}) ?? { kind: 'namespace', localName: targetRaw, importedName: targetRaw, targetRaw },
ws,
);
},
@ -101,6 +111,8 @@ const swiftScopeResolver: ScopeResolver = {
arityCompatibility: (callsite, def) => swiftArityCompatibility(def, callsite),
buildMro: (graph, parsedFiles, nodeLookup) => buildSwiftMro(graph, parsedFiles, nodeLookup),
implicitThisWalksMro: true,
stripTypePreservingDecoration: stripSwiftTypePreservingDecoration,
// Methods/properties/init are owned by their enclosing class/struct/
// extension(→extended type)/protocol. Extension members hoist for free
@ -108,6 +120,13 @@ const swiftScopeResolver: ScopeResolver = {
// the extended type.
populateOwners: (parsed: ParsedFile) => populateClassOwnedMembers(parsed),
// An extension has its own Class scope but deliberately does not mint a
// second type def. Its members therefore leave the per-file owner walk with
// a qualified name (`ExtendedType.member`) but no ownerId. Reconcile those
// members after all files are available so extensions declared in sibling
// files work as well as extensions beside the original type.
populateWorkspaceOwners: populateSwiftExtensionOwners,
// `super.method()` dispatches through the superclass chain.
isSuperReceiver: (text) => text.trim() === 'super',
@ -209,6 +228,91 @@ function buildSwiftMro(
return mro;
}
function populateSwiftExtensionOwners(
parsedFiles: readonly ParsedFile[],
ctx?: { readonly fileContents: ReadonlyMap<string, string>; readonly resolutionConfig?: unknown },
): void {
const filesByTarget = groupSwiftFilesBySpmTarget(
parsedFiles,
(parsed) => parsed.filePath,
coerceSwiftTargets(ctx?.resolutionConfig),
);
for (const files of filesByTarget.values()) {
stampSwiftExtensionOwnersInTarget(files);
}
}
function stampSwiftExtensionOwnersInTarget(parsedFiles: readonly ParsedFile[]): void {
const ownersByName = new Map<string, SymbolDefinition[]>();
for (const parsed of parsedFiles) {
for (const def of parsed.localDefs) {
if (!isClassLike(def.type) || def.qualifiedName === undefined) continue;
const bucket = ownersByName.get(def.qualifiedName);
if (bucket === undefined) ownersByName.set(def.qualifiedName, [def]);
else bucket.push(def);
}
}
// Extension members are Function-owned defs whose parent Class minted no
// type def. Nested locals are Function-owned defs whose parent is another
// Function — leave those ownerless so they cannot enter implicit-self CALLS.
for (const parsed of parsedFiles) {
const byId = new Map(parsed.scopes.map((scope) => [scope.id, scope]));
for (const scope of parsed.scopes) {
if (scope.kind !== 'Function') continue;
const parent = scope.parent === null ? undefined : byId.get(scope.parent);
if (parent?.kind !== 'Class') continue;
// A type-decl Class owns the type that opened it (same start line).
// Nested types inside an `extension` are also class-like and live on
// that Class scope — they are not the extended type, so they must
// not suppress stamping `func added` onto `Foo`.
if (parent.ownedDefs.some((d) => isClassLike(d.type) && defDeclaresThisClassScope(parent, d)))
continue;
for (const def of scope.ownedDefs) {
if (def.ownerId !== undefined || def.qualifiedName === undefined) continue;
const dot = def.qualifiedName.lastIndexOf('.');
if (dot <= 0) continue;
const owner = uniqueOwnerForExtensionPrefix(ownersByName, def.qualifiedName.slice(0, dot));
if (owner !== undefined) {
(def as { ownerId?: string }).ownerId = owner.nodeId;
}
}
}
}
}
function defDeclaresThisClassScope(
parent: { readonly range: { readonly startLine: number } },
def: SymbolDefinition,
): boolean {
const line = /#(\d+):/.exec(def.nodeId);
if (line === null) return true;
return Number(line[1]) === parent.range.startLine;
}
function uniqueOwnerForExtensionPrefix(
ownersByName: ReadonlyMap<string, readonly SymbolDefinition[]>,
prefix: string,
): SymbolDefinition | undefined {
const seen = new Set<string>();
const matches: SymbolDefinition[] = [];
const consider = (defs: readonly SymbolDefinition[] | undefined): void => {
if (defs === undefined) return;
for (const def of defs) {
if (seen.has(def.nodeId)) continue;
seen.add(def.nodeId);
matches.push(def);
}
};
consider(ownersByName.get(prefix));
if (!prefix.includes('.')) {
for (const [qn, defs] of ownersByName) {
if (qn !== prefix && qn.endsWith('.' + prefix)) consider(defs);
}
}
return matches.length === 1 ? matches[0] : undefined;
}
function closeProtocols(
seeds: readonly string[],
directImpls: ReadonlyMap<string, readonly string[]>,

View file

@ -0,0 +1,29 @@
/**
* Well-known Apple / Swift SDK module names that must not bind to a
* same-named in-repo folder on the no-manifest path (#2964).
*
* Modeled on `CSHARP_EXTERNAL_ROOTS`. A Package.swift target that uses
* one of these names still wins — this set is only the inferred/null path.
*/
export const SWIFT_SDK_MODULES: ReadonlySet<string> = new Set([
'Foundation',
'UIKit',
'SwiftUI',
'AppKit',
'Combine',
'Dispatch',
'Darwin',
'ObjectiveC',
'Swift',
'CoreFoundation',
'CoreData',
'CoreGraphics',
'XCTest',
'Testing',
'Observation',
'os',
]);
export function isSwiftSdkModule(name: string): boolean {
return SWIFT_SDK_MODULES.has(name);
}

View file

@ -9,36 +9,31 @@
* drops cross-directory same-module edges and can mis-resolve a
* constructor call to a wrong same-simple-named type in another target.
*
* This module duplicates the legacy `groupSwiftFilesByTarget`
* (`languages/swift.ts`) semantics **verbatim** so the registry-primary
* path matches legacy SPM-subtree grouping without touching the legacy
* pipeline (hard constraint: legacy stays byte-identical). The SPM target
* map is threaded in via the `resolutionConfig` channel
* The SPM target map is threaded in via the `resolutionConfig` channel
* (`loadSwiftPackageConfig` → `resolutionConfig` → these hooks); see
* `scope-resolver.ts` and `scope-resolution/pipeline/run.ts`.
*
* NOTE: This intentionally differs from the import-config module's
* leading-`startsWith` (`import-resolvers/configs/swift.ts`): that module
* fans a file out to EVERY matching target (a nested file can belong to
* multiple configured target dirs there), whereas legacy module grouping
* assigns each file to the FIRST matching target only (legacy `break`s) —
* one bucket per file. Do not copy the import-config behavior here.
* Path matching is the same segment-boundary rule as import-config
* (`fileMatchesSwiftTargetDir`). Grouping assigns each file to the FIRST
* matching target; import-config fans a file out to every matching target.
*/
import type { SwiftPackageConfig } from '../../language-config.js';
import { swiftDeclaredTargetPrefix, type SwiftPackageConfig } from '../../language-config.js';
export { coerceDeclaredSwiftTargets } from '../../language-config.js';
const DEFAULT_TARGET = '__default__';
/**
* Group `items` by SPM target subtree, replicating legacy
* `groupSwiftFilesByTarget` semantics exactly:
* Group `items` by SPM target subtree:
*
* - `targets` null/empty (no scanned source dir found) → ALL items go to
* a single `__default__` bucket (single-Xcode-project assumption).
* - Otherwise: a file matches a target when its normalized path either
* starts with `<targetDir>/` (`indexOf === 0`) OR contains it at a `/`
* boundary (`norm[idx - 1] === '/'`). Each file is assigned to the
* FIRST matching target only (one bucket per file, no fan-out).
* starts with `<targetDir>/` or contains `/<targetDir>/` at a segment
* boundary. Using a segment-aware suffix search matters when an earlier,
* non-boundary occurrence of the same text appears in the path (#2931).
* - Each file is assigned to the FIRST matching target only (one bucket per
* file, no fan-out).
* - Files matching no target fall into the `__default__` bucket.
*
* `targets` is `name → directory` (the `SwiftPackageConfig.targets` map).
@ -53,10 +48,9 @@ export function groupSwiftFilesBySpmTarget<T>(
return new Map([[DEFAULT_TARGET, [...items]]]);
}
// Pre-convert target dirs to normalized prefix format once.
const targetPrefixes = [...targets.entries()].map(([name, dir]) => ({
name,
prefix: dir.replace(/\\/g, '/') + '/',
prefix: swiftDeclaredTargetPrefix(dir),
}));
const groups = new Map<string, T[]>();
@ -67,8 +61,7 @@ export function groupSwiftFilesBySpmTarget<T>(
const normalized = rawPath.includes('\\') ? rawPath.replace(/\\/g, '/') : rawPath;
let assigned = false;
for (const { name, prefix } of targetPrefixes) {
const idx = normalized.indexOf(prefix);
if (idx === 0 || (idx > 0 && normalized[idx - 1] === '/')) {
if (pathMatchesTargetPrefix(normalized, prefix)) {
let group = groups.get(name);
if (group === undefined) {
group = [];
@ -102,3 +95,13 @@ export function coerceSwiftTargets(resolutionConfig: unknown): ReadonlyMap<strin
}
return null;
}
function pathMatchesTargetPrefix(normalizedPath: string, prefix: string): boolean {
if (prefix === '') return true;
return normalizedPath.startsWith(prefix) || normalizedPath.includes(`/${prefix}`);
}
/** Segment-boundary membership used by grouping and declared import resolve. */
export function fileMatchesSwiftTargetDir(normalizedPath: string, targetDir: string): boolean {
return pathMatchesTargetPrefix(normalizedPath, swiftDeclaredTargetPrefix(targetDir));
}

View file

@ -10,10 +10,13 @@
* Module identity: Swift has no in-source `package X` marker. The SPM
* target is a directory subtree (`Sources/<Target>/…`). Module membership
* is threaded in via the SPM target map (`ctx.resolutionConfig` →
* `coerceSwiftTargets`) and grouped by `groupSwiftFilesBySpmTarget`,
* replicating legacy `wireSwiftImplicitImports`'s `groupSwiftFilesByTarget`:
* files are grouped by SPM target subtree when a package config is present,
* else ALL Swift files form one module (`__default__`,
* `coerceSwiftTargets`) and grouped by `groupSwiftFilesBySpmTarget`.
* The helper preserves the legacy `wireSwiftImplicitImports` bucketing
* contract for ordinary layouts (first target wins and unmatched files use
* `__default__`) but intentionally fixes #2931's repeated-prefix edge case.
* When a target map is present (declared Package.swift or inferred
* `Sources/*` folders), files are grouped by SPM target subtree;
* otherwise ALL Swift files form one module (`__default__`,
* single-Xcode-project assumption). Every `.swift` file in the same target
* sees its siblings' top-level defs.
*
@ -23,8 +26,9 @@
* finalize and must not be mutated.
*/
import type { BindingRef, ParsedFile, ScopeId, SymbolDefinition } from 'gitnexus-shared';
import type { BindingRef, ParsedFile, Scope, ScopeId, SymbolDefinition } from 'gitnexus-shared';
import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js';
import { isClassLike } from '../../scope-resolution/scope/walkers.js';
import { coerceSwiftTargets, groupSwiftFilesBySpmTarget } from './target-grouping.js';
export function populateSwiftTargetSiblings(
@ -35,8 +39,6 @@ export function populateSwiftTargetSiblings(
readonly resolutionConfig?: unknown;
},
): void {
// Group files by SPM target subtree (the module). No-source-dir → all
// files in one `__default__` bucket.
const targets = coerceSwiftTargets(ctx.resolutionConfig);
const filesByTarget = groupSwiftFilesBySpmTarget(
parsedFiles,
@ -47,7 +49,8 @@ export function populateSwiftTargetSiblings(
const augmentations = indexes.bindingAugmentations as Map<ScopeId, Map<string, BindingRef[]>>;
for (const [, group] of filesByTarget) {
if (group.length < 2) continue; // no siblings to share
populateNestedTypeFragments(group, indexes, augmentations, ctx.fileContents);
if (group.length < 2) continue; // no file siblings to share
const siblings = group.map((parsed) => ({
filePath: parsed.filePath,
defs: [...parsed.localDefs] as SymbolDefinition[],
@ -59,17 +62,328 @@ export function populateSwiftTargetSiblings(
if (receiverModule === undefined) continue;
for (const def of target.defs) {
const name = def.qualifiedName?.split('.').pop() ?? def.qualifiedName ?? '';
if (name === '') continue;
const bucket = getAugmentationBucket(augmentations, receiverModule, name);
if (bucket.some((b) => b.def.nodeId === def.nodeId)) continue;
bucket.push({ def, origin: 'namespace' });
addNamespaceBinding(augmentations, receiverModule, def);
}
}
}
}
}
/**
* A Swift extension is a second lexical fragment of its extended type. Make
* nested types declared by the primary fragment visible from every same-target
* fragment with the same logical owner. Keeping this on class scopes preserves
* lexical precedence when an unrelated top-level type has the same simple name.
*/
function populateNestedTypeFragments(
group: readonly ParsedFile[],
indexes: ScopeResolutionIndexes,
augmentations: Map<ScopeId, Map<string, BindingRef[]>>,
fileContents: ReadonlyMap<string, string>,
): void {
const scopesByOwner = new Map<string, ScopeId[]>();
const lineStartsByFile = new Map<string, readonly number[]>();
for (const parsed of group) {
const source = fileContents.get(parsed.filePath);
let lineStarts: readonly number[] | undefined;
if (source !== undefined) {
lineStarts = lineStartsByFile.get(parsed.filePath);
if (lineStarts === undefined) {
lineStarts = lineStartsOf(source);
lineStartsByFile.set(parsed.filePath, lineStarts);
}
}
for (const scope of parsed.scopes) {
if (scope.kind !== 'Class') continue;
const key = scopeOwnerKey(scope, source, lineStarts);
if (key === undefined) continue;
let scopes = scopesByOwner.get(key);
if (scopes === undefined) {
scopes = [];
scopesByOwner.set(key, scopes);
}
scopes.push(scope.id);
}
}
for (const parsed of group) {
for (const def of parsed.localDefs) {
if (!isClassLike(def.type) || def.ownerId === undefined) continue;
const owner = indexes.defs.byId.get(def.ownerId);
if (owner === undefined) continue;
const targetScopes = scopesByOwner.get(logicalOwnerKey(owner));
if (targetScopes === undefined) continue;
for (const scopeId of targetScopes) {
addNamespaceBinding(augmentations, scopeId, def);
}
}
}
}
function scopeOwnerKey(
scope: Scope,
source: string | undefined,
lineStarts?: readonly number[],
): string | undefined {
const owner = scope.ownedDefs.find((def) => isClassLike(def.type));
if (owner !== undefined) return logicalOwnerKey(owner);
// Extension scopes carry no synthetic class def. Capture generation keeps
// only the trailing owner on members (`Inner.f` for `extension Outer.Inner`),
// so recover the full owner from this scope's declaration text first.
if (source !== undefined) {
const sourceOwner = swiftExtensionOwner(source, scope, lineStarts);
// Do not last-dot-guess: member qualified names are trailing-only, so
// `Inner.make` would key `Inner` instead of `Outer.Inner`.
if (sourceOwner === undefined) return undefined;
const representative = firstBoundDefinition(scope);
return logicalOwnerKey({
...(representative ?? {
nodeId: sourceOwner,
filePath: scope.filePath,
type: 'Class',
qualifiedName: sourceOwner,
}),
qualifiedName: sourceOwner,
});
}
// Hand-built fixtures and old cached shapes may have no source text. Keep
// the conservative member-prefix fallback, rejecting inconsistent owners.
let inferredOwner: string | undefined;
for (const refs of scope.bindings.values()) {
for (const { def } of refs) {
const qualifiedName = def.qualifiedName;
if (qualifiedName === undefined) continue;
const separator = qualifiedName.lastIndexOf('.');
if (separator <= 0) continue;
const candidate = logicalOwnerKey({
...def,
qualifiedName: qualifiedName.slice(0, separator),
});
if (inferredOwner !== undefined && inferredOwner !== candidate) return undefined;
inferredOwner = candidate;
}
}
return inferredOwner;
}
function firstBoundDefinition(scope: Scope): SymbolDefinition | undefined {
for (const refs of scope.bindings.values()) {
const first = refs[0]?.def;
if (first !== undefined) return first;
}
return undefined;
}
/**
* Read `extension Outer.Inner` from the class-scope source range.
* Access modifiers and attributes (`public`, `@MainActor`, `@available`)
* may precede the keyword, so the match is not start-anchored. Attribute
* message strings and comments must not supply a false `extension Type`.
* Owner segments use Unicode identifier characters so `Café.Container`
* is not truncated to `Caf`.
*/
const SWIFT_TYPE_IDENT = String.raw`[\p{ID_Start}_][\p{ID_Continue}]*`;
const EXTENSION_OWNER = new RegExp(
String.raw`\bextension\s+(${SWIFT_TYPE_IDENT}(?:\s*\.\s*${SWIFT_TYPE_IDENT})*)`,
'u',
);
function swiftExtensionOwner(
source: string,
scope: Scope,
lineStarts?: readonly number[],
): string | undefined {
const declaration = sliceScopeRange(source, scope.range, lineStarts ?? lineStartsOf(source));
if (declaration === undefined) return undefined;
const cleaned = cleanExtensionHeader(declaration);
return EXTENSION_OWNER.exec(cleaned)?.[1]?.replace(/\s+/g, '');
}
/** `Scope.range` is 1-based on lines. Columns are Tree-sitter UTF-8 bytes. */
function lineStartsOf(source: string): number[] {
const starts = [0, 0];
for (let index = 0; index < source.length; index += 1) {
if (source[index] === '\n') starts.push(index + 1);
}
return starts;
}
function sliceScopeRange(
source: string,
range: Scope['range'],
starts: readonly number[],
): string | undefined {
const start = starts[range.startLine];
const end = starts[range.endLine];
if (start === undefined || end === undefined) return undefined;
return source.slice(
start + jsOffsetForUtf8Column(source, start, range.startCol),
end + jsOffsetForUtf8Column(source, end, range.endCol),
);
}
/** Convert a Tree-sitter UTF-8 column into a JS string offset on that line. */
function jsOffsetForUtf8Column(source: string, lineStart: number, utf8Column: number): number {
let bytes = 0;
let index = lineStart;
while (index < source.length && bytes < utf8Column) {
if (source[index] === '\n') break;
const codePoint = source.codePointAt(index);
if (codePoint === undefined) break;
bytes += utf8ByteLength(codePoint);
index += codePoint > 0xffff ? 2 : 1;
}
return index - lineStart;
}
function utf8ByteLength(codePoint: number): number {
if (codePoint <= 0x7f) return 1;
if (codePoint <= 0x7ff) return 2;
if (codePoint <= 0xffff) return 3;
return 4;
}
/** Header through the first unquoted `{`, with strings and comments blanked. */
function cleanExtensionHeader(declaration: string): string {
let index = 0;
let cleaned = '';
const blank = (from: number, to: number): void => {
for (let cursor = from; cursor < to; cursor += 1) {
cleaned += declaration[cursor] === '\n' ? '\n' : ' ';
}
};
while (index < declaration.length) {
const current = declaration[index];
if (current === '/' && declaration[index + 1] === '/') {
const end = skipLineComment(declaration, index);
blank(index, end);
index = end;
continue;
}
if (current === '/' && declaration[index + 1] === '*') {
const end = skipBlockComment(declaration, index);
blank(index, end);
index = end;
continue;
}
const pounds = current === '#' ? leadingPounds(declaration, index) : 0;
const quoteAt = index + pounds;
if (declaration.startsWith('"""', quoteAt) || declaration[quoteAt] === '"') {
const end = skipSwiftString(declaration, index);
blank(index, end);
index = end;
continue;
}
if (current === "'") {
const end = skipQuoted(declaration, index, current);
blank(index, end);
index = end;
continue;
}
if (current === '{') break;
cleaned += current;
index += 1;
}
return cleaned;
}
function skipLineComment(text: string, start: number): number {
const newline = text.indexOf('\n', start);
return newline === -1 ? text.length : newline + 1;
}
function skipBlockComment(text: string, start: number): number {
let index = start + 2;
let depth = 1;
while (index < text.length && depth > 0) {
if (text.startsWith('/*', index)) {
depth += 1;
index += 2;
continue;
}
if (text.startsWith('*/', index)) {
depth -= 1;
index += 2;
continue;
}
index += 1;
}
return index;
}
function leadingPounds(text: string, start: number): number {
let count = 0;
while (text[start + count] === '#') count += 1;
return count;
}
function skipSwiftString(text: string, start: number): number {
const pounds = leadingPounds(text, start);
const quoteAt = start + pounds;
if (text.startsWith('"""', quoteAt)) {
return skipDelimitedString(text, quoteAt + 3, `"""${'#'.repeat(pounds)}`, pounds === 0);
}
if (text[quoteAt] === '"') {
return skipDelimitedString(text, quoteAt + 1, `"${'#'.repeat(pounds)}`, pounds === 0);
}
return start + Math.max(pounds, 1);
}
function skipDelimitedString(
text: string,
bodyStart: number,
closer: string,
escapes: boolean,
): number {
let index = bodyStart;
while (index < text.length) {
if (escapes && text[index] === '\\') {
index += 2;
continue;
}
if (text.startsWith(closer, index)) return index + closer.length;
index += 1;
}
return text.length;
}
function skipQuoted(text: string, start: number, quote: string): number {
let index = start + 1;
while (index < text.length) {
if (text[index] === '\\') {
index += 2;
continue;
}
if (text[index] === quote) return index + 1;
index += 1;
}
return text.length;
}
function logicalOwnerKey(def: SymbolDefinition): string {
const qualifiedName = def.qualifiedName ?? def.nodeId;
const namespacePrefix = def.namespacePrefix ?? '';
return `${namespacePrefix.length}:${namespacePrefix}:${qualifiedName}`;
}
function simpleName(def: SymbolDefinition): string {
return def.qualifiedName?.split('.').pop() ?? def.qualifiedName ?? '';
}
function addNamespaceBinding(
augmentations: Map<ScopeId, Map<string, BindingRef[]>>,
scopeId: ScopeId,
def: SymbolDefinition,
): void {
const name = simpleName(def);
if (name === '') return;
const bucket = getAugmentationBucket(augmentations, scopeId, name);
if (bucket.some((binding) => binding.def.nodeId === def.nodeId)) return;
bucket.push({ def, origin: 'namespace' });
}
function getAugmentationBucket(
augmentations: Map<ScopeId, Map<string, BindingRef[]>>,
scopeId: ScopeId,

View file

@ -19,6 +19,12 @@ import type { ToolsOutput } from './tools.js';
import type { StructureOutput } from './structure.js';
import type { ParseOutput } from './parse.js';
import { processProcesses, type ProcessDetectionResult } from '../process-processor.js';
import {
buildProcessDetectionPhaseConfig,
formatWholeFlowsMissingRemedies,
processDetectionEffectiveLimits,
resolveProcessDetectionBudget,
} from '../process-detection-budget.js';
import { generateId } from '../../../lib/utils.js';
import { routeNodeKey } from '../route-extractors/route-path.js';
import { isDev } from '../utils/env.js';
@ -79,11 +85,29 @@ export const processesPhase: PipelinePhase<ProcessesOutput> = {
stats: { filesProcessed: totalFiles, totalFiles, nodesCreated: ctx.graph.nodeCount },
});
const resolvedBudget = resolveProcessDetectionBudget(
{
maxProcesses: ctx.options?.maxProcesses,
maxProcessBranching: ctx.options?.maxProcessBranching,
maxProcessTraceDepth: ctx.options?.maxProcessTraceDepth,
maxEntryPointCandidates: ctx.options?.maxEntryPointCandidates,
},
// Env is resolved in `runFullAnalysis` and threaded on PipelineOptions.
// The phase reads only those fields so unit tests stay isolated from
// the host environment.
{},
);
let symbolCount = 0;
ctx.graph.forEachNode((n) => {
if (n.label !== 'File') symbolCount++;
});
const dynamicMaxProcesses = computeDynamicMaxProcesses(symbolCount);
if (resolvedBudget.maxProcesses === undefined) {
ctx.graph.forEachNode((n) => {
if (n.label !== 'File') symbolCount++;
});
}
const detectionConfig = buildProcessDetectionPhaseConfig(
resolvedBudget,
symbolCount,
computeDynamicMaxProcesses,
);
// R3-6: where the program reaches outward. Already collected by the parse
// phase for FILE-level FETCHES/QUERIES edges; reused here at function
@ -123,7 +147,7 @@ export const processesPhase: PipelinePhase<ProcessesOutput> = {
stats: { filesProcessed: totalFiles, totalFiles, nodesCreated: ctx.graph.nodeCount },
});
},
{ maxProcesses: dynamicMaxProcesses, minSteps: 3 },
detectionConfig,
outwardActionSites,
);
@ -143,8 +167,8 @@ export const processesPhase: PipelinePhase<ProcessesOutput> = {
// "unexplored entry points mean whole flows are missing, while a
// depth-capped trace means a flow is present but shorter than it really is"
// — and it is what keeps the line worth reading. Warning on every counter
// meant warning on every run: this phase overrides only `maxProcesses`, so
// at the shipped defaults (`maxBranching: 4`, `maxTraceDepth: 10`,
// meant warning on every run: at the shipped defaults (`maxBranching: 4`,
// `maxTraceDepth: 10`,
// per-entry trace budget 12) `calleesDropped` fires for any function with
// five callees, `tracesDepthCapped` for any chain deeper than ten, and
// `walksCutByBudget` for any entry point with twelve paths under it. All
@ -171,6 +195,15 @@ export const processesPhase: PipelinePhase<ProcessesOutput> = {
truncation.entryPointCandidatesDropped > 0 ||
truncation.entryPointsUnexplored > 0 ||
truncation.processesDropped > 0;
const effectiveLimits = processDetectionEffectiveLimits(
detectionConfig.maxProcesses,
resolvedBudget,
);
const remedies = formatWholeFlowsMissingRemedies(
truncation,
effectiveLimits,
entryPointCandidates,
);
const shape =
`${truncation.entryPointCandidatesDropped} of ${entryPointCandidates} candidate entry point(s) never ranked in, ` +
`${truncation.entryPointsUnexplored} ranked entry point(s) never traced, ` +
@ -180,13 +213,13 @@ export const processesPhase: PipelinePhase<ProcessesOutput> = {
`${truncation.walksCutByBudget} walk(s) cut by the per-entry trace budget.`;
if (flowsMissing) {
logger.warn(
{ truncation },
{ truncation, effectiveLimits },
`[processes] ${processResult.stats.totalProcesses} flows reported, but whole flows are MISSING: ` +
`${shape} An absent flow does NOT mean the code path does not exist.`,
`${shape}${remedies} An absent flow does NOT mean the code path does not exist.`,
);
} else if (truncation.truncated) {
logger.debug(
{ truncation },
{ truncation, effectiveLimits },
`[processes] ${processResult.stats.totalProcesses} flows reported; every flow found is present, ` +
`but some are shorter than the code path they describe: ${shape}`,
);

View file

@ -251,6 +251,15 @@ export interface PipelineOptions {
* analyze invocations.
*/
workerPoolSize?: number;
/**
* Process-detection budget (#3313). Explicit `maxProcesses` replaces the
* dynamic `symbols / 10` formula; the other three replace compiled defaults.
* Unset fields keep shipped behavior. `0` is rejected upstream — not unlimited.
*/
maxProcesses?: number;
maxProcessBranching?: number;
maxProcessTraceDepth?: number;
maxEntryPointCandidates?: number;
/**
* Number of chunks whose file contents may be read into memory in
* parallel while the worker pool is busy dispatching the current
@ -445,8 +454,8 @@ export const runPipelineFromRepo = async (
const propertyInference = scopeResolutionOutput.propertyInference;
// Presence check, not `!skipGraphPhases`: phases can now be filtered out by
// any `enabledWhen` predicate (streamGraphEmit disables communities/processes
// too), and `getPhaseOutput` THROWS on a phase that was never resolved. Keying
// any `enabledWhen` predicate (`skipGraphPhases` drops communities/processes),
// and `getPhaseOutput` THROWS on a phase that was never resolved. Keying
// off the options flag alone made every filtered-out combination crash here
// rather than return undefined results.
if (results.has('communities') && results.has('processes')) {

View file

@ -0,0 +1,304 @@
/**
* Analyze-time process-detection budget (#3313).
*
* Four knobs control how many execution flows `processProcesses` keeps:
* process count, per-node branching, trace depth, and the ranked entry-point
* pool. Precedence is CLI / explicit `AnalyzeOptions` > `.gitnexusrc` (already
* merged into those fields) > `GITNEXUS_*` env > built-in defaults. Unset
* `maxProcesses` keeps the dynamic `symbols / 10` formula. `0` is invalid, not
* unlimited.
*/
import type { ProcessDetectionConfig, ProcessTruncationStats } from './process-processor.js';
import { parsePositiveIntEnv } from './utils/env.js';
export const PROCESS_DETECTION_BUDGET_DEFAULTS = {
maxProcessBranching: 4,
maxProcessTraceDepth: 10,
maxEntryPointCandidates: 200,
minSteps: 3,
} as const;
export const PROCESS_DETECTION_ENV = {
maxProcesses: 'GITNEXUS_MAX_PROCESSES',
maxProcessBranching: 'GITNEXUS_MAX_PROCESS_BRANCHING',
maxProcessTraceDepth: 'GITNEXUS_MAX_PROCESS_TRACE_DEPTH',
maxEntryPointCandidates: 'GITNEXUS_MAX_ENTRY_POINT_CANDIDATES',
} as const;
export const PROCESS_DETECTION_CLI_FLAGS = {
maxProcesses: '--max-processes',
maxProcessBranching: '--max-process-branching',
maxProcessTraceDepth: '--max-process-trace-depth',
maxEntryPointCandidates: '--max-entry-point-candidates',
} as const;
const BUDGET_KEYS = [
'maxProcesses',
'maxProcessBranching',
'maxProcessTraceDepth',
'maxEntryPointCandidates',
] as const;
export type ProcessDetectionBudgetKey = (typeof BUDGET_KEYS)[number];
export type ProcessDetectionBudgetFields = {
maxProcesses?: number;
maxProcessBranching?: number;
maxProcessTraceDepth?: number;
maxEntryPointCandidates?: number;
};
export type ProcessDetectionBudgetStrings = {
maxProcesses?: string;
maxProcessBranching?: string;
maxProcessTraceDepth?: string;
maxEntryPointCandidates?: string;
};
export type ProcessDetectionStamp = {
/** Explicit override, or `null` when this run used the dynamic formula. */
maxProcesses: number | null;
maxProcessBranching: number;
maxProcessTraceDepth: number;
maxEntryPointCandidates: number;
/**
* In-place FTS park after a derived-layer rewrite (#3322). Missing stamp +
* defaults is a match, so recovery must persist a complete stamp that still
* mismatches until a successful analyze certifies the live Community/Process
* rows. Success writes omit this flag.
*/
uncertified?: true;
};
export type ResolvedProcessDetectionBudget = {
/** Set only when an explicit override won. */
maxProcesses?: number;
maxProcessBranching: number;
maxProcessTraceDepth: number;
maxEntryPointCandidates: number;
overridden: {
maxProcesses: boolean;
maxProcessBranching: boolean;
maxProcessTraceDepth: boolean;
maxEntryPointCandidates: boolean;
};
};
export type ProcessDetectionEffectiveLimits = {
maxProcesses: number;
maxProcessBranching: number;
maxProcessTraceDepth: number;
maxEntryPointCandidates: number;
/**
* Pre-entry gate: `processProcesses` does not start the next entry once
* collected traces already reach `maxProcesses * 2`. One started entry can
* still append every trace `traceFromEntryPoint` returns.
*/
maxProcessTraces: number;
};
export type InvalidBudgetHandler = (knob: string, raw: string) => void;
/** Operator copy when a CLI/rc/env token is rejected. Next precedence still applies. */
export const formatInvalidProcessDetectionOverride = (knob: string, raw: string): string => {
const next = knob.startsWith('GITNEXUS_')
? 'the built-in default'
: 'the next source (env, then the built-in default)';
return `${knob} must be a positive integer (got ${JSON.stringify(raw)}); ignoring it so ${next} applies.`;
};
export const parsePositiveIntegerOverride = (
raw: string | number | undefined | null,
onInvalid?: (raw: string) => void,
): number | undefined => {
if (raw === undefined || raw === null) return undefined;
const text = typeof raw === 'number' ? String(raw) : raw;
const parsed = parsePositiveIntEnv(text);
if (parsed === undefined) onInvalid?.(typeof raw === 'number' ? text : text.trim());
return parsed;
};
export const parseProcessDetectionBudgetStrings = (
raw: ProcessDetectionBudgetStrings,
onInvalid?: InvalidBudgetHandler,
): ProcessDetectionBudgetFields => {
const out: ProcessDetectionBudgetFields = {};
for (const key of BUDGET_KEYS) {
if (raw[key] === undefined) continue;
const parsed = parsePositiveIntegerOverride(raw[key], (invalid) =>
onInvalid?.(PROCESS_DETECTION_CLI_FLAGS[key], invalid),
);
if (parsed !== undefined) out[key] = parsed;
}
return out;
};
const readOverride = (
options: ProcessDetectionBudgetFields,
env: NodeJS.ProcessEnv,
key: ProcessDetectionBudgetKey,
onInvalid?: InvalidBudgetHandler,
): number | undefined => {
const fromOptions = options[key];
if (fromOptions !== undefined) {
const parsed = parsePositiveIntegerOverride(fromOptions, (invalid) =>
onInvalid?.(PROCESS_DETECTION_CLI_FLAGS[key], invalid),
);
if (parsed !== undefined) return parsed;
}
const envName = PROCESS_DETECTION_ENV[key];
const fromEnv = env[envName];
if (fromEnv === undefined) return undefined;
return parsePositiveIntegerOverride(fromEnv, (invalid) => onInvalid?.(envName, invalid));
};
export const resolveProcessDetectionBudget = (
options: ProcessDetectionBudgetFields = {},
env: NodeJS.ProcessEnv = process.env,
onInvalid?: InvalidBudgetHandler,
): ResolvedProcessDetectionBudget => {
const maxProcesses = readOverride(options, env, 'maxProcesses', onInvalid);
const branching = readOverride(options, env, 'maxProcessBranching', onInvalid);
const depth = readOverride(options, env, 'maxProcessTraceDepth', onInvalid);
const entryPoints = readOverride(options, env, 'maxEntryPointCandidates', onInvalid);
return {
...(maxProcesses === undefined ? {} : { maxProcesses }),
maxProcessBranching: branching ?? PROCESS_DETECTION_BUDGET_DEFAULTS.maxProcessBranching,
maxProcessTraceDepth: depth ?? PROCESS_DETECTION_BUDGET_DEFAULTS.maxProcessTraceDepth,
maxEntryPointCandidates:
entryPoints ?? PROCESS_DETECTION_BUDGET_DEFAULTS.maxEntryPointCandidates,
overridden: {
maxProcesses: maxProcesses !== undefined,
maxProcessBranching: branching !== undefined,
maxProcessTraceDepth: depth !== undefined,
maxEntryPointCandidates: entryPoints !== undefined,
},
};
};
export const hasProcessDetectionOverride = (resolved: ResolvedProcessDetectionBudget): boolean =>
resolved.overridden.maxProcesses ||
resolved.overridden.maxProcessBranching ||
resolved.overridden.maxProcessTraceDepth ||
resolved.overridden.maxEntryPointCandidates;
export const toProcessDetectionStamp = (
resolved: ResolvedProcessDetectionBudget,
): ProcessDetectionStamp => ({
maxProcesses: resolved.maxProcesses ?? null,
maxProcessBranching: resolved.maxProcessBranching,
maxProcessTraceDepth: resolved.maxProcessTraceDepth,
maxEntryPointCandidates: resolved.maxEntryPointCandidates,
});
/** Complete stamp that always mismatches until the next successful analyze. */
export const uncertifyProcessDetectionStamp = (
recorded: ProcessDetectionStamp | undefined,
): ProcessDetectionStamp => ({
maxProcesses: recorded?.maxProcesses ?? null,
maxProcessBranching:
recorded?.maxProcessBranching ?? PROCESS_DETECTION_BUDGET_DEFAULTS.maxProcessBranching,
maxProcessTraceDepth:
recorded?.maxProcessTraceDepth ?? PROCESS_DETECTION_BUDGET_DEFAULTS.maxProcessTraceDepth,
maxEntryPointCandidates:
recorded?.maxEntryPointCandidates ?? PROCESS_DETECTION_BUDGET_DEFAULTS.maxEntryPointCandidates,
uncertified: true,
});
const isCompleteStamp = (
recorded: ProcessDetectionStamp | undefined,
): recorded is ProcessDetectionStamp =>
recorded !== undefined &&
(recorded.maxProcesses === null ||
(typeof recorded.maxProcesses === 'number' && Number.isInteger(recorded.maxProcesses))) &&
Number.isInteger(recorded.maxProcessBranching) &&
Number.isInteger(recorded.maxProcessTraceDepth) &&
Number.isInteger(recorded.maxEntryPointCandidates);
export const processDetectionBudgetMismatch = (
recorded: ProcessDetectionStamp | undefined,
resolved: ResolvedProcessDetectionBudget,
): boolean => {
if (recorded?.uncertified === true) return true;
if (!isCompleteStamp(recorded)) {
// Legacy meta: same defaults as today's shipped behavior stay a match so
// an upgrade backfills the stamp instead of re-detecting. Any explicit
// override is a mismatch — otherwise a budget-only raise on a pre-#3313
// index would preserve the sampled Community/Process layer.
return hasProcessDetectionOverride(resolved);
}
const stamp = toProcessDetectionStamp(resolved);
return (
recorded.maxProcesses !== stamp.maxProcesses ||
recorded.maxProcessBranching !== stamp.maxProcessBranching ||
recorded.maxProcessTraceDepth !== stamp.maxProcessTraceDepth ||
recorded.maxEntryPointCandidates !== stamp.maxEntryPointCandidates
);
};
export const buildProcessDetectionPhaseConfig = (
resolved: ResolvedProcessDetectionBudget,
symbolCount: number,
computeDynamicMaxProcesses: (n: number) => number,
): Pick<
ProcessDetectionConfig,
'maxProcesses' | 'maxBranching' | 'maxTraceDepth' | 'maxEntryPointCandidates' | 'minSteps'
> => ({
maxProcesses: resolved.maxProcesses ?? computeDynamicMaxProcesses(symbolCount),
maxBranching: resolved.maxProcessBranching,
maxTraceDepth: resolved.maxProcessTraceDepth,
maxEntryPointCandidates: resolved.maxEntryPointCandidates,
minSteps: PROCESS_DETECTION_BUDGET_DEFAULTS.minSteps,
});
export const processDetectionEffectiveLimits = (
maxProcesses: number,
resolved: ResolvedProcessDetectionBudget,
): ProcessDetectionEffectiveLimits => ({
maxProcesses,
maxProcessBranching: resolved.maxProcessBranching,
maxProcessTraceDepth: resolved.maxProcessTraceDepth,
maxEntryPointCandidates: resolved.maxEntryPointCandidates,
maxProcessTraces: maxProcesses * 2,
});
export const formatWholeFlowsMissingRemedies = (
truncation: Pick<
ProcessTruncationStats,
'entryPointCandidatesDropped' | 'entryPointsUnexplored' | 'processesDropped'
>,
limits: ProcessDetectionEffectiveLimits,
observedEntryPointCandidates: number,
): string => {
const parts: string[] = [];
if (truncation.entryPointCandidatesDropped > 0) {
parts.push(
`${PROCESS_DETECTION_CLI_FLAGS.maxEntryPointCandidates} ` +
`(this run ranked ${observedEntryPointCandidates} candidate(s))`,
);
}
if (truncation.entryPointsUnexplored > 0 || truncation.processesDropped > 0) {
parts.push(
`${PROCESS_DETECTION_CLI_FLAGS.maxProcesses} ` +
`(this run used ${limits.maxProcesses}; next entry is skipped once ` +
`collected traces reach ${limits.maxProcessTraces})`,
);
}
return parts.length === 0 ? '' : ` Raise ${parts.join('; ')}.`;
};
export const formatProcessDetectionBudgetBanner = (
resolved: ResolvedProcessDetectionBudget,
): string | null => {
if (!hasProcessDetectionOverride(resolved)) return null;
const maxProcesses = resolved.overridden.maxProcesses
? String(resolved.maxProcesses)
: 'dynamic (max(20, round(symbols/10)))';
return (
` Process-detection budget: maxProcesses=${maxProcesses}, ` +
`maxProcessBranching=${resolved.maxProcessBranching}, ` +
`maxProcessTraceDepth=${resolved.maxProcessTraceDepth}, ` +
`maxEntryPointCandidates=${resolved.maxEntryPointCandidates}`
);
};

View file

@ -17,6 +17,7 @@ import { CommunityMembership } from './community-processor.js';
import { calculateEntryPointScore, isTestFile } from './entry-point-scoring.js';
import { SupportedLanguages } from 'gitnexus-shared';
import { isDev } from './utils/env.js';
import { PROCESS_DETECTION_BUDGET_DEFAULTS } from './process-detection-budget.js';
import { logger } from '../logger.js';
// ============================================================================
@ -25,16 +26,18 @@ import { logger } from '../logger.js';
export interface ProcessDetectionConfig {
maxTraceDepth: number; // Maximum steps to trace (default: 10)
maxBranching: number; // Max branches to follow per node (default: 3)
maxProcesses: number; // Maximum processes to detect (default: 50)
minSteps: number; // Minimum steps for a valid process (default: 2)
maxBranching: number; // Max branches to follow per node (default: 4)
maxProcesses: number; // Maximum processes to detect (default: 75)
minSteps: number; // Minimum steps for a valid process (default: 3)
maxEntryPointCandidates: number; // Ranked entry-point pool (default: 200)
}
const DEFAULT_CONFIG: ProcessDetectionConfig = {
maxTraceDepth: 10,
maxBranching: 4,
export const DEFAULT_CONFIG: ProcessDetectionConfig = {
maxTraceDepth: PROCESS_DETECTION_BUDGET_DEFAULTS.maxProcessTraceDepth,
maxBranching: PROCESS_DETECTION_BUDGET_DEFAULTS.maxProcessBranching,
maxProcesses: 75,
minSteps: 3, // 3+ steps = genuine multi-hop flow (2-step is just "A calls B")
minSteps: PROCESS_DETECTION_BUDGET_DEFAULTS.minSteps, // 3+ steps = genuine multi-hop flow (2-step is just "A calls B")
maxEntryPointCandidates: PROCESS_DETECTION_BUDGET_DEFAULTS.maxEntryPointCandidates,
};
// ============================================================================
@ -89,7 +92,7 @@ export interface ProcessTruncationStats {
truncated: boolean;
/**
* Scoring candidates that never reached the trace loop because
* `findEntryPoints` keeps only the top `ENTRY_POINT_CANDIDATE_LIMIT`.
* `findEntryPoints` keeps only the top `maxEntryPointCandidates`.
* Counted BEFORE the slice, so it sees what `entryPointsFound` cannot.
*/
entryPointCandidatesDropped: number;
@ -166,11 +169,17 @@ export const processProcesses = async (
for (const n of knowledgeGraph.iterNodes()) nodeMap.set(n.id, n);
// Declared before Step 1 because `findEntryPoints` has a ceiling of its own
// (see `ENTRY_POINT_CANDIDATE_LIMIT`) and reports it through the same record.
// (`cfg.maxEntryPointCandidates`, default 200) and reports it through the same record.
const truncation = emptyTruncation();
// Step 1: Find entry points (functions that call others but have few callers)
const entryPoints = findEntryPoints(knowledgeGraph, reverseCallsEdges, callsEdges, truncation);
const entryPoints = findEntryPoints(
knowledgeGraph,
reverseCallsEdges,
callsEdges,
truncation,
cfg.maxEntryPointCandidates,
);
onProgress?.(`Found ${entryPoints.length} entry points, tracing flows...`, 20);
@ -199,7 +208,7 @@ export const processProcesses = async (
// the remainder are not "no flows found" — they were never looked at.
//
// Counted over the list `findEntryPoints` RETURNS, which is already capped at
// `ENTRY_POINT_CANDIDATE_LIMIT`; candidates beyond that cap are invisible here
// `maxEntryPointCandidates`; candidates beyond that cap are invisible here
// by construction and are reported separately as
// `entryPointCandidatesDropped`.
truncation.entryPointsUnexplored = entryPoints.length - tracedEntryPoints;
@ -480,13 +489,6 @@ const buildReverseCallsGraph = (graph: KnowledgeGraph): AdjacencyList => {
return adj;
};
/**
* How many ranked candidates survive to be traced. Everything below this line
* is discarded — see `ProcessTruncationStats.entryPointCandidatesDropped`, the
* counter that exists because this cap spent a release being invisible.
*/
const ENTRY_POINT_CANDIDATE_LIMIT = 200;
/**
* Find functions/methods that are good entry points for tracing.
*
@ -495,7 +497,9 @@ const ENTRY_POINT_CANDIDATE_LIMIT = 200;
* 2. Export status (exported/public functions rank higher)
* 3. Name patterns (handle*, on*, *Controller, etc.)
*
* Test files are excluded entirely.
* Test files are excluded entirely. How many ranked candidates survive to be
* traced is `maxEntryPointCandidates` (default 200) — see
* `ProcessTruncationStats.entryPointCandidatesDropped`.
*/
const findEntryPoints = (
graph: KnowledgeGraph,
@ -509,6 +513,7 @@ const findEntryPoints = (
* to unwrap a counter to ask for entry points.
*/
truncation?: ProcessTruncationStats,
maxEntryPointCandidates: number = DEFAULT_CONFIG.maxEntryPointCandidates,
): string[] => {
const symbolTypes = new Set<NodeLabel>(['Function', 'Method']);
const entryPointCandidates: {
@ -576,15 +581,15 @@ const findEntryPoints = (
// Limit to prevent explosion — and SAY SO. This is the ceiling that decides
// how much of a repository process detection ever looks at: on anything with
// more than 200 scoring candidates the reported flows are a sample of the
// top-ranked ones, and every downstream count (`entryPointsFound`,
// `entryPointsUnexplored`) is computed over the survivors, so none of them can
// see what was cut here.
if (truncation !== undefined && sorted.length > ENTRY_POINT_CANDIDATE_LIMIT) {
truncation.entryPointCandidatesDropped = sorted.length - ENTRY_POINT_CANDIDATE_LIMIT;
// more scoring candidates than `maxEntryPointCandidates` the reported flows
// are a sample of the top-ranked ones, and every downstream count
// (`entryPointsFound`, `entryPointsUnexplored`) is computed over the
// survivors, so none of them can see what was cut here.
if (truncation !== undefined && sorted.length > maxEntryPointCandidates) {
truncation.entryPointCandidatesDropped = sorted.length - maxEntryPointCandidates;
}
return sorted.slice(0, ENTRY_POINT_CANDIDATE_LIMIT).map((c) => c.id);
return sorted.slice(0, maxEntryPointCandidates).map((c) => c.id);
};
// ============================================================================
@ -822,7 +827,7 @@ const compareOrderKeys = (a: string, b: string): number => (a < b ? -1 : a > b ?
* joined strings, so an O(n log n) sort performs O(n log n) joins of
* O(depth x id-length) characters each.
*
* `n` is bounded here (`ENTRY_POINT_CANDIDATE_LIMIT` entry points x the
* `n` is bounded here (`maxEntryPointCandidates` entry points x the
* per-entry trace budget), so the cost is small and once-per-analyze: measured
* at the ceiling, 23,851 comparisons performed 70,524 joins, and end-to-end
* `processProcesses` at 80,000 functions / 640k CALLS went 456 -> 555 ms. It is

View file

@ -4,7 +4,7 @@
* (RFC §5.3 + §3.2 Phase 1; Ring 2 PKG #919).
*
* Exactly one entry point: `extract(matches, filePath, provider) → ParsedFile`.
* Runs a five-pass pipeline over the matches. Each pass is internal; the
* Runs a seven-pass pipeline over the matches. Each pass is internal; the
* public contract is the output `ParsedFile`.
*
* ## Design principles
@ -22,7 +22,7 @@
* don't overlap) are enforced by `buildScopeTree` from Ring 2 SHARED
* (#912). Malformed inputs throw `ScopeTreeInvariantError`.
*
* ## The five passes
* ## The seven passes
*
* 1. **Build scope tree.** Walk `@scope.*` matches. For each, consult
* `provider.resolveScopeKind` (default: suffix of the capture name).
@ -60,6 +60,10 @@
* one `ReferenceSite` per match. Classify call form via
* `provider.classifyCallForm` (default: the capture's sub-tag if
* present; else `'free'`).
* 6. **Collect callable-value-flow facts.** Independent of Pass 5 so
* existing reference-site extraction stays byte-identical.
* 7. **Preserve call-result assignment identity.** Untyped
* `let lhs = call()` facts used by exact-callee return-type replay.
*
* ## What gets attached where
*
@ -80,6 +84,7 @@ import type {
CallableFlowOperand,
CallableFlowPassingMode,
CallableFlowSite,
CallResultAssignmentSite,
Capture,
CaptureMatch,
ImportEdge,
@ -134,7 +139,7 @@ export type ScopeExtractorHooks = Pick<
// ─── Public entry point ─────────────────────────────────────────────────────
/**
* Drive the five extraction passes and return a `ParsedFile`.
* Drive the seven extraction passes and return a `ParsedFile`.
*
* Throws `ScopeTreeInvariantError` (from #912) when the provider emits
* captures that violate structural scope invariants (e.g., overlapping
@ -240,6 +245,15 @@ export function extract(
const callableFlowSites: CallableFlowSite[] = [];
pass6CollectCallableFlows(partitioned.callableFlow, positionIndex, filePath, callableFlowSites);
// ── Pass 7: preserve call-result assignment identity ───────────────
const callResultAssignmentSites: CallResultAssignmentSite[] = [];
pass7CollectCallResultAssignments(
partitioned.callResultAssignment,
positionIndex,
filePath,
callResultAssignmentSites,
);
// Freeze Scope drafts into final shape and return.
const frozenScopes = scopeDrafts.map(draftToScope);
return Object.freeze({
@ -252,6 +266,9 @@ export function extract(
...(callableFlowSites.length > 0
? { callableFlowSites: Object.freeze(callableFlowSites.slice()) }
: {}),
...(callResultAssignmentSites.length > 0
? { callResultAssignmentSites: Object.freeze(callResultAssignmentSites.slice()) }
: {}),
});
}
@ -264,6 +281,7 @@ interface Partitioned {
readonly typeBinding: readonly CaptureMatch[];
readonly reference: readonly CaptureMatch[];
readonly callableFlow: readonly CaptureMatch[];
readonly callResultAssignment: readonly CaptureMatch[];
}
/**
@ -283,6 +301,7 @@ function partitionByTopic(matches: readonly CaptureMatch[]): Partitioned {
const typeBinding: CaptureMatch[] = [];
const reference: CaptureMatch[] = [];
const callableFlow: CaptureMatch[] = [];
const callResultAssignment: CaptureMatch[] = [];
for (const match of matches) {
for (const topic of topicsOf(match)) {
@ -305,14 +324,32 @@ function partitionByTopic(matches: readonly CaptureMatch[]): Partitioned {
case 'callable-flow':
callableFlow.push(match);
break;
case 'call-result-assignment':
callResultAssignment.push(match);
break;
}
}
}
return { scope, declaration, import_, typeBinding, reference, callableFlow };
return {
scope,
declaration,
import_,
typeBinding,
reference,
callableFlow,
callResultAssignment,
};
}
type Topic = 'scope' | 'declaration' | 'import' | 'type-binding' | 'reference' | 'callable-flow';
type Topic =
| 'scope'
| 'declaration'
| 'import'
| 'type-binding'
| 'reference'
| 'callable-flow'
| 'call-result-assignment';
function topicsOf(match: CaptureMatch): ReadonlySet<Topic> {
const topics = new Set<Topic>();
@ -323,6 +360,7 @@ function topicsOf(match: CaptureMatch): ReadonlySet<Topic> {
else if (name.startsWith('@type-binding.')) topics.add('type-binding');
else if (name.startsWith('@reference.')) topics.add('reference');
else if (name.startsWith('@callable-flow.')) topics.add('callable-flow');
else if (name.startsWith('@call-result-assignment.')) topics.add('call-result-assignment');
}
return topics;
}
@ -598,17 +636,25 @@ function pass2AttachDeclarations(
// fact is the failure mode this subsystem rejects everywhere else.
//
// Copying the field onto BOTH twins makes the outcome identical whichever
// one wins. Deliberately narrow — only `typeParameters`, the one field with
// an asymmetric twin today. Widening this to "merge all metadata" would
// change what every existing duplicate resolves to, which is a different
// change with a different blast radius and no evidence behind it yet.
// one wins. Deliberately narrow — only metadata whose asymmetric twins have
// executable regressions (`typeParameters` and `returnType`). Widening this
// to "merge all metadata" would change what every existing duplicate
// resolves to, which is a different change with a different blast radius
// and no evidence behind it yet.
const first = firstDefByNodeId.get(def.nodeId);
if (first === undefined) {
firstDefByNodeId.set(def.nodeId, def);
} else if (first.typeParameters === undefined && def.typeParameters !== undefined) {
first.typeParameters = def.typeParameters;
} else if (def.typeParameters === undefined && first.typeParameters !== undefined) {
def.typeParameters = first.typeParameters;
} else {
if (first.typeParameters === undefined && def.typeParameters !== undefined) {
first.typeParameters = def.typeParameters;
} else if (def.typeParameters === undefined && first.typeParameters !== undefined) {
def.typeParameters = first.typeParameters;
}
if (first.returnType === undefined && def.returnType !== undefined) {
first.returnType = def.returnType;
} else if (def.returnType === undefined && first.returnType !== undefined) {
def.returnType = first.returnType;
}
}
// Find the innermost scope that contains the declaration's anchor range.
@ -1652,6 +1698,24 @@ function pass6CollectCallableFlows(
}
}
// ─── Pass 7: collect call-result assignment identity ──────────────────────
function pass7CollectCallResultAssignments(
matches: readonly CaptureMatch[],
positionIndex: ReturnType<typeof buildPositionIndex>,
filePath: string,
out: CallResultAssignmentSite[],
): void {
for (const match of matches) {
const call = match['@call-result-assignment.call'];
const lhs = match['@call-result-assignment.lhs'];
if (call === undefined || lhs === undefined || !nonEmpty(lhs.text)) continue;
const inScope = positionIndex.atPosition(filePath, call.range.startLine, call.range.startCol);
if (inScope === undefined) continue;
out.push({ callSite: call.range, inScope, lhs: lhs.text });
}
}
function callableFlowKind(match: CaptureMatch): CallableFlowKind | undefined {
return CALLABLE_FLOW_KINDS.find((kind) => match[`@callable-flow.${kind}`] !== undefined);
}

View file

@ -692,7 +692,10 @@ export interface ScopeResolver {
*/
readonly populateWorkspaceOwners?: (
parsedFiles: readonly ParsedFile[],
ctx: { readonly fileContents: ReadonlyMap<string, string> },
ctx: {
readonly fileContents: ReadonlyMap<string, string>;
readonly resolutionConfig?: unknown;
},
) => void;
/**
@ -965,6 +968,11 @@ export interface ScopeResolver {
*/
readonly freeCallsRequireInstanceOwnership?: boolean;
/** Whether an unqualified call inside a type may dispatch to an inherited
* instance method. Languages such as Swift allow implicit-self lookup,
* while Python/JavaScript/PHP require an explicit receiver. */
readonly implicitThisWalksMro?: boolean;
/**
* When true, a constructor-form call `Type(...)` links to the Class def
* itself rather than its explicit Constructor def. Default

View file

@ -121,6 +121,7 @@ export function emitFreeCallFallback(
* contains the method's owner. See
* `ScopeResolver.freeCallsRequireInstanceOwnership`. */
readonly freeCallsRequireInstanceOwnership?: boolean;
readonly implicitThisWalksMro?: boolean;
readonly recordResolutionOutcome?: ResolutionOutcomeRecorder;
/** Call sites owned by a later precise pass (for example callable-value-flow). */
readonly skipSites?: ReadonlySet<string>;
@ -314,6 +315,7 @@ export function emitFreeCallFallback(
conversionRankFn: options.conversionRankFn,
conversionOnlyArgTypePrefixes: options.conversionOnlyArgTypePrefixes,
constraintCompatibility: options.constraintCompatibility,
implicitThisWalksMro: options.implicitThisWalksMro,
});
fnDefFromImplicitThis = fnDef !== undefined;
}
@ -1167,11 +1169,14 @@ export function pickUniqueGlobalClass(
* pick a method member by name with overload narrowing on arity +
* argument types. Returns undefined if there's no enclosing class,
* no matching method, OR narrowing leaves multiple compatible
* candidates — in the multi-candidate case, picking
* `candidates[0]` would emit a high-confidence CALLS edge whose
* target depends on registration order rather than a defensible
* resolution. Mirrors `pickUniqueGlobalCallable`'s uniqueness check
* in the same file (Codex PR #1497 review, finding 2).
* candidates — except when those survivors are a protocol/interface
* requirement plus exactly one extension witness, in which case the
* witness (the default body) is returned. An inherited class, struct,
* or enum member still wins over a protocol-extension default.
* Picking `candidates[0]` would emit a high-confidence CALLS edge
* whose target depends on registration order rather than a
* defensible resolution. Mirrors `pickUniqueGlobalCallable`'s
* uniqueness check in the same file (Codex PR #1497 review, finding 2).
*
* Exported for unit testing — language-agnostic logic, exercised
* via synthetic stubs in `pick-implicit-this-overload.test.ts`. The
@ -1191,6 +1196,7 @@ export function pickImplicitThisOverload(
readonly conversionRankFn?: ConversionRankFn;
readonly conversionOnlyArgTypePrefixes?: readonly string[];
readonly constraintCompatibility?: ScopeResolver['constraintCompatibility'];
readonly implicitThisWalksMro?: boolean;
},
): SymbolDefinition | undefined {
// Find the enclosing Class scope by walking parents.
@ -1211,22 +1217,118 @@ export function pickImplicitThisOverload(
const classDefId = workspaceIndex.classScopeIdToDefId.get(classScopeId);
if (classDefId === undefined) return undefined;
const overloads = model.methods.lookupAllByOwner(classDefId, site.name);
if (overloads.length === 0) return undefined;
if (overloads.length === 1) return overloads[0];
// Bare calls in an instance method use the same implicit receiver as
// `self.member()`. Prefer declarations on the enclosing type; when the
// language opts into MRO implicit-this, walk inherited owners
// nearest-first and arity-narrow per owner. Compatible-but-ambiguous
// on a nearer ancestor fail-closes — do not fall through to a farther
// override. Falling through to the global name lookup makes inherited
// defaults depend on file order.
const own = model.methods.lookupAllByOwner(classDefId, site.name);
const ownPicked = pickUniqueImplicitThisCandidate(own, site, hookCtx, workspaceIndex);
if (ownPicked !== undefined) return ownPicked;
if (own.length > 0) {
const ownCompatible = narrowOverloadCandidates(own, site.arity, site.argumentTypes, {
argumentTypeClasses: site.argumentTypeClasses,
conversionRankFn: hookCtx?.conversionRankFn,
conversionOnlyArgTypePrefixes: hookCtx?.conversionOnlyArgTypePrefixes,
constraintCompatibility: hookCtx?.constraintCompatibility,
});
if (ownCompatible.length > 0) return undefined;
}
if (hookCtx?.implicitThisWalksMro !== true) return undefined;
// Narrow on arity + argument types. Require a UNIQUE survivor —
// ambiguous narrowing (multiple compatible candidates with no
// disambiguating signal) leaves the call unresolved rather than
// routing to an arbitrary first overload by registration order.
for (const ownerId of scopes.methodDispatch?.mroFor(classDefId) ?? []) {
const inherited = model.methods.lookupAllByOwner(ownerId, site.name);
const inheritedPicked = pickUniqueImplicitThisCandidate(
inherited,
site,
hookCtx,
workspaceIndex,
);
if (inheritedPicked !== undefined) return inheritedPicked;
if (inherited.length > 0) {
const inheritedCompatible = narrowOverloadCandidates(
inherited,
site.arity,
site.argumentTypes,
{
argumentTypeClasses: site.argumentTypeClasses,
conversionRankFn: hookCtx?.conversionRankFn,
conversionOnlyArgTypePrefixes: hookCtx?.conversionOnlyArgTypePrefixes,
constraintCompatibility: hookCtx?.constraintCompatibility,
},
);
if (inheritedCompatible.length > 0) return undefined;
}
}
return undefined;
}
function pickUniqueImplicitThisCandidate(
overloads: readonly SymbolDefinition[],
site: {
readonly arity?: number;
readonly argumentTypes?: readonly string[];
readonly argumentTypeClasses?: readonly import('gitnexus-shared').ParameterTypeClass[];
},
hookCtx:
| {
readonly conversionRankFn?: ConversionRankFn;
readonly conversionOnlyArgTypePrefixes?: readonly string[];
readonly constraintCompatibility?: ScopeResolver['constraintCompatibility'];
}
| undefined,
workspaceIndex: WorkspaceResolutionIndex,
): SymbolDefinition | undefined {
if (overloads.length === 0) return undefined;
const candidates = narrowOverloadCandidates(overloads, site.arity, site.argumentTypes, {
argumentTypeClasses: site.argumentTypeClasses,
conversionRankFn: hookCtx?.conversionRankFn,
conversionOnlyArgTypePrefixes: hookCtx?.conversionOnlyArgTypePrefixes,
constraintCompatibility: hookCtx?.constraintCompatibility,
});
if (candidates.length !== 1) return undefined;
return candidates[0];
if (candidates.length === 0) return undefined;
if (candidates.length === 1) return candidates[0];
const witnesses = preferExtensionWitnesses(candidates, workspaceIndex);
return witnesses.length === 1 ? witnesses[0] : undefined;
}
function preferExtensionWitnesses(
candidates: readonly SymbolDefinition[],
workspaceIndex: WorkspaceResolutionIndex,
): readonly SymbolDefinition[] {
const onOwnerType: SymbolDefinition[] = [];
const extensionWitnesses: SymbolDefinition[] = [];
for (const def of candidates) {
const ownerId = def.ownerId;
if (ownerId === undefined) {
extensionWitnesses.push(def);
continue;
}
const ownerScope = workspaceIndex.classScopeByDefId?.get(ownerId);
const livesOnOwner =
ownerScope?.ownedDefs.some((owned) => owned.nodeId === def.nodeId) === true;
if (livesOnOwner) onOwnerType.push(def);
else extensionWitnesses.push(def);
}
if (extensionWitnesses.length === 0 || onOwnerType.length === 0) return candidates;
const concrete = onOwnerType.filter((def) => !isProtocolLikeOwner(def.ownerId, workspaceIndex));
if (concrete.length > 0) return concrete;
return extensionWitnesses;
}
function isProtocolLikeOwner(
ownerId: string | undefined,
workspaceIndex: WorkspaceResolutionIndex,
): boolean {
if (ownerId === undefined) return false;
const ownerScope = workspaceIndex.classScopeByDefId?.get(ownerId);
if (ownerScope === undefined) return false;
return ownerScope.ownedDefs.some(
(owned) =>
owned.nodeId === ownerId && (owned.type === 'Protocol' || owned.type === 'Interface'),
);
}
/**

View file

@ -24,6 +24,7 @@
*/
import type { ParsedFile, RegistryProviders } from 'gitnexus-shared';
import type { TypeRef } from 'gitnexus-shared';
import type { KnowledgeGraph } from '../../../graph/types.js';
import { generateId } from '../../../../lib/utils.js';
import { lookupOwnedMembersByOwner } from '../../model/owned-members-lookup.js';
@ -84,6 +85,7 @@ import {
import { emitReturnShapeMemberAccesses } from '../passes/return-shape-members.js';
import { emitImportedValueReferences } from '../passes/imported-value-refs.js';
import {
calleeIdPosKey,
createCalleeIdAccumulator,
type CalleeIdAccumulator,
} from '../graph-bridge/callee-id-sink.js';
@ -103,6 +105,58 @@ import { buildWorkspaceResolutionIndex } from '../workspace-index.js';
import type { ResolutionOutcome, ResolutionOutcomeRecorder } from '../resolution-outcome.js';
import { logHeapProbe } from '../../utils/heap-probe.js';
import { parseTruthyEnv } from '../../utils/env.js';
/**
* Join extraction-time assignment identity to Phase-4's exact resolved callee.
* Ambiguous dispatch and missing return annotations deliberately produce no
* binding. Extraction emits these facts only for untyped declarations, so an
* existing entry here is necessarily inference/mirroring and may be corrected;
* explicit annotations never enter this join.
*/
export function applyPreciseCallResultBindings(
parsedFiles: readonly ParsedFile[],
indexes: ReturnType<typeof finalizeScopeModel>,
workspaceIndex: ReturnType<typeof buildWorkspaceResolutionIndex>,
calleeIds: CalleeIdAccumulator,
nodeLookup: ReturnType<typeof buildGraphNodeLookup>,
): number {
let updated = 0;
const returnTypeByGraphId = new Map<string, TypeRef>();
for (const parsed of parsedFiles) {
for (const def of parsed.localDefs) {
const returnType = workspaceIndex.declaredReturnTypeByCallableId.get(def.nodeId);
if (returnType === undefined) continue;
const graphId = resolveDefGraphId(def.filePath, def, nodeLookup);
if (graphId !== undefined) returnTypeByGraphId.set(graphId, returnType);
}
}
for (const parsed of parsedFiles) {
const resolvedByPosition = calleeIds.get(parsed.filePath);
if (resolvedByPosition === undefined) continue;
for (const assignment of parsed.callResultAssignmentSites ?? []) {
const targets = resolvedByPosition.get(
calleeIdPosKey(assignment.callSite.startLine, assignment.callSite.startCol),
);
if (targets === undefined || targets.size !== 1) continue;
const targetId = targets.values().next().value as string | undefined;
if (targetId === undefined) continue;
const returnType = returnTypeByGraphId.get(targetId);
if (returnType === undefined) continue;
const scope = indexes.scopeTree.getScope(assignment.inScope);
if (scope === undefined) continue;
(scope.typeBindings as Map<string, TypeRef>).set(assignment.lhs, {
rawName: returnType.rawName,
...(returnType.declaredSpelling !== undefined
? { declaredSpelling: returnType.declaredSpelling }
: {}),
declaredAtScope: assignment.inScope,
source: 'assignment-inferred',
});
updated++;
}
}
return updated;
}
import { isValueDefinitionLabel } from '../../utils/ast-helpers.js';
import { TransitionalScopeTree } from '../../../../storage/scope-index-store.js';
import { forceGc } from '../../../../storage/parsedfile-store.js';
@ -551,7 +605,6 @@ export function runScopeResolution(
const undecidedSatisfaction: UndecidedSatisfaction[] = [];
const recordResolutionOutcome: ResolutionOutcomeRecorder = (outcome) => {
resolutionOutcomes.push(outcome);
input.recordResolutionOutcome?.(outcome);
};
const PROF = process.env.PROF_SCOPE_RESOLUTION === '1';
const tStart = PROF ? process.hrtime.bigint() : 0n;
@ -638,7 +691,10 @@ export function runScopeResolution(
'sr-extract-end',
`lang=${provider.language} parsedFiles=${parsedFiles.length} preExtractedHits=${preExtractedHits} skipped=${filesSkipped}`,
);
provider.populateWorkspaceOwners?.(parsedFiles, { fileContents: getFileContents() });
provider.populateWorkspaceOwners?.(parsedFiles, {
fileContents: getFileContents(),
resolutionConfig: input.resolutionConfig,
});
provider.populateWorkspaceReferences?.(parsedFiles, {
fileContents: getFileContents(),
treeCache,
@ -844,7 +900,9 @@ export function runScopeResolution(
// views that delegate to it (out-of-core scope index) — the index pins no Scope objects, so the
// disk seal can reclaim them. Byte-identical: the view returns the same Scope
// the resident tree holds (or a value-identical revived one in disk mode).
const workspaceIndex = buildWorkspaceResolutionIndex(parsedFiles, indexes.scopeTree);
const workspaceIndex = buildWorkspaceResolutionIndex(parsedFiles, indexes.scopeTree, {
stripTypePreservingDecoration: provider.stripTypePreservingDecoration,
});
logHeapProbe('sr-post-workspaceIndex', `lang=${provider.language}`);
// Cross-file implicit-namespace visibility (C#). Must run before
@ -1011,6 +1069,12 @@ export function runScopeResolution(
const deferredIndirectCollection = collectDeferredIndirectCollection(emitParsedFiles, indexes);
const deferredIndirectSites = deferredIndirectCollection.sites;
const callableArgumentSites = new Set<string>();
const callResultAssignmentSites = new Set<string>();
for (const parsed of emitParsedFiles) {
for (const site of parsed.callResultAssignmentSites ?? []) {
callResultAssignmentSites.add(callableFlowSiteKey(parsed.filePath, site.callSite));
}
}
if (input.pdg !== true && deferredIndirectSites.size > 0) {
for (const parsed of emitParsedFiles) {
for (const site of parsed.callableFlowSites ?? []) {
@ -1025,11 +1089,14 @@ export function runScopeResolution(
// propagation. Populated below at every CALLS emit path before dedup; the CFG
// join still consumes it only inside the `input.pdg` block.
const calleeIdAccumulator: CalleeIdAccumulator | undefined =
input.pdg === true || deferredIndirectSites.size > 0
input.pdg === true || deferredIndirectSites.size > 0 || callResultAssignmentSites.size > 0
? createCalleeIdAccumulator(
input.pdg === true
? undefined
: (filePath, line, col) => callableArgumentSites.has(`${filePath}:${line}:${col}`),
: (filePath, line, col) => {
const key = `${filePath}:${line}:${col}`;
return callableArgumentSites.has(key) || callResultAssignmentSites.has(key);
},
)
: undefined;
const receiverBound = callableFlowOnly
@ -1060,7 +1127,7 @@ export function runScopeResolution(
heritageTypeArguments,
},
);
const receiverExtras = receiverBound.emitted;
let receiverExtras = receiverBound.emitted;
if (receiverBound.dispatchFanoutSkipped > 0) {
// Never drop dispatch coverage silently (#2829) — same contract as the
// property-dispatch cap below. An interface member over the cap loses real
@ -1112,6 +1179,7 @@ export function runScopeResolution(
isFileLocalDef: provider.isFileLocalDef,
isBuiltInName: provider.languageProvider.isBuiltInName,
freeCallsRequireInstanceOwnership: provider.freeCallsRequireInstanceOwnership === true,
implicitThisWalksMro: provider.implicitThisWalksMro === true,
isCallableVisibleFromCaller: provider.isCallableVisibleFromCaller,
resolveAdlCandidates: provider.resolveAdlCandidates,
resolveQualifiedFreeCall: provider.resolveQualifiedFreeCall,
@ -1123,6 +1191,63 @@ export function runScopeResolution(
skipSites: deferredIndirectSites,
},
);
const replayedCallResultBindings =
callableFlowOnly || calleeIdAccumulator === undefined
? 0
: applyPreciseCallResultBindings(
emitParsedFiles,
indexes,
workspaceIndex,
calleeIdAccumulator,
postHeritageNodeLookup,
);
if (replayedCallResultBindings > 0) {
const handledBeforeReplay = new Set(handledSites);
const replayedReceiverBound = emitReceiverBoundCalls(
graph,
indexes,
emitParsedFiles,
postHeritageNodeLookup,
handledSites,
provider,
workspaceIndex,
readonlyModel,
{
calleeIdSink: calleeIdAccumulator,
isBuiltInName: provider.languageProvider.isBuiltInName,
heritageTypeArguments,
},
);
receiverExtras += replayedReceiverBound.emitted;
if (replayedReceiverBound.dispatchFanoutSkipped > 0) {
logger.warn(
{
lang: provider.language,
dispatchFanoutSkipped: replayedReceiverBound.dispatchFanoutSkipped,
dispatchFanoutSkippedNames: replayedReceiverBound.dispatchFanoutSkippedNames,
fanoutCap: MAX_INTERFACE_DISPATCH_FANOUT,
replay: true,
},
'interface-dispatch: members over the fan-out cap dropped implementors (their CALLS edges were not emitted)',
);
}
const resolvedOnReplay = new Set<string>();
for (const key of handledSites) {
if (!handledBeforeReplay.has(key)) resolvedOnReplay.add(key);
}
if (resolvedOnReplay.size > 0) {
for (let i = resolutionOutcomes.length - 1; i >= 0; i--) {
const outcome = resolutionOutcomes[i];
if (
outcome.kind === 'suppressed' &&
outcome.reason === 'receiver-unresolved' &&
resolvedOnReplay.has(callableFlowSiteKey(outcome.filePath, outcome.range))
) {
resolutionOutcomes.splice(i, 1);
}
}
}
}
const referenceSkipSites = new Set(handledSites);
for (const key of deferredIndirectSites) referenceSkipSites.add(key);
const { emitted, skipped } = callableFlowOnly
@ -1688,6 +1813,8 @@ export function runScopeResolution(
logHeapProbe('sr-end', `lang=${provider.language} parsedFiles=${parsedFiles.length}`);
for (const outcome of resolutionOutcomes) input.recordResolutionOutcome?.(outcome);
return {
filesProcessed: parsedFiles.length,
filesSkipped,

View file

@ -13,7 +13,7 @@
* on `SemanticModel`, and for the builder itself.
*/
import type { Scope, ScopeId, SymbolDefinition } from 'gitnexus-shared';
import type { Scope, ScopeId, SymbolDefinition, TypeRef } from 'gitnexus-shared';
export interface WorkspaceResolutionIndex {
/** Class def `nodeId` → that class's `Scope`. */
@ -36,4 +36,10 @@ export interface WorkspaceResolutionIndex {
* killer). "First module-local callable in `moduleScopeByFile` order" is the
* exact semantics the old scan returned, so it is byte-identical. */
readonly exportedCallableByName: ReadonlyMap<string, SymbolDefinition>;
/** Exact callable def `nodeId` → its declared return type binding.
* Built from definitions owned by each callable's Function scope, so methods
* declared in extension/partial scopes remain addressable even when those
* scopes deliberately own no separate class-like definition. */
readonly declaredReturnTypeByCallableId: ReadonlyMap<string, TypeRef>;
}

View file

@ -39,7 +39,14 @@
* Build cost is O(totalScopes). Read-only after construction.
*/
import type { ParsedFile, Scope, ScopeId, ScopeTree, SymbolDefinition } from 'gitnexus-shared';
import type {
ParsedFile,
Scope,
ScopeId,
ScopeTree,
SymbolDefinition,
TypeRef,
} from 'gitnexus-shared';
import type { WorkspaceResolutionIndex } from './workspace-index-types.js';
import { isClassLike } from './scope/walkers.js';
@ -107,11 +114,15 @@ class ScopeByKeyView<K> implements ReadonlyMap<K, Scope> {
export function buildWorkspaceResolutionIndex(
parsedFiles: readonly ParsedFile[],
scopeTree?: ScopeTree,
options?: {
readonly stripTypePreservingDecoration?: (typeName: string) => string | undefined;
},
): WorkspaceResolutionIndex {
const classScopeIdByDefId = new Map<string, ScopeId>();
const classScopeIdToDefId = new Map<ScopeId, string>();
const moduleScopeIdByFile = new Map<string, ScopeId>();
const exportedCallableByName = new Map<string, SymbolDefinition>();
const declaredReturnTypeByCallableId = new Map<string, TypeRef>();
// Back-compat (no scopeTree): keep the direct Scope-object maps.
const classScopeByDefIdDirect = scopeTree === undefined ? new Map<string, Scope>() : undefined;
const moduleScopeByFileDirect = scopeTree === undefined ? new Map<string, Scope>() : undefined;
@ -138,6 +149,24 @@ export function buildWorkspaceResolutionIndex(
}
for (const scope of parsed.scopes) {
if (scope.kind === 'Function' && scope.parent !== null) {
for (const def of scope.ownedDefs) {
if (def.type !== 'Function' && def.type !== 'Method' && def.type !== 'Constructor') {
continue;
}
if (def.returnType !== undefined) {
const stripped = options?.stripTypePreservingDecoration?.(def.returnType);
const rawName = stripped ?? def.returnType;
declaredReturnTypeByCallableId.set(def.nodeId, {
rawName,
...(rawName !== def.returnType ? { declaredSpelling: def.returnType } : {}),
declaredAtScope: scope.id,
source: 'return-annotation',
});
continue;
}
}
}
if (scope.kind !== 'Class') continue;
const cd = scope.ownedDefs.find((d) => isClassLike(d.type));
if (cd !== undefined) {
@ -157,5 +186,11 @@ export function buildWorkspaceResolutionIndex(
? moduleScopeByFileDirect!
: new ScopeByKeyView(moduleScopeIdByFile, scopeTree);
return { classScopeByDefId, classScopeIdToDefId, moduleScopeByFile, exportedCallableByName };
return {
classScopeByDefId,
classScopeIdToDefId,
moduleScopeByFile,
exportedCallableByName,
declaredReturnTypeByCallableId,
};
}

View file

@ -34,7 +34,16 @@ import type { GraphEmitManifest } from './graph-emit-sink.js';
import type { PdgEmitManifest } from './pdg-emit-sink.js';
import { PDG_EDGE_TYPES } from './pdg-emit-sink.js';
import { getNodeLabel as deriveNodeLabel, type WriteStreamFactory } from './rel-pair-routing.js';
import { EMBEDDABLE_LABELS, type CachedEmbedding } from '../embeddings/types.js';
import { EMBEDDABLE_LABELS } from '../embeddings/types.js';
import {
abortCachedEmbeddingsBuilder,
createCachedEmbeddingsBuilder,
emptyCachedEmbeddingsSnapshot,
finalizeCachedEmbeddingsSnapshot,
ingestCachedEmbeddingRow,
type CachedEmbeddingsSnapshot,
type LoadCachedEmbeddingsOptions,
} from '../embeddings/embedding-restore-spill.js';
import {
extensionManager,
getFtsCapability,
@ -2069,28 +2078,31 @@ export const getLbugStats = async (): Promise<{
/**
* Load cached embeddings from LadybugDB before a rebuild.
* Returns all embedding vectors so they can be re-inserted after the graph is reloaded,
* avoiding expensive re-embedding of unchanged nodes.
*
* Streams `CodeEmbedding` rows with `hasNext`/`getNext` under `withConnLock`
* (#2264, #3306). Vectors are spilled to a temp Float32 file once the table
* exceeds the in-memory row limit so incremental analyze cannot OOM the V8
* heap by materializing every `number[]` up front. Small tables still return
* in-RAM `embeddings` for existing callers/tests.
*
* Detects old schema (no chunkIndex column) and returns empty cache to trigger rebuild.
*/
export const loadCachedEmbeddings = async (): Promise<{
embeddingNodeIds: Set<string>;
embeddings: CachedEmbedding[];
}> => {
export const loadCachedEmbeddings = async (
options?: LoadCachedEmbeddingsOptions,
): Promise<CachedEmbeddingsSnapshot> => {
const c = conn;
if (!c) {
return { embeddingNodeIds: new Set(), embeddings: [] };
return emptyCachedEmbeddingsSnapshot();
}
// The whole read runs inside the connection lock (#2264 review P2). It's safe
// today only by call-ordering (loadCachedEmbeddings runs before the WAL driver
// starts), but the lock makes it robust to future reordering — a concurrent
// CHECKPOINT on the singleton connection is the documented corruption trigger.
// Leaf read: no nested withConnLock-wrapped helpers inside.
// Leaf read: no nested withConnLock-wrapped helpers inside. Do NOT call
// `streamQuery` here — that path is unlocked and would race a CHECKPOINT.
return withConnLock(async () => {
const embeddingNodeIds = new Set<string>();
const embeddings: CachedEmbedding[] = [];
const builder = createCachedEmbeddingsBuilder(options);
try {
// Schema migration detection: query with new columns to verify schema version.
// Old schema only had (nodeId, embedding); new schema adds (id, chunkIndex, startLine, endLine, contentHash).
@ -2104,51 +2116,46 @@ export const loadCachedEmbeddings = async (): Promise<{
);
await readQueryRows(check);
} catch {
return { embeddingNodeIds: new Set(), embeddings: [] };
abortCachedEmbeddingsBuilder(builder);
return emptyCachedEmbeddingsSnapshot();
}
// Try to read contentHash alongside chunk columns
let rows: any;
let queryResult: lbug.QueryResult | lbug.QueryResult[] | undefined;
let hasContentHash = true;
try {
rows = await c.query(
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.chunkIndex AS chunkIndex, e.startLine AS startLine, e.endLine AS endLine, e.embedding AS embedding, e.contentHash AS contentHash`,
);
} catch (err: any) {
// Fallback for legacy DBs without contentHash column
const msg = err?.message ?? '';
if (isMissingColumnOrTableError(msg)) {
hasContentHash = false;
rows = await c.query(
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.chunkIndex AS chunkIndex, e.startLine AS startLine, e.endLine AS endLine, e.embedding AS embedding`,
try {
queryResult = await c.query(
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.chunkIndex AS chunkIndex, e.startLine AS startLine, e.endLine AS endLine, e.embedding AS embedding, e.contentHash AS contentHash`,
);
} else {
throw err;
} catch (err: any) {
// Fallback for legacy DBs without contentHash column
const msg = err?.message ?? '';
if (isMissingColumnOrTableError(msg)) {
hasContentHash = false;
queryResult = await c.query(
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.chunkIndex AS chunkIndex, e.startLine AS startLine, e.endLine AS endLine, e.embedding AS embedding`,
);
} else {
throw err;
}
}
}
for (const row of await readQueryRows(rows)) {
const nodeId = String(row.nodeId ?? row[0] ?? '');
if (!nodeId) continue;
embeddingNodeIds.add(nodeId);
const embedding = row.embedding ?? row[4];
if (embedding) {
embeddings.push({
nodeId,
chunkIndex: Number(row.chunkIndex ?? row[1] ?? 0),
startLine: Number(row.startLine ?? row[2] ?? 0),
endLine: Number(row.endLine ?? row[3] ?? 0),
embedding: Array.isArray(embedding)
? embedding.map(Number)
: Array.from(embedding as any).map(Number),
contentHash: hasContentHash ? (row.contentHash ?? row[5] ?? undefined) : undefined,
});
const results = Array.isArray(queryResult) ? queryResult : [queryResult];
const result = results[0];
while (await result.hasNext()) {
const row = await result.getNext();
ingestCachedEmbeddingRow(builder, row, hasContentHash);
}
return finalizeCachedEmbeddingsSnapshot(builder);
} catch (err) {
abortCachedEmbeddingsBuilder(builder);
throw err;
} finally {
if (queryResult) await closeQueryResults(queryResult);
}
} catch {
/* embedding table may not exist */
} catch (err) {
abortCachedEmbeddingsBuilder(builder);
throw err;
}
return { embeddingNodeIds, embeddings };
});
};

View file

@ -198,6 +198,13 @@ import {
nodeTablesForIncrementalDelete,
shouldPreservePersistedDerivedGraph,
} from './incremental/derived-writeback.js';
import {
formatInvalidProcessDetectionOverride,
processDetectionBudgetMismatch,
resolveProcessDetectionBudget,
toProcessDetectionStamp,
uncertifyProcessDetectionStamp,
} from './ingestion/process-detection-budget.js';
import { NODE_TABLES } from './lbug/schema.js';
import {
loadParseCache,
@ -221,9 +228,20 @@ import {
} from '../storage/git.js';
import { isGitNexusManagedPath } from '../storage/gitnexus-managed-paths.js';
import { getMaxFileSizeBytes } from './ingestion/utils/max-file-size.js';
import type { CachedEmbedding } from './embeddings/types.js';
import {
cacheRowCount,
discardScopedEmbeddingSpills,
disposeEmbeddingSpill,
withEmbeddingSpillScope,
emptyCachedEmbeddingsSnapshot,
EmbeddingSpillReader,
materializeCachedEmbeddings,
normalizeCachedEmbeddings,
snapshotEmbeddingDims,
type CachedEmbeddingsSnapshot,
} from './embeddings/embedding-restore-spill.js';
import { generateAIContextFiles } from '../cli/ai-context.js';
import { sanitizeDetectedBranch } from '../cli/analyze-config.js';
import { formatRejectedBranchForLog, sanitizeDetectedBranch } from './git-ref.js';
import {
EMBEDDING_TABLE_NAME,
EMBEDDING_DIMS,
@ -474,8 +492,9 @@ export interface AnalyzeOptions {
pdgEmitChunkSize?: number;
/** Streamed structural graph emit (#2680). Honored only on a full rebuild
* (`force === true`). May also be enabled via `GITNEXUS_STREAM_GRAPH_EMIT`.
* Trades community detection, process extraction and PDG taint summaries for
* a ~2.9x reduction of in-memory graph heap. */
* The sink answers a complete relationship read, so community detection,
* process extraction, and PDG taint summaries still run; streaming reduces
* in-memory graph heap (~2.9x) by keeping those edges on disk. */
streamGraphEmit?: boolean;
/**
* Default branch threaded into generated AGENTS.md / CLAUDE.md so the
@ -517,6 +536,16 @@ export interface AnalyzeOptions {
* removed); `undefined` defers to the env / auto-formula fallback.
*/
workerPoolSize?: number;
/**
* Process-detection budget overrides (#3313). Threaded to
* `PipelineOptions` without mutating `process.env`. Unset fields fall
* back to `GITNEXUS_*` env, then shipped defaults / the dynamic
* `maxProcesses` formula.
*/
maxProcesses?: number;
maxProcessBranching?: number;
maxProcessTraceDepth?: number;
maxEntryPointCandidates?: number;
/**
* Extra fetch-wrapper function names to treat as HTTP consumers, forwarded to
* `PipelineOptions.fetchWrappers` (#1589/#1852 residual). Sourced from the CLI
@ -1041,12 +1070,16 @@ export const pdgModeMismatch = (recorded: RepoMeta['pdg'], options: PdgOptions):
* directory (#2658). `metaDir` — not `getStoragePaths(repoPath, options.branch)`
* — is the lock scope: a `--branch X` that owns the flat slot resolves to the
* flat `.gitnexus`, so scoping off the raw option would lock the wrong dir.
* `rejectedDetectedBranch` is log-only (the detect-reject warning after lock
* settle); it does not change placement.
*/
interface WriteTarget {
storagePath: string;
repoHasGit: boolean;
currentCommit: string;
checkedOutBranch: string | null;
/** Raw checkout name when git returned one the branch-name rules reject. */
rejectedDetectedBranch: string | null;
branchLabel: string | null;
placement: { branch?: string };
lbugPath: string;
@ -1078,9 +1111,12 @@ async function resolveWriteTarget(repoPath: string, options: AnalyzeOptions): Pr
// validated (#2106 R1): a git ref the branch-name rules forbid becomes `null`
// → the flat slot, matching that a later `--branch <that-ref>` query would
// also be rejected. A normal ref round-trips index-time/query-time labels.
const checkedOutBranch = repoHasGit
? (sanitizeDetectedBranch(getCurrentBranch(repoPath)) ?? null)
: null;
// Keep the raw rejected name so `runFullAnalysis` can warn once after the
// lock settles. Detached / non-git / empty detect stay `null` here and silent.
const rawDetectedBranch = repoHasGit ? getCurrentBranch(repoPath) : null;
const checkedOutBranch = sanitizeDetectedBranch(rawDetectedBranch) ?? null;
const rejectedDetectedBranch =
rawDetectedBranch != null && checkedOutBranch === null ? rawDetectedBranch : null;
// Analyze indexes the working tree, not an arbitrary ref. An explicit
// `--branch X` while a DIFFERENT branch Y is checked out would write Y's
// content into X's slot, corrupting X (#2106). Refuse the mismatch. Detached
@ -1101,6 +1137,7 @@ async function resolveWriteTarget(repoPath: string, options: AnalyzeOptions): Pr
repoHasGit,
currentCommit,
checkedOutBranch,
rejectedDetectedBranch,
branchLabel,
placement,
lbugPath,
@ -1166,59 +1203,69 @@ export async function runFullAnalysis(
let writeTarget = await resolveWriteTarget(repoPath, options);
let lock = await acquireIndexLock(writeTarget.metaDir, acquireOpts);
try {
requireExclusiveIndexLock(
lock,
`Cannot acquire the index lock at ${writeTarget.metaDir}; refusing an unlocked analysis.`,
);
// #2658 review H2: acquireIndexLock can wait up to the timeout ceiling,
// during which git HEAD/branch — and thus the resolved write slot — may
// change (a commit lands, a branch is switched, or another writer adopts the
// flat slot). The pre-wait snapshot must NOT be reused: re-resolve UNDER the
// lock so the freshness check (`existingMeta.lastCommit === currentCommit`)
// and the meta stamps see current git state, honoring the module's "re-check
// freshness after acquiring" contract. If the slot itself moved we hold the
// WRONG lock — release and re-acquire the correct one. Bounded so a
// pathologically churning checkout can't loop forever; after the cap we
// proceed on the current lock. The loop is INSIDE the try so a re-resolve
// that throws (e.g. a `--branch` that stopped matching the now-switched
// checkout) still releases the held lock via `finally` (no leak).
const MAX_RELOCK = 3;
for (let attempt = 0; attempt < MAX_RELOCK; attempt++) {
// Never pass the pre-lock storagePath as already-validated: requireStoragePath
// must run again under the lock so a now-foreign slot aborts (and finally
// still releases the lock).
const fresh = await resolveWriteTarget(repoPath, options);
if (fresh.metaDir === writeTarget.metaDir) {
writeTarget = fresh; // same slot — adopt the freshly-read commit/branch/placement
break;
}
log(
`Index write target moved while waiting for the lock ` +
`(${writeTarget.metaDir} → ${fresh.metaDir}); re-acquiring the correct slot.`,
);
lock.release();
writeTarget = fresh;
lock = await acquireIndexLock(fresh.metaDir, acquireOpts);
return withEmbeddingSpillScope(async () => {
try {
requireExclusiveIndexLock(
lock,
`Cannot acquire the index lock at ${fresh.metaDir}; refusing an unlocked analysis.`,
`Cannot acquire the index lock at ${writeTarget.metaDir}; refusing an unlocked analysis.`,
);
if (attempt === MAX_RELOCK - 1) {
log('Index write target still moving after repeated re-acquire; proceeding on this lock.');
// #2658 review H2: acquireIndexLock can wait up to the timeout ceiling,
// during which git HEAD/branch — and thus the resolved write slot — may
// change (a commit lands, a branch is switched, or another writer adopts the
// flat slot). The pre-wait snapshot must NOT be reused: re-resolve UNDER the
// lock so the freshness check (`existingMeta.lastCommit === currentCommit`)
// and the meta stamps see current git state, honoring the module's "re-check
// freshness after acquiring" contract. If the slot itself moved we hold the
// WRONG lock — release and re-acquire the correct one. Bounded so a
// pathologically churning checkout can't loop forever; after the cap we
// proceed on the current lock. The loop is INSIDE the try so a re-resolve
// that throws (e.g. a `--branch` that stopped matching the now-switched
// checkout) still releases the held lock via `finally` (no leak).
const MAX_RELOCK = 3;
for (let attempt = 0; attempt < MAX_RELOCK; attempt++) {
// Never pass the pre-lock storagePath as already-validated: requireStoragePath
// must run again under the lock so a now-foreign slot aborts (and finally
// still releases the lock).
const fresh = await resolveWriteTarget(repoPath, options);
if (fresh.metaDir === writeTarget.metaDir) {
writeTarget = fresh; // same slot — adopt the freshly-read commit/branch/placement
break;
}
log(
`Index write target moved while waiting for the lock ` +
`(${writeTarget.metaDir} → ${fresh.metaDir}); re-acquiring the correct slot.`,
);
lock.release();
writeTarget = fresh;
lock = await acquireIndexLock(fresh.metaDir, acquireOpts);
requireExclusiveIndexLock(
lock,
`Cannot acquire the index lock at ${fresh.metaDir}; refusing an unlocked analysis.`,
);
if (attempt === MAX_RELOCK - 1) {
log(
'Index write target still moving after repeated re-acquire; proceeding on this lock.',
);
}
}
if (writeTarget.rejectedDetectedBranch) {
log(
`Warning: checkout "${formatRejectedBranchForLog(writeTarget.rejectedDetectedBranch)}" is not a usable index label; continuing.`,
);
}
return await runFullAnalysisInner(
repoPath,
options,
callbacks,
writeTarget,
contentRetention,
runnerIdentityAtBootstrap,
);
} finally {
discardScopedEmbeddingSpills();
lock.release();
}
return await runFullAnalysisInner(
repoPath,
options,
callbacks,
writeTarget,
contentRetention,
runnerIdentityAtBootstrap,
);
} finally {
lock.release();
}
});
}
async function runFullAnalysisInner(
@ -2037,13 +2084,37 @@ async function runFullAnalysisInner(
options = { ...options, force: true };
}
// Process-detection budget (#3313). Resolve CLI/options then env here so
// MCP/server jobs honor GITNEXUS_* without a CLI merge. Compare against
// the persisted stamp BEFORE the already-up-to-date fast path: a clean
// same-commit raise must re-detect flows rather than return the sampled
// index. Does NOT set force — incremental empty-diff + skip derived
// preserve is enough.
const processDetectionBudget = resolveProcessDetectionBudget(
{
maxProcesses: options.maxProcesses,
maxProcessBranching: options.maxProcessBranching,
maxProcessTraceDepth: options.maxProcessTraceDepth,
maxEntryPointCandidates: options.maxEntryPointCandidates,
},
process.env,
(knob, raw) => {
log(formatInvalidProcessDetectionOverride(knob, raw));
},
);
const processDetectionMismatch = processDetectionBudgetMismatch(
existingMeta?.processDetection,
processDetectionBudget,
);
// ── Early-return: already up to date ──────────────────────────────
if (
existingMeta &&
!existingMeta.embeddingCheckpoint &&
!options.force &&
existingMeta.lastCommit === currentCommit &&
!ftsModeChanged
!ftsModeChanged &&
!processDetectionMismatch
) {
// Non-git folders have currentCommit = '' — always rebuild since we can't detect changes
if (currentCommit !== '') {
@ -2103,6 +2174,8 @@ async function runFullAnalysisInner(
// later read on a host where it loads — which is a legitimate, common
// state, and the invariant `analyzer-identity-cli.test.ts` pins.
if (!dirty && !indexedContentChanged && !healUnregistered) {
const processDetectionStamp =
existingMeta.processDetection ?? toProcessDetectionStamp(processDetectionBudget);
if (options.registryName) {
await registerRepo(repoPath, existingMeta, {
name: options.registryName,
@ -2154,7 +2227,11 @@ async function runFullAnalysisInner(
// documented Docker :ro workflow (#1549) — degrades to a warning.
try {
await adoptFlatBranchLabel(repoPath, branchLabel, storagePath);
await saveMeta(metaDir, { ...existingMeta, branch: branchLabel });
await saveMeta(metaDir, {
...existingMeta,
branch: branchLabel,
processDetection: processDetectionStamp,
});
} catch (err) {
// EACCES/EPERM also arise from ownership problems and transient
// Windows locks, so keep the real error visible alongside the
@ -2167,12 +2244,26 @@ async function runFullAnalysisInner(
// Discriminator-only restamp (flag↔env). `existingMeta` already
// carries the folded skipReason; persist it without a write plan.
try {
await saveMeta(metaDir, existingMeta);
await saveMeta(metaDir, {
...existingMeta,
processDetection: processDetectionStamp,
});
} catch (err) {
log(
`Warning: could not restamp the FTS skip reason (${formatMetaWriteFailureReason(err)}); will retry on the next run.`,
);
}
} else if (!existingMeta.processDetection) {
try {
await saveMeta(metaDir, {
...existingMeta,
processDetection: processDetectionStamp,
});
} catch (err) {
log(
`Warning: could not backfill the process-detection stamp (${formatMetaWriteFailureReason(err)}); will retry on the next run.`,
);
}
}
await ensureGitNexusIgnored(repoPath, storagePath);
return {
@ -2214,8 +2305,18 @@ async function runFullAnalysisInner(
// The default-preserve branch is what makes a routine `analyze` (e.g. a
// post-commit hook) safe: a multi-minute embedding pass is no longer
// silently dropped just because the caller omitted `--embeddings`.
let cachedEmbeddingNodeIds = new Set<string>();
let cachedEmbeddings: CachedEmbedding[] = [];
let cachedSnapshot: CachedEmbeddingsSnapshot = emptyCachedEmbeddingsSnapshot();
const adoptCachedEmbeddings = (raw: CachedEmbeddingsSnapshot): void => {
cachedSnapshot = normalizeCachedEmbeddings(raw);
};
const discardCachedEmbeddings = (): void => {
disposeEmbeddingSpill(cachedSnapshot.spill);
cachedSnapshot = emptyCachedEmbeddingsSnapshot();
};
const discardCachedEmbeddingSpill = (): void => {
disposeEmbeddingSpill(cachedSnapshot.spill);
cachedSnapshot = { ...cachedSnapshot, spill: undefined };
};
const existingEmbeddingCount = existingMeta?.stats?.embeddings ?? 0;
const {
@ -2250,7 +2351,7 @@ async function runFullAnalysisInner(
// of the predicted `willTryIncremental`). The post-pipeline branch may
// disagree with the prediction (e.g. when the pipeline produces zero
// File nodes, `isIncremental` flips false and the full-rebuild path
// wipes the DB) — loading unconditionally is cheap insurance against
// wipes the DB) — loading unconditionally is insurance against
// silently dropping embeddings on a mispredicted run. The re-insert
// step gates itself on the actual `isIncremental` value to avoid
// PK-conflicts when the incremental writeback path keeps the rows.
@ -2264,9 +2365,7 @@ async function runFullAnalysisInner(
try {
progress('embeddings', 0, 'Caching embeddings...');
await initAnalysisLbug(lbugPath);
const cached = await loadCachedEmbeddings();
cachedEmbeddingNodeIds = cached.embeddingNodeIds;
cachedEmbeddings = cached.embeddings;
adoptCachedEmbeddings(await loadCachedEmbeddings());
await closeLbug();
} catch (err: any) {
// Surface cache-load failures explicitly: silently swallowing here would
@ -2277,8 +2376,7 @@ async function runFullAnalysisInner(
`(${err?.message ?? String(err)}). ` +
`Embeddings will not be preserved on this run.`,
);
cachedEmbeddingNodeIds = new Set<string>();
cachedEmbeddings = [];
discardCachedEmbeddings();
try {
await closeLbug();
} catch {
@ -2352,6 +2450,16 @@ async function runFullAnalysisInner(
{
parseCache,
workerPoolSize: options.workerPoolSize,
maxProcesses: processDetectionBudget.maxProcesses,
maxProcessBranching: processDetectionBudget.overridden.maxProcessBranching
? processDetectionBudget.maxProcessBranching
: undefined,
maxProcessTraceDepth: processDetectionBudget.overridden.maxProcessTraceDepth
? processDetectionBudget.maxProcessTraceDepth
: undefined,
maxEntryPointCandidates: processDetectionBudget.overridden.maxEntryPointCandidates
? processDetectionBudget.maxEntryPointCandidates
: undefined,
// CFG/PDG opt-in (#2081 M1). PipelineOptions.pdg fans out to the worker
// build gate (workerData.pdg) and the scope-resolution emit gate.
pdg: options.pdg === true,
@ -2391,6 +2499,7 @@ async function runFullAnalysisInner(
},
);
} catch (err) {
discardCachedEmbeddingSpill();
await removeColdParseRebuildDir(coldParseRebuildDir, true);
throw err;
}
@ -2482,7 +2591,8 @@ async function runFullAnalysisInner(
skipDerivedGraphPhases &&
isIncremental &&
!!hashDiff &&
shouldPreservePersistedDerivedGraph(hashDiff);
shouldPreservePersistedDerivedGraph(hashDiff) &&
!processDetectionMismatch;
if (skipDerivedGraphPhases && !preserveDerivedLayer) {
progress('communities', 58, 'Detecting code communities and flows...');
await pipelineResult.runDeferredDerivedPhases?.();
@ -2565,17 +2675,21 @@ async function runFullAnalysisInner(
);
// Set the dirty flag BEFORE any destructive DB mutation. Cleared on
// success at the meta-save step. Scoped to this branch's meta.json.
const now = Date.now();
await saveMeta(metaDir, {
...existingMeta!,
incrementalInProgress: {
startedAt: now,
updatedAt: now,
phase: 'pre-write',
toWriteCount: hashDiff.toWrite.length,
directWriteCount: hashDiff.toWrite.length,
},
});
// POSIX atomic incremental mutates the copy, so a live dirty stamp would
// force-rebuild a healthy index after a crash before swap.
if (!atomicIncremental) {
const now = Date.now();
await saveMeta(metaDir, {
...existingMeta!,
incrementalInProgress: {
startedAt: now,
updatedAt: now,
phase: 'pre-write',
toWriteCount: hashDiff.toWrite.length,
directWriteCount: hashDiff.toWrite.length,
},
});
}
if (atomicIncremental) {
// Stage the live index into the temp so the in-place delete/writeback
// below mutates the COPY, and the end-of-run swap publishes it atomically.
@ -2627,6 +2741,7 @@ async function runFullAnalysisInner(
try {
await wipeLbugDbFiles(buildPath);
} catch (error) {
discardCachedEmbeddingSpill();
if (liveIndexMutationStarted) recordLiveIndexMutationRisk(error);
throw error;
}
@ -2655,6 +2770,7 @@ async function runFullAnalysisInner(
try {
await initAnalysisLbug(buildPath);
} catch (error) {
discardCachedEmbeddingSpill();
if (liveIndexMutationStarted) recordLiveIndexMutationRisk(error);
throw error;
}
@ -2976,7 +3092,7 @@ async function runFullAnalysisInner(
const extensionForcedRebuild = !embeddingRowDmlSafe || !ftsRowDmlSafe;
// `!options.dropEmbeddings` (H1): this rescue reads the rows back OUT of
// the DB, so it must never fire on the one path whose entire purpose is to
// destroy them. `--drop-embeddings` deliberately leaves `cachedEmbeddings`
// destroy them. `--drop-embeddings` deliberately leaves `cachedSnapshot`
// empty (`deriveEmbeddingMode` returns `shouldLoadCache: false` for it by
// construction — see the four-mode comment at the cache-load site), and its
// `options.force = true` conversion sits INSIDE
@ -2991,12 +3107,16 @@ async function runFullAnalysisInner(
// while rows survive ⇒ `hasExisting` false ⇒ `shouldLoadCache` false), i.e.
// it would fix the wipe by deleting the safeguard. Covers
// `--drop-embeddings --embeddings` too — the rescue repopulates
// `cachedEmbeddingNodeIds`, which Phase 4 hands `runEmbeddingPipeline` as
// `cachedSnapshot.embeddingNodeIds`, which Phase 4 hands `runEmbeddingPipeline` as
// the already-embedded set, so the very nodes the user asked to REGENERATE
// would be skipped.
if (extensionForcedRebuild && !options.dropEmbeddings && cachedEmbeddings.length === 0) {
if (
extensionForcedRebuild &&
!options.dropEmbeddings &&
cacheRowCount(cachedSnapshot) === 0
) {
// The escalation below WIPES the DB files, and Phase 3.5 restores
// embedding rows from `cachedEmbeddings` — which is only populated when
// embedding rows from `cachedSnapshot` — which is only populated when
// `deriveEmbeddingMode` saw `meta.stats.embeddings > 0`. A DB whose meta
// under-reports its embeddings (meta restored from an older run, or a
// count that never got stamped) would therefore have every vector
@ -3004,14 +3124,21 @@ async function runFullAnalysisInner(
// while the DB is still intact — a plain MATCH, which needs no VECTOR
// extension. Rows whose owning node is gone are dropped by Phase 3.5's
// live-graph filter, exactly as on any other wiped path.
const rescued = await loadCachedEmbeddings();
if (rescued.embeddings.length > 0) {
cachedEmbeddings = rescued.embeddings;
cachedEmbeddingNodeIds = rescued.embeddingNodeIds;
try {
adoptCachedEmbeddings(await loadCachedEmbeddings());
if (cacheRowCount(cachedSnapshot) > 0) {
log(
`Preserving ${cacheRowCount(cachedSnapshot)} embedding row(s) across the forced rebuild ` +
`(the index metadata did not account for them).`,
);
}
} catch (err: any) {
log(
`Preserving ${rescued.embeddings.length} embedding row(s) across the forced rebuild ` +
`(the index metadata did not account for them).`,
`Warning: could not load cached embeddings ` +
`(${err?.message ?? String(err)}). ` +
`Embeddings will not be preserved on this run.`,
);
discardCachedEmbeddings();
}
}
// Hoisted out of the `||` below (§5.D): the size verdict has to be KNOWN
@ -3492,6 +3619,11 @@ async function runFullAnalysisInner(
lastCommit: '',
indexedAt: new Date().toISOString(),
};
// #3322: persist uncertified *before* CREATE_FTS_INDEX. Park keeps this
// stamp; it must not invent one on every FTS-only crash. Missing stamp +
// shipped defaults is a match, so a budget-mismatch derived rewrite that
// dies in FTS would otherwise recertify the rewritten Community/Process
// rows on a flagless retry.
await saveMeta(metaDir, {
...base,
incrementalInProgress: buildFtsDirtyStamp({
@ -3499,6 +3631,11 @@ async function runFullAnalysisInner(
writePlan: 'in-place',
checkpointSucceeded: boundaryCheckpointSucceeded,
}),
...(processDetectionMismatch
? {
processDetection: uncertifyProcessDetectionStamp(base.processDetection),
}
: {}),
});
}
@ -3650,26 +3787,27 @@ async function runFullAnalysisInner(
// propagates errors (a completed writeback means a deterministic
// delete outcome) and this process holds the exclusive DB lock (no
// concurrent writer).
// The per-batch try/catch stays as a last-resort guard only — it no
// longer fires on the happy path.
// Materialize runs outside the insert catch so a spill I/O failure is not
// treated as a benign PK conflict. Any node with a failed restore batch is
// marked stale in the Phase 4 map so leftover chunks are deleted and rembedded.
let restoredEmbeddingCount = 0;
if (cachedEmbeddings.length > 0) {
const cachedDims = cachedEmbeddings[0].embedding.length;
const restoreFailedNodeIds = new Set<string>();
if (cacheRowCount(cachedSnapshot) > 0) {
const cachedDims = snapshotEmbeddingDims(cachedSnapshot);
const { EMBEDDING_DIMS } = await import('./lbug/schema.js');
if (cachedDims !== EMBEDDING_DIMS) {
if (cachedDims !== undefined && cachedDims !== EMBEDDING_DIMS) {
// Dimensions changed (e.g. switched embedding model) — discard cache and re-embed all
log(
`Embedding dimensions changed (${cachedDims}d -> ${EMBEDDING_DIMS}d), discarding cache`,
);
cachedEmbeddings = [];
cachedEmbeddingNodeIds = new Set();
discardCachedEmbeddings();
} else {
const { batchInsertEmbeddings: batchInsert } =
await import('./embeddings/embedding-pipeline.js');
// (1) Live-graph filter — the FULL pipeline graph (always produced),
// NOT the incremental subgraph, or unchanged files' rows would be
// dropped from the restore set.
const liveEmbeddings = cachedEmbeddings.filter(
const liveEmbeddings = cachedSnapshot.rows.filter(
(e) => pipelineResult.graph.getNode(e.nodeId) !== undefined,
);
// (2) Restore-scope filter (see the discipline note above).
@ -3682,13 +3820,36 @@ async function runFullAnalysisInner(
});
progress('embeddings', 88, `Restoring ${rowsToRestore.length} cached embeddings...`);
const EMBED_BATCH = 200;
for (const batch of chunk(rowsToRestore, EMBED_BATCH)) {
try {
await batchInsert(executeWithReusedStatement, batch);
restoredEmbeddingCount += batch.length;
} catch {
/* last-resort guard — conflict-free by construction above */
let spillReader: EmbeddingSpillReader | undefined;
try {
for (const batch of chunk(rowsToRestore, EMBED_BATCH)) {
let materialized;
try {
if (!spillReader && cachedSnapshot.spill && cachedSnapshot.embeddings.length === 0) {
spillReader = new EmbeddingSpillReader(cachedSnapshot.spill);
}
materialized = materializeCachedEmbeddings(cachedSnapshot, batch, spillReader);
} catch (err) {
for (const row of batch) restoreFailedNodeIds.add(row.nodeId);
log(
`Warning: could not materialize ${batch.length} cached embedding(s) for restore ` +
`(${(err as Error).message}); those nodes will be re-embedded if this run generates embeddings.`,
);
continue;
}
try {
await batchInsert(executeWithReusedStatement, materialized);
restoredEmbeddingCount += batch.length;
} catch (err) {
for (const row of batch) restoreFailedNodeIds.add(row.nodeId);
log(
`Warning: could not restore ${batch.length} cached embedding(s) ` +
`(${(err as Error).message}); those nodes will be re-embedded if this run generates embeddings.`,
);
}
}
} finally {
spillReader?.close();
}
// Legacy-orphan sweep (FIX 3, finder B): the live-graph filter's
@ -3705,7 +3866,7 @@ async function runFullAnalysisInner(
// sweep failure must never fail a completed writeback, so the whole
// sweep warns-and-continues.
if (deletedFilePathsForRestore !== null) {
const orphanRowIds = cachedEmbeddings
const orphanRowIds = cachedSnapshot.rows
.filter((e) => pipelineResult.graph.getNode(e.nodeId) === undefined)
.map((e) => `${e.nodeId}:${e.chunkIndex}`);
if (orphanRowIds.length > 0) {
@ -3734,6 +3895,9 @@ async function runFullAnalysisInner(
}
}
}
// Vectors are on disk only to survive the wipe/delete. After restore,
// drop the spill so Phase 4 does not keep a multi-GB temp file open.
discardCachedEmbeddingSpill();
// ── Phase 4: Embeddings (90–98%) ──────────────────────────────────
const stats = await getLbugStats();
@ -3983,9 +4147,16 @@ async function runFullAnalysisInner(
const embeddingIdentity = embeddingIdentityForRun;
// Build a Map<nodeId, contentHash> from cached embeddings for incremental mode
let existingEmbeddings: Map<string, string> | undefined;
if (cachedEmbeddingNodeIds.size > 0) {
if (cachedSnapshot.embeddingNodeIds.size > 0) {
existingEmbeddings = new Map<string, string>();
for (const e of cachedEmbeddings) {
for (const e of cachedSnapshot.rows) {
if (restoreFailedNodeIds.has(e.nodeId)) {
// Any failed batch for this node: mark stale so Phase 4 DELETEs
// leftover chunks and re-embeds. Omitting the id would treat the
// node as new and PK-conflict on rows that already restored.
existingEmbeddings.set(e.nodeId, STALE_HASH_SENTINEL);
continue;
}
existingEmbeddings.set(e.nodeId, e.contentHash ?? STALE_HASH_SENTINEL);
}
}
@ -4062,7 +4233,7 @@ async function runFullAnalysisInner(
progress('embeddings', scaled, label);
},
{},
cachedEmbeddingNodeIds.size > 0 ? cachedEmbeddingNodeIds : undefined,
cachedSnapshot.embeddingNodeIds.size > 0 ? cachedSnapshot.embeddingNodeIds : undefined,
existingEmbeddings,
{
forceReembedNodeIds: pendingEmbeddingNodeIds,
@ -4471,6 +4642,7 @@ async function runFullAnalysisInner(
// stamp after an on→off flip; the next pdgModeMismatch then compares
// off==off and incremental eligibility is restored.
pdg: resolvePdgConfig(options),
processDetection: toProcessDetectionStamp(processDetectionBudget),
};
// Re-resolve at the commit boundary. Long analyses can overlap an npm
// upgrade, rebuilt dist tree, or native dependency replacement; stamping
@ -4744,6 +4916,7 @@ async function runFullAnalysisInner(
}
}
await removeColdParseRebuildDir(coldParseRebuildDir, true);
discardCachedEmbeddingSpill();
if (liveIndexMutationStarted) {
// Preserve the original error identity/prototype: callers distinguish
// IndexLockTimeoutError and other domain failures with `instanceof`.

View file

@ -769,7 +769,9 @@ import { copyV8CacheIfPresent, tryLoadV8Cache, writeV8CacheFile } from './v8-sid
// v101 (#3294 review): Rust bare-keyword glob imports retain crate/self/super
// instead of an empty target path; restricted pub(...) imports are no longer
// captured as unrestricted reexports. Re-extract both facts on warm indexes.
const SCHEMA_BUMP = 101;
// v102: ParsedFile gained callResultAssignmentSites; old durable shards do
// not carry the exact assignment identity required by return-type replay.
const SCHEMA_BUMP = 102;
const GITNEXUS_PKG_VERSION = (() => {
try {
// package.json sits at gitnexus/package.json — two levels up from

View file

@ -53,6 +53,7 @@ import path from 'node:path';
import v8 from 'node:v8';
import vm from 'node:vm';
import type {
CallResultAssignmentSite,
CallableFlowSite,
ParsedFile,
ReferenceSite,
@ -352,20 +353,26 @@ export const loadParsedFilesForPaths = async (
rejectedFiles++;
continue;
}
const assignments = sanitizeCallResultAssignmentSites(pf.callResultAssignmentSites);
if (assignments === undefined) {
rejectedFiles++;
continue;
}
const chains = sanitizeReceiverChains(pf.referenceSites);
if (chains === undefined) {
rejectedFiles++;
continue;
}
if (flow.dropped === 0 && chains.dropped === 0) {
if (flow.dropped === 0 && assignments.dropped === 0 && chains.dropped === 0) {
out.set(pf.filePath, pf);
} else {
droppedSites += flow.dropped;
droppedSites += flow.dropped + assignments.dropped;
droppedChains += chains.dropped;
filesWithDroppedSites++;
out.set(pf.filePath, {
...pf,
...(flow.dropped === 0 ? {} : { callableFlowSites: flow.sites }),
...(assignments.dropped === 0 ? {} : { callResultAssignmentSites: assignments.sites }),
...(chains.dropped === 0 ? {} : { referenceSites: chains.sites }),
});
}
@ -401,6 +408,19 @@ function sanitizeCallableFlowSites(
return { sites, dropped: value.length - sites.length };
}
function sanitizeCallResultAssignmentSites(
value: unknown,
): { sites: readonly CallResultAssignmentSite[] | undefined; dropped: number } | undefined {
if (value === undefined) return { sites: undefined, dropped: 0 };
if (!Array.isArray(value)) return undefined;
const bounded =
value.length > MAX_CALLABLE_FLOW_SITES_PER_FILE
? value.slice(0, MAX_CALLABLE_FLOW_SITES_PER_FILE)
: value;
const sites = bounded.filter(isValidCallResultAssignmentSite);
return { sites, dropped: value.length - sites.length };
}
/**
* Untrusted-boundary handling for the compact `receiverChain` on a reference site.
*
@ -503,6 +523,15 @@ function isValidCallableFlowSite(value: unknown): value is CallableFlowSite {
}
}
function isValidCallResultAssignmentSite(value: unknown): value is CallResultAssignmentSite {
return (
isRecord(value) &&
isValidRange(value.callSite) &&
isBoundedString(value.inScope) &&
isBoundedString(value.lhs)
);
}
function isValidOperand(value: unknown): boolean {
if (!isRecord(value)) return false;
return (

View file

@ -617,6 +617,22 @@ export interface RepoMeta {
*/
hasCallSummary?: boolean;
};
/**
* The process-detection budget this index's Community/Process rows were
* built under (#3313). Compared on the next analyze so a budget-only
* config change re-detects flows without `--force`. `maxProcesses: null`
* means the dynamic `symbols / 10` formula. Absent on pre-#3313 metas:
* that absence matches default/dynamic knobs (backfill) and mismatches
* any explicit override.
*/
processDetection?: {
maxProcesses: number | null;
maxProcessBranching: number;
maxProcessTraceDepth: number;
maxEntryPointCandidates: number;
/** Live Community/Process rows are not certified under this stamp (#3322). */
uncertified?: true;
};
}
/**

View file

@ -0,0 +1,5 @@
public extension Outer.Container {
static func makeEntry() -> Entry {
Entry(id: 1, text: "sample")
}
}

View file

@ -0,0 +1,3 @@
struct Entry {
let enabled: Bool
}

View file

@ -0,0 +1,8 @@
public enum Outer {
public struct Container {
public struct Entry {
let id: Int
let text: String
}
}
}

View file

@ -0,0 +1,13 @@
struct OtherStore { let value: Int }
struct OtherScenario {
private func makeStore(currentValue: Int) -> OtherStore { OtherStore(value: currentValue) }
}
enum OtherValue {
private static func makeValue(year: Int, month: Int, day: Int) -> Int { -1 }
}
enum OtherItems {
private static func insertItem(into store: Store) -> String { "decoy" }
}

View file

@ -0,0 +1,8 @@
struct Scenario: ScenarioSupport {
func run() -> String {
let store = makeStore()
let value = makeValue(year: 2, month: 3, day: 4)
let item = insertItem(into: store, count: 1)
return "\(store.worker.execute()):\(value):\(item)"
}
}

View file

@ -0,0 +1,13 @@
protocol ScenarioSupport {}
struct Worker {
func execute() -> String { "worker" }
}
struct Store { let worker: Worker }
extension ScenarioSupport {
func makeStore(observer: Int? = nil) -> Store { Store(worker: Worker()) }
func makeValue(year: Int, month: Int, day: Int) -> Int { year + month + day }
func insertItem(into store: Store, count: Int = 1) -> String { "item" }
}

View file

@ -1,31 +1,31 @@
{
"swift-abstract-dispatch/Sources/App.swift": {
"captureGroups": 10,
"digest": "0ec4bb27184ce68dca2ef6bef58c23353b363be9aa483d0bbea5321bccf85ae2"
"captureGroups": 12,
"digest": "29e6acb3afba30eca11eb947812e834196a01dceac7eff131710405b9f5963fa"
},
"swift-abstract-dispatch/Sources/Repository.swift": {
"captureGroups": 26,
"digest": "d8a8c59b7236d6b25a063b8584637d5c89a86cf022776272f80aeb4f1e4826b9"
"digest": "6978cdb9fc70bdba0b640d547e2fb0c4e21c96463303e91612ef30ba1597b3c4"
},
"swift-await-try/App.swift": {
"captureGroups": 13,
"digest": "7ba2042b0cb0a6e5c5d459dc460c509ace477b0ce97bec0cc8aaf981f320e69a"
"captureGroups": 15,
"digest": "6d3bef6d2c98dfaf64c6cb728a54aad87efcc16d7a8ce5aeb1cc2e1207b4d1a8"
},
"swift-await-try/Models.swift": {
"captureGroups": 21,
"digest": "df2a06182275e023c6f40e978f60a0ff933fccc711d7c15f4a2434e8e282a653"
"digest": "c00efbc5b57cce03a5847b304f0956a424ccc345309c8907cd12e5c32449ced1"
},
"swift-call-result-binding/App.swift": {
"captureGroups": 7,
"digest": "979534a46ee420a7a5dfd6b029c9fe5d67ca1b00f3bdc5e8977d72a1c316b75d"
"captureGroups": 8,
"digest": "ae788e37585c57c13bdb29cce48a0fbd1bcdbfa91ba415bdbc675945e500919d"
},
"swift-call-result-binding/Models.swift": {
"captureGroups": 15,
"digest": "d312b4003cbe7f3eec8e1b6f924b0e1afb7023ca086ec26ef0ad69078cc67925"
"digest": "f192ae885ea3de848f3b93acf7672d91b0118800cb0c6513628a6c09c26fadbf"
},
"swift-child-extends-parent/Sources/App.swift": {
"captureGroups": 10,
"digest": "9532b3989cd104dd468deb8d671e727cdf944ebb858e002072875519a014e968"
"captureGroups": 11,
"digest": "4841a7a147f5cac64ed70af845534845bf34701af4bc55a6a9f02fb09ad4b716"
},
"swift-child-extends-parent/Sources/Child.swift": {
"captureGroups": 4,
@ -33,55 +33,55 @@
},
"swift-child-extends-parent/Sources/Parent.swift": {
"captureGroups": 7,
"digest": "09d26410736d5ca5ec8dfe598212e1eefc64075bc2199fda170bb08a478392a4"
"digest": "705670271274ef3f484a4974aba743d8871c8e0ba41afd8bcf07a48225c12880"
},
"swift-class-func-receiver/Service.swift": {
"captureGroups": 27,
"digest": "5d7e5b278bdc94cc2014ebf2f8c975afb10aa80bc2e1bd62ec5559a102fc8e41"
},
"swift-constructor-fallback/App.swift": {
"captureGroups": 7,
"digest": "782cf85fcfe24b2baea6226f09226d23adc2bab591380bd33600cfe8509dbb83"
"captureGroups": 8,
"digest": "c5b2e7f8ea76636a45d46ecee6b44f0687fc6fe2f983bab0d98f8ce2a5c16828"
},
"swift-constructor-fallback/Service.swift": {
"captureGroups": 7,
"digest": "5744bba4c9b821b2264dea36de3e864723f8ca1be9face21ec9de55117a9e7a5"
"digest": "083dd54d6999b18bd96ee51503d5b63957e49f981199022d27112cf9478f89e0"
},
"swift-constructor-type-inference/Models/Repo.swift": {
"captureGroups": 15,
"digest": "e8ea96d49fc2113e29b88961686e39472b92481e929b3f887d53b65201e2e7bb"
"digest": "e5f29bb46ce39f7fcf98a53378ca00825134350afb6c81dd9fe4ff07e1201348"
},
"swift-constructor-type-inference/Models/User.swift": {
"captureGroups": 15,
"digest": "fe1039891719c640bfa7d362040460addcf744eed1350521cbcc7c9399395087"
"digest": "349b1dc04ef00d74058c7e8e7431cd19529c77ffeac5dac76ee7fe5a4ff39f53"
},
"swift-constructor-type-inference/Services/App.swift": {
"captureGroups": 12,
"digest": "a9c40a2dcb51c6e6a9adbaa869183eed90a81302729d4d7a0c300084ea2978e1"
"captureGroups": 14,
"digest": "54cf81184c5017a96c983ee6c5e0e46efa99b1fd4eb67a77d911a5303094173a"
},
"swift-enum-members/Direction.swift": {
"captureGroups": 18,
"digest": "bd0bf779b08305fcd7dfadfcad45319123393aadc17044784314a7b34cb9e6ae"
"digest": "9eb1209d4e89040515e21c0efa17490c58232aa8f5f853ca210e6c1e0470d4cc"
},
"swift-export-visibility/App.swift": {
"captureGroups": 10,
"digest": "6be49190e8a120ed1c3c3812afaa717f60fc12dae82d4b1b7158a358c53ea75a"
"captureGroups": 11,
"digest": "a0ef5984650c2220865999fba38d0803533f1c41d9941031d2b8c25e894ae84f"
},
"swift-export-visibility/Visible.swift": {
"captureGroups": 15,
"digest": "c8515702b94de2f9e3ad009daec61bad6f9328fb39796e57ec03348438446c4c"
"digest": "a0ef7c53a9cc19836afbe94c9bd5f32cabf1ce2756bee4e36267b5a64d1b66f0"
},
"swift-extension-dedup/App.swift": {
"captureGroups": 7,
"digest": "f7062b53f33ec548a2ba5836ef5507b4c662d74ac4935dc654e6f95081f6fcbd"
"captureGroups": 8,
"digest": "4c9583519283afe6f7b80261cc2105781b86ea90bae69882468a71276bbe162c"
},
"swift-extension-dedup/Product.swift": {
"captureGroups": 15,
"digest": "c0f08c41829db294b2c89876a329cfffd7f3fb0a4b2bca315ae867a1d263a068"
"digest": "aaa046017c54533f938ec03669a80ef4b541fd32dad968ec7a339a6438f860b3"
},
"swift-extension-dedup/ProductExtensions.swift": {
"captureGroups": 8,
"digest": "228f2f32494a51fb371c754d9ae11a94e7181a14d875a339c631e26f82df542e"
"digest": "faa5f45d1bbad8478902814d100d776681b2e9015c7bfd519b432effdfb6f915"
},
"swift-field-types/App.swift": {
"captureGroups": 7,
@ -89,7 +89,7 @@
},
"swift-field-types/Models.swift": {
"captureGroups": 20,
"digest": "cad215dd5ce4d0c93056ec5600663b8d17eebbefa06d09675fb64392196531ab"
"digest": "cb40d6eee9315b77bb6f1b1f82644a3636e7ef9aa30cf5edddd55d845b832486"
},
"swift-for-loop-inference/App.swift": {
"captureGroups": 5,
@ -105,39 +105,39 @@
},
"swift-if-let-guard-let/Models.swift": {
"captureGroups": 19,
"digest": "51d6986620c07de553b2c8d848d908899a96c7174a7ae56bf68c7de930c8d5c5"
"digest": "9cf6d8e36976de137d378e2822e0883c7c33a869bc6615b632ee1b7433fde89e"
},
"swift-implicit-imports/App.swift": {
"captureGroups": 7,
"digest": "5a423f851ee81956f48209e618c71668394e2a7312d812656a7f326233bc1387"
"captureGroups": 8,
"digest": "c2a0718ef2ef0f686a36c247bc7a37a64dbfc38f49c400dcd41462050a2e75f4"
},
"swift-implicit-imports/Models.swift": {
"captureGroups": 7,
"digest": "995e604bfd03b42eaeabff130fd1671e524832c64ee52f01cd193a88283d3764"
"digest": "aff5d14d229b15b0806531ebd6db27600b358a6a6fd57a481893e702e3839eb7"
},
"swift-init-cross-file/User.swift": {
"captureGroups": 18,
"digest": "13d60237fb85ddeba695f3e5f0a202c327309bf159a04e696531c419c42e8cf1"
"digest": "005c808719efd792c7ee8f73b863047db27ba57dd7c3c94e0e550186c4274c4e"
},
"swift-init-cross-file/main.swift": {
"captureGroups": 8,
"digest": "3007e4849cf77cc6cb360a1dbf70abec414a408076020f4730d01b8a80c1ddd3"
"captureGroups": 9,
"digest": "c4d442b67247406de3b1158b958f6f3e294f11c0bbb360ed32489582769a1187"
},
"swift-member-write-access/App.swift": {
"captureGroups": 14,
"digest": "f3c28a41c270cea237598bd20c688ba6ab791e3701e85c66bfe3aff7ecfb1ec8"
"digest": "7be68b7e510fefac99c94a4ebab3eb96556a0e587155b6c82b7602705e270987"
},
"swift-member-write-access/Models.swift": {
"captureGroups": 26,
"digest": "376359897f8706015f702b2f3346ad4ca4e78e9f11ae6425746ae57116c2cc51"
"digest": "0eb01f35af949541bf63f781c0d23d2b890a09a858575d2d0922093657bb7661"
},
"swift-method-enrichment/Sources/Animal.swift": {
"captureGroups": 24,
"digest": "48bd53d1e69d4a48569985296f5aa44d4b41ff474566445f4e3d33390165c84d"
"digest": "49eec3df81637274541eb67fd2a640d22ec2166a17854d3f7be179294eee1386"
},
"swift-method-enrichment/Sources/App.swift": {
"captureGroups": 10,
"digest": "906c1f026eb183306a19eff5fafd888fc8982434e2ab07bc886100b99653359b"
"captureGroups": 13,
"digest": "af60edc088ece4298751041cd8cb3fdf3c2a3c03f14f78fcead821b275260f7c"
},
"swift-multi-if-let/App.swift": {
"captureGroups": 17,
@ -145,19 +145,19 @@
},
"swift-multi-if-let/Models.swift": {
"captureGroups": 24,
"digest": "e6add0cb3679cb2800094e41d4afc66c5d714c19ffefc38edfb5686ed9d46d51"
"digest": "b681c53b823b275dd517d29244b8f1b70094e69159d9e49193b6d9fd117385cc"
},
"swift-multidir-target/Package.swift": {
"captureGroups": 5,
"digest": "7c27b1db8b34bf3e81963c29f9f51f1226c1012b7a028c6207209914412b9d5f"
"captureGroups": 6,
"digest": "b7401530c522ff647a7a1d958a35314f669d5dfad2b39578fb9866d2f2d76e58"
},
"swift-multidir-target/Sources/Alpha/Core/User.swift": {
"captureGroups": 7,
"digest": "a67cb60787680595b43af3fdd371ad84237e17fa7de928610a8c14e8ac062043"
},
"swift-multidir-target/Sources/Alpha/Entry/App.swift": {
"captureGroups": 7,
"digest": "3f5df4e88a54d05032cb00c999551cb4f41e0782616ec0cdbf353f3b80a4404d"
"captureGroups": 8,
"digest": "6670c64ae61885694d980ab798e5bab0664bf6b95727ce8a70b08749790b8052"
},
"swift-multidir-target/Sources/Beta/Core/User.swift": {
"captureGroups": 7,
@ -168,8 +168,20 @@
"digest": "356dcd75eb93636607d406e26ce0e62dfa1fe9c78dcd1f950adca304a14d3ad1"
},
"swift-multifolder-nopackage/Services/App.swift": {
"captureGroups": 8,
"digest": "469f895fc72297a08eb0c7ddbf7fdbb812c9296076b4797cd9262e2b7ab59a94"
},
"swift-nested-constructor-extension/Builder.swift": {
"captureGroups": 7,
"digest": "7ed3ccfa724fd58d3c9e10a176101364f2a01f20a7a87f9e3d76b715e4a307d8"
"digest": "8e4e01966ad669ef2e8ea6fc032053540588507e75ecabe142ab5f0b35211cb2"
},
"swift-nested-constructor-extension/Standalone.swift": {
"captureGroups": 5,
"digest": "4ac71fb5f08cea3ab32a23ae41dd3bf3a215880cbb826b83d1c7427fa837d3c2"
},
"swift-nested-constructor-extension/Types.swift": {
"captureGroups": 11,
"digest": "489630892c43db4b9440e6de387d91f3b84ed82702cfffab11db2e6afcce5da1"
},
"swift-nested-extension/Extension.swift": {
"captureGroups": 7,
@ -180,28 +192,40 @@
"digest": "2738a4d77473a2641166b95bc1107f99afca55e9b9d4186eb102e053d378d8ab"
},
"swift-overload-dispatch/App.swift": {
"captureGroups": 7,
"digest": "db8515d0d61419e827767ced61b6bb1949e672b8df11c35cc96882d60f007e07"
"captureGroups": 8,
"digest": "da644f9163f61ef3ec0a000d31dd93b5f0dd11b33def9d96ffe425ff4288b922"
},
"swift-overload-dispatch/Repository.swift": {
"captureGroups": 11,
"digest": "07c6509becf6ae01ae336c53ddeeff38b486bea4a7facad0edf129eb67401408"
"digest": "0dadccb34bc2cc5150bc7d2ac878c2d8f6e1526738e5b5a6301161ff2fe6ae3f"
},
"swift-overload-dispatch/SqlRepository.swift": {
"captureGroups": 28,
"digest": "59d11755ba6796355751608bbf74bd03b99edd6c601c51dcd3e46ff33a4af659"
"digest": "891d106fd673247c89ffbcaf7fe4f6103e2ed79cac09bcf5c3933385c77c287f"
},
"swift-parent-resolution/Sources/Models/BaseModel.swift": {
"captureGroups": 7,
"digest": "0d9056d5dbd480f04ad915c628c81a71e9c1b4ec8981d8027d3a4edcd0f06427"
"digest": "0dbf9f9aff651e5231d87b31d74d0c9d37b30ad6ecdf9e1913cb0da3539bbe00"
},
"swift-parent-resolution/Sources/Models/Serializable.swift": {
"captureGroups": 6,
"digest": "502ce9136303235cb8e1cdd459afeb4bb76c658d186c63731a3dbb3c42fadac9"
"digest": "5551bfbb93cacb230bb635880af1d87fcaffdff6c54173ff1db5aa84079cef36"
},
"swift-parent-resolution/Sources/Models/User.swift": {
"captureGroups": 10,
"digest": "9bc07345adb74d5e2b4711354175825ef00918dbd61785e472d091f14567f9d5"
"digest": "014f2ea08b5a45692b8fbd2dc6cc3e16004908fe6aa229535f59c59c7a4c6a93"
},
"swift-protocol-extension-implicit-self/AUnrelated.swift": {
"captureGroups": 33,
"digest": "2df68574e940bc9d384517dc8ab845525e05a0cd8612973360e4c5c2619622dc"
},
"swift-protocol-extension-implicit-self/Scenario.swift": {
"captureGroups": 23,
"digest": "e0713e3117f5c668b2a96fde7c1e62457e433b9072b8ed4ad9b96af501f0cf39"
},
"swift-protocol-extension-implicit-self/Support.swift": {
"captureGroups": 41,
"digest": "60b645833f81961448817fad0c4c5431b0cb2d19e5d37aa711a65aa600e7438d"
},
"swift-protocol-property/Repository.swift": {
"captureGroups": 7,
@ -213,34 +237,34 @@
},
"swift-qualified-base/Sources/Outer.swift": {
"captureGroups": 9,
"digest": "1ff59349dc4d45a5e1f6fdebf19caf65427d3904eddbc50efeb0879e31574a17"
"digest": "0c6af8f37f15553ed2d6199689994606ae580026e7a608d534cea5834021c89d"
},
"swift-return-type-inference/App.swift": {
"captureGroups": 21,
"digest": "ec0a14090e0240bdd52a955cf7c95c2bc994085ca1482911dd882c5cdaa17c81"
"captureGroups": 23,
"digest": "6e342715fd1d2fa576a713fdf3bdd4f1367566182442b8398f6b35b861946a8d"
},
"swift-return-type-inference/Models.swift": {
"captureGroups": 29,
"digest": "bc539da3c97ac77543b7a1cce457c6ec0f8072c1e5861bfc027cf8fe8f4b541c"
"digest": "1616d99b6c529274c0fe33645b58c69a507162353de96694d90d7539e6c69569"
},
"swift-return-type/App.swift": {
"captureGroups": 7,
"digest": "979534a46ee420a7a5dfd6b029c9fe5d67ca1b00f3bdc5e8977d72a1c316b75d"
"captureGroups": 8,
"digest": "ae788e37585c57c13bdb29cce48a0fbd1bcdbfa91ba415bdbc675945e500919d"
},
"swift-return-type/Models.swift": {
"captureGroups": 21,
"digest": "dc230a5f6085165fb717632f62b65fd3c99825b5fcc7a3a597a8d9143627729e"
"digest": "f3daaadcce524d8c44ef4077bb858a1124e3f9ef9525d5c84fb1288f79e97c4a"
},
"swift-self-this-resolution/Sources/Models/Repo.swift": {
"captureGroups": 7,
"digest": "61fe84aa9826726ee02bce54c62507cc4f27e56e9be2d57c10f55dd8043e13d9"
"digest": "44f6a8907accee7b3dfc6d35de6a70d3e8a154e1bc602f8545aec3bcd2f9af9c"
},
"swift-self-this-resolution/Sources/Models/User.swift": {
"captureGroups": 11,
"digest": "1841837c805e19746138dbb80ccdfccbce483d68d1e7ea08d92daec255cfbdc7"
"digest": "133a8a1186742b19b7408703262fed82ebdf2aa48b9728ae78aa3b26b5b06428"
},
"synthetic:dao-20": {
"captureGroups": 361,
"digest": "aa1747e2f297f5a5a71bd0361273f5505adf93db3d4a2ae3111916bc109c1c7b"
"digest": "56dde1e182164a4ad86de0254e7b683596d9b0340ec941fc09cdff598f16c6dc"
}
}

View file

@ -0,0 +1,6 @@
/**
* Matches the detect-reject warning emitted by `run-analyze` after lock settle.
* Shared so unit and integration tests keep one regex.
*/
export const isDetectRejectWarning = (message: string): boolean =>
/^Warning:.*not a usable index label.*continuing\.$/.test(message);

View file

@ -4,18 +4,17 @@
* Those sets are module-private, and exporting them purely to be testable would
* widen a production surface to satisfy a test — the call
* `receiver-twin-list-drift.test.ts` documents. So the guards read the source
* instead, through the TypeScript parser the repo already vendors and already
* uses this way (`literal-collectors.ts`, `query-determinism-guard.test.ts`,
* `cli-index-help.test.ts`).
* instead, through `@babel/parser` (`parse-typescript-source.ts`).
*
* Using the real parser is what makes the guards trustworthy. A text scanner has
* to decide whether a delimiter opens a comment or sits inside a string, and it
* gets that wrong in both directions here: the ignore-list comments quote paths
* and carry an apostrophe (`Next.js's`), while a glob string such as `'** / *'`
* contains a comment-open sequence. It also has to guess which bracket belongs
* to the declaration rather than to a type annotation. Each of those is a way to
* silently read fewer members — and a guard that quietly stops seeing members is
* the exact defect these guards exist to catch.
* Using an AST parser for TypeScript syntax is what makes the guards
* trustworthy. A text scanner has to decide whether a delimiter opens a comment
* or sits inside a string, and it gets that wrong in both directions here: the
* ignore-list comments quote paths and carry an apostrophe (`Next.js's`), while
* a glob string such as `'** / *'` contains a comment-open sequence. It also
* has to guess which bracket belongs to the declaration rather than to a type
* annotation. Each of those is a way to silently read fewer members — and a
* guard that quietly stops seeing members is the exact defect these guards
* exist to catch.
*
* `setEntries` therefore refuses anything that is not a plain list of string
* literals, rather than skipping the members it cannot resolve.
@ -23,7 +22,8 @@
import { readFileSync } from 'node:fs';
import path from 'node:path';
import { fileURLToPath } from 'node:url';
import ts from 'typescript';
import * as t from '@babel/types';
import { forEachChild, nodeText, parseTypeScript } from './parse-typescript-source.js';
const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..', '..', '..');
@ -55,46 +55,44 @@ export const readSource = (file: string): string => readFileSync(file, 'utf8');
* concatenation, a computed value).
*/
export const setEntries = (source: string, setName: string): string[] => {
const sourceFile = ts.createSourceFile(
'ignore-set-source.ts',
source,
ts.ScriptTarget.Latest,
true,
);
const { ast } = parseTypeScript('ignore-set-source.ts', source);
let elements: ts.NodeArray<ts.Expression> | undefined;
const visit = (node: ts.Node): void => {
let elements: t.ArrayExpression['elements'] | undefined;
const visit = (node: t.Node): void => {
if (
elements === undefined &&
ts.isVariableDeclaration(node) &&
ts.isIdentifier(node.name) &&
node.name.text === setName &&
node.initializer !== undefined &&
ts.isNewExpression(node.initializer) &&
node.initializer.arguments?.length === 1 &&
ts.isArrayLiteralExpression(node.initializer.arguments[0])
t.isVariableDeclarator(node) &&
t.isIdentifier(node.id) &&
node.id.name === setName &&
node.init !== undefined &&
node.init !== null &&
t.isNewExpression(node.init) &&
node.init.arguments.length === 1 &&
t.isArrayExpression(node.init.arguments[0])
) {
elements = node.initializer.arguments[0].elements;
elements = node.init.arguments[0].elements;
return;
}
ts.forEachChild(node, visit);
forEachChild(node, visit);
};
visit(sourceFile);
visit(ast);
if (elements === undefined) {
throw new Error(`${setName} is not declared as \`new Set([...])\` — update this test`);
}
const unresolvable = elements.filter((element) => !ts.isStringLiteral(element));
const unresolvable = elements.filter((element) => !t.isStringLiteral(element));
if (unresolvable.length > 0) {
const first = unresolvable[0];
const excerpt = first && typeof first === 'object' ? nodeText(source, first) : String(first);
throw new Error(
`${setName} holds ${unresolvable.length} member(s) that are not plain string literals ` +
`(first: \`${unresolvable[0].getText(sourceFile)}\`). A source-reading guard cannot resolve ` +
`(first: \`${excerpt}\`). A source-reading guard cannot resolve ` +
`those, so switch this set to a runtime assertion rather than letting the guard see fewer members.`,
);
}
return elements.map((element) => (element as ts.StringLiteral).text);
return elements.map((element) => (element as t.StringLiteral).value);
};
/**

View file

@ -9,7 +9,7 @@
* FOUR modes (plan KTD3):
* 1. Config reflection — import `*-extractors/configs/*.ts`, read each
* config-shaped export's node-type-array keys. Exact `config.language`.
* 2. AST scan (`typescript` parser, no type-checker) over the EXTRACTION
* 2. AST scan (TypeScript 7 program SourceFile; no Checker gating) over the EXTRACTION
* surface — `*-extractors/**`, every `languages/<lang>/captures.ts`, and
* `export-detection.ts`. Collected BY CONSUMPTION SITE: `<n>.type === '..'`,
* `childForFieldName('..')` (capturing the receiver node type when an
@ -26,7 +26,7 @@
* `ingestion/` (e.g. `type-env.ts`). These files MIX SyntaxNode `.type` with
* resolved-symbol `.type` (kinds like 'Class'), so a literal is collected
* ONLY when its `.type` / `childForFieldName` receiver resolves to a
* tree-sitter SyntaxNode (via the TS TypeChecker). Per-`languages/<lang>/`
* tree-sitter SyntaxNode (via the TypeScript 7 Checker). Per-`languages/<lang>/`
* files tag to that one grammar; shared (non-`languages/<lang>/`) files tag
* to the full gated set (valid-if-any).
*
@ -35,8 +35,8 @@
*
* Test-only file: allowed to name languages.
*/
import ts from 'typescript';
import { readFileSync, readdirSync, existsSync, statSync } from 'node:fs';
import * as ts from './ts7-ast.js';
import { readdirSync, existsSync, statSync, realpathSync } from 'node:fs';
import { join } from 'node:path';
import { fileURLToPath } from 'node:url';
import { SupportedLanguages } from '../../src/config/supported-languages.js';
@ -449,11 +449,16 @@ function receiverNodeTypeOf(call: ts.CallExpression, sf: ts.SourceFile): string
return receiverMutatedIn(recvText, scope) ? undefined : found;
}
function scanFile(file: string): ScanResult {
function scanFile(
file: string,
built: { program: ts.Program; checker: ts.Checker } | null,
): ScanResult {
const relPath = rel(file);
const langs = fileLanguages(relPath);
const src = readFileSync(file, 'utf8');
const sf = ts.createSourceFile(file, src, ts.ScriptTarget.Latest, true);
const empty: ScanResult = { nodeTypes: [], fields: [] };
if (!built) return empty;
const sf = programSourceFile(built.program, file);
if (!sf) return empty;
const nodeTypes: CollectedNodeType[] = [];
const fields: CollectedField[] = [];
@ -558,8 +563,9 @@ function scanFile(file: string): ScanResult {
function collectInCodeLiterals(): ScanResult {
const nodeTypes: CollectedNodeType[] = [];
const fields: CollectedField[] = [];
const built = buildProgram();
for (const file of mode2Files()) {
const r = scanFile(file);
const r = scanFile(file, built);
nodeTypes.push(...r.nodeTypes);
fields.push(...r.fields);
}
@ -571,7 +577,7 @@ function collectInCodeLiterals(): ScanResult {
// binding/interpret/arity/import-decomposer/...), the production path for
// migrated languages. These files mix SyntaxNode `.type` (grammar nodes) with
// resolved-symbol `.type` (kinds like 'Class'); a naive scan would false-
// positive on the latter. So this mode uses the TS TypeChecker to collect a
// positive on the latter. So this mode uses the TypeScript 7 Checker to collect a
// literal ONLY when its `.type` receiver / childForFieldName target resolves to
// a tree-sitter SyntaxNode. Per-language dir => grammar (no cross-lang ambiguity).
// ---------------------------------------------------------------------------
@ -613,25 +619,35 @@ function resolutionLayerFiles(): { file: string; langs: SupportedLanguages[] }[]
}
let _program: ts.Program | null = null;
let _checker: ts.TypeChecker | null = null;
function buildProgram(
rootFiles: string[],
): { program: ts.Program; checker: ts.TypeChecker } | null {
let _checker: ts.Checker | null = null;
function buildProgram(): { program: ts.Program; checker: ts.Checker } | null {
if (_program && _checker) return { program: _program, checker: _checker };
try {
const cfg = ts.readConfigFile(join(REPO_ROOT, 'tsconfig.json'), ts.sys.readFile);
const parsed = ts.parseJsonConfigFileContent(cfg.config ?? {}, ts.sys, REPO_ROOT);
const options: ts.CompilerOptions = { ...parsed.options, noEmit: true, skipLibCheck: true };
_program = ts.createProgram(rootFiles, options);
_checker = _program.getTypeChecker();
const api = new ts.API({ cwd: REPO_ROOT });
const snapshot = api.updateSnapshot({
openProjects: [join(REPO_ROOT, 'tsconfig.json')],
});
const project = snapshot.getProjects()[0];
if (!project) return null;
_program = project.program;
_checker = project.checker;
return { program: _program, checker: _checker };
} catch {
return null;
}
}
const sourceFileByPath = new Map<string, ts.SourceFile | undefined>();
function programSourceFile(program: ts.Program, file: string): ts.SourceFile | undefined {
if (sourceFileByPath.has(file)) return sourceFileByPath.get(file);
const sf = program.getSourceFile(file) ?? program.getSourceFile(realpathSync(file));
sourceFileByPath.set(file, sf);
return sf;
}
/** True when `node`'s resolved type is (or includes) a tree-sitter SyntaxNode. */
function isSyntaxNodeReceiver(checker: ts.TypeChecker, node: ts.Node): boolean {
function isSyntaxNodeReceiver(checker: ts.Checker, node: ts.Node): boolean {
try {
const s = checker.typeToString(checker.getTypeAtLocation(node));
return /\bSyntaxNode\b/.test(s);
@ -647,7 +663,7 @@ function collectResolutionLayerLiterals(): ScanResult {
const nodeTypes: CollectedNodeType[] = [];
const fields: CollectedField[] = [];
const entries = resolutionLayerFiles();
const built = buildProgram(entries.map((e) => e.file));
const built = buildProgram();
if (!built) {
resolutionLayerProgramOk = false;
return { nodeTypes, fields };
@ -655,7 +671,7 @@ function collectResolutionLayerLiterals(): ScanResult {
const { program, checker } = built;
for (const { file, langs } of entries) {
const sf = program.getSourceFile(file);
const sf = programSourceFile(program, file);
if (!sf) continue;
const relPath = rel(file);
const constMembers = new Map<string, string[]>();

View file

@ -0,0 +1,162 @@
/**
* Parse-only TypeScript AST walks for tests.
*
* TypeScript 7.0 does not ship the classic Compiler API (`createSourceFile`).
* These guards only need syntax, so they parse with `@babel/parser` (already a
* CLI dependency) rather than spawning the native TypeScript 7 program API.
*/
import { parse } from '@babel/parser';
import {
VISITOR_KEYS,
isBinaryExpression,
isNode,
isStringLiteral,
isTemplateLiteral,
type Comment,
type File,
type MemberExpression,
type Node,
type OptionalMemberExpression,
type TemplateLiteral,
} from '@babel/types';
export type AstNode = Node & { parent?: AstNode };
export interface ParsedSource {
ast: File & { parent?: AstNode };
source: string;
fileName: string;
}
const PARSE_PLUGINS: NonNullable<Parameters<typeof parse>[1]>['plugins'] = [
'typescript',
'explicitResourceManagement',
'importAttributes',
'decoratorAutoAccessors',
['decorators', { decoratorsBeforeExport: true }],
];
export function forEachChild(node: Node, visit: (child: AstNode) => void): void {
const keys = VISITOR_KEYS[node.type] ?? [];
for (const key of keys) {
const value = (node as unknown as Record<string, unknown>)[key];
if (Array.isArray(value)) {
for (const item of value) {
if (isNode(item)) visit(item);
}
} else if (isNode(value)) {
visit(value);
}
}
}
/** Direct descendants in source order (does not include `node` itself). */
export function collectDescendants(node: Node): AstNode[] {
const out: AstNode[] = [];
const visit = (child: AstNode): void => {
out.push(child);
forEachChild(child, visit);
};
forEachChild(node, visit);
return out;
}
export function staticMemberName(
node: MemberExpression | OptionalMemberExpression,
): string | undefined {
return node.computed || node.property.type !== 'Identifier' ? undefined : node.property.name;
}
function templateLiteralText(node: TemplateLiteral): string | undefined {
if (node.expressions.length > 0) return undefined;
return node.quasis.map((quasi) => quasi.value.cooked ?? quasi.value.raw).join('');
}
/** Compile-time string from a literal, template without holes, or `+` chain. */
export function staticStringValue(node: Node | undefined | null): string | undefined {
if (!node) return undefined;
if (isStringLiteral(node)) return node.value;
if (isTemplateLiteral(node)) return templateLiteralText(node);
if (isBinaryExpression(node) && node.operator === '+') {
const left = staticStringValue(node.left);
const right = staticStringValue(node.right);
if (left !== undefined && right !== undefined) return `${left}${right}`;
}
return undefined;
}
function attachParents(node: AstNode, parent?: AstNode): void {
node.parent = parent;
forEachChild(node, (child) => attachParents(child, node));
}
export function parseTypeScript(fileName: string, source: string): ParsedSource {
const ast = parse(source, {
sourceFilename: fileName,
sourceType: 'unambiguous',
plugins: PARSE_PLUGINS,
errorRecovery: true,
attachComment: true,
ranges: true,
}) as File & { parent?: AstNode };
attachParents(ast);
return { ast, source, fileName };
}
export function nodeStart(node: Node): number {
return node.start ?? 0;
}
export function nodeEnd(node: Node): number {
return node.end ?? 0;
}
export function nodeText(source: string, node: Node): string {
return source.slice(nodeStart(node), nodeEnd(node));
}
let lineAtSource = '';
let lineAtStarts: number[] = [0];
function lineStarts(source: string): number[] {
if (source === lineAtSource) return lineAtStarts;
const starts = [0];
for (let i = 0; i < source.length; i++) {
if (source[i] === '\n') starts.push(i + 1);
}
lineAtSource = source;
lineAtStarts = starts;
return starts;
}
export function lineAt(source: string, position: number): number {
if (position <= 0) return 1;
const starts = lineStarts(source);
const pos = Math.min(position, source.length);
let lo = 0;
let hi = starts.length - 1;
while (lo < hi) {
const mid = (lo + hi + 1) >> 1;
if (starts[mid] <= pos) lo = mid;
else hi = mid - 1;
}
return lo + 1;
}
export interface CommentRange {
pos: number;
end: number;
}
function toRange(comment: Comment): CommentRange | undefined {
if (comment.start == null || comment.end == null) return undefined;
return { pos: comment.start, end: comment.end };
}
export function leadingCommentRanges(node: Node): CommentRange[] {
return (node.leadingComments ?? []).map(toRange).filter((range) => range !== undefined);
}
export function trailingCommentRanges(node: Node): CommentRange[] {
return (node.trailingComments ?? []).map(toRange).filter((range) => range !== undefined);
}

View file

@ -0,0 +1,48 @@
/**
* TypeScript 7 AST + Checker surface for test-only walks that need types.
*
* TypeScript 7.0 dropped the classic Compiler API (`createProgram` /
* `getTypeChecker`). The native `typescript/unstable/sync` client exposes the
* same Checker methods against the Go compiler, and `typescript/unstable/ast`
* is the matching syntax tree.
*/
export { SyntaxKind } from 'typescript/unstable/ast';
export type {
CallExpression,
Expression,
Node,
PropertyAccessExpression,
SourceFile,
StringLiteral,
} from 'typescript/unstable/ast';
export {
isArrayLiteralExpression,
isArrowFunction,
isAsExpression,
isBinaryExpression,
isCallExpression,
isCaseClause,
isConstructorDeclaration,
isFunctionDeclaration,
isFunctionExpression,
isGetAccessorDeclaration,
isIdentifier,
isIfStatement,
isMethodDeclaration,
isNewExpression,
isPostfixUnaryExpression,
isPrefixUnaryExpression,
isPropertyAccessExpression,
isSetAccessorDeclaration,
isStringLiteral,
isStringLiteralLikeNode as isStringLiteralLike,
isSwitchStatement,
isVariableDeclaration,
} from 'typescript/unstable/ast/is';
export { API, type Checker, type Program } from 'typescript/unstable/sync';
import type { Node } from 'typescript/unstable/ast';
export function forEachChild(node: Node, visitor: (child: Node) => void): void {
node.forEachChild(visitor);
}

View file

@ -32,12 +32,20 @@ const FIXTURE_SRC = path.resolve(testDir, '..', 'fixtures', 'mini-repo');
let MINI_REPO: string;
let tmpParent: string;
let suiteGitnexusHome: string;
/** False when setup analyze fell back to `--skip-fts` after CREATE_FTS_INDEX native-aborted. */
let ftsIndexed = false;
function cliEnv(extraEnv: Record<string, string> = {}) {
return {
...process.env,
GITNEXUS_HOME: suiteGitnexusHome,
NODE_OPTIONS: `${process.env.NODE_OPTIONS || ''} --max-old-space-size=8192`.trim(),
// Cold parse-worker loads every tree-sitter grammar before the ready
// handshake. The default 5s budget classifies that as a deterministic
// crash-loop on a loaded WSL/CI host (status 1) or the 60s spawnSync
// timeout kills the child first (status null). Sibling integration
// suites pin 60s.
GITNEXUS_WORKER_READY_TIMEOUT_MS: process.env.GITNEXUS_WORKER_READY_TIMEOUT_MS || '60000',
...extraEnv,
};
}
@ -52,6 +60,22 @@ function runCliRaw(extraArgs: string[], cwd: string, timeoutMs = 30000) {
});
}
function isNativeAbort(result: ReturnType<typeof runCliRaw>): boolean {
return (
result.signal === 'SIGSEGV' ||
result.signal === 'SIGABRT' ||
result.signal === 'SIGBUS' ||
result.status === 139
);
}
function isFatalAnalyzeHarness(result: ReturnType<typeof runCliRaw>): boolean {
return (
result.stderr?.includes('Worker script not found') === true ||
result.stderr?.includes('deterministic crash-loop') === true
);
}
/**
* Parse stdout as JSON, returning null on failure (e.g., text output).
*/
@ -116,14 +140,39 @@ beforeAll(() => {
},
});
// Run analyze to populate .gitnexus/ index (required for all tool commands)
const analyzeResult = runCliRaw(['analyze', '--force'], MINI_REPO, 60000);
if (analyzeResult.status !== 0) {
// Index once so every --limit command has a registered repo. Match cli-e2e:
// a tiny fixture analyzes in seconds on a quiet machine, but spawnSync
// status null is SIGTERM from the timeout under load (not an analyze
// exit). Retry timeouts; alreadyUpToDate makes a repeat cheap.
//
// CREATE_FTS_INDEX can SIGSEGV the analyze process on some WSL/native
// hosts (status null, signal SIGSEGV) even when `doctor` reports FTS
// LOAD-able. Do not retry that path — rebuild with --skip-fts so graph
// tools still run. query --limit needs BM25 and is skipped in that case.
let analyzeResult: ReturnType<typeof runCliRaw> | undefined;
for (let attempt = 0; attempt < 3; attempt++) {
analyzeResult = runCliRaw(['analyze', '--force'], MINI_REPO, 90_000);
if (analyzeResult.status === 0) {
ftsIndexed = true;
break;
}
if (isFatalAnalyzeHarness(analyzeResult) || isNativeAbort(analyzeResult)) break;
}
if (
analyzeResult &&
!ftsIndexed &&
isNativeAbort(analyzeResult) &&
!isFatalAnalyzeHarness(analyzeResult)
) {
analyzeResult = runCliRaw(['analyze', '--force', '--skip-fts'], MINI_REPO, 90_000);
}
if (!analyzeResult || analyzeResult.status !== 0) {
const err = analyzeResult?.error;
throw new Error(
`Analyze failed (status ${analyzeResult.status}):\nstdout: ${analyzeResult.stdout}\nstderr: ${analyzeResult.stderr}`,
`Analyze failed (status ${analyzeResult?.status}, signal ${analyzeResult?.signal}, error ${err?.message ?? 'none'}):\nstdout: ${analyzeResult?.stdout}\nstderr: ${analyzeResult?.stderr}`,
);
}
});
}, 300_000);
afterAll(() => {
if (tmpParent) cleanupTempDirSync(tmpParent);
@ -357,6 +406,18 @@ describe('CLI --limit flag E2E', () => {
// ─── query ──────────────────────────────────────────────────────────────
describe('query --limit', () => {
beforeEach((ctx) => {
if (ftsIndexed) return;
if (process.env.GITNEXUS_REQUIRE_FTS === '1') {
throw new Error(
'GITNEXUS_REQUIRE_FTS=1 but setup analyze native-aborted during CREATE_FTS_INDEX; ' +
'query --limit cannot be verified without BM25.',
);
}
ctx.skip(
'query --limit needs BM25; CREATE_FTS_INDEX native-aborted and analyze fell back to --skip-fts',
);
});
it('truncates processes to --limit 1', () => {
// "message" matches logMessage / createLogEntry / formatLogEntry → 4 processes
const limited = runJson<QueryResult>([

View file

@ -139,11 +139,16 @@ describe('CLI update notice subprocess behavior', () => {
preload,
`import fs from 'node:fs';
Object.defineProperty(process.stderr, 'isTTY', { value: true, configurable: true });
globalThis.fetch = async () => {
// Mark the detached child at --import time, before tsx compiles the CLI.
// Writing this from fetch() raced a 30s poll against cold boot + lock
// acquisition on a loaded default-project worker (status: poll timeout).
if (process.argv.includes('__update-check')) {
fs.writeFileSync(${JSON.stringify(started)}, '');
}
globalThis.fetch = async () => {
// Bounded so an abandoned child (test failed before releasing, temp home
// already deleted) still exits instead of spinning forever.
const deadline = Date.now() + 60_000;
const deadline = Date.now() + 90_000;
while (!fs.existsSync(${JSON.stringify(release)}) && Date.now() < deadline) {
await new Promise((resolve) => setTimeout(resolve, 25));
}
@ -176,19 +181,21 @@ globalThis.fetch = async () => {
});
});
// The parent already exited above, so reaching a still-parked child proves
// the refresh outlived it and was never awaited.
// The parent already exited above. `started` is written by the child's
// --import hook, so this wait is "did the detached process actually
// start?" not "did tsx finish compiling the CLI?" The fetch mock still
// parks until `release` so the cache cannot appear before we unblock it.
const cache = path.join(home, 'update-check.json');
await expect.poll(() => fs.existsSync(started), { timeout: 30_000, interval: 50 }).toBe(true);
expect(fs.existsSync(cache)).toBe(false);
fs.writeFileSync(release, '');
await expect.poll(() => fs.existsSync(cache), { timeout: 30_000, interval: 50 }).toBe(true);
await expect.poll(() => fs.existsSync(cache), { timeout: 60_000, interval: 50 }).toBe(true);
expect(JSON.parse(fs.readFileSync(cache, 'utf8'))).toMatchObject({
latestVersion: '99.0.0',
registry: 'https://registry.npmjs.org',
});
}, 90_000);
}, 120_000);
it('prints the localized notice on a forced-TTY stderr and keeps stdout clean', () => {
const home = tempHome();

View file

@ -0,0 +1,113 @@
/**
* Real-DB coverage for #3306: loadCachedEmbeddings must stream CodeEmbedding
* rows instead of getAll()+map(Number) of the whole table, and must be able
* to spill vectors so incremental analyze does not keep every embedding in
* the V8 heap.
*/
import fs from 'node:fs';
import path from 'node:path';
import { afterEach, describe, expect, it } from 'vitest';
import { createTempDir, type TestDBHandle } from '../helpers/test-db.js';
import { EMBEDDING_DIMS } from '../../src/core/lbug/schema.js';
import { batchInsertEmbeddings } from '../../src/core/embeddings/embedding-pipeline.js';
import {
disposeEmbeddingSpill,
materializeCachedEmbeddings,
} from '../../src/core/embeddings/embedding-restore-spill.js';
describe('loadCachedEmbeddings streaming (#3306)', () => {
let tmp: TestDBHandle | undefined;
afterEach(async () => {
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
try {
await adapter.closeLbug();
} catch {
/* already closed */
}
await tmp?.cleanup();
tmp = undefined;
});
async function seedDb(rowCount: number) {
tmp = await createTempDir('gitnexus-lbug-');
const dbPath = path.join(tmp.dbPath, 'lbug');
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
await adapter.initLbug(dbPath);
const rows = Array.from({ length: rowCount }, (_, i) => ({
nodeId: `Function:src/f${i}.ts:fn${i}:1`,
chunkIndex: 0,
startLine: 1,
endLine: 3,
embedding: Array.from({ length: EMBEDDING_DIMS }, (__, d) => (d === 0 ? i + 1 : 0)),
contentHash: `hash-${i}`,
}));
await batchInsertEmbeddings(adapter.executeWithReusedStatement, rows);
return { adapter, rows };
}
it('materializes a small table in RAM (skip-fts / mock-compatible shape)', async () => {
const { adapter, rows } = await seedDb(3);
const cached = await adapter.loadCachedEmbeddings();
expect(cached.spill).toBeUndefined();
expect(cached.embeddings).toHaveLength(3);
expect(cached.rows).toHaveLength(3);
expect(cached.embeddingNodeIds.size).toBe(3);
expect(cached.embeddings.map((e) => e.nodeId).sort()).toEqual(rows.map((r) => r.nodeId).sort());
expect(cached.embeddings.find((e) => e.nodeId === rows[1]!.nodeId)?.embedding[0]).toBe(2);
});
it('streams into a spill file when the in-memory limit is 0 and restores a subset', async () => {
const { adapter, rows } = await seedDb(12);
const cached = await adapter.loadCachedEmbeddings({ inMemoryRowLimit: 0 });
try {
expect(cached.embeddings).toEqual([]);
expect(cached.spill?.rowCount).toBe(12);
expect(cached.rows).toHaveLength(12);
const wanted = new Set([rows[0]!.nodeId, rows[5]!.nodeId, rows[10]!.nodeId]);
const subset = materializeCachedEmbeddings(
cached,
cached.rows.filter((meta) => wanted.has(meta.nodeId)),
);
expect(subset).toHaveLength(3);
const byId = new Map(subset.map((row) => [row.nodeId, row]));
expect(byId.get(rows[0]!.nodeId)?.embedding[0]).toBe(1);
expect(byId.get(rows[5]!.nodeId)?.embedding[0]).toBe(6);
expect(byId.get(rows[10]!.nodeId)?.embedding[0]).toBe(11);
expect(byId.get(rows[0]!.nodeId)?.contentHash).toBe(rows[0]!.contentHash);
} finally {
disposeEmbeddingSpill(cached.spill);
}
});
it('flips from RAM to spill once a non-zero in-memory limit is crossed', async () => {
const { adapter, rows } = await seedDb(8);
const cached = await adapter.loadCachedEmbeddings({ inMemoryRowLimit: 4 });
try {
expect(cached.embeddings).toEqual([]);
expect(cached.spill?.rowCount).toBe(8);
expect(cached.rows).toHaveLength(8);
expect(fs.statSync(cached.spill!.path).size).toBe(12 + 8 * EMBEDDING_DIMS * 4);
const wanted = new Set([rows[2]!.nodeId, rows[3]!.nodeId]);
const subset = materializeCachedEmbeddings(
cached,
cached.rows.filter((meta) => wanted.has(meta.nodeId)),
);
expect(subset).toHaveLength(2);
const byId = new Map(subset.map((row) => [row.nodeId, row]));
expect(byId.get(rows[2]!.nodeId)?.embedding[0]).toBe(3);
expect(byId.get(rows[3]!.nodeId)?.embedding[0]).toBe(4);
} finally {
disposeEmbeddingSpill(cached.spill);
}
});
it('surfaces a spill write failure instead of adopting an empty snapshot', async () => {
const { adapter } = await seedDb(3);
const spillDir = path.join(tmp!.dbPath, 'not-a-directory');
fs.writeFileSync(spillDir, 'x');
await expect(adapter.loadCachedEmbeddings({ inMemoryRowLimit: 0, spillDir })).rejects.toThrow(
/ENOTDIR|not a directory|ENOSPC|EACCES/i,
);
});
});

View file

@ -5,6 +5,7 @@ import path from 'path';
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
import { getStoragePaths, loadMeta, listRegisteredRepos } from '../../src/storage/repo-manager.js';
import { createTempDir } from '../helpers/test-db.js';
import { isDetectRejectWarning } from '../helpers/detect-reject-warning.js';
/**
* #2106/#2354 — branch handling end-to-end. Proves that a plain analyze
@ -220,7 +221,12 @@ describe('multi-branch analyze (#2106)', () => {
execFileSync('git', ['branch', '-M', 'feat`x'], { cwd: repo, stdio: 'pipe' });
const { runFullAnalysis } = await import('../../src/core/run-analyze.js');
await runFullAnalysis(repo, {}, { onProgress: () => {} });
const logs: string[] = [];
await runFullAnalysis(
repo,
{},
{ onProgress: () => {}, onLog: (message) => logs.push(message) },
);
// The forbidden ref was normalized to null → flat slot, no branch field,
// and no branches/ sub-directory created for an unqueryable slug.
@ -228,6 +234,9 @@ describe('multi-branch analyze (#2106)', () => {
expect(existsSync(flat.lbugPath)).toBe(true);
expect((await loadMeta(flat.storagePath))?.branch).toBeUndefined();
expect(existsSync(path.join(flat.storagePath, 'branches'))).toBe(false);
const warnings = logs.filter(isDetectRejectWarning);
expect(warnings).toHaveLength(1);
expect(warnings[0]).toContain('feat`x');
} finally {
await tmp.cleanup();
}

View file

@ -116,19 +116,7 @@ describe.skipIf(!swiftAvailable)('a function-local callable keeps its own node (
});
});
/**
* KNOWN GAP, pinned so it cannot be mistaken for part of the fix above.
*
* `helper(x, x)` inside `run` still targets the one-argument METHOD instead of
* the two-argument local. That is decided upstream of the graph bridge: the
* free-call binding hands `emitFreeCallFallback` the class-member def
* (`def:src/app.swift#5:4:Method:helper`), never the local's
* (`def:src/app.swift#9:8:Method:helper`), so no def→node mapping can correct
* it — both defs carry the same `qualifiedName` and the same label, and the
* binding walk picks the member. Recorded here rather than fixed because it
* lives in the scope walk, not in `resolveDefGraphId`.
*/
it('KNOWN GAP: the call to the local still binds to the same-named method', () => {
expect(targetsFrom(RUN)).toEqual([METHOD_HELPER]);
it('binds the call in run to the function-local rather than the same-named method', () => {
expect(targetsFrom(RUN)).toEqual([LOCAL_HELPER]);
});
});

View file

@ -305,6 +305,52 @@ describe.skipIf(!swiftAvailable)('Swift extension deduplication', () => {
});
});
// ---------------------------------------------------------------------------
// Protocol-extension implicit self (issue #3273): a conforming type may call
// default implementation methods without an explicit receiver. Resolution
// must traverse the protocol's extension surface instead of falling back to
// unrelated same-named private methods elsewhere in the module.
// ---------------------------------------------------------------------------
describe.skipIf(!swiftAvailable)('Swift protocol-extension implicit self (#3273)', () => {
let result: PipelineResult;
beforeAll(async () => {
result = await runPipelineFromRepo(
path.join(FIXTURES, 'swift-protocol-extension-implicit-self'),
() => {},
);
}, 60000);
it('resolves unqualified helper calls to the protocol extension', () => {
const calls = getRelationships(result, 'CALLS').filter((c) => c.source === 'run');
for (const target of ['makeStore', 'makeValue', 'insertItem']) {
const call = calls.find((c) => c.target === target);
expect(call?.targetFilePath).toBe('Support.swift');
}
});
it('keeps unrelated private same-name methods unreachable', () => {
const calls = getRelationships(result, 'CALLS').filter((c) => c.source === 'run');
expect(calls.some((c) => c.targetFilePath === 'AUnrelated.swift')).toBe(false);
});
it('preserves the downstream call through the extension helper return type', () => {
const calls = getRelationships(result, 'CALLS');
const execute = calls.find((c) => c.source === 'run' && c.target === 'execute');
expect(execute?.targetFilePath).toBe('Support.swift');
expect(
result.resolutionOutcomes.some(
(outcome) =>
outcome.kind === 'suppressed' &&
outcome.reason === 'receiver-unresolved' &&
outcome.filePath === 'Scenario.swift' &&
outcome.name === 'execute',
),
).toBe(false);
});
});
// ---------------------------------------------------------------------------
// Constructor fallback: Swift constructors look like free function calls
// (no `new` keyword). The resolver retries with constructor form when
@ -1251,6 +1297,36 @@ describe.skipIf(!swiftAvailable)('Swift nested-type extension (extension Foo.Bar
});
});
// ---------------------------------------------------------------------------
// A bare constructor inside an extension must prefer a nested type owned by
// the extended type over an unrelated top-level type with the same short name.
// ---------------------------------------------------------------------------
describe.skipIf(!swiftAvailable)(
'Swift nested constructor lookup in a public qualified extension (#3262)',
() => {
let result: PipelineResult;
beforeAll(async () => {
result = await runPipelineFromRepo(
path.join(FIXTURES, 'swift-nested-constructor-extension'),
() => {},
);
}, 60000);
it('resolves Entry(id:text:) to Outer.Container.Entry and not the top-level Entry', () => {
const entryCalls = getRelationships(result, 'CALLS').filter(
(call) => call.source === 'makeEntry' && call.target === 'Entry',
);
expect(entryCalls.map((call) => call.rel.targetId)).toEqual(['Struct:Types.swift:Entry']);
expect(entryCalls.some((call) => call.rel.targetId === 'Struct:Standalone.swift:Entry')).toBe(
false,
);
});
},
);
// ---------------------------------------------------------------------------
// F75: protocol property requirements (`var title: String { get }`) are
// extracted as Property symbols owned by the protocol. Before the fix these

View file

@ -133,6 +133,33 @@ describe('analyze-config (.gitnexusrc support, #243)', () => {
expect(() => loadAnalyzeConfig(dir)).toThrow(/Unknown key "defalutBranch"/);
});
it('parses process-detection budget keys as numeric strings (#3313)', async () => {
await writeRc(
JSON.stringify({
maxProcesses: 40,
maxProcessBranching: '2',
maxProcessTraceDepth: 8,
maxEntryPointCandidates: 400,
}),
);
expect(loadAnalyzeConfig(dir)).toEqual({
maxProcesses: '40',
maxProcessBranching: '2',
maxProcessTraceDepth: '8',
maxEntryPointCandidates: '400',
});
});
it('lets a nested analyze block override flat process-detection keys (#3313)', async () => {
await writeRc(
JSON.stringify({
maxProcesses: 80,
analyze: { maxProcesses: 25 },
}),
);
expect(loadAnalyzeConfig(dir)).toEqual({ maxProcesses: '25' });
});
it('accepts embeddingBaseUrl / embeddingModel but rejects embeddingDims (CLI/env-only)', async () => {
// URL + MODEL are read lazily at runtime, so they are valid config keys.
await writeRc(JSON.stringify({ embeddingBaseUrl: 'http://h/v1', embeddingModel: 'm' }));

View file

@ -237,4 +237,18 @@ describe('analyzeCommand .gitnexusrc wiring (#243)', () => {
expect(refreshBaseRefLineMock).toHaveBeenCalledTimes(1);
expect(refreshBaseRefLineMock).toHaveBeenCalledWith(dir, 'develop', expect.any(Object));
});
it('threads .gitnexusrc process-detection knobs and lets CLI win (AE3)', async () => {
await writeRc({ maxProcesses: '40', maxEntryPointCandidates: 300 });
const { analyzeCommand } = await import('../../src/cli/analyze.js');
await analyzeCommand(dir, { maxProcesses: '25' });
expect(runFullAnalysisMock.mock.calls[0][1]).toEqual(
expect.objectContaining({
maxProcesses: 25,
maxEntryPointCandidates: 300,
}),
);
});
});

View file

@ -0,0 +1,133 @@
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
const runFullAnalysisMock = vi.fn();
vi.mock('../../src/core/run-analyze.js', () => ({
runFullAnalysis: runFullAnalysisMock,
}));
vi.mock('../../src/core/lbug/lbug-adapter.js', () => ({
closeLbug: vi.fn(async () => undefined),
closeLbugBeforeExit: vi.fn(async () => undefined),
isLbugReady: vi.fn(() => false),
}));
vi.mock('../../src/storage/repo-manager.js', () => ({
getStoragePaths: vi.fn(() => ({ storagePath: '.gitnexus', lbugPath: '.gitnexus/lbug' })),
getGlobalRegistryPath: vi.fn(() => 'registry.json'),
RegistryNameCollisionError: class RegistryNameCollisionError extends Error {},
AnalysisNotFinalizedError: class AnalysisNotFinalizedError extends Error {},
assertAnalysisFinalized: vi.fn(async () => undefined),
}));
vi.mock('../../src/storage/git.js', () => ({
getGitRoot: vi.fn(() => '/repo'),
hasGitDir: vi.fn(() => true),
}));
vi.mock('../../src/core/ingestion/utils/max-file-size.js', () => ({
getMaxFileSizeBannerMessage: vi.fn(() => null),
}));
describe('analyzeCommand process-detection budget (#3313)', () => {
const ORIGINAL_NODE_OPTIONS = process.env.NODE_OPTIONS;
beforeEach(() => {
vi.resetModules();
runFullAnalysisMock.mockReset();
process.exitCode = undefined;
process.env.NODE_OPTIONS = `${process.env.NODE_OPTIONS ?? ''} --max-old-space-size=8192`.trim();
});
afterEach(() => {
if (ORIGINAL_NODE_OPTIONS === undefined) {
delete process.env.NODE_OPTIONS;
} else {
process.env.NODE_OPTIONS = ORIGINAL_NODE_OPTIONS;
}
vi.unstubAllEnvs();
});
const upToDate = {
repoName: 'repo',
repoPath: '/repo',
stats: {},
alreadyUpToDate: true,
};
it('threads the four CLI flags through runFullAnalysis without env mutation', async () => {
const { analyzeCommand } = await import('../../src/cli/analyze.js');
runFullAnalysisMock.mockResolvedValue(upToDate);
const before = {
processes: process.env.GITNEXUS_MAX_PROCESSES,
branching: process.env.GITNEXUS_MAX_PROCESS_BRANCHING,
depth: process.env.GITNEXUS_MAX_PROCESS_TRACE_DEPTH,
entries: process.env.GITNEXUS_MAX_ENTRY_POINT_CANDIDATES,
};
await analyzeCommand(undefined, {
maxProcesses: '25',
maxProcessBranching: '2',
maxProcessTraceDepth: '8',
maxEntryPointCandidates: '400',
});
expect(runFullAnalysisMock).toHaveBeenCalledWith(
expect.any(String),
expect.objectContaining({
maxProcesses: 25,
maxProcessBranching: 2,
maxProcessTraceDepth: 8,
maxEntryPointCandidates: 400,
}),
expect.any(Object),
);
expect(process.env.GITNEXUS_MAX_PROCESSES).toBe(before.processes);
expect(process.env.GITNEXUS_MAX_PROCESS_BRANCHING).toBe(before.branching);
expect(process.env.GITNEXUS_MAX_PROCESS_TRACE_DEPTH).toBe(before.depth);
expect(process.env.GITNEXUS_MAX_ENTRY_POINT_CANDIDATES).toBe(before.entries);
});
it.each(['0', 'abc', '-4'])(
'warns and continues when --max-processes is %s (AE5)',
async (value) => {
const { _captureLogger } = await import('../../src/core/logger.js');
const cap = _captureLogger();
try {
const { analyzeCommand } = await import('../../src/cli/analyze.js');
runFullAnalysisMock.mockResolvedValue(upToDate);
await analyzeCommand(undefined, { maxProcesses: value });
expect(process.exitCode).toBeUndefined();
expect(runFullAnalysisMock).toHaveBeenCalledWith(
expect.any(String),
expect.not.objectContaining({ maxProcesses: expect.any(Number) }),
expect.any(Object),
);
expect(
cap.records().some((r) => {
const msg = String(r.msg ?? '');
return (
msg.includes('--max-processes must be a positive integer') &&
msg.includes('next source (env, then the built-in default)')
);
}),
).toBe(true);
} finally {
cap.restore();
}
},
);
it('leaves option fields unset so runFullAnalysis can honor env-only overrides', async () => {
const { analyzeCommand } = await import('../../src/cli/analyze.js');
runFullAnalysisMock.mockResolvedValue(upToDate);
vi.stubEnv('GITNEXUS_MAX_PROCESSES', '80');
await analyzeCommand(undefined, {});
const opts = runFullAnalysisMock.mock.calls[0][1] as { maxProcesses?: number };
expect(opts.maxProcesses).toBeUndefined();
});
});

View file

@ -8,6 +8,8 @@ import { runWebBuild, shouldBuildWeb, shouldPreserveWebOutput } from '../../scri
/** Default prepare/build stay CLI-only; the web UI ships only via prepack --web. */
const REPO_ROOT = path.resolve(__dirname, '../../..');
const CLI_TSC_JS = 'node ../gitnexus/node_modules/typescript/lib/tsc.js';
const WEB_TSC_JS = 'node ../gitnexus-web/node_modules/typescript/lib/tsc.js';
const PACKAGE_JSON = JSON.parse(
readFileSync(path.join(REPO_ROOT, 'gitnexus/package.json'), 'utf8'),
) as { scripts?: Record<string, string> };
@ -283,7 +285,7 @@ describe('workflows that need the web UI install it themselves', () => {
describe('setup-gitnexus job budget', () => {
it('does not npm-ci gitnexus-shared (TypeScript 7 optional-platform install stalls CI)', () => {
const shared = setupGitnexus.runs?.steps?.find((step) => step.name === 'Build gitnexus-shared');
expect(String(shared?.run)).toBe('node ../gitnexus/node_modules/typescript/lib/tsc.js');
expect(String(shared?.run)).toBe(CLI_TSC_JS);
expect(String(shared?.run)).not.toContain('.bin');
expect(shared?.if).toContain("lifecycle-scripts == 'false'");
expect(
@ -314,12 +316,28 @@ describe('setup-gitnexus job budget', () => {
expect(String(setupNode?.with?.['cache-dependency-path'])).toBe(
'gitnexus-web/package-lock.json',
);
expect(String(shared?.run)).toBe('node ../gitnexus-web/node_modules/typescript/lib/tsc.js');
expect(String(shared?.run)).toBe(WEB_TSC_JS);
expect(String(shared?.run)).not.toContain('.bin');
expect(String(shared?.run)).not.toContain('npm ci');
expect(webInstall?.env?.PLAYWRIGHT_SKIP_BROWSER_DOWNLOAD).toBe('1');
});
it('Vercel installs web first and compiles shared with the web TypeScript', () => {
const vercel = JSON.parse(
readFileSync(path.join(REPO_ROOT, 'gitnexus-web/vercel.json'), 'utf8'),
) as { installCommand?: string };
const install = String(vercel.installCommand);
// Vercel runs installCommand with NODE_ENV=production, so npm ci drops
// typescript unless --include=dev is on that install (not a later step).
expect(install).toContain('npm ci --include=dev');
expect(install).toContain('gitnexus-shared');
expect(install.indexOf('npm ci --include=dev')).toBeLessThan(
install.indexOf('gitnexus-shared'),
);
expect(install).toContain(WEB_TSC_JS);
expect(install).not.toMatch(/gitnexus-shared[^&]*npm (?:ci|install)/);
});
it('quality typecheck skips prepare/postinstall so tsc --noEmit fits in 10 minutes', () => {
const job = qualityJobs.typecheck;
const setup = job?.steps?.find((step) => step.uses === './.github/actions/setup-gitnexus');
@ -327,6 +345,28 @@ describe('setup-gitnexus job budget', () => {
expect(setup?.with?.['lifecycle-scripts']).toBe('false');
});
it('web app tsconfig typechecks React JSX on TypeScript 7 without baseUrl', () => {
const tsconfig = JSON.parse(
readFileSync(path.join(REPO_ROOT, 'gitnexus-web/tsconfig.app.json'), 'utf8'),
) as {
compilerOptions?: {
baseUrl?: string;
jsx?: string;
jsxImportSource?: string;
lib?: string[];
rootDir?: string;
types?: string[];
};
};
const options = tsconfig.compilerOptions ?? {};
expect(options.baseUrl).toBeUndefined();
expect(options.jsx).toBe('react-jsx');
expect(options.jsxImportSource).toBe('react');
expect(options.lib).toEqual(expect.arrayContaining(['ESNext', 'DOM', 'DOM.Iterable']));
expect(options.rootDir).toBe('./src');
expect(options.types).toEqual(['vite/client']);
});
it('quality typecheck-web can finish a cold web install instead of canceling before cache save', () => {
expect(qualityJobs['typecheck-web']?.['timeout-minutes']).toBe(15);
});

View file

@ -4,8 +4,14 @@ import os from 'node:os';
import path from 'node:path';
import { fileURLToPath } from 'node:url';
import { Command, Option } from 'commander';
import * as ts from 'typescript';
import * as t from '@babel/types';
import { afterEach, describe, expect, it } from 'vitest';
import {
forEachChild,
parseTypeScript,
staticMemberName,
staticStringValue,
} from '../helpers/parse-typescript-source.js';
import { CLI_SPAWN_PREFIX } from '../helpers/cli-entry.js';
import { localizeCliHelp } from '../../src/cli/help-i18n.js';
import { setCliLanguage, type SupportedCliLanguage } from '../../src/cli/i18n/index.js';
@ -69,17 +75,6 @@ const allHelpCommands = [
['group', 'contracts'],
];
function staticStringValue(node: ts.Node | undefined): string | undefined {
if (!node) return undefined;
if (ts.isStringLiteralLike(node) || ts.isNoSubstitutionTemplateLiteral(node)) return node.text;
if (ts.isBinaryExpression(node) && node.operatorToken.kind === ts.SyntaxKind.PlusToken) {
const left = staticStringValue(node.left);
const right = staticStringValue(node.right);
if (left !== undefined && right !== undefined) return `${left}${right}`;
}
return undefined;
}
function extractRegisteredHelpDescriptions(): string[] {
const descriptions = new Set<string>();
const sourceFiles = ['src/cli/index.ts', 'src/cli/group.ts'];
@ -87,27 +82,30 @@ function extractRegisteredHelpDescriptions(): string[] {
for (const relativePath of sourceFiles) {
const filePath = path.join(repoRoot, relativePath);
const source = fs.readFileSync(filePath, 'utf8');
const sourceFile = ts.createSourceFile(filePath, source, ts.ScriptTarget.Latest, true);
const { ast } = parseTypeScript(filePath, source);
function visit(node: ts.Node): void {
if (ts.isCallExpression(node) && ts.isPropertyAccessExpression(node.expression)) {
const method = node.expression.name.text;
const description =
method === 'description'
? staticStringValue(node.arguments[0])
: method === 'option' || method === 'requiredOption'
? staticStringValue(node.arguments[1])
: undefined;
function visit(node: t.Node): void {
if (
t.isCallExpression(node) &&
(t.isMemberExpression(node.callee) || t.isOptionalMemberExpression(node.callee))
) {
const method = staticMemberName(node.callee);
let description: string | undefined;
if (method === 'description') {
description = staticStringValue(node.arguments[0]);
} else if (method === 'option' || method === 'requiredOption') {
description = staticStringValue(node.arguments[1]);
}
if (description && /[A-Za-z]/.test(description)) {
descriptions.add(description.replace(/\s+/g, ' ').trim());
}
}
ts.forEachChild(node, visit);
forEachChild(node, visit);
}
visit(sourceFile);
visit(ast);
}
return [...descriptions].filter((description) => description.length > 0).sort();
@ -198,10 +196,13 @@ describe('CLI help surface', () => {
expect(result.stdout).toContain('外部索引根目录');
expect(result.stdout).toContain('GITNEXUS_CONTENT_RETENTION=full');
expect(result.stdout).toContain('源码文本保留策略');
expect(result.stdout).toContain('当参数和对应环境变量同时提供时,参数优先。');
expect(result.stdout).toContain(
'CLI 参数优先于 `.gitnexusrc`,后者优先于环境变量,环境变量优先于内置默认值。',
);
expect(result.stdout).toContain('提示:`.gitnexusignore` 支持 `.gitignore` 风格的取反。');
expect(result.stdout).not.toContain('Environment variables:');
expect(result.stdout).not.toContain('Flags override the corresponding env vars');
expect(result.stdout).not.toContain('当参数和对应环境变量同时提供时,参数优先。');
});
it('analyze help documents the external storage root layout', () => {
@ -215,6 +216,10 @@ describe('CLI help surface', () => {
expect(result.stdout).toContain('GITNEXUS_CONTENT_RETENTION=full');
expect(result.stdout).toContain('Source-text retention profile');
expect(result.stdout).toContain('<repo-basename>-<canonical-path-hash>/');
expect(result.stdout).toContain(
'CLI flags take precedence over `.gitnexusrc`, which takes precedence over env vars, which take precedence over built-in defaults.',
);
expect(result.stdout).not.toContain('Flags override the corresponding env vars');
});
it('query help keeps advanced search options without importing analyze deps', () => {

View file

@ -0,0 +1,274 @@
import { existsSync, statSync, writeFileSync } from 'node:fs';
import os from 'node:os';
import path from 'node:path';
import { afterEach, describe, expect, it, vi } from 'vitest';
import {
abortCachedEmbeddingsBuilder,
cacheRowCount,
createCachedEmbeddingsBuilder,
DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT,
discardLiveEmbeddingSpills,
discardScopedEmbeddingSpills,
disposeEmbeddingSpill,
EmbeddingSpillReader,
finalizeCachedEmbeddingsSnapshot,
ingestCachedEmbeddingRow,
materializeCachedEmbeddings,
normalizeCachedEmbeddings,
readSpillVectors,
resolveEmbeddingCacheInMemoryRowLimit,
snapshotEmbeddingDims,
withEmbeddingSpillScope,
} from '../../src/core/embeddings/embedding-restore-spill.js';
const DIMS = 8;
function vector(fill: number): number[] {
return Array.from({ length: DIMS }, () => fill);
}
function row(id: string, fill: number, hash = `hash-${id}`) {
return {
nodeId: id,
chunkIndex: 0,
startLine: 1,
endLine: 2,
embedding: vector(fill),
contentHash: hash,
};
}
describe('embedding-restore-spill (#3306)', () => {
const spills: Array<{ path: string }> = [];
afterEach(() => {
for (const spill of spills) disposeEmbeddingSpill(spill);
spills.length = 0;
});
it('keeps small tables in RAM and does not leave a spill file', () => {
const builder = createCachedEmbeddingsBuilder({
inMemoryRowLimit: 4,
spillDir: os.tmpdir(),
});
ingestCachedEmbeddingRow(builder, row('n1', 0.25), true);
ingestCachedEmbeddingRow(builder, row('n2', 0.5), true);
const snapshot = finalizeCachedEmbeddingsSnapshot(builder);
expect(snapshot.spill).toBeUndefined();
expect(existsSync(builder.writer.path)).toBe(false);
expect(snapshot.embeddings).toHaveLength(2);
expect(snapshot.rows).toHaveLength(2);
expect(snapshot.embeddings[0]?.embedding[0]).toBeCloseTo(0.25);
expect(cacheRowCount(snapshot)).toBe(2);
expect(snapshotEmbeddingDims(snapshot)).toBe(DIMS);
});
it('spills vectors once the in-memory limit is exceeded and materializes a subset', () => {
const builder = createCachedEmbeddingsBuilder({
inMemoryRowLimit: 2,
spillDir: os.tmpdir(),
});
for (let i = 0; i < 5; i++) {
ingestCachedEmbeddingRow(builder, row(`n${i}`, i + 1), true);
}
const snapshot = finalizeCachedEmbeddingsSnapshot(builder);
if (snapshot.spill) spills.push(snapshot.spill);
expect(snapshot.embeddings).toEqual([]);
expect(snapshot.spill?.rowCount).toBe(5);
expect(existsSync(snapshot.spill!.path)).toBe(true);
expect(snapshot.embeddingNodeIds.size).toBe(5);
const subset = materializeCachedEmbeddings(snapshot, snapshot.rows.slice(1, 3));
expect(subset).toHaveLength(2);
expect(subset[0]?.nodeId).toBe('n1');
expect(subset[0]?.embedding[0]).toBeCloseTo(2);
expect(subset[1]?.embedding[0]).toBeCloseTo(3);
});
it('always spills when the in-memory limit is 0 (no Number[] table in RAM)', () => {
const builder = createCachedEmbeddingsBuilder({
inMemoryRowLimit: 0,
spillDir: os.tmpdir(),
});
ingestCachedEmbeddingRow(builder, row('only', 0.75), true);
const snapshot = finalizeCachedEmbeddingsSnapshot(builder);
if (snapshot.spill) spills.push(snapshot.spill);
expect(snapshot.embeddings).toEqual([]);
expect(snapshot.rows).toHaveLength(1);
expect(materializeCachedEmbeddings(snapshot, snapshot.rows)[0]?.embedding[0]).toBeCloseTo(0.75);
});
it('aborts an unfinished builder without leaking a spill file', () => {
const builder = createCachedEmbeddingsBuilder({
inMemoryRowLimit: 0,
spillDir: os.tmpdir(),
});
ingestCachedEmbeddingRow(builder, row('n1', 1), true);
expect(existsSync(builder.writer.path)).toBe(true);
abortCachedEmbeddingsBuilder(builder);
expect(existsSync(builder.writer.path)).toBe(false);
});
it('normalizes mock {embeddings} payloads so Phase 3.5 can restore without a spill', () => {
const snapshot = normalizeCachedEmbeddings({
embeddingNodeIds: new Set(['Function:a:foo']),
embeddings: [
{
nodeId: 'Function:a:foo',
chunkIndex: 0,
startLine: 0,
endLine: 3,
embedding: vector(0.1),
contentHash: 'stub',
},
],
});
expect(snapshot.rows).toHaveLength(1);
expect(snapshot.rows[0]?.vectorIndex).toBe(0);
const restored = materializeCachedEmbeddings(snapshot, snapshot.rows);
expect(restored[0]?.contentHash).toBe('stub');
expect(restored[0]?.embedding).toHaveLength(DIMS);
});
it('defaults the in-memory row limit to 2048 and honors GITNEXUS_EMBEDDING_CACHE_IN_MEMORY_LIMIT', () => {
expect(DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT).toBe(2048);
vi.stubEnv('GITNEXUS_EMBEDDING_CACHE_IN_MEMORY_LIMIT', '');
expect(resolveEmbeddingCacheInMemoryRowLimit()).toBe(2048);
vi.stubEnv('GITNEXUS_EMBEDDING_CACHE_IN_MEMORY_LIMIT', '0');
expect(resolveEmbeddingCacheInMemoryRowLimit()).toBe(0);
vi.stubEnv('GITNEXUS_EMBEDDING_CACHE_IN_MEMORY_LIMIT', '12');
expect(resolveEmbeddingCacheInMemoryRowLimit()).toBe(12);
vi.stubEnv('GITNEXUS_EMBEDDING_CACHE_IN_MEMORY_LIMIT', 'nope');
expect(resolveEmbeddingCacheInMemoryRowLimit()).toBe(2048);
vi.unstubAllEnvs();
expect(resolveEmbeddingCacheInMemoryRowLimit(7)).toBe(7);
expect(resolveEmbeddingCacheInMemoryRowLimit(-1)).toBe(2048);
});
it('spills above the default 2048-row limit with header-plus-body size 12 + N * D * 4', () => {
const n = DEFAULT_EMBEDDING_CACHE_IN_MEMORY_ROW_LIMIT + 1;
const builder = createCachedEmbeddingsBuilder({ spillDir: os.tmpdir() });
for (let i = 0; i < n; i++) {
ingestCachedEmbeddingRow(builder, row(`n${i}`, 1), true);
}
const snapshot = finalizeCachedEmbeddingsSnapshot(builder);
if (snapshot.spill) spills.push(snapshot.spill);
expect(snapshot.embeddings).toEqual([]);
expect(snapshot.spill?.rowCount).toBe(n);
expect(statSync(snapshot.spill!.path).size).toBe(12 + n * DIMS * 4);
});
it('rejects a dim mismatch once spilling and rejects a short or bad-magic header', () => {
const builder = createCachedEmbeddingsBuilder({
inMemoryRowLimit: 0,
spillDir: os.tmpdir(),
});
ingestCachedEmbeddingRow(builder, row('a', 1), true);
expect(() =>
ingestCachedEmbeddingRow(builder, { ...row('b', 2), embedding: [1, 2, 3] }, true),
).toThrow(/dim mismatch/);
abortCachedEmbeddingsBuilder(builder);
const shortPath = path.join(os.tmpdir(), `gitnexus-embed-restore-short-${process.pid}.bin`);
writeFileSync(shortPath, Buffer.from('NOPE'));
spills.push({ path: shortPath });
expect(() => readSpillVectors({ path: shortPath, dims: DIMS, rowCount: 1 }, [0])).toThrow(
/invalid embedding spill header/,
);
const badMagic = Buffer.alloc(12);
badMagic.write('NOPE', 0, 4, 'ascii');
badMagic.writeUInt8(1, 4);
badMagic.writeUInt32LE(DIMS, 5);
const badPath = path.join(os.tmpdir(), `gitnexus-embed-restore-bad-${process.pid}.bin`);
writeFileSync(badPath, badMagic);
spills.push({ path: badPath });
expect(() => readSpillVectors({ path: badPath, dims: DIMS, rowCount: 1 }, [0])).toThrow(
/invalid embedding spill header/,
);
});
it('reuses an open spill reader across materialize batches', () => {
const builder = createCachedEmbeddingsBuilder({
inMemoryRowLimit: 0,
spillDir: os.tmpdir(),
});
for (let i = 0; i < 4; i++) {
ingestCachedEmbeddingRow(builder, row(`n${i}`, i + 1), true);
}
const snapshot = finalizeCachedEmbeddingsSnapshot(builder);
if (snapshot.spill) spills.push(snapshot.spill);
const reader = new EmbeddingSpillReader(snapshot.spill!);
try {
const first = materializeCachedEmbeddings(snapshot, snapshot.rows.slice(0, 2), reader);
const second = materializeCachedEmbeddings(snapshot, snapshot.rows.slice(2, 4), reader);
expect(first[0]?.embedding[0]).toBeCloseTo(1);
expect(second[1]?.embedding[0]).toBeCloseTo(4);
} finally {
reader.close();
}
});
it('scoped discard unlinks only spills created in that analyze run', async () => {
const other = createCachedEmbeddingsBuilder({
inMemoryRowLimit: 0,
spillDir: os.tmpdir(),
});
ingestCachedEmbeddingRow(other, row('other', 1), true);
const otherSnapshot = finalizeCachedEmbeddingsSnapshot(other);
if (otherSnapshot.spill) spills.push(otherSnapshot.spill);
expect(existsSync(otherSnapshot.spill!.path)).toBe(true);
await withEmbeddingSpillScope(async () => {
const builder = createCachedEmbeddingsBuilder({
inMemoryRowLimit: 0,
spillDir: os.tmpdir(),
});
ingestCachedEmbeddingRow(builder, row('scoped', 2), true);
const snapshot = finalizeCachedEmbeddingsSnapshot(builder);
expect(existsSync(snapshot.spill!.path)).toBe(true);
discardScopedEmbeddingSpills();
expect(existsSync(snapshot.spill!.path)).toBe(false);
expect(existsSync(otherSnapshot.spill!.path)).toBe(true);
});
});
it('unlinks a finished spill that was not disposed', () => {
const builder = createCachedEmbeddingsBuilder({
inMemoryRowLimit: 0,
spillDir: os.tmpdir(),
});
ingestCachedEmbeddingRow(builder, row('n1', 1), true);
const snapshot = finalizeCachedEmbeddingsSnapshot(builder);
expect(snapshot.spill).toBeDefined();
expect(existsSync(snapshot.spill!.path)).toBe(true);
discardLiveEmbeddingSpills();
expect(existsSync(snapshot.spill!.path)).toBe(false);
});
it('throws when materializing a meta row with no matching vector', () => {
const snapshot = normalizeCachedEmbeddings({
embeddings: [
{
nodeId: 'Function:a:foo',
chunkIndex: 0,
startLine: 0,
endLine: 3,
embedding: vector(0.1),
contentHash: 'stub',
},
],
});
expect(() =>
materializeCachedEmbeddings(snapshot, [
{
nodeId: 'Function:missing:bar',
chunkIndex: 0,
startLine: 0,
endLine: 1,
contentHash: 'x',
vectorIndex: 99,
},
]),
).toThrow(/missing cached embedding Function:missing:bar:0/);
});
});

View file

@ -1,5 +1,10 @@
import { describe, expect, it } from 'vitest';
import { InvalidBranchError, validateBranchName } from '../../src/core/git-ref.js';
import {
InvalidBranchError,
formatRejectedBranchForLog,
sanitizeDetectedBranch,
validateBranchName,
} from '../../src/core/git-ref.js';
describe('core/git-ref', () => {
it('throws InvalidBranchError with name "InvalidBranchError"', () => {
@ -12,4 +17,46 @@ describe('core/git-ref', () => {
expect((err as Error).name).toBe('InvalidBranchError');
}
});
it('sanitizeDetectedBranch returns the trimmed name for a legal branch', () => {
expect(sanitizeDetectedBranch('develop')).toBe('develop');
expect(sanitizeDetectedBranch(' feature/foo-bar ')).toBe('feature/foo-bar');
});
it('sanitizeDetectedBranch returns undefined for null, empty, or whitespace', () => {
expect(sanitizeDetectedBranch(null)).toBeUndefined();
expect(sanitizeDetectedBranch(undefined)).toBeUndefined();
expect(sanitizeDetectedBranch('')).toBeUndefined();
expect(sanitizeDetectedBranch(' ')).toBeUndefined();
});
it('sanitizeDetectedBranch swallows InvalidBranchError and does not throw', () => {
expect(sanitizeDetectedBranch('feat`x')).toBeUndefined();
expect(sanitizeDetectedBranch('main`evil')).toBeUndefined();
expect(sanitizeDetectedBranch('HEAD')).toBeUndefined();
expect(() => sanitizeDetectedBranch('feat`x')).not.toThrow();
});
it('formatRejectedBranchForLog keeps backticks visible and escapes bidi, quotes, and line separators', () => {
expect(formatRejectedBranchForLog('feat`x')).toBe('feat`x');
expect(formatRejectedBranchForLog('a"b')).toBe('a\\"b');
expect(formatRejectedBranchForLog(`ok${'\u202e'}bad`)).toBe('ok\\u202ebad');
expect(formatRejectedBranchForLog(`zw${'\u200b'}sp`)).toBe('zw\\u200bsp');
expect(formatRejectedBranchForLog(`foo${'\u2028'}bar`)).toBe('foo\\u2028bar');
expect(formatRejectedBranchForLog(`foo${'\u2029'}bar`)).toBe('foo\\u2029bar');
});
it('formatRejectedBranchForLog escapes NBSP, NEL, and other C1 as \\uXXXX', () => {
expect(formatRejectedBranchForLog(`foo${'\u00a0'}bar`)).toBe('foo\\u00a0bar');
const nelBacktick = formatRejectedBranchForLog(`feat${'\u0085'}\``);
expect(nelBacktick).toContain('\\u0085');
expect(nelBacktick).toContain('`');
expect(nelBacktick).not.toContain('\u0085');
expect(formatRejectedBranchForLog(`x${'\u009f'}y`)).toBe('x\\u009fy');
});
it('sanitizeDetectedBranch rejects git-legal U+2028/U+2029 as whitespace', () => {
expect(sanitizeDetectedBranch(`foo${'\u2028'}bar`)).toBeUndefined();
expect(sanitizeDetectedBranch(`foo${'\u2029'}bar`)).toBeUndefined();
});
});

View file

@ -4,7 +4,16 @@ import * as fs from 'node:fs';
import * as path from 'node:path';
import * as os from 'node:os';
import { fileURLToPath } from 'node:url';
import ts from 'typescript';
import * as t from '@babel/types';
import {
type AstNode,
collectDescendants,
lineAt,
nodeStart,
nodeText,
parseTypeScript,
staticMemberName,
} from '../../helpers/parse-typescript-source.js';
import type {
ContractRegistry,
ExtractedContract,
@ -182,65 +191,63 @@ describe('syncGroup when one extractor fails partway through a repo', () => {
*/
const SYNC_SOURCE_PATH = fileURLToPath(new URL('../../../src/core/group/sync.ts', import.meta.url));
/** Every node under `node`, in source order. No branching, so nothing is skippable. */
function descendants(node: ts.Node): ts.Node[] {
const out: ts.Node[] = [];
const visit = (n: ts.Node): void => {
out.push(n);
n.forEachChild(visit);
};
node.forEachChild(visit);
return out;
}
/** `const <name>: StoredContract[] = []` — the per-repo staging buffer. */
function isStagingBufferDeclaration(node: ts.Node): node is ts.VariableDeclaration {
function isStagingBufferDeclaration(node: t.Node): node is t.VariableDeclarator {
if (!t.isVariableDeclarator(node) || !t.isIdentifier(node.id)) return false;
const annotation = node.id.typeAnnotation;
if (
!annotation ||
!t.isTSTypeAnnotation(annotation) ||
!t.isTSArrayType(annotation.typeAnnotation)
) {
return false;
}
const elementType = annotation.typeAnnotation.elementType;
return (
ts.isVariableDeclaration(node) &&
node.type !== undefined &&
ts.isArrayTypeNode(node.type) &&
ts.isTypeReferenceNode(node.type.elementType) &&
ts.isIdentifier(node.type.elementType.typeName) &&
node.type.elementType.typeName.text === 'StoredContract' &&
node.initializer !== undefined &&
ts.isArrayLiteralExpression(node.initializer) &&
node.initializer.elements.length === 0 &&
ts.isVariableDeclarationList(node.parent) &&
(node.parent.flags & ts.NodeFlags.Const) !== 0
t.isTSTypeReference(elementType) &&
t.isIdentifier(elementType.typeName) &&
elementType.typeName.name === 'StoredContract' &&
node.init !== undefined &&
node.init !== null &&
t.isArrayExpression(node.init) &&
node.init.elements.length === 0 &&
t.isVariableDeclaration((node as AstNode).parent) &&
((node as AstNode).parent as t.VariableDeclaration).kind === 'const'
);
}
/** `x.apply(dest, args)` — an argument-limited append in non-spread clothing. */
function isApplyCall(call: ts.CallExpression): boolean {
return ts.isPropertyAccessExpression(call.expression) && call.expression.name.text === 'apply';
function isApplyCall(call: t.CallExpression): boolean {
return (
(t.isMemberExpression(call.callee) || t.isOptionalMemberExpression(call.callee)) &&
staticMemberName(call.callee) === 'apply'
);
}
function describeCall(sourceFile: ts.SourceFile, call: ts.CallExpression): string {
const { line } = sourceFile.getLineAndCharacterOfPosition(call.getStart(sourceFile));
return `${line + 1}: ${call.getText(sourceFile).replace(/\s+/g, ' ')}`;
function describeCall(source: string, call: t.CallExpression): string {
const line = lineAt(source, nodeStart(call));
return `${line}: ${nodeText(source, call).replace(/\s+/g, ' ')}`;
}
describe('the per-repo staging append in sync.ts', () => {
it('appends the staged contracts without spreading them into a call', () => {
const source = fs.readFileSync(SYNC_SOURCE_PATH, 'utf-8');
const sourceFile = ts.createSourceFile(
SYNC_SOURCE_PATH,
source,
ts.ScriptTarget.Latest,
true,
ts.ScriptKind.TS,
);
const { ast } = parseTypeScript(SYNC_SOURCE_PATH, source);
const allNodes = descendants(sourceFile);
const allNodes = collectDescendants(ast);
const stagingBuffers = allNodes.filter(isStagingBufferDeclaration);
// One staging buffer, or this gate no longer knows which code it guards.
expect(stagingBuffers.map((d) => d.name.getText(sourceFile))).toHaveLength(1);
const stagingNames = stagingBuffers.map((d) => d.name.getText(sourceFile));
const stagingNames = stagingBuffers.map((d) =>
t.isIdentifier(d.id) ? d.id.name : nodeText(source, d.id),
);
expect(stagingNames).toHaveLength(1);
// The block the buffer is declared in — the per-repo loop body.
// VariableDeclarator → VariableDeclaration → BlockStatement (Babel has no
// extra VariableStatement wrapper).
const declaringBlocks = stagingBuffers
.map((d) => d.parent.parent.parent) // declaration → list → statement → block
.filter(ts.isBlock);
.map((d) => (d as AstNode).parent?.parent)
.filter((node): node is t.BlockStatement => t.isBlockStatement(node));
expect(declaringBlocks).toHaveLength(1);
// The extractor try-block: a DIRECT statement of that block whose `try` reads
@ -249,22 +256,22 @@ describe('the per-repo staging append in sync.ts', () => {
// ancestor reads the buffer too. Widening to "any try that mentions it" pulls
// in the entire function body, manifest-window spreads and all.
const extractorTryBlocks = declaringBlocks.flatMap((block) =>
block.statements
.filter(ts.isTryStatement)
block.body
.filter((statement): statement is t.TryStatement => t.isTryStatement(statement))
.filter((statement) =>
descendants(statement.tryBlock).some(
(n) => ts.isIdentifier(n) && stagingNames.includes(n.text),
collectDescendants(statement.block).some(
(n) => t.isIdentifier(n) && stagingNames.includes(n.name),
),
)
.map((statement) => statement.tryBlock),
.map((statement) => statement.block),
);
expect(extractorTryBlocks).toHaveLength(1);
const unboundedAppends = extractorTryBlocks.flatMap((block) =>
descendants(block)
.filter(ts.isCallExpression)
.filter((call) => call.arguments.some(ts.isSpreadElement) || isApplyCall(call))
.map((call) => describeCall(sourceFile, call)),
collectDescendants(block)
.filter((node): node is t.CallExpression => t.isCallExpression(node))
.filter((call) => call.arguments.some((arg) => t.isSpreadElement(arg)) || isApplyCall(call))
.map((call) => describeCall(source, call)),
);
// Every staged contract must reach `autoContracts` through a bounded loop:

View file

@ -485,6 +485,31 @@ describe('swiftPackageStrategy', () => {
const result = swiftPackageStrategy('Foundation', 'App.swift', ctx);
expect(result).toBeNull();
});
it('does not resolve inferred grouping folders when the declaration is empty', () => {
const ctx = makeCtx(['Sources/Foundation/Thing.swift', 'Sources/App/main.swift'], {
swiftPackageConfig: {
origin: 'package.swift',
targets: new Map([
['Foundation', 'Sources/Foundation'],
['App', 'Sources/App'],
]),
declaredTargets: new Map(),
},
});
expect(swiftPackageStrategy('Foundation', 'Sources/App/main.swift', ctx)).toBeNull();
expect(swiftPackageStrategy('App', 'Sources/App/main.swift', ctx)).toBeNull();
});
it('ignores an inferred directories origin', () => {
const ctx = makeCtx(['Sources/Models/User.swift'], {
swiftPackageConfig: {
origin: 'directories',
targets: new Map([['Models', 'Sources/Models']]),
},
});
expect(swiftPackageStrategy('Models', 'Sources/App/main.swift', ctx)).toBeNull();
});
});
describe('rubyRequireStrategy', () => {

View file

@ -19,7 +19,7 @@
*/
import { writeFile, readFile } from 'fs/promises';
import { describe, it, expect } from 'vitest';
import { afterEach, describe, expect, it, vi } from 'vitest';
import {
getStoragePaths,
saveMeta,
@ -35,92 +35,109 @@ import { seedEmbeddingsForFiles } from '../helpers/embedding-seed.js';
const setupMiniRepo = () => setupSharedMiniRepo('gitnexus-incr-dirty-rec-');
async function parkCrashSidecarsAndRebuild(options?: { alwaysSpill?: boolean }) {
vi.stubEnv('GITNEXUS_WORKER_READY_TIMEOUT_MS', '60000');
if (options?.alwaysSpill) {
vi.stubEnv('GITNEXUS_EMBEDDING_CACHE_IN_MEMORY_LIMIT', '0');
}
const repo = await setupMiniRepo();
try {
const { runFullAnalysis } = await import('../../src/core/run-analyze.js');
await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} });
// Seed real embeddings BEFORE the tamper (tri-review 4669518496 / U5):
// with meta.stats.embeddings = 0 the recovery run derived
// shouldLoadCache=false and never opened the DB pre-wipe — this test
// was vacuous about the exact open the parking protects. Seeded rows +
// a stats stamp route the recovery (which runs force:true internally,
// so forceRegenerate → shouldLoadCache) through the REAL
// embedding-cache preservation open on the just-parked DB.
const { storagePath, lbugPath } = getStoragePaths(repo.dbPath);
const seededIdsByFile = await seedEmbeddingsForFiles(
repo.dbPath,
['src/handler.ts', 'src/logger.ts'],
1,
);
const seededNodeIds = [...seededIdsByFile.values()].flat();
expect(seededNodeIds.length).toBeGreaterThan(0);
// Simulate a crashed incremental writeback: dirty flag in meta plus
// leftover sidecars whose bytes must never be replayed. 8KB puts the
// WAL above the tiny-orphan threshold — the state the sidecar
// preflight deliberately leaves in place for engine replay.
const meta = await loadMeta(storagePath);
const tampered: RepoMeta = {
...meta!,
stats: { ...meta!.stats, embeddings: seededNodeIds.length },
incrementalInProgress: {
startedAt: Date.now() - 60_000,
toWriteCount: 12,
phase: 'load-graph',
},
};
await saveMeta(storagePath, tampered);
const walGarbage = Buffer.alloc(8192, 0xab);
const shadowGarbage = Buffer.alloc(4096, 0xcd);
await writeFile(`${lbugPath}.wal`, walGarbage);
await writeFile(`${lbugPath}.shadow`, shadowGarbage);
const logs: string[] = [];
// embeddingsNodeLimit: 1 (KTD9): the recovery runs force:true
// internally, and the seeded stats would otherwise route Phase 4 into
// a real embedder in CI — the 1-node cap suppresses generation while
// leaving the preserve/restore path fully live. On linux the
// wipe-and-restore vector-index seam then fires for real (statically
// linked VECTOR): a CREATE_VECTOR_INDEX over the restored rows is
// expected and harmless here.
const recovered = await runFullAnalysis(
repo.dbPath,
{ skipAgentsMd: true, embeddingsNodeLimit: 1 },
{ onProgress: () => {}, onLog: (m) => logs.push(m) },
);
expect(recovered.alreadyUpToDate).toBeUndefined();
// Both sidecars were parked verbatim (renamed, never deleted) before
// any open could replay them…
expect(Buffer.compare(await readFile(`${lbugPath}.wal.dirty-recovery`), walGarbage)).toBe(0);
expect(Buffer.compare(await readFile(`${lbugPath}.shadow.dirty-recovery`), shadowGarbage)).toBe(
0,
);
const joinedLogs = logs.join('\n');
expect(joinedLogs).toContain('Parked lbug.wal.dirty-recovery, lbug.shadow.dirty-recovery');
// …the run traversed the REAL pre-wipe preservation open — recovery's
// internal force on an embedded repo upgrades to regenerate mode, whose
// banner only prints when existingEmbeddingCount was read from the
// seeded stats and the cache-load path engaged…
expect(joinedLogs).toContain(
`--force on a repo with ${seededNodeIds.length} existing embeddings`,
);
// …with generation itself cap-suppressed (no embedder in CI):
expect(joinedLogs).toContain('exceeds the 1-node safety cap');
// …and the rebuild completed into a clean index: dirty flag cleared,
// and the seeded embeddings survived the park → open → wipe → restore
// round-trip (the strongest signal the preservation open really ran:
// the DB was wiped, so these rows can only come from the cache load).
const after = await loadMeta(storagePath);
expect(after!.incrementalInProgress).toBeUndefined();
expect(after!.stats?.embeddings).toBe(seededNodeIds.length);
} finally {
vi.unstubAllEnvs();
await repo.cleanup();
}
}
describe('runFullAnalysis — dirty-flag recovery sidecar parking (#2409)', () => {
afterEach(() => {
vi.unstubAllEnvs();
});
it('parks the crashed run WAL/shadow sidecars before reopening, then rebuilds clean', async () => {
const repo = await setupMiniRepo();
try {
const { runFullAnalysis } = await import('../../src/core/run-analyze.js');
await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} });
await parkCrashSidecarsAndRebuild();
}, 300_000);
// Seed real embeddings BEFORE the tamper (tri-review 4669518496 / U5):
// with meta.stats.embeddings = 0 the recovery run derived
// shouldLoadCache=false and never opened the DB pre-wipe — this test
// was vacuous about the exact open the parking protects. Seeded rows +
// a stats stamp route the recovery (which runs force:true internally,
// so forceRegenerate → shouldLoadCache) through the REAL
// embedding-cache preservation open on the just-parked DB.
const { storagePath, lbugPath } = getStoragePaths(repo.dbPath);
const seededIdsByFile = await seedEmbeddingsForFiles(
repo.dbPath,
['src/handler.ts', 'src/logger.ts'],
1,
);
const seededNodeIds = [...seededIdsByFile.values()].flat();
expect(seededNodeIds.length).toBeGreaterThan(0);
// Simulate a crashed incremental writeback: dirty flag in meta plus
// leftover sidecars whose bytes must never be replayed. 8KB puts the
// WAL above the tiny-orphan threshold — the state the sidecar
// preflight deliberately leaves in place for engine replay.
const meta = await loadMeta(storagePath);
const tampered: RepoMeta = {
...meta!,
stats: { ...meta!.stats, embeddings: seededNodeIds.length },
incrementalInProgress: {
startedAt: Date.now() - 60_000,
toWriteCount: 12,
phase: 'load-graph',
},
};
await saveMeta(storagePath, tampered);
const walGarbage = Buffer.alloc(8192, 0xab);
const shadowGarbage = Buffer.alloc(4096, 0xcd);
await writeFile(`${lbugPath}.wal`, walGarbage);
await writeFile(`${lbugPath}.shadow`, shadowGarbage);
const logs: string[] = [];
// embeddingsNodeLimit: 1 (KTD9): the recovery runs force:true
// internally, and the seeded stats would otherwise route Phase 4 into
// a real embedder in CI — the 1-node cap suppresses generation while
// leaving the preserve/restore path fully live. On linux the
// wipe-and-restore vector-index seam then fires for real (statically
// linked VECTOR): a CREATE_VECTOR_INDEX over the restored rows is
// expected and harmless here.
const recovered = await runFullAnalysis(
repo.dbPath,
{ skipAgentsMd: true, embeddingsNodeLimit: 1 },
{ onProgress: () => {}, onLog: (m) => logs.push(m) },
);
expect(recovered.alreadyUpToDate).toBeUndefined();
// Both sidecars were parked verbatim (renamed, never deleted) before
// any open could replay them…
expect(Buffer.compare(await readFile(`${lbugPath}.wal.dirty-recovery`), walGarbage)).toBe(0);
expect(
Buffer.compare(await readFile(`${lbugPath}.shadow.dirty-recovery`), shadowGarbage),
).toBe(0);
const joinedLogs = logs.join('\n');
expect(joinedLogs).toContain('Parked lbug.wal.dirty-recovery, lbug.shadow.dirty-recovery');
// …the run traversed the REAL pre-wipe preservation open — recovery's
// internal force on an embedded repo upgrades to regenerate mode, whose
// banner only prints when existingEmbeddingCount was read from the
// seeded stats and the cache-load path engaged…
expect(joinedLogs).toContain(
`--force on a repo with ${seededNodeIds.length} existing embeddings`,
);
// …with generation itself cap-suppressed (no embedder in CI):
expect(joinedLogs).toContain('exceeds the 1-node safety cap');
// …and the rebuild completed into a clean index: dirty flag cleared,
// and the seeded embeddings survived the park → open → wipe → restore
// round-trip (the strongest signal the preservation open really ran:
// the DB was wiped, so these rows can only come from the cache load).
const after = await loadMeta(storagePath);
expect(after!.incrementalInProgress).toBeUndefined();
expect(after!.stats?.embeddings).toBe(seededNodeIds.length);
} finally {
await repo.cleanup();
}
it('restores seeded embeddings from a spill after wipe when the in-memory limit is 0', async () => {
await parkCrashSidecarsAndRebuild({ alwaysSpill: true });
}, 300_000);
});

View file

@ -279,8 +279,10 @@ describe('PARSE_CACHE_VERSION', () => {
// Moved 99 -> 100 for #3253: retain absolute Rust import qualifiers.
// Moved 100 -> 101 for #3294 review: retain keyword glob paths and distinguish
// restricted pub(...) imports from unrestricted reexports.
it('pins SCHEMA_BUMP to 101 so concurrent bumps cannot silently collide (#2766, #3015, #3088, #2885, #3128, #2865, #3130, #1432, #3161, #3179, #3219, #3190, #3253)', () => {
expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(101);
// Moved 101 -> 102 for #3273: preserve exact call-result assignment facts
// required by post-resolution Swift return-type replay.
it('pins SCHEMA_BUMP to 102 so concurrent bumps cannot silently collide (#2766, #3015, #3088, #2885, #3128, #2865, #3130, #1432, #3161, #3179, #3219, #3190, #3253, #3273)', () => {
expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(102);
expect(PARSE_CACHE_BUCKET_COUNT).toBe(128);
// The PREVIOUS version must fail the reuse gate, not merely differ from the
// current one — a hardcoded number outside the conflict hunk rebases cleanly
@ -288,7 +290,7 @@ describe('PARSE_CACHE_VERSION', () => {
// Every nearby historical or in-flight value is rejected.
for (const taken of [
59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, 81,
82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100,
82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101,
]) {
expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).not.toBe(taken);
}

Some files were not shown because too many files have changed in this diff Show more