From 28f3b99822c68c6ea42bb603512b00e736375698 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Fri, 12 Jun 2026 07:13:33 +0100 Subject: [PATCH 01/16] chore(devcontainer): simplify Dockerfile and devcontainer.json by removing version args for AI CLIs (#2174) --- .devcontainer/Dockerfile | 83 +++++++-------------------------- .devcontainer/devcontainer.json | 21 --------- .devcontainer/post-create.sh | 15 +++--- 3 files changed, 22 insertions(+), 97 deletions(-) diff --git a/.devcontainer/Dockerfile b/.devcontainer/Dockerfile index 620a5483c..c57c0d6aa 100644 --- a/.devcontainer/Dockerfile +++ b/.devcontainer/Dockerfile @@ -22,26 +22,10 @@ # --format '{{json .Manifest.Digest}}' FROM mcr.microsoft.com/devcontainers/typescript-node@sha256:7c2e711a4f7b02f32d2da16192d5e05aa7c95279be4ce889cff5df316f251c1d -# Build args. We deliberately set no version defaults here. devcontainer.json -# `build.args` is the single source of truth for versions. A standalone -# `docker build .devcontainer/` (for example, a CI smoke test) must pass each -# version with --build-arg. Without a default, the build fails loudly instead of -# silently drifting from the version pinned in devcontainer.json. -ARG CLAUDE_CODE_VERSION -ARG CODEX_VERSION -# Cursor is pinned by version plus a per-arch tarball sha256 hash. The install -# step below verifies that hash. All three values live in devcontainer.json -# build.args. They follow the same rule as the others: one source of truth, and -# no default so the build fails loudly if a value is missing. -ARG CURSOR_VERSION -ARG CURSOR_SHA256_X64 -ARG CURSOR_SHA256_ARM64 # Bun is installed via the official remote script (bun.sh/install), pinned by -# version. UNLIKE Cursor and the npm packages, this install path runs an -# UNVERIFIED remote script — there is no tarball-hash check. Chosen explicitly -# at request time over the pin-by-sha256 alternative for install-script -# simplicity. To harden later, switch to a pinned tarball + per-arch sha256 in -# the Cursor style (release artifacts at github.com/oven-sh/bun/releases). +# version. Claude Code and Cursor also use official install scripts (no version +# to pin). To harden Bun: switch to a pinned tarball + per-arch sha256 +# (release artifacts at github.com/oven-sh/bun/releases). ARG BUN_VERSION ARG TZ=UTC ARG USERNAME=node @@ -50,10 +34,7 @@ ARG USERNAME=node # read them. We deliberately do not set CLAUDE_CONFIG_DIR here. Its one true # value lives in devcontainer.json `containerEnv`, and the runtime value wins # anyway. -ENV CLAUDE_CODE_VERSION=${CLAUDE_CODE_VERSION} \ - CODEX_VERSION=${CODEX_VERSION} \ - CURSOR_VERSION=${CURSOR_VERSION} \ - BUN_VERSION=${BUN_VERSION} \ +ENV BUN_VERSION=${BUN_VERSION} \ BUN_INSTALL=/home/${USERNAME}/.bun \ TZ=${TZ} \ DEVCONTAINER=true \ @@ -86,51 +67,19 @@ RUN mkdir -p \ USER ${USERNAME} -# Install Claude Code and the Codex CLI globally, as the `node` user. The base -# image sets /usr/local/share/npm-global as the npm-global prefix and makes the -# `npm` group writable by `node`. So `npm install -g` works without sudo. Both -# versions come from build args. To upgrade, bump them in devcontainer.json and -# rebuild. -RUN npm install -g \ - @anthropic-ai/claude-code@${CLAUDE_CODE_VERSION} \ - @openai/codex@${CODEX_VERSION} +# Install Claude Code via the official native installer. Downloads the latest +# self-contained binary for the running platform and places it at +# ~/.local/bin/claude — no Node.js runtime dependency, no version to pin. +RUN curl -fsSL https://claude.ai/install.sh | bash -# Install the Cursor CLI. It is pinned and hash-verified, and we run no remote -# script. The cursor.com/install script just detects os/arch, downloads a -# versioned tarball from -# downloads.cursor.com/lab////agent-cli-package.tar.gz, -# extracts it, and symlinks `agent`/`cursor-agent` into ~/.local/bin. We do that -# ourselves against a PINNED version plus a per-arch sha256 hash. So the build -# runs no unverified remote code. This matches how we pin the base image and npm -# packages by digest (issue #1451). The download is fail-closed: if the hash -# does not match, the build aborts. -# -# To bump: set CURSOR_VERSION and both CURSOR_SHA256_* in devcontainer.json -# build.args. Get each arch's hash with: -# curl -fSL https://downloads.cursor.com/lab//linux//agent-cli-package.tar.gz | sha256sum -# -# TARGETARCH is the per-platform build arg that BuildKit sets automatically. It -# must be (re)declared in this stage to be visible. When the build is a -# non-BuildKit `docker build`, TARGETARCH is unset, so we fall back to `dpkg -# --print-architecture`. -ARG TARGETARCH -RUN set -eux; \ - arch="${TARGETARCH:-$(dpkg --print-architecture)}"; \ - case "$arch" in \ - amd64) cursor_arch=x64; cursor_sha="${CURSOR_SHA256_X64}";; \ - arm64) cursor_arch=arm64; cursor_sha="${CURSOR_SHA256_ARM64}";; \ - *) echo "unsupported architecture for Cursor: $arch" >&2; exit 1;; \ - esac; \ - url="https://downloads.cursor.com/lab/${CURSOR_VERSION}/linux/${cursor_arch}/agent-cli-package.tar.gz"; \ - curl -fSL --retry 3 --max-time 120 -o /tmp/cursor.tgz "$url"; \ - echo "${cursor_sha} /tmp/cursor.tgz" | sha256sum -c -; \ - dir="/home/${USERNAME}/.local/share/cursor-agent/versions/${CURSOR_VERSION}"; \ - install -d "$dir" "/home/${USERNAME}/.local/bin"; \ - tar --strip-components=1 -xzf /tmp/cursor.tgz -C "$dir"; \ - test -x "$dir/cursor-agent"; \ - ln -sf "$dir/cursor-agent" "/home/${USERNAME}/.local/bin/agent"; \ - ln -sf "$dir/cursor-agent" "/home/${USERNAME}/.local/bin/cursor-agent"; \ - rm -f /tmp/cursor.tgz +# Install the Codex CLI globally via npm. No version pinned — @latest at build +# time. (Codex has no native binary installer; npm is the canonical method.) +RUN npm install -g @openai/codex + +# Install the Cursor agent CLI via the official install script. Downloads the +# latest agent-cli-package for the running platform and places `cursor-agent` +# and `agent` into ~/.local/bin — no version or hash to pin. +RUN curl -fsSL https://cursor.com/install | bash # Install Bun via the official remote installer, pinned by version. The first # positional arg to `bash` is the release tag (`bun-vX.Y.Z`), so a specific diff --git a/.devcontainer/devcontainer.json b/.devcontainer/devcontainer.json index a7160c8d5..7b0f7e219 100644 --- a/.devcontainer/devcontainer.json +++ b/.devcontainer/devcontainer.json @@ -14,16 +14,6 @@ "dockerfile": "Dockerfile", "context": ".", "args": { - "CLAUDE_CODE_VERSION": "2.1.156", - "CODEX_VERSION": "0.134.0", - // Cursor: a pinned version plus one sha256 hash per CPU arch. The - // Dockerfile checks the tarball against the hash at build time, so it - // never runs a remote install script. Bump all three values together. - // Re-hash each arch with: - // curl -fSL https://downloads.cursor.com/lab//linux//agent-cli-package.tar.gz | sha256sum - "CURSOR_VERSION": "2026.05.28-a70ca7c", - "CURSOR_SHA256_X64": "7f8b6a09393e0b84b288cc6952b292fc98d15775f644cc01b0b9aa4f04b268df", - "CURSOR_SHA256_ARM64": "05a0ab361e038729aba25fe7f407531b3e8432912e499d0bffdf1dda0e7833e9", // Bun: pinned by version. Installed by the official bun.sh/install // script, which accepts the release tag as its first positional arg // (`bash -s bun-vX.Y.Z`). UNLIKE Cursor, the install path runs an @@ -324,17 +314,6 @@ // dependency explicit instead of silently following the default. "containerEnv": { "CODEX_HOME": "/home/node/.codex", - "DISABLE_AUTOUPDATER": "1", - // post-create.sh removes `installMethod` from the seeded ~/.claude.json so - // the npm-global binary detects its own install method. This is a backup - // safeguard for Claude Code issue #17289. The install-checks routine probes - // ~/.local/bin/claude just because that directory EXISTS. It does exist - // here, because Cursor drops agent and cursor-agent symlinks there. So even - // when installMethod is non-native, the routine reports a false "claude - // command not found at ~/.local/bin/claude". DISABLE_AUTOUPDATER does NOT - // turn that routine off. DISABLE_INSTALLATION_CHECKS is its dedicated kill - // switch. - "DISABLE_INSTALLATION_CHECKS": "1", "HISTFILE": "/commandhistory/.zsh_history" }, diff --git a/.devcontainer/post-create.sh b/.devcontainer/post-create.sh index 58c9f8892..fce605153 100644 --- a/.devcontainer/post-create.sh +++ b/.devcontainer/post-create.sh @@ -141,15 +141,12 @@ sync_from_host /host/.codex/config.toml /home/node/.codex/config.toml 644 # Seed $HOME/.claude.json from the host, but NOT as a straight copy. That file # mixes two kinds of state. Some is portable account and onboarding state we # want to keep: hasCompletedOnboarding, oauthAccount, userID, projects, -# tipsHistory. The rest describes how Claude is installed on the host, and that -# part is never valid here. This image installs Claude with `npm install -g`, -# but the host's `installMethod` (for example "native") makes Claude look for -# ~/.local/bin/claude and fail with -# "claude command not found at /home/node/.local/bin/claude". The fix strips the -# machine-specific fields and forces hasCompletedOnboarding, while handling a -# host file that isn't a JSON object. That logic lives in seed-claude-config.cjs -# so it can be unit-tested and prettier-checked -# (translate-plugin-registries.test.cjs). +# tipsHistory. The rest describes how Claude is installed on the HOST, and that +# part is never valid here — for example the host's `installMethod` value only +# makes sense for the host's binary. The fix strips the machine-specific fields +# and forces hasCompletedOnboarding, while handling a host file that isn't a +# JSON object. That logic lives in seed-claude-config.cjs so it can be +# unit-tested and prettier-checked (translate-plugin-registries.test.cjs). node "$SCRIPT_DIR/seed-claude-config.cjs" # Codex auth. Some hosts store credentials in the OS keyring instead of on disk From 14397dd4aa99ed35f27bcc3f97952afa05aa6552 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Fri, 12 Jun 2026 07:35:09 +0100 Subject: [PATCH 02/16] feat(taint): intra-procedural taint analysis (#2083) (#2164) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(taint): harvest occurrence-tagged call/member sites on StatementFacts (#2083 U1) Worker-side site harvest in TsHarvester: call/new/member-read records with dotted callee paths, receiver slots, per-argument occurrence tagging with nested-site links, per-declarator resultDefs, spread/template/require-literal markers. hasTaintSafeSites validation seam. The pdg parse-cache chunk-key namespace is versioned (pdg:1 -> pdg:2) instead of a global SCHEMA_BUMP so flag-off users keep warm caches; bench fingerprints re-baselined for the three call-bearing scenarios (straight-line/dense-bindings byte-unchanged). * feat(taint): built-in TS/JS source/sink/sanitizer model + site matcher (#2083 U2) Typed spec (kind taxonomy; sanitizers carry neutralizes-kinds), the canonical Express/Node model, and matchFunctionSites: ESM alias/namespace + require- literal callee resolution, bare-name fallback restricted to true globals, sanitizers module-or-global only (never user-shadowable by name), spread/ template arg-position rules, deterministic taintModelVersion. * feat(taint): pure intra-procedural taint propagation engine (#2083 U3) Two-rule model (statement-local + du-fact worklist) with per-taint neutralized-kind exclusion sets: sanitizers exclude only the sink kinds they neutralize (escape(req.body) suppresses res.send but still fires db.query; exec(path.basename(t)) fires), intersection-over-paths so a bypass occurrence keeps the taint live, kill locality on resultDefs, propagate-through args+receiver with viaCall hops, one path per finding, deterministic caps, coverage-gap statuses. Test-first: 38 scenarios on real harvested CFGs. * feat(taint): thread taint caps + model version through pdg config/meta (#2083 U5) resolvePdgConfig gains maxTaintFindingsPerFunction (200), maxTaintHops (32), and the taintModelVersion digest; RepoMeta.pdg + RunScopeResolutionInput surfaces added. The key-union comparator trips full writeback on M2->M3 upgrade and on model-version change without --force (mode-flip tested). No CLI flags or rc keys (programmatic parity with the other caps). * feat(taint): in-phase taint emit with sparse TAINTED/SANITIZES edges (#2083 U4) run.ts pdg window: match-first fast path (solver only when a function has both a matched source and sink) -> computeReachingDefs with the shared RD fact derivation -> computeTaintFlows -> per-finding TAINTED (versioned hop-encoded reason via the shared path codec, statement-level occurrence identity) + per-kill SANITIZES, dedup-before-budget, truncate-and-warn. All emit counters surfaced (aggregate warn for gaps/drops, debug for volume); PROF gains taint=. Flag-off golden untouched. * feat(mcp): explain tool for persisted taint findings (#2083 U6) Anchorless calls enumerate the sparse TAINTED table (bounded, deterministic, limit-clamped); anchored calls (file or symbol via resolveSymbolCandidates) return full decoded hop detail. sinkKind rides a version-1 codec header (1;|hops — no other persisted channel exists; U4/U6 ship together). RepoMeta.pdg probe yields a no-taint-layer note instead of an error. TAINTED/SANITIZES pinned OUT of VALID_RELATION_TYPES (KTD9a negative- membership tests); generators + canonical skill docs + mirrors updated. * test(taint): acceptance fixture battery, snapshots, and bench gates (#2083 U7) pdg-repo taint-cases fixtures complete the six plan shapes; committed findings/kills snapshot via a shared pure-path harness that also feeds the AE2 exact-equality assertion (stored TAINTED == pure-path findings, the no-explosion gate). New taint-dense bench scenario with four --check gates: per-function findings pinned AT the cap, absolute reason-byte + site-bytes disk ceilings (the load-bearing R10 gate), zero-match pass < 0.5x match- dense, N-linearity. Pre-existing scenario baselines untouched. * refactor(taint): share one pointKey helper across propagate + emit (#2083 review) Extract pointKey(ProgramPoint) to cfg/reaching-defs.ts (colon-separated, matching the codebase block:stmt id convention) and import it in both propagate.ts and emit.ts, replacing the two divergent locals (':' vs '.'). Edge-id material now uses the colon form; ids are in-memory only and no test asserts the pointKey segment shape. * fix(taint): discriminate taint state by source occurrence (#2083 review) Two distinct sources flowing into one variable at one def point no longer collapse to a single TAINTED edge: the taint-state key gains a root source-occurrence discriminator ({point, siteIndex} — the same fields recordFinding's identity uses, excluding kind). Def->use fact lookup keys on the source-independent (binding, def-point) portion. Same-source multi-path flows still share one state so their exclusion sets intersect (the raw arm soundly wins); termination holds (finite keys, monotone shrink, no cross-source ping-pong). Restores the KTD6 identity contract. * fix(mcp): route dotted symbol names in explain to symbol resolution (#2083 review) The fileish classifier matched any dotted name (UserController.create) as a file via its extension-like suffix, so symbol resolution never ran and the tool returned a silent empty file-anchored result. Tighten the classifier to require a path separator or a real source extension (derived from the resolver's EXTENSIONS list, multi-language), so dotted/bare names route to resolveSymbolCandidates (found / ambiguous / not-found). * fix(mcp): gate explain no-taint-layer note on taintModelVersion (#2083 review) An M1/M2-era --pdg index has meta.pdg defined (BasicBlock/REACHING_DEF recorded) but no taintModelVersion and zero TAINTED rows. The probe keyed on generic meta.pdg presence, so explain returned the generic empty note instead of the actionable 'no taint layer — run analyze' hint. Gate on meta.pdg?.taintModelVersion (the field M3 stamps) so an M2-era index gets the layer hint; a taint-stamped index with no findings still gets the generic note. * fix(taint): sequence-expression value flows only the final operand (#2083 review) A comma expression in value position (exec((log(x), 'safe'))) default- descended, fanning every operand's occurrences into the enclosing sink argument — over-tainting exec's arg 0 with x. Add an explicit walkValue case that records earlier operands' uses with occurrence fan-out suppressed (new FactAccumulator.suppressOccurrences) and routes only the last operand through the value path. Sites-layer only; defs/uses/mayDefs byte-identical (cfg + reaching-defs snapshots unchanged). * perf(taint): FIFO head-cursor worklist + dedup before chainHops (#2083 review) Replace queue.shift() (O(N) dequeue) with a strict-FIFO head cursor plus order-preserving prefix reclamation; FIFO is load-bearing because chainHops reads the live taints map whose parent/source/viaCall are rewritten order-sensitively on monotone shrink, so hop determinism is dequeue-order contingent. Extract findingKey() and dedup-check before chainHops in the justify branch — already-recorded identities discard their hop chain (first write wins), so the ancestry walk was pure waste. The else kill branch is untouched. Findings + hops byte-identical (snapshot unchanged). * perf(taint): O(1) member-read dedup via composite-key set (#2083 review) addMemberRead rescanned the whole per-statement sites array per call to dedup by (object, property, parent) — O(n^2) on member-read-dense statements. Track a composite-key Set alongside sites for O(1) dedup. (The require-literal join is already O(sites) with a no-op body on non-require sites, so no early-exit is needed there.) Behavior identical: harvest + model-match + taint snapshots unchanged. * refactor(taint): drop test-only export; source taint caps via emit.ts (#2083 review) Remove the sanitizerNeutralizes export (its only consumers were two test assertions — inlined to entry.neutralizes membership). Re-export the DEFAULT_PDG_MAX_TAINT_* caps from emit.ts and point run.ts at emit.ts, so the pipeline's taint dependency surface is the single orchestration module rather than reaching into propagate.ts. * test(taint): extract the shared TS CFG/taint test harness (#2083 review) The parse/collectFunctions/cfgOf/cfgsOf/importsFor harness was copied byte-for-byte across four suites (harvest, model-match, propagate, taint-emit). Promote it to test/helpers/ts-cfg-harness.ts and import it. site-safety/reaching-defs carry a structurally different inlined builder and are left as-is. Pure extraction, no assertion changes. * test(mcp): harden explain limit-rejection battery (#2083 review) Add NaN, Infinity, -Infinity, and a numeric string to the out-of-bounds limit cases — a regression fence over the interpolated LIMIT, confirming the Number.isInteger guard rejects every non-integer/non-finite/string input before it reaches the query. --- .../skills/gitnexus/gitnexus-guide/SKILL.md | 13 +- .../skills/gitnexus-guide/SKILL.md | 13 +- gitnexus/bench/cfg/baselines.json | 25 +- gitnexus/bench/cfg/measure.mjs | 184 ++++ gitnexus/skills/gitnexus-guide.md | 11 + gitnexus/src/cli/ai-context.ts | 1 + gitnexus/src/cli/skill-gen.ts | 3 + gitnexus/src/core/ingestion/cfg/emit.ts | 13 +- .../src/core/ingestion/cfg/reaching-defs.ts | 10 + gitnexus/src/core/ingestion/cfg/types.ts | 93 ++ .../cfg/visitors/typescript-harvest.ts | 454 ++++++++- gitnexus/src/core/ingestion/pipeline.ts | 16 + .../scope-resolution/pipeline/phase.ts | 2 + .../scope-resolution/pipeline/run.ts | 146 ++- gitnexus/src/core/ingestion/taint/emit.ts | 297 ++++++ gitnexus/src/core/ingestion/taint/match.ts | 374 ++++++++ .../src/core/ingestion/taint/path-codec.ts | 267 ++++++ .../src/core/ingestion/taint/propagate.ts | 880 ++++++++++++++++++ .../src/core/ingestion/taint/site-safety.ts | 103 ++ .../ingestion/taint/source-sink-config.ts | 127 ++- .../ingestion/taint/source-sink-registry.ts | 10 +- .../core/ingestion/taint/typescript-model.ts | 108 +++ gitnexus/src/core/run-analyze.ts | 32 +- gitnexus/src/mcp/local/local-backend.ts | 278 +++++- gitnexus/src/mcp/resources.ts | 3 + gitnexus/src/mcp/tools.ts | 55 ++ gitnexus/src/storage/parse-cache.ts | 11 +- gitnexus/src/storage/repo-manager.ts | 18 + gitnexus/test/helpers/taint-fixture.ts | 123 +++ gitnexus/test/helpers/ts-cfg-harness.ts | 68 ++ .../__snapshots__/taint-snapshot.test.ts.snap | 102 ++ .../cfg/fixtures/pdg-repo/taint-cases.ts | 61 ++ .../integration/cfg/fixtures/pdg-repo/vuln.ts | 19 + .../test/integration/cfg/pipeline-pdg.test.ts | 75 +- .../integration/cfg/taint-snapshot.test.ts | 144 +++ .../integration/cfg/worker-roundtrip.test.ts | 87 ++ .../test/integration/taint-explain.test.ts | 344 +++++++ gitnexus/test/unit/cfg/harvest.test.ts | 295 +++++- gitnexus/test/unit/pdg-mode-flip.test.ts | 50 + gitnexus/test/unit/run-analyze.test.ts | 31 +- gitnexus/test/unit/security.test.ts | 9 + gitnexus/test/unit/taint/model-match.test.ts | 315 +++++++ gitnexus/test/unit/taint/path-codec.test.ts | 305 ++++++ gitnexus/test/unit/taint/propagate.test.ts | 686 ++++++++++++++ gitnexus/test/unit/taint/site-safety.test.ts | 172 ++++ .../unit/taint/source-sink-registry.test.ts | 13 +- gitnexus/test/unit/taint/taint-emit.test.ts | 338 +++++++ gitnexus/test/unit/tools.test.ts | 37 +- 48 files changed, 6725 insertions(+), 96 deletions(-) create mode 100644 gitnexus/src/core/ingestion/taint/emit.ts create mode 100644 gitnexus/src/core/ingestion/taint/match.ts create mode 100644 gitnexus/src/core/ingestion/taint/path-codec.ts create mode 100644 gitnexus/src/core/ingestion/taint/propagate.ts create mode 100644 gitnexus/src/core/ingestion/taint/site-safety.ts create mode 100644 gitnexus/src/core/ingestion/taint/typescript-model.ts create mode 100644 gitnexus/test/helpers/taint-fixture.ts create mode 100644 gitnexus/test/helpers/ts-cfg-harness.ts create mode 100644 gitnexus/test/integration/cfg/__snapshots__/taint-snapshot.test.ts.snap create mode 100644 gitnexus/test/integration/cfg/fixtures/pdg-repo/taint-cases.ts create mode 100644 gitnexus/test/integration/cfg/fixtures/pdg-repo/vuln.ts create mode 100644 gitnexus/test/integration/cfg/taint-snapshot.test.ts create mode 100644 gitnexus/test/integration/taint-explain.test.ts create mode 100644 gitnexus/test/unit/taint/model-match.test.ts create mode 100644 gitnexus/test/unit/taint/path-codec.test.ts create mode 100644 gitnexus/test/unit/taint/propagate.test.ts create mode 100644 gitnexus/test/unit/taint/site-safety.test.ts create mode 100644 gitnexus/test/unit/taint/taint-emit.test.ts diff --git a/.claude/skills/gitnexus/gitnexus-guide/SKILL.md b/.claude/skills/gitnexus/gitnexus-guide/SKILL.md index 2a1e76b02..a70279222 100644 --- a/.claude/skills/gitnexus/gitnexus-guide/SKILL.md +++ b/.claude/skills/gitnexus/gitnexus-guide/SKILL.md @@ -39,7 +39,8 @@ For any task involving code understanding, debugging, impact analysis, or refact | `check` | Check graph invariants such as circular imports | | `rename` | Multi-file coordinated rename with confidence-tagged edits | | `cypher` | Raw graph queries (read `gitnexus://repo/{name}/schema` first) | -| `list_repos` | Discover indexed repos (paginated — `limit`/`offset`) | +| `explain` | Persisted taint findings — source→sink data flows (needs `analyze --pdg`) | +| `list_repos` | Discover indexed repos (paginated — `limit`/`offset`) | ### Paginating `list_repos` @@ -72,6 +73,16 @@ list_repos { offset: 400 } → repos 401–437, hasMore false Notes: `offset` ≥ `total` returns an empty page (with `total` still reported). Out-of-range or malformed `limit`/`offset` (non-integer, `limit` outside `[1, 200]`, `offset < 0`) are rejected with a clear error — `limit` above the max is rejected, not silently capped. The order is deterministic (lower-cased name, then path), so paging never skips or duplicates an entry while the registry is unchanged. +### Taint findings (`explain`) + +`explain` returns intra-procedural taint findings (`TAINTED` edges) recorded by `gitnexus analyze --pdg` — each with a sink category (command-injection, code-injection, path-traversal, sql-injection, xss), source/sink lines, and the ordered hop path with the variable carried on each hop. + +- `explain {}` — enumerate all findings for the repo (bounded by `limit`, deterministic order) +- `explain { target: "src/vuln.ts" }` — findings in a file (suffix path match accepted) +- `explain { target: "runUserCommand" }` — findings in a function (resolved like `context`; ambiguous names return ranked candidates) + +A repo indexed without `--pdg` returns a clear "no taint layer" note. Caveats: findings are intra-procedural only — cross-function, closure/callback, property/field, and implicit flows are not modeled, so the absence of a finding is **not** proof of safety. `SANITIZES` (sanitizer-kill) edges are queryable via `cypher`. + ## Resources Reference Lightweight reads (~100-500 tokens) for navigation: diff --git a/gitnexus-claude-plugin/skills/gitnexus-guide/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-guide/SKILL.md index cacc4e886..a71429f32 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-guide/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-guide/SKILL.md @@ -38,7 +38,8 @@ For any task involving code understanding, debugging, impact analysis, or refact | `detect_changes` | Git-diff impact — what do your current changes affect | | `rename` | Multi-file coordinated rename with confidence-tagged edits | | `cypher` | Raw graph queries (read `gitnexus://repo/{name}/schema` first) | -| `list_repos` | Discover indexed repos (paginated — `limit`/`offset`) | +| `explain` | Persisted taint findings — source→sink data flows (needs `analyze --pdg`) | +| `list_repos` | Discover indexed repos (paginated — `limit`/`offset`) | ### Paginating `list_repos` @@ -71,6 +72,16 @@ list_repos { offset: 400 } → repos 401–437, hasMore false Notes: `offset` ≥ `total` returns an empty page (with `total` still reported). Out-of-range or malformed `limit`/`offset` (non-integer, `limit` outside `[1, 200]`, `offset < 0`) are rejected with a clear error — `limit` above the max is rejected, not silently capped. The order is deterministic (lower-cased name, then path), so paging never skips or duplicates an entry while the registry is unchanged. +### Taint findings (`explain`) + +`explain` returns intra-procedural taint findings (`TAINTED` edges) recorded by `gitnexus analyze --pdg` — each with a sink category (command-injection, code-injection, path-traversal, sql-injection, xss), source/sink lines, and the ordered hop path with the variable carried on each hop. + +- `explain {}` — enumerate all findings for the repo (bounded by `limit`, deterministic order) +- `explain { target: "src/vuln.ts" }` — findings in a file (suffix path match accepted) +- `explain { target: "runUserCommand" }` — findings in a function (resolved like `context`; ambiguous names return ranked candidates) + +A repo indexed without `--pdg` returns a clear "no taint layer" note. Caveats: findings are intra-procedural only — cross-function, closure/callback, property/field, and implicit flows are not modeled, so the absence of a finding is **not** proof of safety. `SANITIZES` (sanitizer-kill) edges are queryable via `cypher`. + ## Resources Reference Lightweight reads (~100-500 tokens) for navigation: diff --git a/gitnexus/bench/cfg/baselines.json b/gitnexus/bench/cfg/baselines.json index 7e3d718d8..0917b6bdd 100644 --- a/gitnexus/bench/cfg/baselines.json +++ b/gitnexus/bench/cfg/baselines.json @@ -9,20 +9,20 @@ "_note": "#2081 M1 / #2082 M2: ONE function, N coalescing statements (extendBlock text accumulation + per-statement fact harvest). Runs at 2000->8000. M2 REWROTE the old 'output is constant 4 blocks' note: statement facts make disk/heap LINEAR in N (a free gate on the harvest payload); TIME still guards the concat path (array-join ~1.0; a genuine O(n^2) re-join accumulation is ~3.8). M2 adds rd_scaling_budget (measured ~0.74) and disk_bytes_large_max -- an ABSOLUTE ceiling ~1.35x the measured indexed-encoding bytes (969,986 at N=8000, ~121 B/stmt); a named-record encoding regression (~4x facts bytes) blows it. Re-baseline the fingerprint only on an intentional CFG/harvest-shape change (the canon now includes statements+bindings)." }, "many-functions": { - "fingerprint": "f3bcc5e6ef4cf58aefe4e7d801a8fea0215494b9688833e501c2afc6df029c1b", + "fingerprint": "d881f60e77f0262bdc1b5c7049aa4acf5071e0eabc536476be293c3a133e626e", "scaling_budget": 1.5, "disk_bytes_budget": 1.2, "heap_budget": 1.3, "rd_scaling_budget": 2.0, - "_note": "#2081 M1 / #2082 M2: N small branchy functions (collect walk + per-function build + per-function solve). Time ~1.0, disk ~1.01, heap ~1.0, rd ~0.86 (solver is per-function; N functions scale linearly)." + "_note": "#2081 M1 / #2082 M2 / #2083 M3 U1: N small branchy functions (collect walk + per-function build + per-function solve). Time ~1.0, disk ~1.01, heap ~1.0, rd ~0.86 (solver is per-function; N functions scale linearly). M3 U1 re-fingerprinted: taint sites join StatementFacts (a()/b() call sites); disk_large 2565641->2721641 (+6.1% measured site-harvest cost at N=2000)." }, "branchy": { - "fingerprint": "5b5886521ab21604df8f78af98c8c28a6be8e64c24f3d67b165c2d96ba2a3d52", + "fingerprint": "936765bba5c3f8fc7058737c48351e03e4e1da7fed448467e8fcc8a0fb7786ce", "scaling_budget": 1.8, "disk_bytes_budget": 1.2, "heap_budget": 1.3, "rd_scaling_budget": 2.0, - "_note": "#2081 M1 / #2082 M2: ONE function, N sequential ifs (block/edge growth in one CFG). Time ~1.1-1.25 (noisiest scenario; budget 1.8 absorbs noise, catches ~4.0 quadratic), disk ~1.03, heap ~1.0, rd ~0.7." + "_note": "#2081 M1 / #2082 M2 / #2083 M3 U1: ONE function, N sequential ifs (block/edge growth in one CFG). Time ~1.1-1.25 (noisiest scenario; budget 1.8 absorbs noise, catches ~4.0 quadratic), disk ~1.03, heap ~1.0, rd ~0.7. M3 U1 re-fingerprinted (s{i}() call sites); disk_large 908964->993854 (+9.3%)." }, "dense-bindings": { "fingerprint": "e4d7eb3c7e8b3772423af25cef391e0e6b68067b554819e81b543439a487403f", @@ -33,12 +33,25 @@ "_note": "#2082 M2: N bindings live across ~N blocks in one loop -- bindings x blocks scale JOINTLY (the solver-lattice stressor). The overlay design measures rd ~5.2 normalized: the OUT spine copy on genning blocks is O(V) per block, which is quadratic when V scales with B (bounded in prod by maxFunctionLines; real functions have V~10-40). Budget 10 deliberately tolerates that known shape and exists to catch the repo's recurring per-item-rescan class (a per-use scan over all defs is O(n^3) here, ratio >=16). If rd drops well below 5, tighten." }, "fact-fanout": { - "fingerprint": "488e63e072d514a9229e21872615e32c7b099ccbd65ec8c045ba517568fd3e5d", + "fingerprint": "83a8243a8aff117f69aeecb39d02a483e6cca70439d75f63e433f4e4ac85578f", "scaling_budget": 1.8, "disk_bytes_budget": 1.2, "heap_budget": 1.3, "rd_scaling_budget": 3.0, "facts_large_max": 16000, - "_note": "#2082 M2: N switch-arm defs of one variable + N later uses -- facts are O(defs x uses) BY SPEC, so the gate is BOUNDEDNESS, not linearity: with the production fact limit engaged (DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION=16000) the materialized fact count stays pinned at the limit as N grows (facts_large_max), and rd time stays bounded (measured ~1.4). Losing the maxFacts early-stop shows as facts_large exploding quadratically." + "_note": "#2082 M2 / #2083 M3 U1: N switch-arm defs of one variable + N later uses -- facts are O(defs x uses) BY SPEC, so the gate is BOUNDEDNESS, not linearity: with the production fact limit engaged (DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION=16000) the materialized fact count stays pinned at the limit as N grows (facts_large_max), and rd time stays bounded (measured ~1.4). Losing the maxFacts early-stop shows as facts_large exploding quadratically. M3 U1 re-fingerprinted (u{i}(x) call sites); disk_large 996737->1107627 (+11.1%)." + }, + "taint-dense": { + "fingerprint": "218a1a0c7e092550c233607c67daa401543a25bf8d3f122899d30cd9c30c3a89", + "scaling_budget": 1.5, + "disk_bytes_budget": 1.2, + "heap_budget": 1.3, + "rd_scaling_budget": 2.0, + "disk_bytes_large_max": 3150000, + "taint_findings_per_fn_pin": 8, + "taint_scaling_budget": 2.0, + "taint_reason_bytes_large_max": 198000, + "taint_zero_match_budget": 0.5, + "_note": "#2083 M3 U7 (R10): N functions, each with 12 req.body sources + a 4-hop chain + 13 eval sinks (13 deduped findings/fn) at 125->500 fns; the zero-match control (inp.payload/evalish) keeps the identical CFG shape with zero model hits. BOUNDEDNESS pin: kept findings/function == 8 (the scenario cap) at BOTH sizes -- above means the cap was lost, below means detection regressed; total findings grow linearly with N by design. disk_bytes_large_max is the LOAD-BEARING site-harvest absolute ceiling (densest sites of the suite; measured 2335772 at N=500, ceiling ~1.35x). taint_reason_bytes_large_max caps the persisted TAINTED reason bytes (measured 146827 = ~37 B/finding, ceiling ~1.35x; blows on hop-encoding bloat or cap loss). taint_zero_match_budget 0.5 vs measured 0.15: the zero-match pass (match gate only, no solver) must stay a small fraction of the match-dense pass. taint scaling measured ~0.93 (per-function work is N-linear); time/disk/heap/rd ratios all ~1.0." } } diff --git a/gitnexus/bench/cfg/measure.mjs b/gitnexus/bench/cfg/measure.mjs index 2451b5d62..a4a2c9eaa 100644 --- a/gitnexus/bench/cfg/measure.mjs +++ b/gitnexus/bench/cfg/measure.mjs @@ -48,6 +48,13 @@ import { computeReachingDefs } from '../../src/core/ingestion/cfg/reaching-defs. import { DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION } from '../../src/core/ingestion/cfg/emit.ts'; import { createTypeScriptCfgVisitor } from '../../src/core/ingestion/cfg/visitors/typescript.ts'; import { getTreeSitterBufferSize } from '../../src/core/ingestion/constants.ts'; +import { buildTaintImportIndex, matchFunctionSites } from '../../src/core/ingestion/taint/match.ts'; +import { TS_JS_TAINT_MODEL } from '../../src/core/ingestion/taint/typescript-model.ts'; +import { + computeTaintFlows, + DEFAULT_PDG_MAX_TAINT_HOPS, +} from '../../src/core/ingestion/taint/propagate.ts'; +import { encodeTaintPath } from '../../src/core/ingestion/taint/path-codec.ts'; const __dirname = path.dirname(fileURLToPath(import.meta.url)); const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); @@ -141,8 +148,51 @@ const SCENARIOS = [ return s + '}\n'; }, }, + { + name: 'taint-dense', + // #2083 M3 U7 (R10): N functions, EACH source/sink-dense — 12 matched + // `req.body` source statements + a 4-hop chained reassignment + 13 `eval` + // sinks per function (13 deduped findings/fn, ABOVE the scenario cap of 8 + // so the cap binds). Functions scale with N, so total findings grow + // linearly BY DESIGN; the boundedness gate is the per-function pin: kept + // findings/function stays EXACTLY at the cap as N grows (a cap loss shows + // as 13). This scenario's sites are the densest of the suite, so its + // ABSOLUTE disk_bytes_large_max is the load-bearing site-harvest ceiling + // (the M2 straight-line carrier has no call sites), and the summed + // encoded TAINTED reason bytes get their own absolute ceiling + // (taint_reason_bytes_large_max). The zero-match control (genZero) keeps + // the identical statement/CFG shape with names OUTSIDE the model + // (inp.payload / evalish) — the match-gate must make unmatched functions + // cost ~nothing (no solver call), gated as zero-time/dense-time ratio. + small: 125, + large: 500, // 4x, like the global sizes — per-fn bodies are ~30 lines + taint: { cap: 8 }, + gen: (n) => genTaintFunctions(n, false), + genZero: (n) => genTaintFunctions(n, true), + }, ]; +// taint-dense generator: `zero` swaps every model-matched name for an +// unmatched one without changing statement count, def/use shape, or CFG. +const TAINT_SOURCES_PER_FN = 12; +const TAINT_CHAIN_HOPS = 4; +function genTaintFunctions(n, zero) { + const recv = zero ? 'inp' : 'req'; + const prop = zero ? 'payload' : 'body'; + const sink = zero ? 'evalish' : 'eval'; + let s = ''; + for (let i = 0; i < n; i++) { + s += `function f${i}(${recv}) {\n`; + for (let j = 0; j < TAINT_SOURCES_PER_FN; j++) s += ` const s${j} = ${recv}.${prop};\n`; + s += ` let c0 = s0 + '!';\n`; + for (let h = 1; h < TAINT_CHAIN_HOPS; h++) s += ` const c${h} = c${h - 1} + '!';\n`; + for (let j = 0; j < TAINT_SOURCES_PER_FN; j++) s += ` ${sink}(s${j});\n`; + s += ` ${sink}(c${TAINT_CHAIN_HOPS - 1});\n`; + s += '}\n'; + } + return s; +} + const SMALL = 500; const LARGE = 2000; // 4× — O(n) ⇒ ratio ~1, O(n²) ⇒ ratio ~4 const REPS = 15; // median over more reps → stabler time signal at small absolute ms @@ -199,6 +249,57 @@ function measureReachingDefs(cfgs, reps, maxFacts) { return { ms: median(samples), facts }; } +// ---- taint pass cost (#2083 M3 U7) ---- + +// Times the EXACT per-function sequence the in-phase emit driver runs on a +// --pdg run for a taint-modeled language: match sites → zero-match fast path +// → computeReachingDefs → computeTaintFlows. `cap` is the scenario's +// maxFindingsPerFunction (deliberately small so the cap BINDS on the dense +// generator). Also sums the encoded TAINTED `reason` bytes for the kept +// findings — the persisted-taint disk posture (R10). +function measureTaint(cfgs, reps, cap) { + const importIndex = buildTaintImportIndex([]); // bench callees are globals + const pass = () => { + let analyzed = 0; + let kept = 0; + let dropped = 0; + let reasonBytes = 0; + for (const c of cfgs) { + const matches = matchFunctionSites(c, TS_JS_TAINT_MODEL, importIndex); + if (!matches.hasSource || !matches.hasSink) continue; + const du = computeReachingDefs(c, { + maxFacts: DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, + }); + const flows = computeTaintFlows(c, du, matches, { + maxFindingsPerFunction: cap, + maxHops: DEFAULT_PDG_MAX_TAINT_HOPS, + }); + if (flows.status !== 'computed') continue; + analyzed++; + kept += flows.findings.length; + dropped += flows.droppedFindings; + for (const f of flows.findings) { + // All structural chars + identifier names are single-byte ASCII, so + // string length IS the byte length (path-codec discipline). + reasonBytes += encodeTaintPath( + f.hops.map((h) => ({ name: h.name, line: h.point.line, viaCall: h.viaCall })), + { truncated: f.hopsTruncated === true, kind: f.sinkKind }, + ).reason.length; + } + } + return { analyzed, kept, dropped, reasonBytes }; + }; + pass(); // warm JIT (uncounted) + const samples = []; + let out; + for (let i = 0; i < reps; i++) { + const start = process.hrtime.bigint(); + out = pass(); + samples.push(Number(process.hrtime.bigint() - start) / 1e6); + } + return { ms: median(samples), ...out }; +} + // ---- memory growth: retained heap of the cfgSideChannel payload ---- // Needs `node --expose-gc` to force collection for a clean delta; without it the @@ -279,7 +380,40 @@ function measureScenario(scenario) { // ratio 0 and the gate would self-disable exactly when the solver is fast. const rdRatio = rdLarge.ms / Math.max(rdSmall.ms, 0.001) / sizeRatio; + // #2083 M3 U7: taint pass cost + boundedness on taint-bearing scenarios. + let taintMetrics = {}; + if (scenario.taint !== undefined) { + const cap = scenario.taint.cap; + const tSmall = measureTaint(small.cfgs, REPS, cap); + const tLarge = measureTaint(large.cfgs, REPS, cap); + const tRatio = tLarge.ms / Math.max(tSmall.ms, 0.001) / sizeRatio; + // Zero-match control: identical CFG shape, no model hits — measures the + // match-gate overhead unmatched functions pay on a real --pdg repo. + const zeroCfgs = collectFunctionCfgs( + parse(scenario.genZero(nLarge)).rootNode, + visitor, + `${scenario.name}-zero.ts`, + NO_CAP, + ).cfgs; + const tZero = measureTaint(zeroCfgs, REPS, cap); + taintMetrics = { + taint_ms_small: Number(tSmall.ms.toFixed(3)), + taint_ms_large: Number(tLarge.ms.toFixed(3)), + taint_scaling_ratio: Number(tRatio.toFixed(3)), + // Boundedness: kept findings PER ANALYZED FUNCTION (total findings grow + // linearly with N by design — the per-function pin is the cap gate). + taint_findings_per_fn_small: tSmall.analyzed > 0 ? tSmall.kept / tSmall.analyzed : 0, + taint_findings_per_fn_large: tLarge.analyzed > 0 ? tLarge.kept / tLarge.analyzed : 0, + taint_dropped_large: tLarge.dropped, + taint_reason_bytes_large: tLarge.reasonBytes, + taint_zero_ms_large: Number(tZero.ms.toFixed(3)), + taint_zero_findings: tZero.kept + tZero.dropped, + taint_zero_match_ratio: Number((tZero.ms / Math.max(tLarge.ms, 0.001)).toFixed(3)), + }; + } + return { + ...taintMetrics, scenario: scenario.name, elapsed_ms_small: Number(small.ms.toFixed(3)), elapsed_ms_large: Number(large.ms.toFixed(3)), @@ -367,6 +501,56 @@ if (!CHECK) { `${base.disk_bytes_large_max} bytes (constant-factor encoding bloat)`, ); } + // #2083 M3 U7 gates — taint boundedness (per-function findings pinned at + // the cap as N grows), an ABSOLUTE ceiling on persisted TAINTED reason + // bytes, taint solve-time scaling, and the zero-match fast path staying + // ~free relative to the match-dense pass. + if (base.taint_findings_per_fn_pin !== undefined) { + for (const side of ['small', 'large']) { + const perFn = r[`taint_findings_per_fn_${side}`]; + if (perFn !== base.taint_findings_per_fn_pin) { + failures.push( + `${r.scenario}: taint findings/function (${side}) ${perFn} != pin ` + + `${base.taint_findings_per_fn_pin} (cap must BIND exactly: above = cap lost, ` + + `below = detection regressed)`, + ); + } + } + if (r.taint_zero_findings !== 0) { + failures.push( + `${r.scenario}: zero-match control produced ${r.taint_zero_findings} findings ` + + `(the control must not match the model — generator drift)`, + ); + } + } + if ( + base.taint_reason_bytes_large_max !== undefined && + r.taint_reason_bytes_large > base.taint_reason_bytes_large_max + ) { + failures.push( + `${r.scenario}: persisted TAINTED reason bytes ${r.taint_reason_bytes_large} > ceiling ` + + `${base.taint_reason_bytes_large_max} (hop-encoding bloat or cap loss)`, + ); + } + if ( + base.taint_scaling_budget !== undefined && + r.taint_scaling_ratio >= base.taint_scaling_budget + ) { + failures.push( + `${r.scenario}: taint scaling ratio ${r.taint_scaling_ratio} >= budget ` + + `${base.taint_scaling_budget} (ms ${r.taint_ms_small}->${r.taint_ms_large})`, + ); + } + if ( + base.taint_zero_match_budget !== undefined && + r.taint_zero_match_ratio >= base.taint_zero_match_budget + ) { + failures.push( + `${r.scenario}: zero-match taint time is ${r.taint_zero_match_ratio} of the match-dense ` + + `pass, >= budget ${base.taint_zero_match_budget} (the match gate must keep unmatched ` + + `functions ~free — no solver call)`, + ); + } // Heap gate only when measured (--expose-gc present) AND a budget exists. if ( base.heap_budget !== undefined && diff --git a/gitnexus/skills/gitnexus-guide.md b/gitnexus/skills/gitnexus-guide.md index a54337879..a71429f32 100644 --- a/gitnexus/skills/gitnexus-guide.md +++ b/gitnexus/skills/gitnexus-guide.md @@ -38,6 +38,7 @@ For any task involving code understanding, debugging, impact analysis, or refact | `detect_changes` | Git-diff impact — what do your current changes affect | | `rename` | Multi-file coordinated rename with confidence-tagged edits | | `cypher` | Raw graph queries (read `gitnexus://repo/{name}/schema` first) | +| `explain` | Persisted taint findings — source→sink data flows (needs `analyze --pdg`) | | `list_repos` | Discover indexed repos (paginated — `limit`/`offset`) | ### Paginating `list_repos` @@ -71,6 +72,16 @@ list_repos { offset: 400 } → repos 401–437, hasMore false Notes: `offset` ≥ `total` returns an empty page (with `total` still reported). Out-of-range or malformed `limit`/`offset` (non-integer, `limit` outside `[1, 200]`, `offset < 0`) are rejected with a clear error — `limit` above the max is rejected, not silently capped. The order is deterministic (lower-cased name, then path), so paging never skips or duplicates an entry while the registry is unchanged. +### Taint findings (`explain`) + +`explain` returns intra-procedural taint findings (`TAINTED` edges) recorded by `gitnexus analyze --pdg` — each with a sink category (command-injection, code-injection, path-traversal, sql-injection, xss), source/sink lines, and the ordered hop path with the variable carried on each hop. + +- `explain {}` — enumerate all findings for the repo (bounded by `limit`, deterministic order) +- `explain { target: "src/vuln.ts" }` — findings in a file (suffix path match accepted) +- `explain { target: "runUserCommand" }` — findings in a function (resolved like `context`; ambiguous names return ranked candidates) + +A repo indexed without `--pdg` returns a clear "no taint layer" note. Caveats: findings are intra-procedural only — cross-function, closure/callback, property/field, and implicit flows are not modeled, so the absence of a finding is **not** proof of safety. `SANITIZES` (sanitizer-kill) edges are queryable via `cypher`. + ## Resources Reference Lightweight reads (~100-500 tokens) for navigation: diff --git a/gitnexus/src/cli/ai-context.ts b/gitnexus/src/cli/ai-context.ts index 641e7ee94..4bf117974 100644 --- a/gitnexus/src/cli/ai-context.ts +++ b/gitnexus/src/cli/ai-context.ts @@ -179,6 +179,7 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s - **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. - When exploring unfamiliar code, use \`query({query: "concept"})\` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. - When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use \`context({name: "symbolName"})\`. +- For security review, \`explain({target: "fileOrSymbol"})\` lists taint findings (source→sink flows; needs \`analyze --pdg\`). ## Never Do diff --git a/gitnexus/src/cli/skill-gen.ts b/gitnexus/src/cli/skill-gen.ts index f2fb42838..61569e6a4 100644 --- a/gitnexus/src/cli/skill-gen.ts +++ b/gitnexus/src/cli/skill-gen.ts @@ -654,6 +654,9 @@ const renderSkillMarkdown = ( `2. \`query({query: "${community.label.toLowerCase()}"})\` \u2014 find related execution flows`, ); lines.push('3. Read key files listed above for implementation details'); + lines.push( + '4. `explain({target: ""})` — persisted taint findings (source→sink data flows), when indexed with `--pdg`', + ); lines.push(''); return lines.join('\n'); diff --git a/gitnexus/src/core/ingestion/cfg/emit.ts b/gitnexus/src/core/ingestion/cfg/emit.ts index 3b1a1e50d..dc682d5b0 100644 --- a/gitnexus/src/core/ingestion/cfg/emit.ts +++ b/gitnexus/src/core/ingestion/cfg/emit.ts @@ -65,7 +65,13 @@ export interface CfgEmitResult { cappedFunctions: number; } -const basicBlockId = ( +/** + * The single BasicBlock id template (module doc). Exported for the M3 taint + * emit path (taint/emit.ts), whose TAINTED/SANITIZES edges must address the + * SAME persisted block nodes — a re-derived copy of this template would + * silently dangle the moment either drifted. + */ +export const basicBlockId = ( filePath: string, functionStartLine: number, functionStartColumn: number, @@ -259,9 +265,10 @@ export interface ReachingDefEmitResult { * Stable identity for a binding inside edge ids (#2082 M2 KTD3/KTD9): * `name:declLine:declCol` for declared bindings, `name@module` for synthetic * ones. Distinct same-name bindings never share a key; identifier characters - * cannot contain the id separators. + * cannot contain the id separators. Exported for the M3 taint emit path — + * TAINTED/SANITIZES ids key bindings with the same discipline. */ -const bindingKey = (b: BindingEntry): string => +export const bindingKey = (b: BindingEntry): string => b.synthetic ? `${b.name}@module` : `${b.name}:${b.declLine}:${b.declColumn}`; /** diff --git a/gitnexus/src/core/ingestion/cfg/reaching-defs.ts b/gitnexus/src/core/ingestion/cfg/reaching-defs.ts index 14c7b3745..7b7f7233c 100644 --- a/gitnexus/src/core/ingestion/cfg/reaching-defs.ts +++ b/gitnexus/src/core/ingestion/cfg/reaching-defs.ts @@ -43,6 +43,16 @@ export interface ProgramPoint { readonly line: number; } +/** + * Canonical `block:stmt` string key for a program point. Colon-separated to + * match the codebase's `blockIndex:stmtIndex` id conventions. Shared by the + * taint propagation engine (dedup/state keys) and the taint emit path + * (persisted edge-id material) so the two never drift. + */ +export function pointKey(p: ProgramPoint): string { + return `${p.blockIndex}:${p.stmtIndex}`; +} + /** One def→use fact: the definition at `def` reaches the use at `use`. */ export interface DefUseFact { /** Index into {@link FunctionDefUse.bindings}. */ diff --git a/gitnexus/src/core/ingestion/cfg/types.ts b/gitnexus/src/core/ingestion/cfg/types.ts index ed789ec30..53988e687 100644 --- a/gitnexus/src/core/ingestion/cfg/types.ts +++ b/gitnexus/src/core/ingestion/cfg/types.ts @@ -42,6 +42,92 @@ export interface BindingEntry { readonly synthetic?: boolean; } +/** + * One occurrence of a binding inside a call/new site's argument position + * (#2083 M3 U1). A bare `number` is a DIRECT occurrence (binding index into + * {@link FunctionCfg.bindings}); a `[bindingIdx, viaSiteIdx]` tuple marks an + * occurrence that reaches this argument THROUGH the nested site at + * `viaSiteIdx` (an index into the SAME statement's {@link StatementFacts.sites} + * array). The tag is load-bearing for sanitizer interposition (plan KTD4a): + * a flat per-arg binding set cannot distinguish `exec(escape(x))` (kill) from + * `exec(x)` (finding) — the single most common safe pattern would + * false-positive without it. + */ +export type SiteArgOccurrence = number | readonly [number, number]; + +/** + * One call site, constructor call, or value-position member read harvested + * from a statement (#2083 M3 U1, plan KTD2). Worker-side substrate for the M3 + * taint pass: the M2 facts carry no expression structure, and the main thread + * cannot re-parse (the #1983 OOM shape). Spec-AGNOSTIC — records structure + * only, never source/sink/sanitizer-ness (matching is a main-thread concern). + * + * Integer indices: binding fields (`receiver`/`object`/`resultDefs`/arg + * occurrences) index {@link FunctionCfg.bindings}; site references (`parent`, + * via-tags) index the OWNING statement's `sites` array. JSON-plain; NO field + * here may be named `nodeId` (durable parsedfile-store reviver hazard — see + * {@link BindingEntry}). + */ +export interface SiteRecord { + readonly kind: 'call' | 'new' | 'member-read'; + /** + * Dotted callee path for call/new sites whose callee chain is rooted at an + * identifier/`this`/`super` (`child_process.exec`, `req.body.toString`). + * Optional chaining is normalized (`a?.b()` ⇒ `a.b`); string-literal + * subscripts fold into the path (`cp["exec"]` ⇒ `cp.exec`). Absent when the + * chain is not statically resolvable (dynamic key, call-rooted chain). + */ + readonly callee?: string; + /** + * Binding index of the callee chain's ROOT identifier when the callee is a + * member chain (`userInput.trim()` ⇒ `userInput`). Method calls launder + * taint without it (plan KTD5 receiver-position TITO). Absent for bare + * calls (`exec(x)`) and non-identifier roots. + */ + readonly receiver?: number; + /** + * Per-argument-position occurrence entries (trailing empty positions are + * trimmed; absent when no argument carries a binding occurrence). For + * `template: true` sites every substitution occurrence aggregates at + * position 0 (tagged templates have no positional argument list). + */ + readonly args?: ReadonlyArray; + /** + * Bindings defined by a declarator/assignment whose ENTIRE value (after + * unwrapping parens/`await`/`as`/`!`) is this call — `const b = escape(t)` + * ⇒ `[b]`. Per-declarator: `const a = t, b = escape(t)` attaches `[b]` + * only. Kill placement (KTD4b) keys on this: a sanitizer kills exactly the + * defs that receive its result directly. + */ + readonly resultDefs?: readonly number[]; + /** + * `[siteIdx, argIdx]` of the innermost enclosing call/new site argument + * position this site occurs in (`exec(escape(x))` ⇒ escape's parent is + * `[execSiteIdx, 0]`). Absent for top-level sites. + */ + readonly parent?: readonly [number, number]; + /** + * Index of the FIRST spread argument (`exec(...args)` ⇒ 0). Presence means + * position matching must degrade soundly (any sink position ≥ this index — + * plan KTD2/U2). A number (not boolean) because the matcher needs the index. + */ + readonly spread?: number; + /** Tagged-template call (`sql\`…${id}\``) — argument positions are not positional. */ + readonly template?: boolean; + /** + * String-literal first argument when the callee is bare `require` — + * CommonJS aliases resolve like ESM imports on the main thread (KTD7). + */ + readonly requireArg?: string; + /** Member read: binding index of the object root (`req.body` ⇒ `req`). */ + readonly object?: number; + /** + * Member read: property name (`req.body` ⇒ `'body'`; `req["body"]` + * included; dynamic `req[key]` is never recorded — documented KTD10 FN). + */ + readonly property?: string; +} + /** * Def/use facts for one harvested statement (or construct header), in * execution order within its block (#2082 M2 U1). `defs`/`uses` are indices @@ -57,12 +143,19 @@ export interface BindingEntry { * treating them as must-defs would falsely kill the prior def on the * not-taken path (a taint false negative on core JS idioms). Optional — * absent means none. + * + * `sites` (#2083 M3 U1): call/member-read structure for the taint pass — + * see {@link SiteRecord}. Optional and omit-when-empty; absent on pre-M3 + * channels and on statements with no calls or member reads. Sites inside + * nested functions are NOT recorded (consistent with def/use invisibility — + * the enclosing `arr.forEach(...)` call IS, with receiver `arr`). */ export interface StatementFacts { readonly line: number; readonly defs: readonly number[]; readonly uses: readonly number[]; readonly mayDefs?: readonly number[]; + readonly sites?: readonly SiteRecord[]; } /** A basic block: a maximal straight-line run of statements between leaders. */ diff --git a/gitnexus/src/core/ingestion/cfg/visitors/typescript-harvest.ts b/gitnexus/src/core/ingestion/cfg/visitors/typescript-harvest.ts index 81823c97a..f1ea3abc0 100644 --- a/gitnexus/src/core/ingestion/cfg/visitors/typescript-harvest.ts +++ b/gitnexus/src/core/ingestion/cfg/visitors/typescript-harvest.ts @@ -39,7 +39,7 @@ * parsedfile-store reviver dedups objects keyed on that field name. */ import type { SyntaxNode } from '../../utils/ast-helpers.js'; -import type { BindingEntry, StatementFacts } from '../types.js'; +import type { BindingEntry, SiteArgOccurrence, SiteRecord, StatementFacts } from '../types.js'; /** Node types that own a nested CFG — their subtrees are opaque to harvesting. */ const NESTED_FUNCTION_TYPES = new Set([ @@ -83,6 +83,31 @@ const TYPE_CONTEXT_TYPES = new Set([ 'asserts_annotation', ]); +/** + * Wrappers that don't change which VALUE flows through them (#2083 M3 U1) — + * unwrapped when resolving call-result attribution (`const b = (await + * escape(t))!` still attaches `resultDefs: [b]` to the escape site) and + * member-chain roots. Distinct from {@link TsHarvester.unwrapLvalue}, which is + * the narrower LVALUE set. + */ +const VALUE_WRAPPER_TYPES = new Set([ + 'parenthesized_expression', + 'non_null_expression', + 'as_expression', + 'satisfies_expression', + 'await_expression', +]); + +/** Literal text of a `string` node (concatenated fragments; raw escapes kept). */ +const stringLiteralText = (node: SyntaxNode): string => { + let out = ''; + for (let i = 0; i < node.namedChildCount; i++) { + const c = node.namedChild(i); + if (c?.type === 'string_fragment' || c?.type === 'escape_sequence') out += c.text; + } + return out; +}; + interface Scope { readonly parent: Scope | null; /** name → binding index */ @@ -112,6 +137,16 @@ export class TsHarvester { * here falsely kills the prior def on the not-taken path). */ private conditionalDepth = 0; + /** + * Call/new node id → bindings whose declarator/assignment VALUE is exactly + * that call (#2083 M3 U1). Registered by the declarator/assignment handlers + * BEFORE the value walk, consumed by {@link visitCall} when it reaches the + * node — the indirection keeps result-def attribution per-declarator + * (`const a = t, b = escape(t)` attaches `[b]` to the escape site only) and + * top-level-only (`const c = cond ? escape(b) : b` attaches nothing — the + * bypass occurrence must keep `c` taintable, plan KTD4a). + */ + private readonly resultDefTargets = new Map(); constructor(private readonly fnNode: SyntaxNode) { this.fnId = fnNode.id; @@ -458,7 +493,9 @@ export class TsHarvester { // live def (`x = source(); var x; sink(x)` must keep source→sink; // tri-review P2). `let`/`const` declarators genuinely initialize. if (name && (value || t === 'lexical_declaration')) { + const snap = acc.defSnapshot(); this.walkDefPattern(name, acc); + if (value) this.registerResultDefs(value, acc.defsSince(snap)); } if (value) this.walkValue(value, acc); } @@ -466,7 +503,11 @@ export class TsHarvester { case 'assignment_expression': { const left = node.childForFieldName('left'); const right = node.childForFieldName('right'); - if (left) this.walkDefPattern(this.unwrapLvalue(left), acc); + if (left) { + const snap = acc.defSnapshot(); + this.walkDefPattern(this.unwrapLvalue(left), acc); + if (right) this.registerResultDefs(right, acc.defsSince(snap)); + } if (right) this.walkValue(right, acc); return; } @@ -552,6 +593,39 @@ export class TsHarvester { if (body) this.walkValue(body, acc); return; } + case 'call_expression': + // #2083 M3 U1: explicit case (previously default-descended) — same + // uses, plus a taint-site record. MUST keep defs/uses byte-identical. + this.visitCall(node, acc, 'call'); + return; + case 'new_expression': + this.visitCall(node, acc, 'new'); + return; + case 'member_expression': + case 'subscript_expression': + // #2083 M3 U1: value-position member chain — same uses as the old + // default descent (root identifier + dynamic subscript indices), plus + // a member-read site for the innermost identifier-rooted access. + this.walkChain(node, acc, false); + return; + case 'sequence_expression': { + // Comma operator: only the LAST operand's value flows. Earlier operands + // are evaluated for side effects — record their uses but suppress + // occurrence fan-out so `exec((log(x), 'safe'))` does not taint exec's + // arg 0 with `x` (review fix). Defs/uses stay byte-identical to the old + // default descent; only the sites layer narrows. + const operands: SyntaxNode[] = []; + for (let i = 0; i < node.namedChildCount; i++) { + const c = node.namedChild(i); + if (c) operands.push(c); + } + const last = operands.length - 1; + operands.forEach((op, i) => { + if (i === last) this.walkValue(op, acc); + else acc.suppressOccurrences(() => this.walkValue(op, acc)); + }); + return; + } default: for (let i = 0; i < node.namedChildCount; i++) { const c = node.namedChild(i); @@ -593,8 +667,11 @@ export class TsHarvester { case 'member_expression': case 'subscript_expression': // Property/element write — NOT a scalar def (KTD4); its identifiers - // (object, computed key) are uses. - this.walkValue(node, acc); + // (object, computed key) are uses. WRITE position (#2083 M3 U1): the + // written access itself is not a value read — no member-read site for + // it (`obj.p = q` records nothing; `req.body.x = v`'s mid-chain LOAD + // of `req.body` still does). + this.walkChain(node, acc, true); return; default: for (let i = 0; i < node.namedChildCount; i++) { @@ -603,6 +680,201 @@ export class TsHarvester { } } } + + // ── taint-site harvest (#2083 M3 U1) ──────────────────────────────────── + + /** Strip value-transparent wrappers (`(x)`, `x!`, `x as T`, `await x`). */ + private unwrapValueWrappers(node: SyntaxNode): SyntaxNode { + let n = node; + while (VALUE_WRAPPER_TYPES.has(n.type)) { + const inner = n.namedChild(0); + if (!inner) break; + n = inner; + } + return n; + } + + /** + * When `value`'s root (after unwrapping) is a call/new node, remember that + * its site should carry `resultDefs: defs` — consumed by {@link visitCall} + * once the value walk reaches the node. + */ + private registerResultDefs(value: SyntaxNode, defs: readonly number[]): void { + if (defs.length === 0) return; + const root = this.unwrapValueWrappers(value); + if (root.type === 'call_expression' || root.type === 'new_expression') { + this.resultDefTargets.set(root.id, [...defs]); + } + } + + /** + * Explicit call/new handler: records a call site (callee path, receiver, + * per-arg occurrence entries, spread/template markers, require literal, + * result defs) while reproducing EXACTLY the uses the old default descent + * recorded — callee chain root + dynamic subscript indices + arguments. + */ + private visitCall(node: SyntaxNode, acc: FactAccumulator, kind: 'call' | 'new'): void { + const calleeNode = node.childForFieldName(kind === 'new' ? 'constructor' : 'function'); + const argsNode = node.childForFieldName('arguments'); + const siteIdx = acc.openCallSite(kind); + acc.pushFrame(siteIdx); + let calleePath: string | undefined; + if (calleeNode) { + const callee = this.unwrapValueWrappers(calleeNode); + if (callee.type === 'identifier') { + // The callee NAME is a statement-level use but NOT a value occurrence + // flowing into any enclosing argument — `exec(escape(x))` must not + // put the `escape` binding itself into exec's arg 0 (only x, tagged + // via the escape site). Receiver-chain roots DO fan out (KTD5 TITO). + acc.addUseWithoutOccurrence(this.resolve(callee)); + calleePath = callee.text; + } else if (callee.type === 'member_expression' || callee.type === 'subscript_expression') { + // skipFinalRead: the final access IS the callee, carried by the + // dotted path — recording it as a member read would double-count. + // Mid-chain reads (`req.body` inside `req.body.toString()`) ARE + // recorded (plan KTD2). + const chain = this.walkChain(callee, acc, true); + calleePath = chain.path; + if (chain.rootIdx !== undefined) acc.setSiteReceiver(siteIdx, chain.rootIdx); + } else { + // Call-rooted chains, IIFEs, function expressions — no dotted path; + // the walk still records uses and nested sites. + this.walkValue(callee, acc); + } + if (calleePath !== undefined) acc.setSiteCallee(siteIdx, calleePath); + } + const resultDefs = this.resultDefTargets.get(node.id); + if (resultDefs !== undefined) acc.setSiteResultDefs(siteIdx, resultDefs); + if (argsNode?.type === 'template_string') { + // Tagged template (`sql\`…${id}\``): the `arguments` field is a + // template_string, not an arguments node — substitution occurrences + // aggregate at position 0 and the site is marked non-positional. + acc.setSiteTemplate(siteIdx); + acc.setFrameArg(0); + this.walkValue(argsNode, acc); + } else if (argsNode) { + let pos = 0; + for (let i = 0; i < argsNode.namedChildCount; i++) { + const arg = argsNode.namedChild(i); + if (!arg || arg.type === 'comment') continue; + acc.setFrameArg(pos); + if (arg.type === 'spread_element') { + acc.setSiteSpread(siteIdx, pos); + const inner = arg.namedChild(0); + if (inner) this.walkValue(inner, acc); + } else { + if (kind === 'call' && pos === 0 && calleePath === 'require' && arg.type === 'string') { + // CommonJS `require('lit')` — record the literal so the matcher + // resolves require'd aliases like ESM imports (plan KTD7). + acc.setSiteRequireArg(siteIdx, stringLiteralText(arg)); + } + this.walkValue(arg, acc); + } + pos++; + } + } + acc.popFrame(); + } + + /** + * Member/subscript chain walk shared by value position, write position, and + * callee position. Use-recording is identical to the old default descent + * (chain-root identifier once, dynamic subscript index expressions, full + * walk of non-identifier roots) — NO double-recording. Member-read sites: + * at most ONE per chain — the INNERMOST access — and only when the chain + * root is an identifier and the access's key is static (`.prop` or a + * string-literal subscript); `skipFinalRead` suppresses it when that access + * is the final one (callee / write target). Optional chaining (`?.`) never + * appears in the output (field-based traversal normalizes it); dynamic + * computed keys record nothing (documented KTD10 FN). + */ + private walkChain( + node: SyntaxNode, + acc: FactAccumulator, + skipFinalRead: boolean, + ): { path?: string; rootIdx?: number } { + // Collect accesses outer→inner (unshift), then resolve the root. + const accesses: Array<{ prop?: string; dynamicIndex?: SyntaxNode }> = []; + let cur: SyntaxNode = this.unwrapValueWrappers(node); + for (;;) { + if (cur.type === 'member_expression') { + const prop = cur.childForFieldName('property'); + accesses.unshift({ prop: prop?.text }); + const obj = cur.childForFieldName('object'); + if (!obj) break; + cur = this.unwrapValueWrappers(obj); + } else if (cur.type === 'subscript_expression') { + const index = cur.childForFieldName('index'); + if (index?.type === 'string') { + accesses.unshift({ prop: stringLiteralText(index) }); + } else { + accesses.unshift({ dynamicIndex: index ?? undefined }); + } + const obj = cur.childForFieldName('object'); + if (!obj) break; + cur = this.unwrapValueWrappers(obj); + } else { + break; + } + } + let rootIdx: number | undefined; + let rootSegment: string | undefined; + if (cur.type === 'identifier') { + rootIdx = this.resolve(cur); + acc.addUse(rootIdx); + rootSegment = cur.text; + } else if (cur.type === 'this' || cur.type === 'super') { + rootSegment = cur.text; // path segment only — `this`/`super` never bind + } else { + this.walkValue(cur, acc); // call-rooted etc. — uses + nested sites + } + // Dynamic subscript index expressions are real value reads (old default + // descent walked them) — inner→outer matches the old recording order. + for (const a of accesses) { + if (a.dynamicIndex) this.walkValue(a.dynamicIndex, acc); + } + const innermost = accesses[0]; + if ( + rootIdx !== undefined && + innermost?.prop !== undefined && + !(skipFinalRead && accesses.length === 1) + ) { + acc.addMemberRead(rootIdx, innermost.prop); + } + const path = + rootSegment !== undefined && accesses.every((a) => a.prop !== undefined) + ? [rootSegment, ...accesses.map((a) => a.prop as string)].join('.') + : undefined; + return { path, rootIdx }; + } +} + +/** Mutable build-time view of a {@link SiteRecord}. */ +interface MutableSite { + kind: SiteRecord['kind']; + parent?: [number, number]; + callee?: string; + receiver?: number; + args?: SiteArgOccurrence[][]; + resultDefs?: number[]; + spread?: number; + template?: boolean; + requireArg?: string; + object?: number; + property?: string; +} + +/** + * One open call/new site during the walk (#2083 M3 U1). `argIdx` is the + * argument position currently being walked, or -1 while outside any argument + * (callee walk) — occurrences recorded then do NOT land in this frame's args + * (they still fan out to enclosing arg-active frames, via-tagged through this + * frame's site: the receiver of a nested call flows into the outer argument + * through that call). + */ +interface SiteFrame { + siteIdx: number; + argIdx: number; } /** Ordered, deduplicating def/use collector for one statement record. */ @@ -613,6 +885,13 @@ class FactAccumulator { private readonly defSeen = new Set(); private readonly useSeen = new Set(); private readonly mayDefSeen = new Set(); + /** Taint sites recorded for this statement (#2083 M3 U1). */ + private readonly sites: MutableSite[] = []; + /** Composite (object|property|parent) keys of recorded member-read sites, so + * dedup is O(1) instead of a rescan of `sites` per read. */ + private readonly memberReadKeys = new Set(); + /** Stack of open call/new sites — the occurrence fan-out targets. */ + private readonly frames: SiteFrame[] = []; constructor(private readonly line: number) {} @@ -630,6 +909,17 @@ class FactAccumulator { } addUse(idx: number): void { + // Occurrence fan-out happens BEFORE the statement-level dedup: `exec(x, x)` + // records x at BOTH arg positions even though `uses` lists it once. + this.recordOccurrence(idx); + this.addUseWithoutOccurrence(idx); + } + + /** + * Statement-level use that is NOT a value occurrence in any open site + * argument — bare callee names only (#2083 M3 U1, see visitCall). + */ + addUseWithoutOccurrence(idx: number): void { if (this.useSeen.has(idx)) return; this.useSeen.add(idx); this.uses.push(idx); @@ -643,6 +933,147 @@ class FactAccumulator { return this.uses.length; } + // ── site machinery (#2083 M3 U1) ───────────────────────────────────────── + + /** `[defs.length, mayDefs.length]` marker for {@link defsSince}. */ + defSnapshot(): readonly [number, number] { + return [this.defs.length, this.mayDefs.length]; + } + + /** Binding indices def'd (must- OR may-) since the snapshot was taken. */ + defsSince(snap: readonly [number, number]): number[] { + return [...this.defs.slice(snap[0]), ...this.mayDefs.slice(snap[1])]; + } + + /** Open a call/new site; parent = innermost enclosing argument position. */ + openCallSite(kind: 'call' | 'new'): number { + const site: MutableSite = { kind }; + const parent = this.innermostArgPosition(); + if (parent) site.parent = parent; + this.sites.push(site); + return this.sites.length - 1; + } + + pushFrame(siteIdx: number): void { + this.frames.push({ siteIdx, argIdx: -1 }); + } + + popFrame(): void { + this.frames.pop(); + } + + /** Set the argument position the top frame is currently walking. */ + setFrameArg(argIdx: number): void { + const top = this.frames[this.frames.length - 1]; + if (top) top.argIdx = argIdx; + } + + /** + * Run `fn` with all open arg frames temporarily detached (argIdx = -1), so + * identifier reads inside still record USES but do NOT fan occurrences into + * the enclosing sink-argument position. Used for the non-value operands of a + * sequence (comma) expression — only the final operand's value flows. + */ + suppressOccurrences(fn: () => void): void { + const saved = this.frames.map((f) => f.argIdx); + for (const f of this.frames) f.argIdx = -1; + try { + fn(); + } finally { + this.frames.forEach((f, i) => { + f.argIdx = saved[i]; + }); + } + } + + setSiteCallee(siteIdx: number, callee: string): void { + this.sites[siteIdx].callee = callee; + } + + setSiteReceiver(siteIdx: number, receiver: number): void { + this.sites[siteIdx].receiver = receiver; + } + + setSiteResultDefs(siteIdx: number, resultDefs: readonly number[]): void { + this.sites[siteIdx].resultDefs = [...resultDefs]; + } + + setSiteSpread(siteIdx: number, firstSpreadArg: number): void { + const site = this.sites[siteIdx]; + if (site.spread === undefined) site.spread = firstSpreadArg; + } + + setSiteTemplate(siteIdx: number): void { + this.sites[siteIdx].template = true; + } + + setSiteRequireArg(siteIdx: number, literal: string): void { + this.sites[siteIdx].requireArg = literal; + } + + /** + * Record a value-position member read. Exact duplicates within the + * statement (same object/property/parent position) dedup; reads at + * DIFFERENT argument positions stay distinct (`exec(req.body, req.body)` + * is two occurrences — KTD6 finding identity needs both). + */ + addMemberRead(object: number, property: string): void { + const parent = this.innermostArgPosition(); + const dedupKey = `${object}|${property}|${parent ? `${parent[0]}:${parent[1]}` : 'top'}`; + if (this.memberReadKeys.has(dedupKey)) return; + this.memberReadKeys.add(dedupKey); + const site: MutableSite = { kind: 'member-read' }; + if (parent) site.parent = parent; + site.object = object; + site.property = property; + this.sites.push(site); + } + + private innermostArgPosition(): [number, number] | undefined { + for (let i = this.frames.length - 1; i >= 0; i--) { + const f = this.frames[i]; + if (f.argIdx >= 0) return [f.siteIdx, f.argIdx]; + } + return undefined; + } + + /** + * Fan a binding occurrence out to every arg-active open frame. The entry is + * via-tagged with the site of the IMMEDIATELY nested frame when one exists: + * `exec(escape(x))` puts a plain `x` in escape's arg 0 and `[x, escapeIdx]` + * in exec's arg 0 — the KTD4a interposition substrate. + */ + private recordOccurrence(idx: number): void { + for (let i = this.frames.length - 1; i >= 0; i--) { + const f = this.frames[i]; + if (f.argIdx < 0) continue; + const via = i + 1 < this.frames.length ? this.frames[i + 1].siteIdx : undefined; + this.pushArgEntry(f.siteIdx, f.argIdx, idx, via); + } + } + + private pushArgEntry( + siteIdx: number, + argIdx: number, + bindingIdx: number, + via: number | undefined, + ): void { + const site = this.sites[siteIdx]; + const args = (site.args ??= []); + while (args.length <= argIdx) args.push([]); + const list = args[argIdx]; + // Dedup exact (binding, via) pairs per position — `f(x + x)` is one entry; + // `f(x + g(x))` keeps the plain AND the via-tagged entry (distinct paths). + for (const e of list) { + const match = + typeof e === 'number' + ? via === undefined && e === bindingIdx + : via !== undefined && e[0] === bindingIdx && e[1] === via; + if (match) return; + } + list.push(via === undefined ? bindingIdx : [bindingIdx, via]); + } + finish(): StatementFacts { return { line: this.line, @@ -651,6 +1082,21 @@ class FactAccumulator { // Optional field stays absent when empty — keeps the serialized // side-channel payload lean (most statements have no may-defs). ...(this.mayDefs.length > 0 ? { mayDefs: this.mayDefs } : {}), + // Sites likewise omit-when-empty (#2083 M3 U1): flag-off runs never + // harvest, and most fact-bearing statements carry no calls. + ...(this.sites.length > 0 ? { sites: this.sites.map(finalizeSite) } : {}), }; } } + +/** Trim trailing empty arg positions; drop `args` entirely when all-empty. */ +const finalizeSite = (site: MutableSite): SiteRecord => { + const args = site.args; + if (args !== undefined) { + let end = args.length; + while (end > 0 && args[end - 1].length === 0) end--; + if (end === 0) delete site.args; + else if (end < args.length) site.args = args.slice(0, end); + } + return site as SiteRecord; +}; diff --git a/gitnexus/src/core/ingestion/pipeline.ts b/gitnexus/src/core/ingestion/pipeline.ts index f1f832632..d8b9fa16b 100644 --- a/gitnexus/src/core/ingestion/pipeline.ts +++ b/gitnexus/src/core/ingestion/pipeline.ts @@ -82,6 +82,22 @@ export interface PipelineOptions { * programmatic / server path only, like the M1 caps. */ pdgMaxReachingDefEdgesPerFunction?: number; + /** + * Per-function taint findings cap for the scope-resolution taint pass + * (#2083 M3). `undefined` ⇒ `DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION` + * (200); `0` ⇒ no cap (unlimited). Emit-time-only — NOT folded into the + * parse-cache chunk key; recorded resolved in `RepoMeta.pdg` so a cap + * change forces a full writeback. No CLI flag or rc key (KTD8) — + * programmatic / server path only, like the other pdg caps. + */ + pdgMaxTaintFindingsPerFunction?: number; + /** + * Per-finding taint hop cap (#2083 M3, KTD6 — bounds the persisted + * hop-encoded `reason`). `undefined` ⇒ `DEFAULT_PDG_MAX_TAINT_HOPS` (32); + * `0` ⇒ no cap (unlimited). Same emit-time-only / RepoMeta-stamped / + * no-CLI-flag discipline as `pdgMaxTaintFindingsPerFunction`. + */ + pdgMaxTaintHops?: number; /** * Request parsing with the worker pool disabled. The sequential parser was * removed — the worker pool is the sole parse path — so setting this now diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts index 7d0fd811e..dc0339c42 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts @@ -354,6 +354,8 @@ export const scopeResolutionPhase: PipelinePhase = { pdg: ctx.options?.pdg === true, pdgMaxEdgesPerFunction: ctx.options?.pdgMaxEdgesPerFunction, pdgMaxReachingDefEdgesPerFunction: ctx.options?.pdgMaxReachingDefEdgesPerFunction, + pdgMaxTaintFindingsPerFunction: ctx.options?.pdgMaxTaintFindingsPerFunction, + pdgMaxTaintHops: ctx.options?.pdgMaxTaintHops, recordResolutionOutcome: (outcome) => { resolutionOutcomes.push(outcome); }, diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts index e8d5fe6b5..2398e5ae6 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts @@ -40,7 +40,16 @@ import { isEmitSafeCfg, DEFAULT_MAX_CFG_EDGES_PER_FUNCTION, DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, + REACHING_DEF_FACTS_PER_EDGE_CAP, } from '../../cfg/emit.js'; +import { + emitFileTaint, + DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION, + DEFAULT_PDG_MAX_TAINT_HOPS, + type TaintEmitLimits, +} from '../../taint/emit.js'; +import { registerBuiltinTaintModels } from '../../taint/typescript-model.js'; +import { getSourceSinkConfig } from '../../taint/source-sink-registry.js'; import type { FunctionCfg } from '../../cfg/types.js'; import { resolveDefGraphId } from '../graph-bridge/ids.js'; import { buildPopulatedMethodDispatch } from '../graph-bridge/method-dispatch.js'; @@ -273,6 +282,14 @@ interface RunScopeResolutionInput { /** Per-function REACHING_DEF edge cap (#2082 M2). `undefined` ⇒ * {@link DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION}; `0` ⇒ no cap. */ readonly pdgMaxReachingDefEdgesPerFunction?: number; + /** Per-function taint findings cap (#2083 M3, consumed by the U4 taint + * emit step in the pdg window). `undefined` ⇒ + * `DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION` (200); `0` ⇒ no cap. */ + readonly pdgMaxTaintFindingsPerFunction?: number; + /** Per-finding taint hop cap (#2083 M3 KTD6 — bounds the hop-encoded + * `reason`; consumed by the U4 taint emit step). `undefined` ⇒ + * `DEFAULT_PDG_MAX_TAINT_HOPS` (32); `0` ⇒ no cap. */ + readonly pdgMaxTaintHops?: number; /** * Optional graph-node lookup built ONCE by the caller and shared across * every language pass. `buildGraphNodeLookup` scans the whole graph and is @@ -713,6 +730,11 @@ export function runScopeResolution( // pair can't bracket them; without this accumulator the M2 cost would // silently disappear into `emit=` and field regressions would be invisible. let pdgMs = 0; + // M3 (#2083 U4): accumulated taint time (match + taint-side solve + + // propagate + TAINTED/SANITIZES emit), a sibling of `pdgMs` for the same + // reason — it interleaves per file inside `emit=`, so only an accumulator + // can bracket it. Printed as the PROF `taint=` segment. + let taintMs = 0; if (input.pdg === true) { let cfgBlocks = 0; let cfgEdges = 0; @@ -721,6 +743,47 @@ export function runScopeResolution( let rdDropped = 0; let rdFacts = 0; let rdTruncated = 0; + // ── M3 taint setup (#2083 U4) ──────────────────────────────────────── + // Explicit model-registration seam (idempotent, cheap) — the registry + // stays empty on non-pdg runs, preserving default-run parity. The + // registry is keyed by `SupportedLanguages` enum VALUES ('typescript' / + // 'javascript'), and `ScopeResolver.language` IS a `SupportedLanguages` + // member registered under those same constants — the join is direct + // equality, no mapping table. A language without a registered spec + // (python, go, …) skips taint entirely: no work, no warn spam (KTD8). + registerBuiltinTaintModels(); + const taintSpec = getSourceSinkConfig(provider.language); + // Taint-side solver fact cap: the SAME derivation emitFileReachingDefs + // uses for the RD projection (edge cap × headroom factor, 0 ⇒ unlimited), + // so taint coverage and RD coverage truncate together — a function is + // never a taint coverage gap while its RD projection computed, and the + // RD layer's per-function truncation warn already names it. + const rdEdgeCap = + input.pdgMaxReachingDefEdgesPerFunction ?? DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION; + const taintLimits: TaintEmitLimits = { + maxFindingsPerFunction: + input.pdgMaxTaintFindingsPerFunction ?? DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION, + maxHops: input.pdgMaxTaintHops ?? DEFAULT_PDG_MAX_TAINT_HOPS, + maxFacts: rdEdgeCap > 0 ? rdEdgeCap * REACHING_DEF_FACTS_PER_EDGE_CAP : 0, + }; + // Cross-file aggregate of EVERY TaintEmitResult counter (the M2 emit + // result shipped with two fields dropped on the floor — R4 forbids that + // here; gaps/drops feed the unconditional warn below, volume feeds the + // per-language debug line). + const taintTotals = { + analyzed: 0, + noMatch: 0, + unsafeSites: 0, + gapTruncated: 0, + gapOverflow: 0, + gapNoFacts: 0, + findings: 0, + kills: 0, + dropped: 0, + hopsTruncated: 0, + gapExamples: [] as string[], + dropExamples: [] as string[], + }; for (const pf of emitParsedFiles) { const cfgs = pf.cfgSideChannel; // Defensive: cfgSideChannel is opaque (`unknown`) and crosses the cache / @@ -777,6 +840,39 @@ export function runScopeResolution( rdDropped += rd.droppedEdges; rdFacts += rd.facts; rdTruncated += rd.truncatedFunctions; + + // M3 (#2083 U4): taint over the SAME validated CFGs, inside the SAME + // per-file try (a taint throw costs this file's taint layer only — + // its CFG/REACHING_DEF edges above are already in the graph). Skipped + // entirely when the language has no registered model. + if (taintSpec !== undefined) { + const t1 = PROF ? performance.now() : 0; + const taint = emitFileTaint( + graph, + wellFormed, + pf.parsedImports, + taintSpec, + taintLimits, + (message) => logger.warn(message), // unconditional — R4/R6 + ); + if (PROF) taintMs += performance.now() - t1; + taintTotals.analyzed += taint.functionsAnalyzed; + taintTotals.noMatch += taint.functionsSkippedNoMatch; + taintTotals.unsafeSites += taint.functionsSkippedUnsafeSites; + taintTotals.gapTruncated += taint.functionsCoverageGap.truncated; + taintTotals.gapOverflow += taint.functionsCoverageGap.overflow; + taintTotals.gapNoFacts += taint.functionsCoverageGap['no-facts']; + taintTotals.findings += taint.findingsEmitted; + taintTotals.kills += taint.killsEmitted; + taintTotals.dropped += taint.findingsDropped; + taintTotals.hopsTruncated += taint.hopsTruncatedFindings; + for (const ex of taint.coverageGapExamples) { + if (taintTotals.gapExamples.length < 5) taintTotals.gapExamples.push(ex); + } + for (const ex of taint.droppedExamples) { + if (taintTotals.dropExamples.length < 5) taintTotals.dropExamples.push(ex); + } + } } catch (err) { // Last-resort isolation, mirroring the worker-side per-file try/catch: // a shape the predicate misses must cost this one file's CFG, not @@ -798,9 +894,54 @@ export function runScopeResolution( (cfgDroppedEdges > 0 ? `, ${cfgDroppedEdges} edges dropped (per-function cap)` : '') + `; ${rdEdges} REACHING_DEF edges (${rdFacts} facts)` + (rdDropped > 0 ? `, ${rdDropped} REACHING_DEF edges dropped (per-function cap)` : '') + - (rdTruncated > 0 ? `, ${rdTruncated} function(s) hit the fact limit` : ''), + (rdTruncated > 0 ? `, ${rdTruncated} function(s) hit the fact limit` : '') + + // M3 volume telemetry — only for languages with a registered model. + (taintSpec !== undefined + ? `; taint: ${taintTotals.findings} TAINTED, ${taintTotals.kills} SANITIZES ` + + `(${taintTotals.analyzed} function(s) analyzed, ` + + `${taintTotals.noMatch} skipped: no source/sink match` + + (taintTotals.hopsTruncated > 0 + ? `, ${taintTotals.hopsTruncated} finding(s) with truncated hop paths` + : '') + + `)` + : ''), ); } + // R4: taint coverage gaps and cap drops surface UNCONDITIONALLY (never + // logger.debug, never input.onWarn) at the per-language aggregate, with + // counts and up to 5 example functions. Per-function warns above cover + // the rare/actionable cases (unsafe sites, cap drops); solver-status gaps + // were already per-function-warned by the RD layer (same solver, same + // fact cap), so this aggregate is their single taint-side surface. + if (taintSpec !== undefined) { + const gapCount = + taintTotals.unsafeSites + + taintTotals.gapTruncated + + taintTotals.gapOverflow + + taintTotals.gapNoFacts; + if (gapCount > 0 || taintTotals.dropped > 0) { + const parts: string[] = []; + if (gapCount > 0) { + parts.push( + `${gapCount} function(s) skipped for taint ` + + `(${taintTotals.gapTruncated} fact-limit, ${taintTotals.gapOverflow} overflow, ` + + `${taintTotals.gapNoFacts} no-facts, ${taintTotals.unsafeSites} malformed sites)` + + (taintTotals.gapExamples.length > 0 + ? ` — e.g. ${taintTotals.gapExamples.join(', ')}` + : ''), + ); + } + if (taintTotals.dropped > 0) { + parts.push( + `${taintTotals.dropped} finding(s) dropped by the per-function cap` + + (taintTotals.dropExamples.length > 0 + ? ` — e.g. ${taintTotals.dropExamples.join(', ')}` + : ''), + ); + } + logger.warn(`[taint] lang=${provider.language}: ${parts.join('; ')}`); + } + } } if (PROF) { @@ -813,7 +954,8 @@ export function runScopeResolution( ` resolve=${ns(tPropagate, tResolve).toFixed(0)}ms` + ` emit=${ns(tResolve, tEnd).toFixed(0)}ms` + // pdg ⊆ emit: the M2 reaching-defs share of the emit bucket (#2082 U4). - (input.pdg === true ? ` pdg=${pdgMs.toFixed(0)}ms` : '') + + // taint ⊆ emit likewise: the M3 match+solve+propagate+emit share (#2083 U4). + (input.pdg === true ? ` pdg=${pdgMs.toFixed(0)}ms taint=${taintMs.toFixed(0)}ms` : '') + ` total=${ns(tStart, tEnd).toFixed(0)}ms` + ` (${parsedFiles.length} files)`, ); diff --git a/gitnexus/src/core/ingestion/taint/emit.ts b/gitnexus/src/core/ingestion/taint/emit.ts new file mode 100644 index 000000000..d11694b28 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/emit.ts @@ -0,0 +1,297 @@ +/** + * In-phase taint emission (#2083 M3 U4, plan KTD1/KTD6). + * + * Per-file driver for the M3 taint pass: gate → match → solve → propagate → + * persist sparse `TAINTED` + `SANITIZES` edges. Invoked from the pdg window in + * scope-resolution (`pipeline/run.ts`), immediately after `emitFileReachingDefs` + * inside the SAME per-file try — per-file isolation for free (KTD1). Mirrors + * `emitFileReachingDefs` (cfg/emit.ts) for the budget/dedup/warn discipline and + * the telemetry-result shape. + * + * ## Per-function pipeline (ordering is load-bearing) + * + * 1. `hasTaintSafeSites` — a corrupted-store site annotation degrades to + * SKIP-TAINT-KEEP-RD for this function (counted + warned), never a crash + * (KTD2; the matcher/propagator dereference indices unvalidated). + * 2. `matchFunctionSites` against the language spec (the import index is + * built ONCE per file — imports are a file-level fact). + * 3. ZERO-MATCH FAST PATH: the solver runs only when the function has at + * least one matched source AND one matched sink. In a typical repo almost + * no function has both; an unconditional second `computeReachingDefs` per + * function would ship a near-2× solve cost to every `--pdg` user. + * 4. `computeReachingDefs` with the taint `maxFacts` — by DEFAULT the M2 + * derived `DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION` (deliberate + * reuse, not a new constant: the fact-materialization envelope is a + * memory question, O(defs×uses), orthogonal to the findings cap, and M2 + * already validated exactly this envelope on the same solver in the same + * window). The run.ts caller derives `limits.maxFacts` from the SAME + * RD-edge-cap formula `emitFileReachingDefs` uses, so taint coverage and + * RD coverage truncate together — a function is never `truncated` for one + * layer and `computed` for the other. + * 5. `computeTaintFlows` — a non-`computed` status is a per-function + * COVERAGE GAP (R4: counted by `gapReason`, function skipped entirely, + * never partially analyzed). + * 6. Emit one `TAINTED` edge per finding and one `SANITIZES` edge per kill. + * Kills are emitted even when findings are zero — a fully-sanitized + * function's kills are exactly its evidence of safety. + * + * ## Identity, dedup, budget (KTD6) + * + * Findings carry STATEMENT-LEVEL identity — function anchor + sink kind + + * source occurrence (point/site/object-binding/property) + sink occurrence + * (point/site/arg/binding) — NOT the REACHING_DEF block-level key (block-pair + * conflation would drop `exec(req.body, req.query)`'s second finding). The + * propagation engine dedups by this exact key BEFORE its deterministic cap + * (`maxFindingsPerFunction`) and counts the overflow; this module templates + * the same coordinates into the edge id (binding identity via the shared + * `bindingKey`; the free-text `property` rides LAST so it can never collide + * into another component) and warns with the drop count on truncation. + * + * `reason` carries the versioned hop encoding (`taint/path-codec.ts` — U6's + * `explain` decodes the same module) for `TAINTED`, and the killed binding's + * plain name for `SANITIZES` (M0/S1 queryability verdict, like REACHING_DEF). + * + * ## Warn split (R4 vs noise) + * + * Unsafe-site skips and cap drops warn PER FUNCTION here (rare, actionable — + * mirrors `emitFileReachingDefs`' malformed/cap warns). Solver coverage gaps + * (`truncated`/`overflow`) do NOT re-warn per function: the RD layer already + * warned for the same function with the same solver status (same `maxFacts` + * derivation — see step 4), and a duplicate `[taint]` line per mega-function + * would be pure spam. They are counted (+ exampled) in the result and the + * run.ts caller aggregates them into ONE unconditional `logger.warn` per + * language (R4) — never dropped on the floor (the M2 lesson). + */ + +import type { ParsedImport } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../../graph/types.js'; +import { generateId } from '../../../lib/utils.js'; +import { + basicBlockId, + bindingKey, + DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, +} from '../cfg/emit.js'; +import { computeReachingDefs, pointKey, type ProgramPoint } from '../cfg/reaching-defs.js'; +import type { BindingEntry, FunctionCfg } from '../cfg/types.js'; +import { hasTaintSafeSites } from './site-safety.js'; +import { buildTaintImportIndex, matchFunctionSites } from './match.js'; +import { + computeTaintFlows, + DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION, + DEFAULT_PDG_MAX_TAINT_HOPS, +} from './propagate.js'; + +// Re-exported so the pipeline (run.ts) sources the taint default caps through +// this orchestration module rather than reaching into propagate.ts directly. +export { + DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION, + DEFAULT_PDG_MAX_TAINT_HOPS, +} from './propagate.js'; +import { encodeTaintPath } from './path-codec.js'; +import type { SourceSinkSanitizerSpec } from './source-sink-config.js'; + +/** Cap on example anchors carried per result (aggregate-warn material, R4). */ +const MAX_EXAMPLES = 5; + +export interface TaintEmitLimits { + /** Per-function findings cap (post-dedup). `undefined` ⇒ + * {@link DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION}; `0` ⇒ unlimited. */ + readonly maxFindingsPerFunction?: number; + /** Per-finding hop cap (source-side prefix kept). `undefined` ⇒ + * {@link DEFAULT_PDG_MAX_TAINT_HOPS}; `0` ⇒ unlimited. */ + readonly maxHops?: number; + /** + * Solver fact-materialization cap for the taint-side `computeReachingDefs` + * call. `undefined` ⇒ {@link DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION} + * (the M2 derived default — see the module doc for why it is REUSED rather + * than derived from the findings cap); `0` ⇒ unlimited. + */ + readonly maxFacts?: number; +} + +/** + * Full taint-emit telemetry for one file. EVERY counter is surfaced by the + * run.ts aggregate (the M2 emit result had two fields dropped on the floor — + * the plan names that mistake; don't repeat it). + */ +export interface TaintEmitResult { + /** Functions fully propagated (`computeTaintFlows` returned `computed`). */ + functionsAnalyzed: number; + /** Functions skipped by the zero-match fast path (no solver call). */ + functionsSkippedNoMatch: number; + /** Functions whose `sites` failed {@link hasTaintSafeSites} (skip-taint-keep-RD). */ + functionsSkippedUnsafeSites: number; + /** Source+sink functions skipped on a non-`computed` solver status (R4). */ + functionsCoverageGap: { truncated: number; overflow: number; 'no-facts': number }; + /** TAINTED edges persisted. */ + findingsEmitted: number; + /** SANITIZES edges persisted (emitted even when findings are zero). */ + killsEmitted: number; + /** Findings dropped by the per-function cap (post-dedup), summed. */ + findingsDropped: number; + /** Findings whose persisted hop path is a truncated prefix (hop/byte cap). */ + hopsTruncatedFindings: number; + /** ≤{@link MAX_EXAMPLES} `file:line` anchors of gap/unsafe-site functions. */ + coverageGapExamples: string[]; + /** ≤{@link MAX_EXAMPLES} `file:line` anchors of cap-dropped functions. */ + droppedExamples: string[]; +} + +const pushExample = (list: string[], anchor: string): void => { + if (list.length < MAX_EXAMPLES) list.push(anchor); +}; + +/** + * Run the taint pass over one file's emit-safe CFGs and persist TAINTED + + * SANITIZES edges. `cfgs` MUST already be `isEmitSafeCfg`-filtered (the same + * `wellFormed` array the caller fed `emitFileCfgs`/`emitFileReachingDefs`) — + * block/edge anchors are trusted here; only the M3 `sites` layer is + * re-validated (`hasTaintSafeSites`). Never throws on well-formed input; + * the caller's per-file try isolates the rest. + */ +export function emitFileTaint( + graph: KnowledgeGraph, + cfgs: readonly FunctionCfg[], + parsedImports: readonly ParsedImport[], + spec: SourceSinkSanitizerSpec, + limits?: TaintEmitLimits, + onWarn?: (message: string) => void, +): TaintEmitResult { + const result: TaintEmitResult = { + functionsAnalyzed: 0, + functionsSkippedNoMatch: 0, + functionsSkippedUnsafeSites: 0, + functionsCoverageGap: { truncated: 0, overflow: 0, 'no-facts': 0 }, + findingsEmitted: 0, + killsEmitted: 0, + findingsDropped: 0, + hopsTruncatedFindings: 0, + coverageGapExamples: [], + droppedExamples: [], + }; + + // Imports are a FILE-level fact — build the index once, not per function. + const importIndex = buildTaintImportIndex(parsedImports); + const maxFindingsPerFunction = + limits?.maxFindingsPerFunction ?? DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION; + const maxHops = limits?.maxHops ?? DEFAULT_PDG_MAX_TAINT_HOPS; + const maxFacts = limits?.maxFacts ?? DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION; + + // Defensive cross-CFG id guard: finding identity is unique WITHIN a + // function by construction (the propagation engine dedups), so a repeat can + // only mean two CFGs sharing an anchor — skip, never double-insert. + const seenEdgeIds = new Set(); + + for (const cfg of cfgs) { + const { filePath, functionStartLine, functionStartColumn } = cfg; + const anchor = `${filePath}:${functionStartLine}`; + + if (!hasTaintSafeSites(cfg)) { + result.functionsSkippedUnsafeSites++; + pushExample(result.coverageGapExamples, anchor); + onWarn?.( + `[taint] ${anchor}: malformed site annotations (out-of-range binding/site ` + + `indices) — taint skipped for this function; its CFG and REACHING_DEF ` + + `layers are unaffected`, + ); + continue; + } + + const matches = matchFunctionSites(cfg, spec, importIndex); + if (!matches.hasSource || !matches.hasSink) { + // Zero-match fast path: no solver call (see module doc step 3). + result.functionsSkippedNoMatch++; + continue; + } + + const defUse = computeReachingDefs(cfg, { maxFacts }); + const flows = computeTaintFlows(cfg, defUse, matches, { maxFindingsPerFunction, maxHops }); + if (flows.status === 'coverage-gap') { + // R4: skipped entirely, counted by reason; aggregate-warned by the + // caller (the RD layer already per-function-warned this solver status). + result.functionsCoverageGap[flows.gapReason ?? 'no-facts']++; + pushExample(result.coverageGapExamples, anchor); + continue; + } + result.functionsAnalyzed++; + + const bindings: readonly BindingEntry[] = cfg.bindings ?? []; + const fnAnchor = `${filePath}:${functionStartLine}:${functionStartColumn}`; + const blockId = (p: ProgramPoint): string => + basicBlockId(filePath, functionStartLine, functionStartColumn, p.blockIndex); + const bKey = (idx: number): string => { + const b = bindings[idx]; + return b === undefined ? `#${idx}` : bindingKey(b); + }; + + // SANITIZES — one edge per kill, REGARDLESS of findings (kills can and do + // exist with zero findings: a fully-sanitized flow IS the kill evidence). + for (const kill of flows.kills) { + const id = generateId( + 'SANITIZES', + `${fnAnchor}:${pointKey(kill.sanitizer)}->${pointKey(kill.killedDef)}:` + + bKey(kill.bindingIdx), + ); + if (seenEdgeIds.has(id)) continue; + seenEdgeIds.add(id); + graph.addRelationship({ + id, + type: 'SANITIZES', + sourceId: blockId(kill.sanitizer), + targetId: blockId(kill.killedDef), + confidence: 1.0, + reason: bindings[kill.bindingIdx]?.name ?? `#${kill.bindingIdx}`, + }); + result.killsEmitted++; + } + + // TAINTED — one edge per finding (already deduped + capped upstream). + for (const finding of flows.findings) { + const { source, sink } = finding; + // KTD6 statement-level identity: function anchor + kind + source + // occurrence + sink occurrence + binding keys. The rule-(b) occurrence + // coordinates (site index / arg index) distinguish + // `exec(req.body, req.query)`'s two findings; `property` is free-text + // (string-literal subscripts) and rides LAST so it cannot collide into + // another component. + const id = generateId( + 'TAINTED', + `${fnAnchor}:${finding.sinkKind}:` + + `${pointKey(source.point)}.${source.siteIndex}:${bKey(source.objectBindingIdx)}:` + + `${pointKey(sink.point)}.${sink.siteIndex}.${sink.argIndex}:${bKey(sink.bindingIdx)}:` + + `${sink.entryName}:${source.property}`, + ); + if (seenEdgeIds.has(id)) continue; + seenEdgeIds.add(id); + // `kind` rides the reason's `;` header — the only persisted + // channel for the finding's category (the edge id embedding it is not a + // stored column; `step` is INT32). U6's `explain` decodes it back. + const encoded = encodeTaintPath( + finding.hops.map((h) => ({ name: h.name, line: h.point.line, viaCall: h.viaCall })), + { truncated: finding.hopsTruncated === true, kind: finding.sinkKind }, + ); + if (encoded.truncated) result.hopsTruncatedFindings++; + graph.addRelationship({ + id, + type: 'TAINTED', + sourceId: blockId(source.point), + targetId: blockId(sink.point), + confidence: 1.0, + reason: encoded.reason, + }); + result.findingsEmitted++; + } + + if (flows.droppedFindings > 0) { + result.findingsDropped += flows.droppedFindings; + pushExample(result.droppedExamples, anchor); + onWarn?.( + `[taint] ${anchor}: per-function taint findings cap ` + + `(${maxFindingsPerFunction}) reached — dropped ${flows.droppedFindings} of ` + + `${flows.findings.length + flows.droppedFindings} deduped findings`, + ); + } + } + + return result; +} diff --git a/gitnexus/src/core/ingestion/taint/match.ts b/gitnexus/src/core/ingestion/taint/match.ts new file mode 100644 index 000000000..50c513182 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/match.ts @@ -0,0 +1,374 @@ +/** + * Import-aware taint-site matcher (#2083 M3 U2, plan KTD7). + * + * Classifies a function's harvested {@link SiteRecord}s against a registered + * {@link SourceSinkSanitizerSpec}: which member reads are SOURCES, which + * call/new sites are SINKS (and at which argument positions), and which are + * SANITIZERS. Pure main-thread data work — sites + bindings come from the U1 + * worker harvest, imports from `ParsedFile.parsedImports`; no AST, no I/O. + * + * PRECONDITION: the caller must gate the CFG through `hasTaintSafeSites` + * (taint/site-safety.ts) first — this module dereferences binding/site + * indices without re-validating them. + * + * ## Callee resolution precedence (bare and member-rooted calls) + * + * 1. ESM import join — the callee root's local name is resolved through the + * {@link TaintImportIndex} built from `parsedImports` (`named`/`alias` + * members, `namespace`/default-import module handles); `import { exec as + * run } from 'child_process'` makes `run(c)` resolve to + * `child_process.exec`, and `import * as cp …` makes `cp.exec(c)` resolve + * the same way. + * 2. require-literal join — a binding whose in-function defining site carries + * `requireArg` resolves like a namespace handle (`const cp = + * require('child_process'); cp.exec(c)`). A BARE call of a require-joined + * binding is matched under BOTH interpretations, `.default` (the + * module/default export invoked directly) and `.` + * (non-renamed destructured require — the harvest attaches `resultDefs` + * to destructured bindings without recording the property path, and the + * binding name IS the member name in the non-renamed case). + * 3. Bare-name fallback — TRUE GLOBALS only (`global: true` entries: `eval`, + * `new Function`, `encodeURIComponent`), and only when the name is neither + * import-bound nor shadowed. Conventional receiver names (`req`/`request` + * member-read sources, `res.send`, `.query`/`.execute`) are matched + * name-based by their own mechanisms, never via the global fallback. + * + * ## Shadowing rule (exact) + * + * A name is treated as function-local — blocking import/global resolution — + * iff the function's binding table contains a NON-`synthetic` entry with that + * name (an in-function `function exec(){}` / `const exec = …`). Synthetic + * bindings (kind `module`, `synthetic: true`) are imports, true globals, or + * enclosing-scope captures and do not shadow. Member-call roots use the + * harvested `receiver` binding index directly (no name scan). + * + * ## Documented resolution gaps (direction stated, per plan KTD10) + * + * - MODULE-LEVEL `const cp = require('child_process')`: the binding is + * synthetic inside the function, produces no `ParsedImport`, and its + * defining site lives outside the function's harvested sites — the + * require join cannot see it. Module-mechanism sinks miss (FN) and + * sanitizers don't kill (FP noise — never a false kill, the safe + * direction). Only in-function requires resolve. + * - RENAMED destructured require (`const { exec: run } = require(…)`): + * the dual interpretation resolves `run` to `child_process.run` — no + * match (FN). Non-renamed destructures resolve exactly. + * - CONSERVATIVE shadow scan for bare calls: ANY non-synthetic binding of + * the callee name anywhere in the function blocks import/global + * resolution, even when the shadow is block-scoped elsewhere and the call + * site actually sees the import (FN; rare; safe for sanitizers). + * - MODULE-LEVEL user declarations are indistinguishable from imports in + * the binding table (both synthetic). ESM forbids a module-level + * declaration colliding with an import name, so the import join is + * authoritative when an import exists; a module-level user function + * shadowing a TRUE GLOBAL (e.g. a local `encodeURIComponent`) is not + * detectable and would still match (pathological; accepted). + * - Handle COPIES (`const c2 = cp; c2.exec(…)`) are not followed — joins + * are one level deep (binding → import/require), never through + * assignments (FN). + * - `this.`/`super.`-rooted and call-rooted callee chains have no + * resolvable root: only the syntactic `anyReceiver`/`receivers` + * mechanisms can match them. + * - `reexport`/`wildcard`/`dynamic-*`/`side-effect` imports introduce no + * matcher-visible local binding and are skipped by the index. + */ + +import type { ParsedImport } from 'gitnexus-shared'; +import type { FunctionCfg, SiteRecord } from '../cfg/types.js'; +import type { + SourceSinkSanitizerSpec, + TaintMemberSourceEntry, + TaintSanitizerEntry, + TaintSinkEntry, +} from './source-sink-config.js'; + +/** What a local name imported into the file denotes. */ +export interface TaintImportBinding { + /** Normalized module specifier (`node:` scheme stripped). */ + readonly module: string; + /** + * Exported member bound by a named/aliased import; `undefined` when the + * local name is a MODULE HANDLE (namespace import, or a default import — + * CJS interop makes the default export ≈ the module object). + */ + readonly member?: string; +} + +/** Local name → import provenance for one file. Build once per file (U4). */ +export type TaintImportIndex = ReadonlyMap; + +/** A member-read site matched as a taint source. */ +export interface MatchedSourceRead { + /** Index into the owning statement's `sites` array. */ + readonly siteIndex: number; + readonly entry: TaintMemberSourceEntry; +} + +/** A call/new site matched as a sink. */ +export interface MatchedSinkCall { + /** Index into the owning statement's `sites` array. */ + readonly siteIndex: number; + readonly entry: TaintSinkEntry; + /** + * Positions (indices into `site.args`) that are registered sink positions + * AND carry at least one recorded binding occurrence, after the spread + * rule (a recorded position ≥ `site.spread` matches when any registered + * position ≥ the spread index exists — runtime positions after a spread + * are unknowable) and the template rule (`template: true` aggregates all + * substitutions at position 0 and matches any-position). Never empty — a + * sink whose dangerous positions carry no occurrences cannot produce a + * finding and is not reported. + */ + readonly argPositions: readonly number[]; +} + +/** A call site matched as a sanitizer (import-aware/global only — see module doc). */ +export interface MatchedSanitizerCall { + /** Index into the owning statement's `sites` array. */ + readonly siteIndex: number; + readonly entry: TaintSanitizerEntry; + /** + * Bindings the sanitizer's result defines directly (`const b = escape(t)` + * ⇒ `b`) — U3's kill targets (KTD4b). Empty for value-position sanitizer + * calls (`exec(escape(x))`), whose effect is occurrence INTERPOSITION via + * the site's `parent`/via-tag chain, not a def kill. + */ + readonly resultDefs: readonly number[]; +} + +/** All matches within one statement. Emitted only when at least one list is non-empty. */ +export interface StatementMatches { + readonly blockIndex: number; + readonly statementIndex: number; + readonly line: number; + readonly sources: readonly MatchedSourceRead[]; + readonly sinks: readonly MatchedSinkCall[]; + readonly sanitizers: readonly MatchedSanitizerCall[]; +} + +/** Classified sites for one function, in (block, statement, site, entry) order. */ +export interface FunctionSiteMatches { + readonly statements: readonly StatementMatches[]; + /** Fast-path gates for U4: the solver runs only when both are true. */ + readonly hasSource: boolean; + readonly hasSink: boolean; +} + +const stripNodeScheme = (specifier: string): string => + specifier.startsWith('node:') ? specifier.slice('node:'.length) : specifier; + +/** + * Build the local-name → module/member index from a file's `parsedImports`. + * Only `named`/`alias`/`namespace` kinds bind matcher-visible local names; + * `importedName === 'default'` collapses to a module handle. + */ +export function buildTaintImportIndex(imports: readonly ParsedImport[]): TaintImportIndex { + const index = new Map(); + for (const imp of imports) { + if (imp.kind === 'named' || imp.kind === 'alias') { + const module = stripNodeScheme(imp.targetRaw); + index.set( + imp.localName, + imp.importedName === 'default' ? { module } : { module, member: imp.importedName }, + ); + } else if (imp.kind === 'namespace') { + index.set(imp.localName, { module: stripNodeScheme(imp.targetRaw) }); + } + } + return index; +} + +/** Internal: a callee's resolution — canonical dotted names + syntactic path. */ +interface ResolvedCallee { + /** Syntactic dotted path segments (`cp.exec` ⇒ `['cp','exec']`). */ + readonly path: readonly string[]; + /** Module-resolved canonical names this callee may denote. */ + readonly canonical: readonly string[]; + /** True when the bare root may denote an ECMAScript global (unshadowed, un-imported). */ + readonly globalRoot: boolean; +} + +/** + * Classify a function's harvested sites against a language spec. See the + * module doc for resolution precedence, the shadowing rule, and gaps. + */ +export function matchFunctionSites( + cfg: FunctionCfg, + spec: SourceSinkSanitizerSpec, + imports: TaintImportIndex, +): FunctionSiteMatches { + const bindings = cfg.bindings ?? []; + + // Non-synthetic (in-function-declared) binding indices by name — the + // shadow scan + bare-call require-join lookup. + const nonSyntheticByName = new Map(); + bindings.forEach((b, i) => { + if (b.synthetic === true) return; + const list = nonSyntheticByName.get(b.name); + if (list) list.push(i); + else nonSyntheticByName.set(b.name, [i]); + }); + + // require-literal join: binding index → module specifier. A binding def'd + // by two DIFFERENT require literals is conflicted → dropped (resolving it + // either way could fabricate a sanitizer kill). + const requireByBinding = new Map(); + const conflicted = new Set(); + for (const block of cfg.blocks) { + for (const stmt of block.statements ?? []) { + for (const site of stmt.sites ?? []) { + if (site.requireArg === undefined || site.resultDefs === undefined) continue; + const module = stripNodeScheme(site.requireArg); + for (const def of site.resultDefs) { + if (conflicted.has(def)) continue; + const prior = requireByBinding.get(def); + if (prior === undefined) requireByBinding.set(def, module); + else if (prior !== module) { + requireByBinding.delete(def); + conflicted.add(def); + } + } + } + } + } + + const resolveCallee = (site: SiteRecord): ResolvedCallee | undefined => { + if (site.callee === undefined) return undefined; + const path = site.callee.split('.'); + const root = path[0]; + const rest = path.slice(1); + const canonical: string[] = []; + let globalRoot = false; + + if (site.receiver !== undefined) { + // Member chain with an identifier root — origin known by binding index. + const rb = bindings[site.receiver]; + if (rb.synthetic === true) { + const imp = imports.get(rb.name); + if (imp !== undefined) { + const base = imp.member === undefined ? [imp.module] : [imp.module, imp.member]; + canonical.push([...base, ...rest].join('.')); + } + } else { + const module = requireByBinding.get(site.receiver); + if (module !== undefined) canonical.push([module, ...rest].join('.')); + } + } else if (path.length === 1) { + // Bare call. `this`/`super`/call-rooted chains never get here (those + // are dotted-without-receiver or callee-less). + const locals = nonSyntheticByName.get(root); + if (locals !== undefined) { + // Shadowed by an in-function declaration — only the require join + // applies, under the dual interpretation (module doc). + for (const idx of locals) { + const module = requireByBinding.get(idx); + if (module !== undefined) canonical.push(`${module}.default`, `${module}.${root}`); + } + } else { + const imp = imports.get(root); + if (imp !== undefined) { + canonical.push( + imp.member === undefined ? `${imp.module}.default` : `${imp.module}.${imp.member}`, + ); + } else { + globalRoot = true; + } + } + } + return { path, canonical, globalRoot }; + }; + + /** The spread/template/registered-position rule (see MatchedSinkCall doc). */ + const positionMatches = (entry: TaintSinkEntry, site: SiteRecord, p: number): boolean => { + if (site.template === true) return true; + if (entry.args === undefined) return true; + if (site.spread !== undefined && p >= site.spread) { + const spread = site.spread; + return entry.args.some((q) => q >= spread); + } + return entry.args.includes(p); + }; + + const sinkMechanismHit = ( + entry: TaintSinkEntry, + site: SiteRecord, + r: ResolvedCallee, + ): boolean => { + if (entry.module !== undefined) return r.canonical.includes(`${entry.module}.${entry.name}`); + if (entry.global === true) { + return ( + r.globalRoot && + r.path.length === 1 && + r.path[0] === entry.name && + (entry.newOnly !== true || site.kind === 'new') + ); + } + if (entry.anyReceiver === true) { + return r.path.length >= 2 && r.path[r.path.length - 1] === entry.name; + } + if (entry.receivers !== undefined) { + return r.path.length === 2 && entry.receivers.includes(r.path[0]) && r.path[1] === entry.name; + } + return false; + }; + + // Sanitizers: module + global mechanisms ONLY — never receiver-conventional, + // never bare-name for non-globals (a false kill is the forbidden direction). + const sanitizerMechanismHit = (entry: TaintSanitizerEntry, r: ResolvedCallee): boolean => { + if (entry.module !== undefined) return r.canonical.includes(`${entry.module}.${entry.name}`); + if (entry.global === true) { + return r.globalRoot && r.path.length === 1 && r.path[0] === entry.name; + } + return false; + }; + + const statements: StatementMatches[] = []; + let hasSource = false; + let hasSink = false; + + cfg.blocks.forEach((block, blockIndex) => { + block.statements?.forEach((stmt, statementIndex) => { + const sites = stmt.sites; + if (sites === undefined || sites.length === 0) return; + const sources: MatchedSourceRead[] = []; + const sinks: MatchedSinkCall[] = []; + const sanitizers: MatchedSanitizerCall[] = []; + + sites.forEach((site, siteIndex) => { + if (site.kind === 'member-read') { + if (site.object === undefined || site.property === undefined) return; + const objectName = bindings[site.object].name; + const property = site.property; + for (const entry of spec.sources) { + if (entry.objects.includes(objectName) && entry.properties.includes(property)) { + sources.push({ siteIndex, entry }); + } + } + return; + } + // call / new + const resolved = resolveCallee(site); + if (resolved === undefined) return; + for (const entry of spec.sinks) { + if (!sinkMechanismHit(entry, site, resolved)) continue; + const argPositions: number[] = []; + site.args?.forEach((occurrences, p) => { + if (occurrences.length > 0 && positionMatches(entry, site, p)) argPositions.push(p); + }); + if (argPositions.length > 0) sinks.push({ siteIndex, entry, argPositions }); + } + for (const entry of spec.sanitizers) { + if (!sanitizerMechanismHit(entry, resolved)) continue; + sanitizers.push({ siteIndex, entry, resultDefs: site.resultDefs ?? [] }); + } + }); + + if (sources.length === 0 && sinks.length === 0 && sanitizers.length === 0) return; + hasSource ||= sources.length > 0; + hasSink ||= sinks.length > 0; + statements.push({ blockIndex, statementIndex, line: stmt.line, sources, sinks, sanitizers }); + }); + }); + + return { statements, hasSource, hasSink }; +} diff --git a/gitnexus/src/core/ingestion/taint/path-codec.ts b/gitnexus/src/core/ingestion/taint/path-codec.ts new file mode 100644 index 000000000..a7ceee6f7 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/path-codec.ts @@ -0,0 +1,267 @@ +/** + * Taint-path reason codec (#2083 M3 U4/U6, plan KTD6). + * + * THE one shared encoder/decoder for the hop-encoded `reason` carried on + * persisted `TAINTED` edges: the U4 emit path writes it, the U6 MCP `explain` + * tool reads it. Two hand-rolled copies of a wire format drift — both sides + * MUST import from here. + * + * ## Wire format (version `1`) + * + * ``` + * 1[;]|:[:]|:[:]|…[|~] + * ``` + * + * - One-character version prefix (`TAINT_PATH_CODEC_VERSION`), then an + * OPTIONAL `;` header segment, then ordered source→sink hops, each + * `variable:line[:flags]`. + * - `kind` is the finding's sink category (`SinkKind`, e.g. + * `command-injection`). It rides the reason because it is the ONLY + * persisted channel: the CodeRelation columns are + * `type/confidence/reason/step` — `step` is INT32 and the emit-time edge id + * (which embeds the kind) is not a stored column. The U6 `explain` tool + * reads it for finding classification. Charset `[a-z0-9-]` (printable + * ASCII, disjoint from every structural delimiter); `;` itself is printable + * ASCII and never appears in hop names (identifier charset) or flags. + * U6 deviation note: this header was added by U6 WITHIN version `1` — + * U4 and U6 ship in the same release, so no reason string without the + * header was ever persisted by a released build; the decoder still accepts + * header-less strings (`kind` simply decodes as `undefined`). + * - `flags` is a lowercase-letter set; only `c` (= the hop passed through an + * unmodeled call, KTD5 `viaCall`) is defined today — the rest of the + * alphabet is RESERVED, and the decoder accepts unknown flag letters so a + * future writer's output stays decodable. + * - A trailing `|~` segment is the TRUNCATION MARKER: the encoded path is a + * source-side PREFIX of the real one (hop cap, byte cap, or an unencodable + * hop name). Decoders MUST report it as "path incomplete" — never an error. + * + * ## Delimiter / round-trip discipline (KTD6) + * + * Every structural character (`|`, `:`, `~`, digits, flag letters) is + * printable ASCII: `sanitizeUTF8` (csv-generator.ts) strips control + * characters, lone surrogates, and U+FFFE/FFFF — printable ASCII passes + * through byte-exact, so the encoding survives `escapeCSVField ∘ + * sanitizeUTF8` and the DB load unchanged (pinned by the round-trip test). + * None of the delimiters can appear in a JS identifier. + * + * Hop names are identifier-charset by U1 construction (the harvest records + * binding names), but the encoder DEFENDS anyway: a hop whose name falls + * outside the safe charset (or whose line is not a non-negative integer) is + * never emitted — encoding stops at the offending hop and sets the truncation + * marker, preserving the prefix-of-the-true-path invariant rather than + * corrupting the format. (`#` is in the charset: JS private names are + * `#field`, and the propagation engine's fallback hop names are `#`.) + * + * The byte cap (`TAINT_REASON_MAX_BYTES`, KTD6's "absolute reason-byte cap") + * bounds the persisted reason column regardless of hop caps: overflow drops + * TRAILING hops (keeps the source side) and sets the marker. All structural + * chars and valid names are single-byte ASCII, so `string.length` IS the + * byte length. + */ + +/** One-character format version prefix. Bump on any wire-format change. */ +export const TAINT_PATH_CODEC_VERSION = '1'; + +/** + * Absolute cap on the encoded reason's byte length (KTD6). 4096 comfortably + * holds ~100 hops of realistic identifiers — far beyond the default hop cap + * (32) — while bounding the persisted column even at `maxHops: 0` (unlimited). + */ +export const TAINT_REASON_MAX_BYTES = 4096; + +/** The truncation-marker segment content (rides as a trailing `|~`). */ +export const TAINT_PATH_TRUNCATION_MARKER = '~'; + +/** + * Safe hop-name charset: ASCII identifier characters plus `#` (JS private + * names / the propagation engine's `#` fallback). Deliberately ASCII-only + * — a Unicode identifier is VALID JS but is skipped (truncation marker) rather + * than risking a `sanitizeUTF8` byte change breaking decode (defensive + * simplification; documented FN on path completeness, never on the finding). + */ +const SAFE_NAME = /^[A-Za-z0-9_$#]+$/; + +/** Decoder-side flags charset — `c` defined, the rest reserved (see module doc). */ +const FLAGS = /^[a-z]*$/; + +/** + * Kind-header charset: lowercase + digits + hyphen — covers every `SinkKind` + * label and stays disjoint from the structural delimiters (`;|:~`). + */ +const SAFE_KIND = /^[a-z0-9-]+$/; + +/** Encoder input hop — shape-compatible with `TaintHop` (propagate.ts). */ +export interface TaintPathHopInput { + readonly name: string; + readonly line: number; + readonly viaCall?: boolean; +} + +export interface EncodeTaintPathOptions { + /** + * The hop list is already a truncated prefix (e.g. the propagation engine's + * `hopsTruncated` from its hop cap) — emit the marker even when every hop + * fits. + */ + readonly truncated?: boolean; + /** Byte cap override (tests). Default {@link TAINT_REASON_MAX_BYTES}. */ + readonly maxBytes?: number; + /** + * Finding sink category (`SinkKind`) carried in the `;` header — the + * only persisted channel for it (see the module doc). A value outside the + * `[a-z0-9-]` charset is DROPPED (header omitted), never corrupted into the + * wire string; `SinkKind` is a closed lowercase-hyphen union so this is + * purely defensive. + */ + readonly kind?: string; +} + +export interface EncodedTaintPath { + /** The wire string for the TAINTED edge's `reason` column. */ + readonly reason: string; + /** True when the marker was emitted (caller-flagged, byte cap, or bad hop). */ + readonly truncated: boolean; +} + +export interface DecodedTaintHop { + readonly variable: string; + readonly line: number; + /** The hop passed through an unmodeled call (flag `c`, KTD5). */ + readonly viaCall: boolean; +} + +export interface DecodedTaintPath { + readonly ok: true; + readonly version: string; + /** Finding sink category from the `;` header; absent when not encoded. */ + readonly kind?: string; + /** Ordered source→sink hops (a PREFIX when `truncated`). */ + readonly hops: readonly DecodedTaintHop[]; + /** Path incomplete (trailing `|~`) — informational, NOT an error. */ + readonly truncated: boolean; +} + +/** Typed parse failure — the decoder never throws. */ +export interface TaintPathDecodeFailure { + readonly ok: false; + readonly error: string; +} + +export type TaintPathDecodeResult = DecodedTaintPath | TaintPathDecodeFailure; + +/** + * Encode an ordered hop list into the versioned `reason` wire string. + * Deterministic; never throws. See the module doc for the format and the + * three truncation triggers (caller flag, unencodable hop, byte cap). + */ +export function encodeTaintPath( + hops: readonly TaintPathHopInput[], + options?: EncodeTaintPathOptions, +): EncodedTaintPath { + // Kind header (defensively validated — see EncodeTaintPathOptions.kind). + const kindHeader = + typeof options?.kind === 'string' && SAFE_KIND.test(options.kind) ? `;${options.kind}` : ''; + // Floor: version char + kind header + room for the marker — a smaller cap + // could not hold even the empty truncated path. The header is identity + // material (finding classification), so it is never sacrificed to the byte + // cap; trailing hops are. + const maxBytes = Math.max( + options?.maxBytes ?? TAINT_REASON_MAX_BYTES, + TAINT_PATH_CODEC_VERSION.length + kindHeader.length + 2, + ); + let truncated = options?.truncated === true; + const segments: string[] = []; + let total = TAINT_PATH_CODEC_VERSION.length + kindHeader.length; + for (const hop of hops) { + if ( + typeof hop.name !== 'string' || + !SAFE_NAME.test(hop.name) || + !Number.isInteger(hop.line) || + hop.line < 0 + ) { + // Unencodable hop: drop it AND everything after it so the emitted hops + // stay a faithful source-side prefix (a silent mid-path gap would lie). + truncated = true; + break; + } + const segment = `|${hop.name}:${hop.line}${hop.viaCall === true ? ':c' : ''}`; + if (total + segment.length > maxBytes) { + truncated = true; + break; + } + segments.push(segment); + total += segment.length; + } + if (truncated) { + // Make room for the trailing `|~` marker (drop trailing hops as needed). + while (segments.length > 0 && total + 2 > maxBytes) { + total -= (segments.pop() as string).length; + } + } + const reason = + TAINT_PATH_CODEC_VERSION + + kindHeader + + segments.join('') + + (truncated ? `|${TAINT_PATH_TRUNCATION_MARKER}` : ''); + return { reason, truncated }; +} + +/** + * Decode a `reason` wire string. Returns a typed failure for anything that is + * not a well-formed version-`1` path — never throws. A truncated path decodes + * `ok: true` with `truncated: true` ("path incomplete", per KTD6). + */ +export function decodeTaintPath(reason: unknown): TaintPathDecodeResult { + if (typeof reason !== 'string' || reason.length === 0) { + return { ok: false, error: 'empty or non-string reason' }; + } + const version = reason[0]; + if (version !== TAINT_PATH_CODEC_VERSION) { + return { ok: false, error: `unsupported taint-path version '${version}'` }; + } + let body = reason.slice(1); + // Optional `;` header segment (finding sink category — see module doc). + let kind: string | undefined; + if (body.startsWith(';')) { + const headerEnd = body.indexOf('|'); + kind = headerEnd === -1 ? body.slice(1) : body.slice(1, headerEnd); + if (!SAFE_KIND.test(kind)) { + return { ok: false, error: `invalid kind header '${kind}'` }; + } + body = headerEnd === -1 ? '' : body.slice(headerEnd); + } + const hops: DecodedTaintHop[] = []; + if (body.length === 0) + return { ok: true, version, ...(kind ? { kind } : {}), hops, truncated: false }; + if (!body.startsWith('|')) { + return { ok: false, error: 'malformed body: expected a hop separator after the version' }; + } + const parts = body.slice(1).split('|'); + let truncated = false; + for (let i = 0; i < parts.length; i++) { + const part = parts[i]; + if (part === TAINT_PATH_TRUNCATION_MARKER) { + if (i !== parts.length - 1) { + return { ok: false, error: 'truncation marker not in trailing position' }; + } + truncated = true; + break; + } + const fields = part.split(':'); + if (fields.length < 2 || fields.length > 3) { + return { ok: false, error: `malformed hop segment '${part}'` }; + } + const [name, lineStr, flags = ''] = fields; + if (!SAFE_NAME.test(name)) { + return { ok: false, error: `invalid hop variable '${name}'` }; + } + if (!/^\d+$/.test(lineStr)) { + return { ok: false, error: `invalid hop line '${lineStr}'` }; + } + if (!FLAGS.test(flags)) { + return { ok: false, error: `invalid hop flags '${flags}'` }; + } + hops.push({ variable: name, line: Number(lineStr), viaCall: flags.includes('c') }); + } + return { ok: true, version, ...(kind ? { kind } : {}), hops, truncated }; +} diff --git a/gitnexus/src/core/ingestion/taint/propagate.ts b/gitnexus/src/core/ingestion/taint/propagate.ts new file mode 100644 index 000000000..0330f2d32 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/propagate.ts @@ -0,0 +1,880 @@ +/** + * Pure intra-procedural taint propagation engine (#2083 M3 U3). + * + * Forward taint reachability over one function's reaching-definition facts + * (M2 `computeReachingDefs`) and matched taint sites (U2 `matchFunctionSites`) + * — sources in, findings + sanitizer kills + coverage status out. PURE AND + * DETERMINISTIC, mirroring the reaching-defs contract: no graph, no I/O, no + * logger; insertion-ordered worklist; explicitly sorted outputs; snapshot + * tests and content-derived edge ids (U4) rely on it. + * + * PRECONDITIONS: the caller gates the CFG through `hasTaintSafeSites` + * (taint/site-safety.ts) and the emit-safety checks before calling — this + * module dereferences binding/site/statement indices without re-validating. + * + * ## The two-rule model (plan HTD) + * + * - **Rule (b), statement-local:** a matched SOURCE occurrence (member read) + * whose intra-statement occurrence path — the member-read's `parent` chain — + * reaches a matched SINK argument position produces an immediate single-hop + * finding (`exec(req.body)`). The same statement SEEDS taint: every binding + * the statement defines becomes tainted (see the precision floor below). + * - **Rule (a), worklist:** for each tainted `(binding, defPoint)`, every + * def→use fact delivers the taint to a use statement, where occurrences of + * the binding in matched sink argument positions produce findings and the + * statement's own defs are tainted onward. The fact graph contains genuine + * cycles (loop back-edges, same-statement self-facts) — the visited-set + * discipline below is load-bearing, not defensive. + * + * ## Sanitizer semantics — the KIND-SET exclusion model (KTD4, sharpened) + * + * The plan sketches a binary kill; this module implements the strictly more + * precise SOUND refinement: a taint carries a set of *excluded* (neutralized) + * `SinkKind`s accumulated through sanitizer hops, and a sink fires unless its + * kind is in the taint's exclusion set. A binary kill is the special case + * where the sanitizer neutralizes the sink's kind; the kind-set model + * additionally keeps `const b = escape(req.body); db.query(b)` a FINDING + * (an HTML escaper does not neutralize SQL — un-tainting `b` outright would + * be a suppressed live injection, the forbidden false-negative direction) + * while still suppressing `res.send(b)` (xss IS neutralized). + * + * - **Occurrence interposition (KTD4a):** evaluated over the U1 site + * structure. An occurrence reaching a sink arg / def-feeding position + * through a matched sanitizer site accumulates that sanitizer's + * `neutralizes` kinds on that PATH; a direct occurrence contributes the + * empty set. Per-position narrowing (`entry.args`) is respected; receiver + * flow through a sanitizer is NOT neutralized (the receiver is not the + * sanitized payload), and spread/template positions are never neutralized + * (position unprovable) — both sound-direction choices (under-kill). + * - **Intersection over paths:** a def fed by several occurrence paths + * excludes a kind only when EVERY path neutralizes it + * (`const c = cond ? escape(b) : b` taints `c` with NO exclusions — the + * direct arm's ∅ intersects everything away). Equally, a taint re-derived + * along a second route keeps the INTERSECTION of the exclusion sets and is + * re-processed whenever the set SHRINKS — a less-neutralized taint is + * strictly more dangerous. Exclusion sets only shrink over a finite + * lattice, so the worklist terminates. + * - **Kill locality (KTD4b):** a kill applies to the def the sanitizer + * produces (`SiteRecord.resultDefs`) only; the flowing binding's own taint + * is untouched (`const c = escape(b); exec(b)` still finds `b`'s flow). + * `x = escape(x)` works because taint keys on the DEF POINT: the + * sanitizer statement's def enters the set with the sanitizer's kinds + * excluded, while the seed def keeps flowing wherever the CFG still + * carries it (zero-iteration loops, conditional sanitizers — may-path + * mechanics need no special handling here, kills are absent from facts). + * + * ## Statement-coalescing precision floor (documented FP) + * + * Statement facts conflate multi-declarator statements: a statement that + * uses tainted `b` and defines `c` taints `c` with NO exclusions even when + * the two are textually unrelated (`const a = clean(z), b = g(t)` floor- + * taints `a` from `t` — pinned by a test). The per-declarator `resultDefs` + * precision narrows the EXCLUSION computation (and powers kills) only — a + * def in a call's `resultDefs` is fed exactly through that call, so its + * exclusions come from the paths into it; when the tainted input provably + * never flows into that call, the floor still taints the def (sound) but + * records no kill (a kill requires evidence of flow through the sanitizer). + * + * ## Propagate-through (KTD5) + * + * Taint in any argument or in the receiver of an UNMODELED call flows to + * the call's result defs, marked `viaCall` on the hop so `explain` can + * express lower confidence. An occurrence that reaches the unmodeled call + * only through a sanitizer carries the neutralization through + * (`const y = unknownFn(escape(b))` excludes the sanitizer's kinds — the + * plan's deliberate precision choice over flat-conservative). + * + * ## Kills output + * + * `kills` records every sanitizer that ACTUALLY neutralized kinds on a + * flowing taint — U4 emits `SANITIZES` edges from them. Two shapes share the + * record: result-def kills (`killedDef` = the def the sanitizer produces; + * `bindingIdx` = that def's binding) and value-position interposition kills + * (`exec(escape(x))` — no def exists; `killedDef` = the sink statement's own + * point, `bindingIdx` = the interposed binding). Interposition kills are + * recorded only when the (input, sink, position) produced no finding — a + * bypassed sanitizer (`exec(x + escape(x))`) killed nothing. + */ + +import type { FunctionCfg, SiteRecord, StatementFacts } from '../cfg/types.js'; +import { pointKey } from '../cfg/reaching-defs.js'; +import type { DefUseFact, FunctionDefUse, ProgramPoint } from '../cfg/reaching-defs.js'; +import type { + FunctionSiteMatches, + MatchedSanitizerCall, + MatchedSinkCall, + StatementMatches, +} from './match.js'; +import type { SinkKind, SourceKind } from './source-sink-config.js'; + +/** + * Default per-function findings cap (U5 config resolution; cfg/emit.ts + * DEFAULT_* pattern). Resolved into the RepoMeta `pdg` stamp by + * `resolvePdgConfig` so a cap change trips full writeback; `0` = unlimited + * is preserved like the other pdg caps. 200 is generous — a real function + * with more deduped source→sink findings is a fixture or a disaster, and + * the truncation is deterministic + counted (`droppedFindings`). + */ +export const DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION = 200; + +/** + * Default per-finding hop cap (U5; joins the RepoMeta `pdg` stamp like the + * findings cap). Bounds the persisted `reason` hop encoding (KTD6 pins the + * hop cap in config); 32 intra-procedural def→use hops is far beyond any + * legible path — overflow keeps the source-side prefix and sets + * `hopsTruncated`, parsed downstream as "path incomplete", never an error. + */ +export const DEFAULT_PDG_MAX_TAINT_HOPS = 32; + +export interface TaintLimits { + /** + * Maximum findings per function AFTER dedup; the sorted finding list is + * truncated deterministically and the overflow counted in + * `droppedFindings`. `undefined`/0 ⇒ unlimited. + */ + readonly maxFindingsPerFunction?: number; + /** + * Maximum hops retained per finding (source-side prefix kept); overflow + * sets `hopsTruncated`. `undefined`/0 ⇒ unlimited. + */ + readonly maxHops?: number; +} + +/** One hop of a finding's path — enough for U4's reason codec (name, line, flag). */ +export interface TaintHop { + /** Index into the function's binding table. */ + readonly bindingIdx: number; + /** Resolved binding name (carried so U4 never re-joins the table). */ + readonly name: string; + readonly point: ProgramPoint; + /** The value passed through an unmodeled call to get here (KTD5). */ + readonly viaCall?: boolean; +} + +/** + * The KTD6 rule-(b) source identity material: the matched member-read + * occurrence itself — statement point + site index + object/property. For + * worklist findings this is the ROOT source the taint chain was seeded from. + */ +export interface TaintSourceOccurrence { + readonly point: ProgramPoint; + /** Index into the source statement's `sites` array. */ + readonly siteIndex: number; + readonly objectBindingIdx: number; + readonly property: string; + readonly kind: SourceKind; +} + +/** The sink side of a finding's identity: point + site + argument + binding. */ +export interface TaintSinkOccurrence { + readonly point: ProgramPoint; + /** Index into the sink statement's `sites` array. */ + readonly siteIndex: number; + /** Matched sink argument position the tainted occurrence landed in. */ + readonly argIndex: number; + /** + * The binding whose occurrence reached the sink position (for rule-(b) + * findings: the source member-read's object binding). + */ + readonly bindingIdx: number; + /** The matched sink entry's `name` (e.g. `exec`) — finding classification. */ + readonly entryName: string; +} + +export interface TaintFinding { + readonly sinkKind: SinkKind; + readonly source: TaintSourceOccurrence; + readonly sink: TaintSinkOccurrence; + /** + * Ordered source→sink path, one path per finding (the CodeQL + * `--max-paths=1` convention): the taint chain's def hops followed by the + * sink-use hop. Rule-(b) findings carry the single sink-statement hop. + */ + readonly hops: readonly TaintHop[]; + readonly hopsTruncated?: boolean; +} + +/** A sanitizer that neutralized kinds on a flowing taint — U4's SANITIZES rows. */ +export interface SanitizerKill { + /** Statement point of the sanitizer call site. */ + readonly sanitizer: ProgramPoint; + /** + * The killed def's point (result-def kills — always the sanitizer's own + * statement in the intra-statement model) or the suppressed sink + * statement's point (value-position interposition kills). + */ + readonly killedDef: ProgramPoint; + /** The killed def's binding, or the interposed binding for value-position kills. */ + readonly bindingIdx: number; + /** Sorted, deduped kinds the sanitizer neutralized at that position. */ + readonly neutralized: readonly SinkKind[]; +} + +export interface FunctionTaintResult { + /** + * `computed` — full propagation ran. + * `coverage-gap` — the solver result was not `computed`; the function is + * skipped for findings entirely (R4: never partially + * analyzed), `gapReason` carries the solver status. + */ + readonly status: 'computed' | 'coverage-gap'; + readonly gapReason?: 'truncated' | 'overflow' | 'no-facts'; + /** Deduped (KTD6 identity), deterministically sorted, capped. */ + readonly findings: readonly TaintFinding[]; + readonly kills: readonly SanitizerKill[]; + /** Findings dropped by `maxFindingsPerFunction` (post-dedup). */ + readonly droppedFindings: number; +} + +/** Canonical SinkKind order for deterministic `neutralized` arrays. */ +const KIND_ORDER: readonly SinkKind[] = [ + 'code-injection', + 'command-injection', + 'path-traversal', + 'sql-injection', + 'xss', +]; +const kindRank = new Map(KIND_ORDER.map((k, i) => [k, i])); +const sortKinds = (kinds: Iterable): SinkKind[] => + [...new Set(kinds)].sort((a, b) => (kindRank.get(a) ?? 99) - (kindRank.get(b) ?? 99)); + +const EMPTY_KINDS: ReadonlySet = new Set(); + +/** One intra-statement occurrence path (interposition evidence). */ +interface OccPath { + /** Kinds neutralized along the path (union of traversed sanitizer hops). */ + readonly kinds: ReadonlySet; + /** The path traverses an unmodeled call/new site. */ + readonly viaCall: boolean; + /** Matched sanitizers traversed, with the kinds each contributed. */ + readonly sanitizers: ReadonlyArray<{ siteIndex: number; kinds: readonly SinkKind[] }>; +} + +const DIRECT_PATH: OccPath = { kinds: EMPTY_KINDS, viaCall: false, sanitizers: [] }; + +/** Per-statement match/site context, indexed once per visited statement. */ +interface StmtContext { + readonly point: ProgramPoint; + readonly facts: StatementFacts; + readonly sites: readonly SiteRecord[]; + readonly sinksBySite: ReadonlyMap; + readonly sanitizersBySite: ReadonlyMap; + /** binding → site indices whose `resultDefs` contain it (kill targets). */ + readonly resultDefSites: ReadonlyMap; +} + +/** One tainted (binding, defPoint) with the current minimal exclusion set. */ +interface TaintState { + readonly bindingIdx: number; + readonly point: ProgramPoint; + /** Mutable: only ever SHRINKS (intersection on re-derivation). */ + exclusions: ReadonlySet; + /** Taint-chain parent key, or undefined for seeds. */ + parentKey?: string; + /** Root source occurrence the chain was seeded from. */ + source: TaintSourceOccurrence; + viaCall: boolean; + /** Exclusion-set size at last processing — skips no-op requeues. */ + processedSize: number; +} + +/** + * Compute taint flows for one function. See the module doc for the two-rule + * model, the kind-set exclusion semantics, and the precision floor. + */ +export function computeTaintFlows( + cfg: FunctionCfg, + defUse: FunctionDefUse, + matches: FunctionSiteMatches, + limits?: TaintLimits, +): FunctionTaintResult { + if (defUse.status !== 'computed') { + return { + status: 'coverage-gap', + gapReason: defUse.status, + findings: [], + kills: [], + droppedFindings: 0, + }; + } + + const bindings = defUse.bindings; + + // ── per-statement context (built lazily; statements revisit often) ──────── + const matchByPoint = new Map(); + for (const sm of matches.statements) { + matchByPoint.set(`${sm.blockIndex}:${sm.statementIndex}`, sm); + } + const ctxCache = new Map(); + const contextAt = (blockIndex: number, stmtIndex: number): StmtContext | undefined => { + const key = `${blockIndex}:${stmtIndex}`; + if (ctxCache.has(key)) return ctxCache.get(key); + const facts = cfg.blocks[blockIndex]?.statements?.[stmtIndex]; + let ctx: StmtContext | undefined; + if (facts) { + const sm = matchByPoint.get(key); + const sinksBySite = new Map(); + const sanitizersBySite = new Map(); + for (const s of sm?.sinks ?? []) { + const list = sinksBySite.get(s.siteIndex); + if (list) list.push(s); + else sinksBySite.set(s.siteIndex, [s]); + } + for (const s of sm?.sanitizers ?? []) { + const list = sanitizersBySite.get(s.siteIndex); + if (list) list.push(s); + else sanitizersBySite.set(s.siteIndex, [s]); + } + const resultDefSites = new Map(); + facts.sites?.forEach((site, siteIndex) => { + for (const d of site.resultDefs ?? []) { + const list = resultDefSites.get(d); + if (list) list.push(siteIndex); + else resultDefSites.set(d, [siteIndex]); + } + }); + ctx = { + point: { blockIndex, stmtIndex, line: facts.line }, + facts, + sites: facts.sites ?? [], + sinksBySite, + sanitizersBySite, + resultDefSites, + }; + } + ctxCache.set(key, ctx); + return ctx; + }; + + /** Kinds the matched sanitizers at `siteIndex` neutralize for input position `argPos`. */ + const neutralizedAt = (ctx: StmtContext, siteIndex: number, argPos: number): SinkKind[] => { + const sans = ctx.sanitizersBySite.get(siteIndex); + if (!sans) return []; + const site = ctx.sites[siteIndex]; + // Spread/template positions are never provably the sanitized argument — + // do not neutralize (sound: under-kill). Exact positions check `args`. + if (site.template === true || (site.spread !== undefined && argPos >= site.spread)) return []; + const kinds: SinkKind[] = []; + for (const san of sans) { + if (san.entry.args === undefined || san.entry.args.includes(argPos)) { + kinds.push(...san.entry.neutralizes); + } + } + return sortKinds(kinds); + }; + + /** A call/new site the model does not understand (anything but a matched sanitizer). */ + const isUnmodeledCall = (ctx: StmtContext, siteIndex: number): boolean => { + const site = ctx.sites[siteIndex]; + return site.kind !== 'member-read' && !ctx.sanitizersBySite.has(siteIndex); + }; + + const emerge = (ctx: StmtContext, siteIndex: number, argPos: number, inner: OccPath): OccPath => { + const added = neutralizedAt(ctx, siteIndex, argPos); + return { + kinds: added.length === 0 ? inner.kinds : new Set([...inner.kinds, ...added]), + viaCall: inner.viaCall || isUnmodeledCall(ctx, siteIndex), + sanitizers: + added.length === 0 ? inner.sanitizers : [...inner.sanitizers, { siteIndex, kinds: added }], + }; + }; + + /** + * STRICT occurrence paths of binding `b` flowing OUT of site `siteIndex`'s + * result — found arg entries and the receiver only, each with the site's + * own neutralization/viaCall applied. Empty when `b` provably never flows + * in (the caller falls back to the floor and records NO kill). `guard` + * breaks corrupted-store via cycles (site-safety checks ranges, not + * acyclicity). + */ + const flowsOutOf = ( + ctx: StmtContext, + b: number, + siteIndex: number, + guard: Set, + ): OccPath[] => { + if (guard.has(siteIndex)) return []; + guard.add(siteIndex); + const site = ctx.sites[siteIndex]; + const out: OccPath[] = []; + site.args?.forEach((entries, argPos) => { + for (const e of entries) { + if (typeof e === 'number') { + if (e === b) out.push(emerge(ctx, siteIndex, argPos, DIRECT_PATH)); + } else if (e[0] === b) { + // Via-tagged: the occurrence reaches this position THROUGH the + // nested site. When the nested site shows no recognized channel + // for `b` (callee-chain occurrences — dynamic subscript keys), the + // via-tag is still evidence of flow: fall back to a direct, + // UN-neutralized path (sound: never a false kill). + const inner = flowsOutOf(ctx, b, e[1], guard); + const paths = + inner.length > 0 + ? inner + : [{ ...DIRECT_PATH, viaCall: isUnmodeledCall(ctx, e[1]) } satisfies OccPath]; + for (const p of paths) out.push(emerge(ctx, siteIndex, argPos, p)); + } + } + }); + if (site.receiver === b) { + // Receiver TITO (KTD5): the receiver's value flows through the call + // into its result — but a sanitizer does not neutralize its receiver. + out.push({ ...DIRECT_PATH, viaCall: isUnmodeledCall(ctx, siteIndex) }); + } + guard.delete(siteIndex); + return out; + }; + + /** Occurrence paths of `b` INTO sink position (s, p) — no emergence from s. */ + const pathsIntoPosition = ( + ctx: StmtContext, + b: number, + siteIndex: number, + argPos: number, + ): OccPath[] => { + const entries = ctx.sites[siteIndex].args?.[argPos] ?? []; + const out: OccPath[] = []; + for (const e of entries) { + if (typeof e === 'number') { + if (e === b) out.push(DIRECT_PATH); + } else if (e[0] === b) { + const inner = flowsOutOf(ctx, b, e[1], new Set()); + if (inner.length > 0) out.push(...inner); + else out.push({ ...DIRECT_PATH, viaCall: isUnmodeledCall(ctx, e[1]) }); + } + } + return out; + }; + + /** + * Walk a SOURCE member-read's `parent` chain. Linear (each site has one + * parent); invokes `onPosition` with the accumulated path BEFORE the + * ancestor's own emergence (the value flows INTO the ancestor at that + * position) — sink checks and stop-at-site joins both hang off it. + */ + const climbSourceChain = ( + ctx: StmtContext, + srcSiteIndex: number, + onPosition: (siteIndex: number, argPos: number, sofar: OccPath) => boolean, + ): void => { + const visited = new Set([srcSiteIndex]); + let cur = ctx.sites[srcSiteIndex]; + let sofar: OccPath = DIRECT_PATH; + while (cur.parent) { + const [siteIndex, argPos] = cur.parent; + if (visited.has(siteIndex)) return; // corrupted-store parent cycle + visited.add(siteIndex); + if (onPosition(siteIndex, argPos, sofar)) return; + sofar = emerge(ctx, siteIndex, argPos, sofar); + cur = ctx.sites[siteIndex]; + } + }; + + /** Source path INTO site `target` (with target's emergence), or undefined. */ + const sourceFlowsOutOf = ( + ctx: StmtContext, + srcSiteIndex: number, + target: number, + ): OccPath | undefined => { + let found: OccPath | undefined; + climbSourceChain(ctx, srcSiteIndex, (siteIndex, argPos, sofar) => { + if (siteIndex !== target) return false; + found = emerge(ctx, target, argPos, sofar); + return true; + }); + return found; + }; + + // ── accumulators ────────────────────────────────────────────────────────── + const findingsByIdentity = new Map(); + const killsByIdentity = new Map< + string, + { kill: Omit; kinds: Set } + >(); + + const recordKill = ( + sanitizer: ProgramPoint, + killedDef: ProgramPoint, + bindingIdx: number, + kinds: readonly SinkKind[], + ): void => { + if (kinds.length === 0) return; + const key = `${pointKey(sanitizer)}|${pointKey(killedDef)}|${bindingIdx}`; + const existing = killsByIdentity.get(key); + if (existing) for (const k of kinds) existing.kinds.add(k); + else + killsByIdentity.set(key, { + kill: { sanitizer, killedDef, bindingIdx }, + kinds: new Set(kinds), + }); + }; + + // KTD6 statement-level finding identity: source occurrence + sink occurrence + // + kind (NOT entryName). Computed standalone so the worklist can dedup-check + // BEFORE the cost of chainHops (first write wins; dedup-before-budget). + const findingKey = ( + sinkKind: SinkKind, + source: TaintSourceOccurrence, + sink: Pick, + ): string => + [ + sinkKind, + pointKey(source.point), + source.siteIndex, + source.objectBindingIdx, + source.property, + pointKey(sink.point), + sink.siteIndex, + sink.argIndex, + sink.bindingIdx, + ].join('|'); + + const recordFinding = ( + sinkKind: SinkKind, + source: TaintSourceOccurrence, + sink: TaintSinkOccurrence, + hops: TaintHop[], + hopsTruncated: boolean, + ): void => { + const key = findingKey(sinkKind, source, sink); + if (findingsByIdentity.has(key)) return; + const maxHops = limits?.maxHops && limits.maxHops > 0 ? limits.maxHops : Infinity; + let truncated = hopsTruncated; + let kept = hops; + if (hops.length > maxHops) { + kept = hops.slice(0, maxHops); + truncated = true; + } + findingsByIdentity.set(key, { + sinkKind, + source, + sink, + hops: kept, + ...(truncated ? { hopsTruncated: true } : {}), + }); + }; + + // ── taint state ─────────────────────────────────────────────────────────── + const taints = new Map(); + const queue: string[] = []; + + /** The (binding, def-point) portion of a state key — the def→use fact-table + * lookup key, source-independent. */ + const defKey = (bindingIdx: number, point: ProgramPoint): string => + `${bindingIdx}:${pointKey(point)}`; + /** Full taint-state key: the def-point portion plus a ROOT source-occurrence + * discriminator ({point, siteIndex} — the same source fields recordFinding's + * identity uses, deliberately excluding `kind`). Distinct sources reaching + * one def get distinct states, so a second source is no longer dropped + * (KTD6); same-source multi-path derivations still share a key so their + * exclusion sets intersect (the raw arm soundly wins). */ + const stateKey = ( + bindingIdx: number, + point: ProgramPoint, + source: TaintSourceOccurrence, + ): string => `${defKey(bindingIdx, point)}#${pointKey(source.point)}:${source.siteIndex}`; + + const deriveTaint = ( + bindingIdx: number, + point: ProgramPoint, + exclusions: ReadonlySet, + parentKey: string | undefined, + source: TaintSourceOccurrence, + viaCall: boolean, + ): void => { + const key = stateKey(bindingIdx, point, source); + const existing = taints.get(key); + if (!existing) { + taints.set(key, { + bindingIdx, + point, + exclusions, + parentKey, + source, + viaCall, + processedSize: -1, + }); + queue.push(key); + return; + } + // Monotone shrink: keep the intersection; re-process only when it got + // strictly smaller (a less-neutralized derivation is more dangerous). + const inter = new Set(); + for (const k of existing.exclusions) if (exclusions.has(k)) inter.add(k); + if (inter.size < existing.exclusions.size) { + existing.exclusions = inter; + existing.parentKey = parentKey; + existing.source = source; + existing.viaCall = viaCall; + queue.push(key); + } + }; + + /** Taint-chain hops from seed to `key`, with a cycle guard (re-derivation + * can rewire parents into a loop — truncate instead of spinning). */ + const chainHops = (key: string): { hops: TaintHop[]; truncated: boolean } => { + const reversed: TaintHop[] = []; + const seen = new Set(); + let cur: string | undefined = key; + let truncated = false; + while (cur !== undefined) { + if (seen.has(cur)) { + truncated = true; + break; + } + seen.add(cur); + const t = taints.get(cur); + if (!t) break; + reversed.push({ + bindingIdx: t.bindingIdx, + name: bindings[t.bindingIdx]?.name ?? `#${t.bindingIdx}`, + point: t.point, + ...(t.viaCall ? { viaCall: true } : {}), + }); + cur = t.parentKey; + } + return { hops: reversed.reverse(), truncated }; + }; + + /** Intersection of path kind-sets; viaCall = any path through a call. */ + const summarizePaths = ( + paths: readonly OccPath[], + ): { kinds: ReadonlySet; viaCall: boolean } => { + let kinds: ReadonlySet | undefined; + let viaCall = false; + for (const p of paths) { + viaCall ||= p.viaCall; + if (kinds === undefined) { + kinds = p.kinds; + } else { + const inter = new Set(); + for (const k of kinds) if (p.kinds.has(k)) inter.add(k); + kinds = inter; + } + } + return { kinds: kinds ?? EMPTY_KINDS, viaCall }; + }; + + /** + * Taint every def of `ctx`'s statement from one input. `pathsInto(c)` + * supplies the input's strict occurrence paths into call site `c` — + * defining the resultDefs precision and the kill evidence; an empty list + * means "no provable flow" and the floor applies (taint with the input's + * own exclusions, no kill). + */ + const feedDefs = ( + ctx: StmtContext, + inputExclusions: ReadonlySet, + parentKey: string | undefined, + source: TaintSourceOccurrence, + pathsInto: (siteIndex: number) => OccPath[], + ): void => { + const defs = [...ctx.facts.defs, ...(ctx.facts.mayDefs ?? [])]; + if (defs.length === 0) return; + const seen = new Set(); + for (const d of defs) { + if (seen.has(d)) continue; + seen.add(d); + const rdSites = ctx.resultDefSites.get(d); + let addKinds: ReadonlySet = EMPTY_KINDS; + let viaCall = false; + if (rdSites) { + const paths: OccPath[] = []; + for (const c of rdSites) paths.push(...pathsInto(c)); + if (paths.length > 0) { + const summary = summarizePaths(paths); + addKinds = summary.kinds; + viaCall = summary.viaCall; + for (const p of paths) { + for (const san of p.sanitizers) { + recordKill(ctx.point, ctx.point, d, san.kinds); + } + } + } + // else: floor — tainted with no exclusions added, no kill (the input + // provably never flows into the producing call; conflation FP pinned). + } + const exclusions = + addKinds.size === 0 ? inputExclusions : new Set([...inputExclusions, ...addKinds]); + deriveTaint(d, ctx.point, exclusions, parentKey, source, viaCall); + } + }; + + // ── rule (b) + seeding: statements with matched sources ─────────────────── + for (const sm of matches.statements) { + if (sm.sources.length === 0) continue; + const ctx = contextAt(sm.blockIndex, sm.statementIndex); + if (!ctx) continue; + for (const src of sm.sources) { + const srcSite = ctx.sites[src.siteIndex]; + if (srcSite?.object === undefined || srcSite.property === undefined) continue; + const sourceOcc: TaintSourceOccurrence = { + point: ctx.point, + siteIndex: src.siteIndex, + objectBindingIdx: srcSite.object, + property: srcSite.property, + kind: src.entry.kind, + }; + + // Statement-local sink checks along the member-read's parent chain. + climbSourceChain(ctx, src.siteIndex, (siteIndex, argPos, sofar) => { + for (const sink of ctx.sinksBySite.get(siteIndex) ?? []) { + if (!sink.argPositions.includes(argPos)) continue; + const kind = sink.entry.kind; + if (!sofar.kinds.has(kind)) { + recordFinding( + kind, + sourceOcc, + { + point: ctx.point, + siteIndex, + argIndex: argPos, + bindingIdx: srcSite.object as number, + entryName: sink.entry.name, + }, + [ + { + bindingIdx: srcSite.object as number, + name: bindings[srcSite.object as number]?.name ?? `#${srcSite.object}`, + point: ctx.point, + ...(sofar.viaCall ? { viaCall: true } : {}), + }, + ], + false, + ); + } else { + for (const san of sofar.sanitizers) { + if (san.kinds.includes(kind)) { + recordKill(ctx.point, ctx.point, srcSite.object as number, san.kinds); + } + } + } + } + return false; + }); + + // Seed every def of the statement (precision floor + resultDefs kills). + feedDefs(ctx, EMPTY_KINDS, undefined, sourceOcc, (c) => { + const p = sourceFlowsOutOf(ctx, src.siteIndex, c); + return p ? [p] : []; + }); + } + } + + // ── rule (a): worklist over def→use facts ───────────────────────────────── + const factsByDef = new Map(); + for (const f of defUse.facts) { + const key = defKey(f.bindingIdx, f.def); + const list = factsByDef.get(key); + if (list) list.push(f); + else factsByDef.set(key, [f]); + } + + // Strict-FIFO worklist via a head cursor (not Array.shift, which is O(N) per + // dequeue). FIFO order is load-bearing beyond perf: chainHops reconstructs + // hops from the live `taints` map, whose parentKey/source/viaCall are + // rewritten order-sensitively on monotone shrink — so hop-content + // determinism is contingent on dequeue order matching enqueue order. Do NOT + // sort or reprioritize the worklist. + let head = 0; + while (head < queue.length) { + const key = queue[head++]; + // Reclaim the consumed prefix periodically so the array doesn't grow + // unbounded across a long run (order-preserving — pure memory hygiene). + if (head > 1024 && head * 2 > queue.length) { + queue.splice(0, head); + head = 0; + } + const t = taints.get(key) as TaintState; + if (t.processedSize === t.exclusions.size) continue; // no-op requeue + t.processedSize = t.exclusions.size; + const b = t.bindingIdx; + const E = t.exclusions; + + // Facts are keyed by (binding, def-point) only — look up by the def portion + // of this state, not the source-discriminated state key. + for (const fact of factsByDef.get(defKey(b, t.point)) ?? []) { + const ctx = contextAt(fact.use.blockIndex, fact.use.stmtIndex); + if (!ctx) continue; + + // Sink check: occurrences of `b` at matched sink argument positions. + for (const [siteIndex, sinks] of ctx.sinksBySite) { + for (const sink of sinks) { + const kind = sink.entry.kind; + for (const argPos of sink.argPositions) { + const paths = pathsIntoPosition(ctx, b, siteIndex, argPos); + if (paths.length === 0) continue; + if (E.has(kind)) continue; // suppressed at def time; kill already recorded + const justify = paths.find((p) => !p.kinds.has(kind)); + if (justify) { + const sinkOcc = { + point: ctx.point, + siteIndex, + argIndex: argPos, + bindingIdx: b, + entryName: sink.entry.name, + }; + // Dedup BEFORE chainHops: already-recorded identities discard + // their hop chain anyway (first write wins), so skip the walk. + if (findingsByIdentity.has(findingKey(kind, t.source, sinkOcc))) continue; + const chain = chainHops(key); + chain.hops.push({ + bindingIdx: b, + name: bindings[b]?.name ?? `#${b}`, + point: ctx.point, + ...(justify.viaCall ? { viaCall: true } : {}), + }); + recordFinding(kind, t.source, sinkOcc, chain.hops, chain.truncated); + } else { + // EVERY path interposed — value-position kill(s) held. + for (const p of paths) { + for (const san of p.sanitizers) { + if (san.kinds.includes(kind)) recordKill(ctx.point, ctx.point, b, san.kinds); + } + } + } + } + } + } + + // Def-feed: the use statement's own defs become tainted. + feedDefs(ctx, E, key, t.source, (c) => flowsOutOf(ctx, b, c, new Set())); + } + } + + // ── deterministic assembly ──────────────────────────────────────────────── + const comparePoints = (a: ProgramPoint, b: ProgramPoint): number => + a.blockIndex - b.blockIndex || a.stmtIndex - b.stmtIndex; + + const findings = [...findingsByIdentity.values()].sort( + (a, b) => + comparePoints(a.source.point, b.source.point) || + a.source.siteIndex - b.source.siteIndex || + comparePoints(a.sink.point, b.sink.point) || + a.sink.siteIndex - b.sink.siteIndex || + a.sink.argIndex - b.sink.argIndex || + a.sink.bindingIdx - b.sink.bindingIdx || + (kindRank.get(a.sinkKind) ?? 99) - (kindRank.get(b.sinkKind) ?? 99), + ); + const maxFindings = + limits?.maxFindingsPerFunction && limits.maxFindingsPerFunction > 0 + ? limits.maxFindingsPerFunction + : Infinity; + const kept = findings.length > maxFindings ? findings.slice(0, maxFindings) : findings; + + const kills = [...killsByIdentity.values()] + .map(({ kill, kinds }) => ({ ...kill, neutralized: sortKinds(kinds) })) + .sort( + (a, b) => + comparePoints(a.sanitizer, b.sanitizer) || + comparePoints(a.killedDef, b.killedDef) || + a.bindingIdx - b.bindingIdx, + ); + + return { + status: 'computed', + findings: kept, + kills, + droppedFindings: findings.length - kept.length, + }; +} diff --git a/gitnexus/src/core/ingestion/taint/site-safety.ts b/gitnexus/src/core/ingestion/taint/site-safety.ts new file mode 100644 index 000000000..f2f1777f9 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/site-safety.ts @@ -0,0 +1,103 @@ +/** + * Taint-site safety validation (#2083 M3 U1, plan KTD2). + * + * Mirrors `hasEmitSafeFacts` (cfg/emit.ts): an untrusted `cfgSideChannel` + * element — possibly from a corrupted durable parsedfile store — must never + * crash the taint pass or fabricate matches from out-of-range indices. The + * degradation contract is per-FUNCTION and one-directional: a CFG whose sites + * fail this check is SKIPPED FOR TAINT ONLY — the BasicBlock/CFG layer and + * the REACHING_DEF projection (guarded by their own checks) are unaffected. + * + * Checked: exactly the indices the taint matcher dereferences — binding + * indices (`receiver`/`object`/`resultDefs`/arg occurrences) against the + * function's binding table, and intra-statement site references (`parent` + * site / via-tags) against the OWNING statement's `sites` array. Site + * references are statement-local by construction (each statement's + * FactAccumulator starts at index 0); a cross-statement reference is + * corruption, not a feature. + * + * Lives in `taint/` (not cfg/emit.ts): U4's taint emit path is the only + * consumer, and the guard must evolve with the matcher that dereferences + * these fields. + */ +import type { FunctionCfg, SiteRecord } from '../cfg/types.js'; + +const SITE_KINDS = new Set(['call', 'new', 'member-read']); + +/** + * Whether a structurally-valid CFG's M3 `sites` annotations are safe to feed + * to the taint matcher/propagator. `true` when no statement carries sites + * (pre-M3 channel, or no calls) — absence is the well-formed empty case. + */ +export const hasTaintSafeSites = (cfg: FunctionCfg): boolean => { + // Sites carry binding indices — a channel with sites but no binding table + // has nothing to range-check them against: reject (checked per statement). + const bindingCount = Array.isArray(cfg.bindings) ? cfg.bindings.length : -1; + for (const block of cfg.blocks) { + const stmts = block.statements; + if (stmts === undefined) continue; + if (!Array.isArray(stmts)) return false; + for (const s of stmts) { + if (s?.sites === undefined) continue; + if (bindingCount < 0) return false; + if (!isSafeSiteList(s.sites, bindingCount)) return false; + } + } + return true; +}; + +const isSafeSiteList = (sites: unknown, bindingCount: number): boolean => { + if (!Array.isArray(sites)) return false; + const siteCount = sites.length; + const bindingInRange = (i: unknown): boolean => + Number.isInteger(i) && (i as number) >= 0 && (i as number) < bindingCount; + const siteInRange = (i: unknown): boolean => + Number.isInteger(i) && (i as number) >= 0 && (i as number) < siteCount; + + for (const site of sites as ReadonlyArray | null | undefined>) { + if (site === null || typeof site !== 'object') return false; + if (typeof site.kind !== 'string' || !SITE_KINDS.has(site.kind)) return false; + if (site.callee !== undefined && typeof site.callee !== 'string') return false; + if (site.receiver !== undefined && !bindingInRange(site.receiver)) return false; + if (site.requireArg !== undefined && typeof site.requireArg !== 'string') return false; + if (site.template !== undefined && typeof site.template !== 'boolean') return false; + if ( + site.spread !== undefined && + (!Number.isInteger(site.spread) || (site.spread as number) < 0) + ) { + return false; + } + if (site.parent !== undefined) { + const p = site.parent; + if (!Array.isArray(p) || p.length !== 2) return false; + if (!siteInRange(p[0])) return false; + if (!Number.isInteger(p[1]) || (p[1] as number) < 0) return false; + } + if (site.resultDefs !== undefined) { + if (!Array.isArray(site.resultDefs) || !site.resultDefs.every(bindingInRange)) return false; + } + if (site.args !== undefined) { + if (!Array.isArray(site.args)) return false; + for (const position of site.args) { + if (!Array.isArray(position)) return false; + for (const entry of position) { + if (typeof entry === 'number') { + if (!bindingInRange(entry)) return false; + } else if (Array.isArray(entry) && entry.length === 2) { + if (!bindingInRange(entry[0]) || !siteInRange(entry[1])) return false; + } else { + return false; + } + } + } + } + if (site.kind === 'member-read') { + // The matcher dereferences both unconditionally on member reads. + if (!bindingInRange(site.object) || typeof site.property !== 'string') return false; + } else { + if (site.object !== undefined && !bindingInRange(site.object)) return false; + if (site.property !== undefined && typeof site.property !== 'string') return false; + } + } + return true; +}; diff --git a/gitnexus/src/core/ingestion/taint/source-sink-config.ts b/gitnexus/src/core/ingestion/taint/source-sink-config.ts index 2b03bf289..5909dcb3e 100644 --- a/gitnexus/src/core/ingestion/taint/source-sink-config.ts +++ b/gitnexus/src/core/ingestion/taint/source-sink-config.ts @@ -1,38 +1,119 @@ /** - * Source/sink/sanitizer config model (issue #2080, taint/PDG substrate M0). + * Source/sink/sanitizer config model (issue #2080 M0 seam, extended by #2083 + * M3 U2). * - * The per-language taint configuration *shape*. M0 ships only the type and an - * (empty) registry seam — no analysis consumes it yet. M3 (#2083, intra-proc - * taint) populates per-language specs and reads them when emitting TAINTED / - * SANITIZES edges. + * The per-language taint configuration *shape*. M0 shipped only the bare + * `{name, args?}` callable matcher and an empty registry seam; M3 U2 extends + * it with the `kind` taxonomy and the resolution-mechanism fields the + * import-aware matcher (`taint/match.ts`) needs, and fills the registry with + * the built-in TS/JS model (`taint/typescript-model.ts`). * - * Kept deliberately minimal: enough for M3 to express "callable X is a - * source / sink / sanitizer, optionally for argument position N" without M0 - * committing to matcher semantics it cannot yet validate. The shape is - * expected to grow (e.g. sanitizer escape conditions, return-position taint) - * when M3 makes contact with real flows; that is a forward-declared-interface - * design choice, not a finished contract. + * Design rule: entries describe WHAT a callable is (category + how its name + * resolves), never HOW matching works — matching semantics (import joins, + * shadow checks, spread/template position rules) live in the matcher so the + * spec stays declarative data that can hash into `taintModelVersion`. */ +/** Categories of taint sources. M3 ships remote HTTP input only. */ +export type SourceKind = 'remote-input'; + /** - * Identifies a callable that participates in taint flow. `name` is matched - * against a resolved callable (simple or qualified name — exact matching - * semantics are M3's call). `args` optionally narrows to specific 0-based - * argument positions that carry taint (for a source/sink) or clear it (for a - * sanitizer); omit to mean "unspecified / all". + * Vulnerability categories for sinks. Sanitizers reference the SAME taxonomy + * via {@link TaintSanitizerEntry.neutralizes}: a sanitizer kill applies only + * when it neutralizes the matched sink's kind (`path.basename` strips + * directories, not shell metacharacters — a kind-blind kill is a suppressed + * live command injection, the forbidden false-negative direction). */ -export interface TaintCallableMatcher { +export type SinkKind = + | 'command-injection' + | 'code-injection' + | 'path-traversal' + | 'sql-injection' + | 'xss'; + +/** + * Identifies a callable that participates in taint flow. `name` is the + * callable's own (unqualified) name — qualification comes from the + * resolution-mechanism fields on the extending entry types, not from dotted + * `name` strings. `args` optionally narrows to specific 0-based argument + * positions that carry taint into a sink (or are cleared by a sanitizer); + * omit to mean "all positions". + */ +export interface TaintCallableMatcher { readonly name: string; readonly args?: readonly number[]; + /** Category label — drives finding classification and sanitizer kind-compat. */ + readonly kind: K; } /** - * The taint configuration for a single language: which callables introduce - * taint (sources), which are dangerous to reach with tainted input (sinks), - * and which clear taint (sanitizers). + * A sink callable. Exactly one resolution mechanism should be set per entry: + * + * - `module` — the callable lives in a package/builtin module; the matcher + * resolves call sites against it import-aware (ESM `parsedImports` aliases, + * namespace handles, and the CommonJS `require('')` join). `name` + * is the exported member (`'exec'` of `'child_process'`); the pseudo-name + * `'default'` denotes invoking the module's default export / the module + * handle itself. + * - `global` — a true ECMAScript global (`eval`, `Function`); matched by bare + * name only when the name is not shadowed by an in-function declaration and + * not bound by an import. `newOnly` further restricts to `new` expressions + * (`new Function(body)`). + * - `anyReceiver` — a method matched on ANY receiver chain by its final + * segment (`.query(sql)` / `.execute(sql)` on whatever the DB handle is + * named) — deliberately name-conventional, like Semgrep's default rules. + * - `receivers` — a method matched only on the listed conventional receiver + * names (`res.send` / `res.write`); exactly `.`, name-based. + */ +export interface TaintSinkEntry extends TaintCallableMatcher { + readonly module?: string; + readonly global?: boolean; + /** Only meaningful with `global`: match `new (…)` sites only. */ + readonly newOnly?: boolean; + readonly anyReceiver?: boolean; + readonly receivers?: readonly string[]; +} + +/** + * A sanitizer callable. Carries the sink kinds it `neutralizes` instead of a + * `kind` of its own. STRICTER resolution than sinks by design: only the + * `module` (import-aware) and `global` mechanisms exist — never a bare-name + * convention — because a sanitizer mis-match is a false KILL (a user's own + * `escape` helper must not suppress findings), while a sink mis-match is + * merely noise. `args` narrows which argument positions are cleared (omit = + * all). + */ +export interface TaintSanitizerEntry { + readonly name: string; + readonly args?: readonly number[]; + readonly neutralizes: readonly SinkKind[]; + readonly module?: string; + readonly global?: boolean; +} + +/** + * A member-read taint source: reading `.` where the object + * is one of the conventional receiver `objects` names (`req`/`request`) and + * the property is one of `properties` (`body`, `query`, …). Matching is + * name-based on the harvested `member-read` site (Semgrep-convention, not + * type-aware — the accepted M3 FP/FN trade recorded in the plan's risk + * table). One entry fans out over the objects × properties product. + */ +export interface TaintMemberSourceEntry { + readonly kind: SourceKind; + readonly objects: readonly string[]; + readonly properties: readonly string[]; +} + +/** + * The taint configuration for a single language: which member reads introduce + * taint (sources), which callables are dangerous to reach with tainted input + * (sinks), and which callables clear it (sanitizers). M3 sources are + * member-read entries only; call-result sources are a forward extension + * (add a union variant), not a missing case. */ export interface SourceSinkSanitizerSpec { - readonly sources: readonly TaintCallableMatcher[]; - readonly sinks: readonly TaintCallableMatcher[]; - readonly sanitizers: readonly TaintCallableMatcher[]; + readonly sources: readonly TaintMemberSourceEntry[]; + readonly sinks: readonly TaintSinkEntry[]; + readonly sanitizers: readonly TaintSanitizerEntry[]; } diff --git a/gitnexus/src/core/ingestion/taint/source-sink-registry.ts b/gitnexus/src/core/ingestion/taint/source-sink-registry.ts index 7fe1f1229..84785615b 100644 --- a/gitnexus/src/core/ingestion/taint/source-sink-registry.ts +++ b/gitnexus/src/core/ingestion/taint/source-sink-registry.ts @@ -1,10 +1,12 @@ /** * Per-language source/sink/sanitizer registry seam (issue #2080). * - * A keyed registry of {@link SourceSinkSanitizerSpec} by language id. M0 stands - * up the empty seam — no language is registered and nothing in the pipeline - * reads it. M3 (#2083) registers per-language specs and queries this registry - * when emitting taint edges. + * A keyed registry of {@link SourceSinkSanitizerSpec} by language id. M0 stood + * up the empty seam; M3 U2 (#2083) fills it with the built-in TS/JS model via + * the EXPLICIT `registerBuiltinTaintModels()` seam in `typescript-model.ts` — + * deliberately not an import side-effect. The U4 taint emit path must call it + * once before the pdg window consumes the registry (idempotent; the registry + * itself stays empty until then, preserving default-run parity). * * The store is module-level (matching the codebase's other per-language * registries). {@link clearSourceSinkRegistry} resets it for test isolation. diff --git a/gitnexus/src/core/ingestion/taint/typescript-model.ts b/gitnexus/src/core/ingestion/taint/typescript-model.ts new file mode 100644 index 000000000..9e08f5ce7 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/typescript-model.ts @@ -0,0 +1,108 @@ +/** + * Built-in TS/JS taint model (#2083 M3 U2, plan KTD7). + * + * The canonical Express/Node source/sink/sanitizer set, registered for the + * `typescript` and `javascript` language ids via the EXPLICIT + * {@link registerBuiltinTaintModels} seam — deliberately not an import + * side-effect, so the U4 emit path controls WHEN registration happens (call + * it once before the pdg window runs; it is idempotent — the registry is + * last-write-wins on the same language id). + * + * `taintModelVersion` is a deterministic digest of the FULL model content + * (entries, kinds, args, modules). It joins the RepoMeta `pdg` stamp in U5 so + * that ANY model change — adding an entry, relabeling a kind — trips full + * writeback on an existing `--pdg` index (R7): persisted findings must never + * outlive the model that produced them. + */ + +import { createHash } from 'node:crypto'; +import { SupportedLanguages } from 'gitnexus-shared'; +import type { SourceSinkSanitizerSpec } from './source-sink-config.js'; +import { registerSourceSinkConfig } from './source-sink-registry.js'; + +/** + * The built-in TS/JS model. Module provenance uses bare specifier names — + * the matcher normalizes the `node:` scheme prefix, so `import { exec } from + * 'node:child_process'` resolves identically. + */ +export const TS_JS_TAINT_MODEL: SourceSinkSanitizerSpec = { + sources: [ + // Express-convention request member reads, matched name-based on the + // receiver (`req`/`request`) — the plan's accepted Semgrep-style trade. + { + kind: 'remote-input', + objects: ['req', 'request'], + properties: ['body', 'query', 'params', 'headers', 'cookies'], + }, + ], + sinks: [ + // Command execution — the command string is argument 0. + { name: 'exec', kind: 'command-injection', args: [0], module: 'child_process' }, + { name: 'execSync', kind: 'command-injection', args: [0], module: 'child_process' }, + { name: 'spawn', kind: 'command-injection', args: [0], module: 'child_process' }, + // Code evaluation. `eval` takes code at 0; `new Function(...)` treats + // EVERY argument as source text (params + body), so `args` is omitted + // (= all positions) rather than pinned to 0. + { name: 'eval', kind: 'code-injection', args: [0], global: true }, + { name: 'Function', kind: 'code-injection', global: true, newOnly: true }, + // Filesystem path consumption — path argument 0. + { name: 'readFile', kind: 'path-traversal', args: [0], module: 'fs' }, + { name: 'readFileSync', kind: 'path-traversal', args: [0], module: 'fs' }, + { name: 'writeFile', kind: 'path-traversal', args: [0], module: 'fs' }, + { name: 'writeFileSync', kind: 'path-traversal', args: [0], module: 'fs' }, + // SQL — `.query(sql)` / `.execute(sql)` member calls on ANY receiver + // (mysql2/pg/knex handles go by many names; receiver-conventional). + { name: 'query', kind: 'sql-injection', args: [0], anyReceiver: true }, + { name: 'execute', kind: 'sql-injection', args: [0], anyReceiver: true }, + // Reflected XSS — Express response writes, conventional receiver `res`. + { name: 'send', kind: 'xss', args: [0], receivers: ['res'] }, + { name: 'write', kind: 'xss', args: [0], receivers: ['res'] }, + ], + sanitizers: [ + // URL-encoding: neutralizes markup injection AND path separators + // (`%2F` is not a separator inside a path component). + { name: 'encodeURIComponent', neutralizes: ['xss', 'path-traversal'], global: true }, + // `escape-html` exports its function as the module default — the + // `'default'` pseudo-name matches the default-imported / require'd + // module handle being invoked directly. + { name: 'default', neutralizes: ['xss'], module: 'escape-html' }, + { name: 'encode', neutralizes: ['xss'], module: 'he' }, + { name: 'basename', neutralizes: ['path-traversal'], module: 'path' }, + { name: 'escape', neutralizes: ['xss'], module: 'validator' }, + ], +}; + +/** + * Deterministic digest of a spec's full content. Key order is canonicalized + * (recursively sorted) so the version reflects CONTENT, not literal layout; + * array order is semantic (entry identity) and intentionally preserved. + */ +export function computeTaintModelVersion(spec: SourceSinkSanitizerSpec): string { + return createHash('sha256').update(canonicalJson(spec)).digest('hex').slice(0, 12); +} + +function canonicalJson(value: unknown): string { + if (Array.isArray(value)) return `[${value.map(canonicalJson).join(',')}]`; + if (value !== null && typeof value === 'object') { + const entries = Object.entries(value as Record) + .filter(([, v]) => v !== undefined) + .sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0)) + .map(([k, v]) => `${JSON.stringify(k)}:${canonicalJson(v)}`); + return `{${entries.join(',')}}`; + } + return JSON.stringify(value); +} + +/** Version stamp of the built-in TS/JS model (joins the RepoMeta pdg key in U5). */ +export const taintModelVersion: string = computeTaintModelVersion(TS_JS_TAINT_MODEL); + +/** + * Register the built-in model for TypeScript and JavaScript. Explicit init + * seam for the U4 emit path (call before the pdg window consumes the + * registry); idempotent. Vue and other TS-adjacent language ids are + * deliberately NOT registered — the M3 scope is TS/JS only. + */ +export function registerBuiltinTaintModels(): void { + registerSourceSinkConfig(SupportedLanguages.TypeScript, TS_JS_TAINT_MODEL); + registerSourceSinkConfig(SupportedLanguages.JavaScript, TS_JS_TAINT_MODEL); +} diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index 66300b8b3..542326a63 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -49,6 +49,11 @@ import { DEFAULT_MAX_CFG_EDGES_PER_FUNCTION, DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, } from './ingestion/cfg/emit.js'; +import { + DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION, + DEFAULT_PDG_MAX_TAINT_HOPS, +} from './ingestion/taint/propagate.js'; +import { taintModelVersion } from './ingestion/taint/typescript-model.js'; import { computeFileHashes, diffFileHashes } from '../storage/file-hash.js'; import { extractChangedSubgraph, @@ -141,6 +146,13 @@ export interface AnalyzeOptions { /** Per-function REACHING_DEF edge cap (#2082 M2). Forwarded to * `PipelineOptions.pdgMaxReachingDefEdgesPerFunction`. */ pdgMaxReachingDefEdgesPerFunction?: number; + /** Per-function taint findings cap (#2083 M3). Forwarded to + * `PipelineOptions.pdgMaxTaintFindingsPerFunction`. No CLI flag or rc key + * (KTD8) — programmatic / server path only, like the other pdg caps. */ + pdgMaxTaintFindingsPerFunction?: number; + /** Per-finding taint hop cap (#2083 M3, KTD6). Forwarded to + * `PipelineOptions.pdgMaxTaintHops`. No CLI flag or rc key (KTD8). */ + pdgMaxTaintHops?: number; /** * Default branch threaded into generated AGENTS.md / CLAUDE.md so the * regression-compare example uses the configured branch instead of a @@ -343,7 +355,12 @@ export const collectBranchCacheKeys = async ( */ type PdgOptions = Pick< AnalyzeOptions, - 'pdg' | 'pdgMaxFunctionLines' | 'pdgMaxEdgesPerFunction' | 'pdgMaxReachingDefEdgesPerFunction' + | 'pdg' + | 'pdgMaxFunctionLines' + | 'pdgMaxEdgesPerFunction' + | 'pdgMaxReachingDefEdgesPerFunction' + | 'pdgMaxTaintFindingsPerFunction' + | 'pdgMaxTaintHops' >; export const resolvePdgConfig = (options: PdgOptions): RepoMeta['pdg'] => @@ -354,6 +371,17 @@ export const resolvePdgConfig = (options: PdgOptions): RepoMeta['pdg'] => maxReachingDefEdgesPerFunction: options.pdgMaxReachingDefEdgesPerFunction ?? DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, + // #2083 M3: taint caps + model identity. The key-union comparator in + // pdgModeMismatch picks these up structurally — an M2-era stamp lacks + // all three, so the first M3 run over an M2 `--pdg` index trips a full + // writeback that populates TAINTED/SANITIZES rows without `--force`. + maxTaintFindingsPerFunction: + options.pdgMaxTaintFindingsPerFunction ?? DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION, + maxTaintHops: options.pdgMaxTaintHops ?? DEFAULT_PDG_MAX_TAINT_HOPS, + // Built-in model digest (KTD7/R7): persisted findings must never + // outlive the model that produced them — ANY model-content change + // ships as a new digest and repopulates the taint edges. + taintModelVersion, } : undefined; @@ -753,6 +781,8 @@ export async function runFullAnalysis( pdgMaxFunctionLines: options.pdgMaxFunctionLines, pdgMaxEdgesPerFunction: options.pdgMaxEdgesPerFunction, pdgMaxReachingDefEdgesPerFunction: options.pdgMaxReachingDefEdgesPerFunction, + pdgMaxTaintFindingsPerFunction: options.pdgMaxTaintFindingsPerFunction, + pdgMaxTaintHops: options.pdgMaxTaintHops, fetchWrappers: options.fetchWrappers, }, ); diff --git a/gitnexus/src/mcp/local/local-backend.ts b/gitnexus/src/mcp/local/local-backend.ts index 025f11bb2..83e46cfeb 100644 --- a/gitnexus/src/mcp/local/local-backend.ts +++ b/gitnexus/src/mcp/local/local-backend.ts @@ -54,8 +54,29 @@ import { import { PhaseTimer } from '../../core/search/phase-timer.js'; import { checkStalenessAsync, checkCwdMatch } from '../../core/git-staleness.js'; import { logger } from '../../core/logger.js'; -import { LIST_REPOS_DEFAULT_LIMIT, LIST_REPOS_MAX_LIMIT } from '../tools.js'; +import { + LIST_REPOS_DEFAULT_LIMIT, + LIST_REPOS_MAX_LIMIT, + EXPLAIN_DEFAULT_LIMIT, + EXPLAIN_MAX_LIMIT, +} from '../tools.js'; import { findImportCycles } from '../../core/graph/import-cycles.js'; +import { decodeTaintPath } from '../../core/ingestion/taint/path-codec.js'; +import { EXTENSIONS } from '../../core/ingestion/import-resolvers/utils.js'; + +/** Real source-file extensions (`.ts`, `.py`, …) from the resolver's list, + * excluding the empty entry and the `/index.*` forms — used to decide whether + * an `explain` target is a file path vs a (possibly dotted) symbol name. */ +const SOURCE_FILE_EXTENSIONS: readonly string[] = EXTENSIONS.filter( + (e) => e.startsWith('.') && !e.includes('/'), +); +/** A target is path-ish if it has a path separator or ends in a known source + * extension. A bare dotted symbol (`UserController.create`) is NOT path-ish. */ +function looksLikeFilePath(target: string): boolean { + if (/[\\/]/.test(target)) return true; + const lower = target.toLowerCase(); + return SOURCE_FILE_EXTENSIONS.some((ext) => lower.endsWith(ext)); +} // AI context generation is CLI-only (gitnexus analyze) // import { generateAIContextFiles } from '../../cli/ai-context.js'; @@ -1243,6 +1264,8 @@ export class LocalBackend { } case 'context': return this.context(repo, params); + case 'explain': + return this.explain(repo, params); case 'impact': return this.impact(repo, params); case 'detect_changes': @@ -2730,6 +2753,259 @@ export class LocalBackend { }; } + /** + * Explain tool (#2083 M3 U6) — persisted taint-finding explanation. + * WAL-aware wrapper mirroring `context`. + */ + private async explain( + repo: RepoHandle, + params: { target?: string; limit?: number }, + ): Promise { + try { + return await this._explainImpl(repo, params); + } catch (err: any) { + const msg = (err instanceof Error ? err.message : String(err)) || 'Explain query failed'; + if (isWalCorruptionError(err)) { + return { + error: msg, + recoverySuggestion: WAL_RECOVERY_SUGGESTION, + }; + } + throw err; + } + } + + /** + * Taint findings are persisted as `TAINTED` rows in CodeRelation whose + * endpoints are BOTH BasicBlock nodes — the label anchor restricts every + * query here to the BasicBlock→BasicBlock partition of the rel table + * (which holds only the sparse, per-function-capped pdg layers), never a + * global symbol-space scan (the S1 verdict; LadybugDB has no rel-property + * index, so the label anchor IS the bound). + * + * Anchoring granularity: + * - file target → BasicBlock id prefix (`BasicBlock::` — the + * shared `basicBlockId` template) with an exact-or-suffix path match so + * `vuln.ts` finds `src/vuln.ts`. + * - symbol target → resolved via `resolveSymbolCandidates` (the context() + * path: ambiguous ⇒ ranked candidates, unknown ⇒ not-found), then the + * file id-prefix PLUS source-block startLine within the symbol's + * [startLine, endLine] span. Findings are intra-procedural, so filtering + * the SOURCE endpoint is sufficient — both endpoints share the function. + * Symbols without a line span degrade to the file-level filter. + * + * The per-finding `sinkKind` and hop path decode from the persisted + * `reason` via the SHARED `taint/path-codec.ts` (the U4 write path encodes + * with the same module — `;` header + ordered `variable:line` hops). + */ + private async _explainImpl( + repo: RepoHandle, + params: { target?: string; limit?: number }, + ): Promise { + await this.ensureInitialized(repo); + + const rawLimit = params.limit ?? EXPLAIN_DEFAULT_LIMIT; + if (!Number.isInteger(rawLimit) || rawLimit < 1 || rawLimit > EXPLAIN_MAX_LIMIT) { + return { + error: `Invalid "limit": expected an integer in [1, ${EXPLAIN_MAX_LIMIT}], got ${JSON.stringify(params.limit)}.`, + }; + } + const limit = rawLimit; + + const NO_TAINT_NOTE = + 'no taint layer — run gitnexus analyze --pdg to record taint findings for this repo'; + + // Cheap meta probe: the TAINT layer exists iff the pdg stamp carries a + // `taintModelVersion` (the field M3 added). An M1/M2-era `--pdg` index has + // `meta.pdg` defined but no taintModelVersion — BasicBlock/REACHING_DEF + // exist, zero TAINTED rows do — so it must surface the no-taint-layer hint, + // not the generic "analyzed, nothing found" note. An unreadable meta (e.g. + // a seeded test DB) falls through to the row-existence probe below. + let pdgStamped: boolean | undefined; + try { + const meta = await loadMeta(path.dirname(repo.lbugPath)); + if (meta) pdgStamped = meta.pdg?.taintModelVersion !== undefined; + } catch { + /* meta unreadable — decide from the DB below */ + } + if (pdgStamped === false) { + return { findings: [], totalFindings: 0, note: NO_TAINT_NOTE }; + } + + // Resolve the optional anchor into a WHERE clause on the SOURCE block. + const target = typeof params.target === 'string' ? params.target.trim() : ''; + let anchorClause = ''; + const queryParams: Record = {}; + let anchor: { file: string; symbol?: string; startLine?: number; endLine?: number } | undefined; + + // Build the anchor as a file filter (used only when `target` is path-ish). + const buildFileAnchor = (): void => { + // Exact path via the BasicBlock id-prefix template, OR a + // path-separator-aligned suffix so partial paths work like context()'s + // file_path hint ("vuln.ts" ⇒ "src/vuln.ts", never "devuln.ts"). + anchorClause = + 'AND (a.id STARTS WITH $idPrefix OR a.filePath = $targetPath OR a.filePath ENDS WITH $targetSuffix)'; + queryParams.idPrefix = `BasicBlock:${target}:`; + queryParams.targetPath = target; + queryParams.targetSuffix = `/${target}`; + anchor = { file: target as string }; + }; + + // Resolve `target` as a symbol into the anchor. Returns an early-return + // payload (not_found / ambiguous) or undefined on success. + const resolveSymbolAnchor = async (): Promise | undefined> => { + const outcome = await this.resolveSymbolCandidates(repo, { name: target as string }, {}); + if (outcome.kind === 'not_found') { + return { error: `Symbol '${target}' not found` }; + } + if (outcome.kind === 'ambiguous') { + return { + status: 'ambiguous', + message: `Found ${outcome.candidates.length} symbols matching '${target}'. Re-call explain with the file path, or disambiguate via context() first.`, + candidates: outcome.candidates.map((c) => ({ + uid: c.id, + name: c.name, + kind: c.type, + filePath: c.filePath, + line: c.startLine, + score: Number(c.score.toFixed(2)), + })), + }; + } + const sym = outcome.symbol; + queryParams.idPrefix = `BasicBlock:${sym.filePath}:`; + anchor = { file: sym.filePath, symbol: sym.name }; + if ( + typeof sym.startLine === 'number' && + typeof sym.endLine === 'number' && + sym.endLine >= sym.startLine + ) { + anchorClause = + 'AND a.id STARTS WITH $idPrefix AND a.startLine >= $symStart AND a.startLine <= $symEnd'; + queryParams.symStart = sym.startLine; + queryParams.symEnd = sym.endLine; + anchor.startLine = sym.startLine; + anchor.endLine = sym.endLine; + } else { + // No usable span — degrade to the file-level filter (documented). + anchorClause = 'AND a.id STARTS WITH $idPrefix'; + } + return undefined; + }; + + // Bounded by construction: the BasicBlock→BasicBlock partition holds only + // the sparse pdg layers, TAINTED rows are per-function-capped at analyze + // time, and the page is LIMIT-bounded (the limit is a validated integer — + // interpolated because LadybugDB does not parameterize LIMIT). + const runAnchoredQuery = async (): Promise<{ rows: unknown[]; totalFindings: number }> => { + const matchClause = ` + MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock) + WHERE r.type = 'TAINTED' ${anchorClause}`; + const [qRows, countRows] = await Promise.all([ + executeParameterized( + repo.lbugPath, + `${matchClause} + RETURN a.id AS sourceBlockId, a.filePath AS file, a.startLine AS sourceStart, + b.startLine AS sinkStart, r.reason AS reason, b.id AS sinkBlockId + ORDER BY sourceBlockId, sinkBlockId, reason + LIMIT ${limit}`, + queryParams, + ), + executeParameterized( + repo.lbugPath, + `${matchClause} + RETURN COUNT(*) AS total`, + queryParams, + ), + ]); + return { + rows: qRows, + totalFindings: Number((countRows[0] as any)?.total ?? (countRows[0] as any)?.[0] ?? 0), + }; + }; + + if (target) { + if (looksLikeFilePath(target)) { + buildFileAnchor(); + } else { + // A bare or dotted symbol name (`UserController.create`) — resolve as a + // symbol rather than silently file-anchoring to an empty result. + const early = await resolveSymbolAnchor(); + if (early) return early; + } + } + + const { rows, totalFindings } = await runAnchoredQuery(); + + if (totalFindings === 0 && pdgStamped === undefined && !target) { + // Meta was unreadable and the repo-wide enumerate found nothing — the + // count above WAS the existence probe; surface the layer hint. + return { findings: [], totalFindings: 0, note: NO_TAINT_NOTE }; + } + if (totalFindings === 0 && pdgStamped === undefined && target) { + // Anchored miss with unreadable meta: one extra bounded probe decides + // "no findings for this anchor" vs "no taint layer at all". + const probe = await executeParameterized( + repo.lbugPath, + `MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock) WHERE r.type = 'TAINTED' RETURN r.reason AS reason LIMIT 1`, + {}, + ); + if (probe.length === 0) { + return { findings: [], totalFindings: 0, note: NO_TAINT_NOTE }; + } + } + + const findings = rows.map((r: any) => { + const sourceBlockId = String(r.sourceBlockId ?? r[0] ?? ''); + const file = String(r.file ?? r[1] ?? ''); + const sourceStart = (r.sourceStart ?? r[2]) as number | undefined; + const sinkStart = (r.sinkStart ?? r[3]) as number | undefined; + const reason = r.reason ?? r[4]; + // basicBlockId = `BasicBlock::::` — + // split from the RIGHT (the filePath may itself contain ':'). + const idParts = sourceBlockId.split(':'); + const fnLine = Number(idParts[idParts.length - 3]); + const decoded = decodeTaintPath(reason); + if (!decoded.ok) { + // Unreadable reason (foreign/corrupt row): surface the finding's + // existence with its block anchors, never throw. + return { + file, + ...(Number.isInteger(fnLine) ? { functionLine: fnLine } : {}), + sinkKind: 'unknown', + source: { line: sourceStart }, + sink: { line: sinkStart }, + hops: [], + pathIncomplete: true, + }; + } + const hops = decoded.hops.map((h) => ({ + variable: h.variable, + line: h.line, + ...(h.viaCall ? { viaCall: true } : {}), + })); + const first = hops[0]; + const last = hops[hops.length - 1]; + return { + file, + ...(Number.isInteger(fnLine) ? { functionLine: fnLine } : {}), + sinkKind: decoded.kind ?? 'unknown', + source: first ? { variable: first.variable, line: first.line } : { line: sourceStart }, + sink: { line: last?.line ?? sinkStart }, + hops, + ...(decoded.truncated ? { pathIncomplete: true } : {}), + }; + }); + + return { + ...(anchor ? { anchor } : {}), + findings, + totalFindings, + ...(totalFindings > findings.length ? { truncated: true } : {}), + note: 'Intra-procedural findings only — cross-function, closure/callback, property/field, and implicit flows are not modeled; absence of a finding is not proof of safety. SANITIZES (kill) edges are queryable via cypher.', + }; + } + /** * Legacy explore — kept for backwards compatibility with resources.ts. * Routes cluster/process types to direct graph queries. diff --git a/gitnexus/src/mcp/resources.ts b/gitnexus/src/mcp/resources.ts index 707a5e047..bb900d924 100644 --- a/gitnexus/src/mcp/resources.ts +++ b/gitnexus/src/mcp/resources.ts @@ -335,6 +335,9 @@ async function getContextResource(backend: LocalBackend, repoName?: string): Pro lines.push(' - query: Process-grouped code intelligence (execution flows related to a concept)'); lines.push(' - context: 360-degree symbol view (categorized refs, process participation)'); lines.push(' - impact: Blast radius analysis (what breaks if you change a symbol)'); + lines.push( + ' - explain: Persisted taint findings — source→sink data flows with per-hop variables (requires analyze --pdg)', + ); lines.push(' - detect_changes: Git-diff impact analysis (what do your changes affect)'); lines.push(' - rename: Multi-file coordinated rename with confidence tags'); lines.push(' - cypher: Raw graph queries'); diff --git a/gitnexus/src/mcp/tools.ts b/gitnexus/src/mcp/tools.ts index 065fa37b3..dafdf7caf 100644 --- a/gitnexus/src/mcp/tools.ts +++ b/gitnexus/src/mcp/tools.ts @@ -62,6 +62,16 @@ const DESTRUCTIVE_TOOL_ANNOTATIONS: ToolAnnotations = { export const LIST_REPOS_DEFAULT_LIMIT = 50; export const LIST_REPOS_MAX_LIMIT = 200; +/** + * Pagination bounds for the `explain` tool (#2083 M3 U6). Findings are sparse + * and capped per function at analyze time, but a large repo can still + * accumulate enough TAINTED rows to blow MCP/LLM token limits — the response + * is page-bounded like `list_repos`. Exported so the backend clamp + * (`local-backend.ts`) and the schema stay a single source of truth. + */ +export const EXPLAIN_DEFAULT_LIMIT = 50; +export const EXPLAIN_MAX_LIMIT = 200; + export const GITNEXUS_TOOLS: ToolDefinition[] = [ { name: 'list_repos', @@ -513,6 +523,50 @@ SERVICE: optional monorepo path prefix (case-sensitive path segments). When "rep required: ['target', 'direction'], }, }, + { + name: 'explain', + description: `Explain persisted taint findings: intra-procedural source→sink data flows (TAINTED edges) recorded by \`gitnexus analyze --pdg\`. + +Each finding carries the sink category (command-injection, code-injection, path-traversal, sql-injection, xss), the source/sink lines, and the ordered hop path with the variable carried on each hop (decoded from the persisted path encoding). + +WHEN TO USE: Security review — "what taint findings exist in this repo / file / function?". Requires the repo to be indexed with \`gitnexus analyze --pdg\`; without that layer the tool returns a clear "no taint layer" note, not an error. + +ANCHORLESS (no "target"): enumerates all persisted findings for the repo — bounded ("limit", deterministic order), with "totalFindings" and a "truncated" flag. +ANCHORED ("target" = file path or symbol/function name): full hop detail for that anchor. A file-ish target (contains "/" or an extension) filters by file; a symbol name resolves like context() — ambiguous names return ranked candidates, unknown names return not-found. Symbol anchoring is line-range granular (findings whose source block starts inside the symbol's span). + +CONTRACT CAVEATS (intra-procedural M3 scope — absent flows are NOT proof of safety): +- Cross-function flows are not modeled (a flow through a helper function is invisible). +- Closure/callback flows are invisible in both directions (e.g. arr.forEach(() => sink(y))). +- Property/field flows are not tracked (obj.x = taint; sink(obj.y) has no chain). +- Guard-style sanitizers (if (isValid(x))) and implicit/control-dependence flows are not modeled. +- CommonJS aliasing is partially modeled (require('') joins resolve; dynamic requires do not). +- Exception-path over-approximation can produce false-positive noise. + +Findings are deliberately NOT part of impact()'s traversal or the web schema — explain is the dedicated taint consumer. SANITIZES (kill) edges are queryable via cypher.`, + annotations: READ_ONLY_TOOL_ANNOTATIONS, + inputSchema: { + type: 'object', + properties: { + target: { + type: 'string', + description: + 'Optional anchor: a file path (e.g. "src/handlers/run.ts" — suffix match accepted) or a symbol/function name (resolved like context()). Omit to enumerate all findings for the repo.', + }, + limit: { + type: 'integer', + description: `Max findings returned (default: ${EXPLAIN_DEFAULT_LIMIT}, max: ${EXPLAIN_MAX_LIMIT}). "totalFindings" reports the full matched count; "truncated" is set when the page is smaller.`, + default: EXPLAIN_DEFAULT_LIMIT, + minimum: 1, + maximum: EXPLAIN_MAX_LIMIT, + }, + repo: { + type: 'string', + description: 'Repository name or path. Omit if only one repo is indexed.', + }, + }, + required: [], + }, + }, { name: 'route_map', description: `Show API route mappings: which components/hooks fetch which API endpoints, and which handler files serve them. @@ -647,6 +701,7 @@ const BRANCH_SCOPED_TOOLS = new Set([ 'cypher', 'context', 'detect_changes', + 'explain', 'check', 'impact', 'rename', diff --git a/gitnexus/src/storage/parse-cache.ts b/gitnexus/src/storage/parse-cache.ts index 837144c5c..5061e635a 100644 --- a/gitnexus/src/storage/parse-cache.ts +++ b/gitnexus/src/storage/parse-cache.ts @@ -184,7 +184,16 @@ export const computeChunkHash = ( // option-blind-key trap. `def` marks an unset (default) value so two // default-cap runs share a key. The emit-time edge cap is deliberately // absent — see the PdgCacheKey doc comment. - const ns = `pdg:1;maxFn=${opts.maxFunctionLines ?? 'def'}`; + // + // NAMESPACE VERSION (`pdg:2`): bumped when the worker-emitted + // `cfgSideChannel` SHAPE changes for pdg-mode runs only — pdg:1→2 in #2083 + // M3 U1 (TsHarvester emits taint `sites` on StatementFacts). Invalidates + // pdg-mode chunks and their durable parsedfile-cache entries; flag-off + // chunk keys never reach this line and stay byte-identical, so non-pdg + // users pay nothing. Deliberately NOT a SCHEMA_BUMP — that gates the whole + // cache version and would force a full cold re-parse on EVERY user (the M1 + // bump comment above records that cost). + const ns = `pdg:2;maxFn=${opts.maxFunctionLines ?? 'def'}`; return sha256Hex(`${ns}\n${joined}`); }; diff --git a/gitnexus/src/storage/repo-manager.ts b/gitnexus/src/storage/repo-manager.ts index 5e9fd972f..a9bfd88a9 100644 --- a/gitnexus/src/storage/repo-manager.ts +++ b/gitnexus/src/storage/repo-manager.ts @@ -158,6 +158,24 @@ export interface RepoMeta { * type for that reason; resolved (always present) on every M2+ write. */ maxReachingDefEdgesPerFunction?: number; + /** + * Per-function taint findings cap, resolved (0 = unlimited; #2083 M3). + * ABSENT on an M1/M2-era stamp — like `maxReachingDefEdgesPerFunction`, + * that absence is what trips `pdgModeMismatch` on the first M3 run and + * forces the full writeback that populates TAINTED/SANITIZES rows. + */ + maxTaintFindingsPerFunction?: number; + /** Per-finding taint hop cap, resolved (0 = unlimited; #2083 M3 KTD6 — + * bounds the persisted hop-encoded `reason`). Optional for the same + * M2-era-stamp upgrade reason as the findings cap. */ + maxTaintHops?: number; + /** + * Digest of the built-in taint model the persisted findings were + * produced under (#2083 M3 KTD7/R7). Any model-content change ships a + * new digest → mismatch → full writeback repopulates taint edges + * without `--force`. Optional: absent on pre-M3 stamps. + */ + taintModelVersion?: string; }; } diff --git a/gitnexus/test/helpers/taint-fixture.ts b/gitnexus/test/helpers/taint-fixture.ts new file mode 100644 index 000000000..38337798f --- /dev/null +++ b/gitnexus/test/helpers/taint-fixture.ts @@ -0,0 +1,123 @@ +/** + * Shared pure-path taint harness over the pdg-repo fixture (#2083 M3 U7). + * + * Runs the SAME per-function pipeline the in-phase emit driver runs — + * collect CFGs → site-safety gate → match → zero-match fast path → + * `computeReachingDefs` → `computeTaintFlows` — but build-free on the main + * thread (parse via tree-sitter directly, like reaching-defs-snapshot). + * + * Two consumers, deliberately fed from ONE module so they cannot drift: + * - `taint-snapshot.test.ts` serializes the per-function results (AE1/AE3); + * - `pipeline-pdg.test.ts` sums findings/kills as the EXPECTED stored-row + * counts for the sparse-persistence gate (AE2): the real worker pipeline + * must persist exactly one TAINTED row per pure-path finding and one + * SANITIZES row per kill — O(findings), never a REACHING_DEF-style + * explosion. + * + * Limits mirror the run.ts derivation: findings/hops caps at their U5 + * defaults, `maxFacts` at the RD-edge-cap formula's default product + * (`DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION` — same number). + */ +import fs from 'fs'; +import path from 'path'; +import Parser from 'tree-sitter'; +import TypeScript from 'tree-sitter-typescript'; +import type { ParsedImport } from 'gitnexus-shared'; +import { collectFunctionCfgs } from '../../src/core/ingestion/cfg/collect.js'; +import { computeReachingDefs } from '../../src/core/ingestion/cfg/reaching-defs.js'; +import { DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION } from '../../src/core/ingestion/cfg/emit.js'; +import { getProvider } from '../../src/core/ingestion/languages/index.js'; +import { SupportedLanguages } from '../../src/config/supported-languages.js'; +import { hasTaintSafeSites } from '../../src/core/ingestion/taint/site-safety.js'; +import { buildTaintImportIndex, matchFunctionSites } from '../../src/core/ingestion/taint/match.js'; +import { TS_JS_TAINT_MODEL } from '../../src/core/ingestion/taint/typescript-model.js'; +import { + computeTaintFlows, + DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION, + DEFAULT_PDG_MAX_TAINT_HOPS, + type FunctionTaintResult, +} from '../../src/core/ingestion/taint/propagate.js'; +import type { FunctionCfg } from '../../src/core/ingestion/cfg/types.js'; + +/** The taint-bearing fixture files (sample.ts is the zero-match control). */ +export const TAINT_FIXTURE_FILES = ['vuln.ts', 'taint-cases.ts', 'sample.ts'] as const; + +/** + * Hand-built `ParsedImport` lists matching each fixture file's import + * statements (the build-free path has no extractor run). MUST stay in sync + * with the fixture sources — the AE2 equality against the real pipeline + * (which uses extracted `parsedImports`) breaks loudly if they drift. + */ +export const TAINT_FIXTURE_IMPORTS: Record = { + 'vuln.ts': [ + { kind: 'named', localName: 'exec', importedName: 'exec', targetRaw: 'child_process' }, + ], + 'taint-cases.ts': [ + { kind: 'named', localName: 'exec', importedName: 'exec', targetRaw: 'child_process' }, + ], + 'sample.ts': [], +}; + +export interface FixtureFunctionTaint { + readonly file: string; + readonly startLine: number; + readonly cfg: FunctionCfg; + /** + * `no-match` — the zero-match fast path skipped the solver entirely; + * `unsafe-sites` — `hasTaintSafeSites` rejected the harvest; + * otherwise the `computeTaintFlows` status (`computed` / `coverage-gap`). + */ + readonly status: 'no-match' | 'unsafe-sites' | FunctionTaintResult['status']; + /** Present iff the solver ran (status `computed` or `coverage-gap`). */ + readonly flows?: FunctionTaintResult; +} + +/** Run the pure taint path over one fixture file. Deterministic. */ +export function computeFixtureFileTaint( + fixtureDir: string, + file: string, +): readonly FixtureFunctionTaint[] { + const visitor = getProvider(SupportedLanguages.TypeScript).cfgVisitor; + if (!visitor) throw new Error('no cfgVisitor for TypeScript'); + const source = fs.readFileSync(path.join(fixtureDir, file), 'utf8'); + const parser = new Parser(); + parser.setLanguage(TypeScript.typescript); + const cfgs = collectFunctionCfgs(parser.parse(source).rootNode, visitor, file).cfgs; + + const imports = TAINT_FIXTURE_IMPORTS[file]; + if (imports === undefined) { + throw new Error(`no FIXTURE_IMPORTS entry for ${file} — add it (see module doc)`); + } + const importIndex = buildTaintImportIndex(imports); + + return cfgs.map((cfg) => { + const base = { file, startLine: cfg.functionStartLine, cfg }; + if (!hasTaintSafeSites(cfg)) return { ...base, status: 'unsafe-sites' as const }; + const matches = matchFunctionSites(cfg, TS_JS_TAINT_MODEL, importIndex); + if (!matches.hasSource || !matches.hasSink) return { ...base, status: 'no-match' as const }; + const defUse = computeReachingDefs(cfg, { + maxFacts: DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, + }); + const flows = computeTaintFlows(cfg, defUse, matches, { + maxFindingsPerFunction: DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION, + maxHops: DEFAULT_PDG_MAX_TAINT_HOPS, + }); + return { ...base, status: flows.status, flows }; + }); +} + +/** Run the pure taint path over the whole fixture battery, in file order. */ +export function computeFixtureTaint(fixtureDir: string): readonly FixtureFunctionTaint[] { + return TAINT_FIXTURE_FILES.flatMap((f) => computeFixtureFileTaint(fixtureDir, f)); +} + +/** Pure-path totals — the AE2 expected stored-row counts. */ +export function fixtureTaintTotals(fixtureDir: string): { findings: number; kills: number } { + let findings = 0; + let kills = 0; + for (const fn of computeFixtureTaint(fixtureDir)) { + findings += fn.flows?.findings.length ?? 0; + kills += fn.flows?.kills.length ?? 0; + } + return { findings, kills }; +} diff --git a/gitnexus/test/helpers/ts-cfg-harness.ts b/gitnexus/test/helpers/ts-cfg-harness.ts new file mode 100644 index 000000000..12f8509c6 --- /dev/null +++ b/gitnexus/test/helpers/ts-cfg-harness.ts @@ -0,0 +1,68 @@ +/** + * Shared TS CFG/taint unit-test harness (#2083 review). + * + * Parses real TypeScript source through the worker-side CFG visitor and the + * scope-capture import interpreter, so taint unit tests run against the exact + * structures the pipeline feeds `computeReachingDefs` / `matchFunctionSites` / + * `computeTaintFlows`, never hand-built mocks. Extracted from the byte-identical + * copies that lived in model-match / propagate / taint-emit / harvest tests. + */ + +import Parser from 'tree-sitter'; +import TypeScript from 'tree-sitter-typescript'; +import type { ParsedImport } from 'gitnexus-shared'; +import type { SyntaxNode } from '../../src/core/ingestion/utils/ast-helpers.js'; +import { + createTypeScriptCfgVisitor, + TS_FUNCTION_TYPES, +} from '../../src/core/ingestion/cfg/visitors/typescript.js'; +import type { FunctionCfg } from '../../src/core/ingestion/cfg/types.js'; +import { emitTsScopeCaptures } from '../../src/core/ingestion/languages/typescript/captures.js'; +import { interpretTsImport } from '../../src/core/ingestion/languages/typescript/interpret.js'; + +const visitor = createTypeScriptCfgVisitor(); + +export function parse(code: string): SyntaxNode { + const parser = new Parser(); + parser.setLanguage(TypeScript.typescript); + return parser.parse(code).rootNode; +} + +export function collectFunctions(root: SyntaxNode): SyntaxNode[] { + const out: SyntaxNode[] = []; + const stack = [root]; + while (stack.length) { + const n = stack.pop() as SyntaxNode; + if (TS_FUNCTION_TYPES.has(n.type)) out.push(n); + for (let i = n.namedChildCount - 1; i >= 0; i--) { + const c = n.namedChild(i); + if (c) stack.push(c); + } + } + return out; +} + +/** The CFG of the function at `index` (default 0). */ +export function cfgOf(code: string, index = 0): FunctionCfg { + const fns = collectFunctions(parse(code)); + const fn = fns[index]; + if (!fn) throw new Error(`no function at index ${index}`); + const cfg = visitor.buildFunctionCfg(fn, 'fixture.ts'); + if (!cfg) throw new Error('buildFunctionCfg returned undefined'); + return cfg; +} + +/** Every function's CFG, in source order. */ +export function cfgsOf(code: string): FunctionCfg[] { + return collectFunctions(parse(code)) + .map((fn) => visitor.buildFunctionCfg(fn, 'fixture.ts')) + .filter((c): c is FunctionCfg => c !== undefined); +} + +/** Real ParsedImports via the TS scope-capture + interpreter path. */ +export function importsFor(src: string): ParsedImport[] { + return emitTsScopeCaptures(src, 'fixture.ts') + .filter((m) => m['@import.statement'] !== undefined) + .map((m) => interpretTsImport(m)) + .filter((p): p is ParsedImport => p !== null); +} diff --git a/gitnexus/test/integration/cfg/__snapshots__/taint-snapshot.test.ts.snap b/gitnexus/test/integration/cfg/__snapshots__/taint-snapshot.test.ts.snap new file mode 100644 index 000000000..0a9c9bff0 --- /dev/null +++ b/gitnexus/test/integration/cfg/__snapshots__/taint-snapshot.test.ts.snap @@ -0,0 +1,102 @@ +// Vitest Snapshot v1, https://vitest.dev/guide/snapshot.html + +exports[`U7 — taint findings/kills snapshot on the pdg-repo fixture battery > matches the committed findings/kills for every fixture function 1`] = ` +[ + { + "dropped": 0, + "file": "vuln.ts", + "findings": [ + "cmd@10->cmd@11:command-injection", + ], + "kills": [], + "startLine": 9, + "status": "computed", + }, + { + "dropped": 0, + "file": "vuln.ts", + "findings": [], + "kills": [ + "value@17<-17:path-traversal,xss", + ], + "startLine": 16, + "status": "computed", + }, + { + "dropped": 0, + "file": "taint-cases.ts", + "findings": [ + "req@15:command-injection", + ], + "kills": [], + "startLine": 14, + "status": "computed", + }, + { + "dropped": 0, + "file": "taint-cases.ts", + "findings": [ + "a@20->b@21->c@22->c@23:command-injection", + ], + "kills": [], + "startLine": 19, + "status": "computed", + }, + { + "dropped": 0, + "file": "taint-cases.ts", + "findings": [ + "text@34->text@38:xss", + ], + "kills": [ + "text@36<-36:path-traversal,xss", + ], + "startLine": 29, + "status": "computed", + }, + { + "dropped": 0, + "file": "taint-cases.ts", + "findings": [ + "cmd@44->cmd@48:command-injection", + ], + "kills": [], + "startLine": 43, + "status": "computed", + }, + { + "dropped": 0, + "file": "taint-cases.ts", + "findings": [ + "raw@54->built@55*->built@56:command-injection", + ], + "kills": [], + "startLine": 53, + "status": "computed", + }, + { + "dropped": 0, + "file": "taint-cases.ts", + "findings": [], + "kills": [], + "startLine": 59, + "status": "no-match", + }, + { + "dropped": 0, + "file": "sample.ts", + "findings": [], + "kills": [], + "startLine": 5, + "status": "no-match", + }, + { + "dropped": 0, + "file": "sample.ts", + "findings": [], + "kills": [], + "startLine": 14, + "status": "no-match", + }, +] +`; diff --git a/gitnexus/test/integration/cfg/fixtures/pdg-repo/taint-cases.ts b/gitnexus/test/integration/cfg/fixtures/pdg-repo/taint-cases.ts new file mode 100644 index 000000000..71f5f1e56 --- /dev/null +++ b/gitnexus/test/integration/cfg/fixtures/pdg-repo/taint-cases.ts @@ -0,0 +1,61 @@ +// Taint acceptance battery (#2083 M3 U7): the plan's six fixture shapes not +// already covered by vuln.ts (which carries the reassignment source→sink flow +// and the must-def sanitized variant). Lives in its OWN file — sample.ts's +// line numbers anchor pre-existing REACHING_DEF assertions and must not +// shift, and vuln.ts's lines anchor the U6 explain-tool assertions. +// +// The taint-snapshot test hand-builds this file's import list (test/helpers/ +// taint-fixture.ts) — keep the imports below in sync with FIXTURE_IMPORTS. + +import { exec } from 'child_process'; + +// Direct source→sink, statement-local rule (b): req.body lands in the exec +// argument with no def anywhere on the statement → single-hop finding. +export function directSourceToSink(req: { body: string }): void { + exec(req.body); +} + +// Multi-hop chain (3+ hops): the taint walks a → b → c → sink. +export function multiHopChain(req: { body: string }): void { + const a = req.body; + const b = a; + const c = b; + exec(c); +} + +// Conditional sanitizer (may-def leg): the encode runs only when !trusted, so +// the unsanitized seed def still reaches the sink → the finding SURVIVES, +// and the sanitizer's own def is killed (one SANITIZES edge, binding `text`). +export function conditionalSanitizer( + req: { query: string }, + res: { send(v: string): void }, + trusted: boolean, +): void { + let text = req.query; + if (!trusted) { + text = encodeURIComponent(text); + } + res.send(text); +} + +// Loop-carried taint: `cmd = cmd + part` feeds itself through the back edge — +// the worklist reaches a fixpoint (monotone visited set) and the sink fires. +export function loopCarried(req: { body: string }, parts: string[]): void { + let cmd = req.body; + for (const part of parts) { + cmd = cmd + part; + } + exec(cmd); +} + +// Through-call (KTD5): `decorate` is unmodeled — taint propagates through to +// its result with the hop marked viaCall (lower-confidence evidence). +export function throughCall(req: { body: string }): void { + const raw = req.body; + const built = decorate(raw); + exec(built); +} + +function decorate(s: string): string { + return 'sh -c ' + s; +} diff --git a/gitnexus/test/integration/cfg/fixtures/pdg-repo/vuln.ts b/gitnexus/test/integration/cfg/fixtures/pdg-repo/vuln.ts new file mode 100644 index 000000000..50ffa1aa4 --- /dev/null +++ b/gitnexus/test/integration/cfg/fixtures/pdg-repo/vuln.ts @@ -0,0 +1,19 @@ +// Taint fixture (#2083 M3 U4): one vulnerable source→sink flow and one +// sanitized variant. Lives in its OWN file — sample.ts's line numbers anchor +// pre-existing REACHING_DEF assertions and must not shift. + +import { exec } from 'child_process'; + +// Vulnerable: req.body (remote-input source) flows unsanitized into +// child_process.exec (command-injection sink) → one TAINTED edge. +export function runUserCommand(req: { body: string }): void { + const cmd = req.body; + exec(cmd); +} + +// Sanitized: encodeURIComponent neutralizes xss before the res.send sink — +// the finding is suppressed and the kill persists as a SANITIZES edge. +export function sendEncoded(req: { query: string }, res: { send(v: string): void }): void { + const value = encodeURIComponent(req.query); + res.send(value); +} diff --git a/gitnexus/test/integration/cfg/pipeline-pdg.test.ts b/gitnexus/test/integration/cfg/pipeline-pdg.test.ts index 69ff6677f..d5d8415b3 100644 --- a/gitnexus/test/integration/cfg/pipeline-pdg.test.ts +++ b/gitnexus/test/integration/cfg/pipeline-pdg.test.ts @@ -4,6 +4,8 @@ import os from 'os'; import path from 'path'; import { runPipelineFromRepo } from '../../../src/core/ingestion/pipeline.js'; import type { PipelineResult } from '../../../src/types/pipeline.js'; +import { decodeTaintPath } from '../../../src/core/ingestion/taint/path-codec.js'; +import { fixtureTaintTotals } from '../../helpers/taint-fixture.js'; // U7 — end-to-end proof that the `--pdg` opt-in reaches BOTH sinks: the parse // worker builds a per-function CFG (workerData.pdg) and scope-resolution emits @@ -17,6 +19,8 @@ function counts(result: PipelineResult): { basicBlocks: number; cfgEdges: number; reachingDefs: number; + tainted: number; + sanitizes: number; } { let basicBlocks = 0; result.graph.forEachNode((n) => { @@ -24,11 +28,15 @@ function counts(result: PipelineResult): { }); let cfgEdges = 0; let reachingDefs = 0; + let tainted = 0; + let sanitizes = 0; for (const rel of result.graph.iterRelationships()) { if (rel.type === 'CFG') cfgEdges++; if (rel.type === 'REACHING_DEF') reachingDefs++; + if (rel.type === 'TAINTED') tainted++; + if (rel.type === 'SANITIZES') sanitizes++; } - return { basicBlocks, cfgEdges, reachingDefs }; + return { basicBlocks, cfgEdges, reachingDefs, tainted, sanitizes }; } const tmpDirs: string[] = []; @@ -69,11 +77,74 @@ describe('U7 — end-to-end --pdg pipeline', () => { } }, 60000); + // M3 (#2083 U4/U7): the taint layer rides the same gate. The fixture's + // vuln.ts carries one vulnerable flow (req.body → child_process.exec) and + // one sanitized variant (encodeURIComponent before res.send); taint-cases.ts + // adds the U7 acceptance battery (direct, multi-hop, conditional-sanitizer, + // loop-carried, through-call). + it('with --pdg on: emits TAINTED + SANITIZES edges with decodable hop reasons', async () => { + const result = await runPipelineFromRepo(freshRepo(), () => {}, { pdg: true }); + const blockIds = new Set(); + result.graph.forEachNode((n) => { + if (n.label === 'BasicBlock') blockIds.add(n.id); + }); + const { tainted, sanitizes, reachingDefs } = counts(result); + expect(tainted).toBeGreaterThan(0); + expect(sanitizes).toBeGreaterThanOrEqual(1); + + // AE2 (AC2) — sparse persistence. The load-bearing O(findings) gate is + // EXACT equality: one TAINTED row per pure-path finding and one SANITIZES + // row per kill over the same fixture, computed through the shared harness + // so the worker pipeline and the snapshot suite cannot drift apart. Any + // REACHING_DEF-style row multiplication (per-fact, per-block-pair, …) + // breaks the equality immediately. + const expected = fixtureTaintTotals(FIXTURE); + expect(expected.findings).toBeGreaterThan(0); + expect(tainted).toBe(expected.findings); + expect(sanitizes).toBe(expected.kills); + // Ratio sanity vs the dense RD projection on the SAME run. The fixture is + // deliberately finding-DENSE (nearly every function is a vulnerable + // acceptance case), so the honest measured ratio here is ~22% (8 taint + // rows vs 37 RD rows) — the < 0.5 bound still catches any per-fact + // explosion (which would multiply taint rows past RD); the representative + // ≪-RD posture on realistic density is gated by the bench taint scenario's + // absolute boundedness/byte ceilings (bench/cfg). + expect(tainted + sanitizes).toBeLessThan(reachingDefs * 0.5); + let sawVulnFlow = false; + for (const rel of result.graph.iterRelationships()) { + if (rel.type !== 'TAINTED' && rel.type !== 'SANITIZES') continue; + // Both endpoints are persisted BasicBlock nodes (the shared id template). + expect(blockIds.has(rel.sourceId)).toBe(true); + expect(blockIds.has(rel.targetId)).toBe(true); + if (rel.type === 'TAINTED') { + // The reason is the versioned hop encoding — decodable by the SHARED + // codec (U6's explain imports the same module), variables carried. + const decoded = decodeTaintPath(rel.reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) { + expect(decoded.hops.length).toBeGreaterThan(0); + for (const hop of decoded.hops) { + expect(hop.variable.length).toBeGreaterThan(0); + expect(hop.line).toBeGreaterThan(0); + } + if (decoded.hops.some((h) => h.variable === 'cmd')) sawVulnFlow = true; + } + } else { + // SANITIZES carries the killed binding's plain name: `value` from + // vuln.ts sendEncoded, `text` from taint-cases.ts conditionalSanitizer. + expect(['value', 'text']).toContain(rel.reason); + } + } + expect(sawVulnFlow).toBe(true); // the req.body → exec flow, via `cmd` + }, 60000); + it('with --pdg off (default): emits zero BasicBlock nodes and zero CFG edges', async () => { const result = await runPipelineFromRepo(freshRepo(), () => {}); - const { basicBlocks, cfgEdges, reachingDefs } = counts(result); + const { basicBlocks, cfgEdges, reachingDefs, tainted, sanitizes } = counts(result); expect(basicBlocks).toBe(0); expect(cfgEdges).toBe(0); expect(reachingDefs).toBe(0); + expect(tainted).toBe(0); + expect(sanitizes).toBe(0); }, 60000); }); diff --git a/gitnexus/test/integration/cfg/taint-snapshot.test.ts b/gitnexus/test/integration/cfg/taint-snapshot.test.ts new file mode 100644 index 000000000..adeb72586 --- /dev/null +++ b/gitnexus/test/integration/cfg/taint-snapshot.test.ts @@ -0,0 +1,144 @@ +import { describe, it, expect } from 'vitest'; +import path from 'path'; +import { + computeFixtureTaint, + TAINT_FIXTURE_FILES, + type FixtureFunctionTaint, +} from '../../helpers/taint-fixture.js'; + +// #2083 M3 U7 acceptance: a committed snapshot of the taint findings/kills on +// the pdg-repo fixture battery (vuln.ts + taint-cases.ts, with sample.ts as +// the zero-match control), mirroring reaching-defs-snapshot. The pure path — +// collect → match → computeReachingDefs → computeTaintFlows — is the SAME +// per-function pipeline the in-phase emit driver runs, so any model/matcher/ +// propagation behavior change shows as a reviewable snapshot diff, never +// silent drift. The fixture battery covers the plan's six shapes: direct +// source→sink (rule-b AND the reassignment form), multi-hop chain, sanitized +// variant (must-def kill suppresses), conditional-sanitizer variant (finding +// survives), loop-carried taint, and through-call (viaCall hop). + +const FIXTURE = path.join(__dirname, 'fixtures', 'pdg-repo'); + +/** + * Deterministic rendering. Findings: `var@line[*]->…->var@line[*]:kind` + * (source-first hop order; `*` marks a viaCall hop — taint passed through an + * unmodeled call). Kills: `binding@defLine<-sanLine:kind[,kind]`. + */ +function serialize(fn: FixtureFunctionTaint): Record { + const bindings = fn.cfg.bindings ?? []; + const bName = (idx: number): string => bindings[idx]?.name ?? `#${idx}`; + return { + file: fn.file, + startLine: fn.startLine, + status: fn.status, + findings: (fn.flows?.findings ?? []).map( + (f) => + f.hops.map((h) => `${h.name}@${h.point.line}${h.viaCall === true ? '*' : ''}`).join('->') + + `:${f.sinkKind}` + + (f.hopsTruncated === true ? ' (truncated)' : ''), + ), + kills: (fn.flows?.kills ?? []).map( + (k) => + `${bName(k.bindingIdx)}@${k.killedDef.line}<-${k.sanitizer.line}:${k.neutralized.join(',')}`, + ), + dropped: fn.flows?.droppedFindings ?? 0, + }; +} + +describe('U7 — taint findings/kills snapshot on the pdg-repo fixture battery', () => { + const results = computeFixtureTaint(FIXTURE); + const blockText = (fn: FixtureFunctionTaint, needle: string): boolean => + fn.cfg.blocks.some((b) => b.text.includes(needle)); + + it('matches the committed findings/kills for every fixture function', () => { + // Every fixture file contributes at least one function; the battery shape + // is pinned so a fixture edit that drops a case fails loudly here, not + // silently in the snapshot. + for (const file of TAINT_FIXTURE_FILES) { + expect(results.some((r) => r.file === file)).toBe(true); + } + expect(results.map(serialize)).toMatchSnapshot(); + }); + + it('every matched fixture function computes (no coverage gaps, no unsafe sites)', () => { + for (const fn of results) { + expect(['computed', 'no-match']).toContain(fn.status); + if (fn.flows) expect(fn.flows.droppedFindings).toBe(0); + } + // sample.ts is the zero-match control: no sources/sinks → fast path. + for (const fn of results.filter((r) => r.file === 'sample.ts')) { + expect(fn.status).toBe('no-match'); + } + }); + + it('AE1 — the source→sink flow IS found; the sanitized variant yields no finding and ≥1 kill', () => { + // vuln.ts runUserCommand: req.body → cmd → exec(cmd). + const vulnerable = results.find((r) => r.file === 'vuln.ts' && blockText(r, 'exec(cmd)'))!; + expect(vulnerable).toBeDefined(); + expect(vulnerable.status).toBe('computed'); + expect(vulnerable.flows!.findings).toHaveLength(1); + expect(vulnerable.flows!.findings[0].sinkKind).toBe('command-injection'); + + // vuln.ts sendEncoded: the must-def encodeURIComponent kill suppresses + // the xss finding entirely; the kill IS the persisted safety evidence. + const sanitized = results.find( + (r) => r.file === 'vuln.ts' && blockText(r, 'encodeURIComponent'), + )!; + expect(sanitized).toBeDefined(); + expect(sanitized.status).toBe('computed'); + expect(sanitized.flows!.findings).toHaveLength(0); + expect(sanitized.flows!.kills.length).toBeGreaterThanOrEqual(1); + expect(sanitized.flows!.kills[0].neutralized).toContain('xss'); + }); + + it('AE1 — the conditional-sanitizer variant survives (may-def leg) with the kill recorded', () => { + const conditional = results.find( + (r) => r.file === 'taint-cases.ts' && blockText(r, 'res.send(text)'), + )!; + expect(conditional).toBeDefined(); + expect(conditional.flows!.findings).toHaveLength(1); + expect(conditional.flows!.findings[0].sinkKind).toBe('xss'); + expect(conditional.flows!.kills.length).toBeGreaterThanOrEqual(1); + }); + + it('AE3 shape — hops are ordered source-first with a variable on every hop', () => { + // Every finding in the battery carries non-empty variables on all hops. + for (const fn of results) { + for (const f of fn.flows?.findings ?? []) { + expect(f.hops.length).toBeGreaterThan(0); + for (const h of f.hops) { + expect(h.name.length).toBeGreaterThan(0); + expect(h.point.line).toBeGreaterThan(0); + } + } + } + // The multi-hop chain (a → b → c → exec(c)): 3+ hops, source-first order. + const chain = results.find((r) => r.file === 'taint-cases.ts' && blockText(r, 'const c = b'))!; + expect(chain).toBeDefined(); + const hops = chain.flows!.findings[0].hops; + expect(hops.length).toBeGreaterThanOrEqual(4); + expect(hops.map((h) => h.name)).toEqual(['a', 'b', 'c', 'c']); + for (let i = 1; i < hops.length; i++) { + expect(hops[i].point.line).toBeGreaterThanOrEqual(hops[i - 1].point.line); + } + }); + + it('loop-carried taint reaches a fixpoint and the sink (terminates, one finding)', () => { + const loop = results.find( + (r) => r.file === 'taint-cases.ts' && blockText(r, 'cmd = cmd + part'), + )!; + expect(loop).toBeDefined(); + expect(loop.status).toBe('computed'); + expect(loop.flows!.findings).toHaveLength(1); + expect(loop.flows!.findings[0].sinkKind).toBe('command-injection'); + }); + + it('through-call taint propagates with the viaCall hop mark (KTD5)', () => { + const through = results.find( + (r) => r.file === 'taint-cases.ts' && blockText(r, 'decorate(raw)'), + )!; + expect(through).toBeDefined(); + expect(through.flows!.findings).toHaveLength(1); + expect(through.flows!.findings[0].hops.some((h) => h.viaCall === true)).toBe(true); + }); +}); diff --git a/gitnexus/test/integration/cfg/worker-roundtrip.test.ts b/gitnexus/test/integration/cfg/worker-roundtrip.test.ts index c46a425ae..e61112ee0 100644 --- a/gitnexus/test/integration/cfg/worker-roundtrip.test.ts +++ b/gitnexus/test/integration/cfg/worker-roundtrip.test.ts @@ -1,3 +1,4 @@ +import { createHash } from 'crypto'; import { describe, it, expect } from 'vitest'; import Parser from 'tree-sitter'; import TypeScript from 'tree-sitter-typescript'; @@ -183,3 +184,89 @@ describe('#2082 M2 — the REACHING_DEF emit cap does NOT perturb the chunk key' expect(withExtra).toBe(base); }); }); + +describe('#2083 M3 U1 — taint sites cross the worker/store boundary intact', () => { + const siteSource = `function handler(req, x) { + const cp = require('child_process'); + const b = req.body; + cp.exec(escape(x), b); + sql\`select \${x}\`; + run(...b); + }`; + + function siteCfgs() { + const { cfgs } = collectFunctionCfgs(tsRoot(siteSource), tsVisitor(), 'sites.ts'); + expect(cfgs).toHaveLength(1); + return cfgs; + } + + function allSites(cfgs: readonly { blocks: readonly { statements?: readonly unknown[] }[] }[]) { + return cfgs.flatMap((c) => + c.blocks.flatMap((b) => + (b.statements ?? []).flatMap((s) => (s as { sites?: unknown[] }).sites ?? []), + ), + ); + } + + it('sites survive the worker JSON boundary (mapReplacer/mapReviver) byte-equal', () => { + const cfgs = siteCfgs(); + expect(allSites(cfgs).length).toBeGreaterThan(0); + const round = JSON.parse(JSON.stringify(cfgs, mapReplacer), mapReviver); + expect(round).toEqual(cfgs); + expect(allSites(round)).toEqual(allSites(cfgs)); + }); + + it('sites survive a frozen re-wrap + the DURABLE store interning reviver (no nodeId-dedup loss)', async () => { + const { makeInterningReviver } = await import('../../../src/storage/parsedfile-store.js'); + const cfgs = siteCfgs(); + // The pipeline deep-freezes ParsedFiles and re-wraps via spread — the CFG + // payload itself rides by reference and must tolerate being frozen. + const deepFreeze = (o: unknown): unknown => { + if (o && typeof o === 'object') { + for (const v of Object.values(o)) deepFreeze(v); + Object.freeze(o); + } + return o; + }; + const frozen = (deepFreeze(cfgs) as typeof cfgs).map((c) => ({ ...c })); + const raw = JSON.stringify(frozen, mapReplacer); + // The durable parsedfile-cache revives with the interning reviver, which + // DEDUPS any object carrying a string `nodeId` field — SiteRecord must + // never trip it (the KTD2 "no field named nodeId" obligation). + const revived = JSON.parse(raw, makeInterningReviver(new Map(), new Map())); + expect(revived).toEqual(frozen); + expect(allSites(revived)).toEqual(allSites(cfgs)); + }); +}); + +describe('#2083 M3 U1 — pdg chunk-key namespace version (flag-off keys untouched)', () => { + const entries = [ + { filePath: 'b.ts', contentHash: 'h2' }, + { filePath: 'a.ts', contentHash: 'h1' }, + ]; + + it('flag-off chunk keys are BYTE-IDENTICAL across the M3 namespace bump (pinned hash)', () => { + // Independent reconstruction of the pre-namespace key format: pdg-off + // keys are sha256 over the sorted filePath:contentHash lines and NOTHING + // else. This pin fails if the version token ever leaks into non-pdg keys + // (which would force a cold re-parse on every flag-off user). + const expected = createHash('sha256').update(Buffer.from('a.ts:h1\nb.ts:h2')).digest('hex'); + expect(computeChunkHash(entries, false)).toBe(expected); + expect(computeChunkHash(entries)).toBe(expected); + }); + + it('pdg-mode keys CHANGED from the M2-era namespace (v1 chunks invalidate on upgrade)', () => { + // The M2-era pdg namespace was `pdg:1;maxFn=` — an M3 binary must not + // serve a v1 chunk (its cfgSideChannel lacks `sites`, so taint would + // silently no-op on warm caches). + const joined = 'a.ts:h1\nb.ts:h2'; + const m2Key = createHash('sha256') + .update(Buffer.from(`pdg:1;maxFn=def\n${joined}`)) + .digest('hex'); + expect(computeChunkHash(entries, { pdg: true })).not.toBe(m2Key); + // and the v2 key is still deterministic + order-independent + expect(computeChunkHash([...entries].reverse(), { pdg: true })).toBe( + computeChunkHash(entries, { pdg: true }), + ); + }); +}); diff --git a/gitnexus/test/integration/taint-explain.test.ts b/gitnexus/test/integration/taint-explain.test.ts new file mode 100644 index 000000000..6cda28eac --- /dev/null +++ b/gitnexus/test/integration/taint-explain.test.ts @@ -0,0 +1,344 @@ +/** + * Integration Tests: MCP `explain` tool (#2083 M3 U6) + * + * End-to-end against a REAL LadybugDB: the pdg-repo fixture is indexed by the + * real pipeline with `--pdg` (workers — requires `node scripts/build.js`), the + * resulting BasicBlock nodes + TAINTED/SANITIZES edges and the fixture's + * Function symbols are persisted into the test DB, and `explain` is exercised + * through the full `callTool` dispatch: + * + * - anchorless enumerate (≥1 finding, decoded hops, deterministic order) + * - anchored by file and by symbol (line-span granularity) + * - sanitized-only function → zero TAINTED findings (its safety evidence is + * the SANITIZES edge, not part of explain's response) + * - unknown symbol → context()-style not-found + * - a repo WITHOUT the taint layer → the "no taint layer" note, not an error + * + * Seeding via the real emit output (not hand-written rows) pins the format + * compatibility between U4's write path and U6's read path — id template, + * `;` reason header, hop encoding. + */ +import { describe, it, expect, beforeAll, vi } from 'vitest'; +import fs from 'fs'; +import os from 'os'; +import path from 'path'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos, loadMeta } from '../../src/storage/repo-manager.js'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; +import { runPipelineFromRepo } from '../../src/core/ingestion/pipeline.js'; + +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), + // No meta.json exists for the seeded test DB — explain's meta probe must + // degrade to the TAINTED-row existence probe (the seeded-DB reality). + loadMeta: vi.fn().mockResolvedValue(null), + }; +}); + +const FIXTURE = path.join(__dirname, 'cfg', 'fixtures', 'pdg-repo'); + +// ─── Block 1: a --pdg index with real taint findings ───────────────── + +withTestLbugDB( + 'taint-explain', + (handle) => { + describe('explain tool against a --pdg index', () => { + let backend: LocalBackend; + + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) { + throw new Error( + 'LocalBackend not initialized — afterSetup did not attach _backend to handle', + ); + } + backend = ext._backend; + }); + + it('anchorless explain enumerates the persisted findings with decoded hops', async () => { + const result = await backend.callTool('explain', {}); + expect(result).not.toHaveProperty('error'); + expect(result.totalFindings).toBeGreaterThanOrEqual(1); + expect(result.findings.length).toBeGreaterThanOrEqual(1); + expect(result.truncated).toBeUndefined(); + // The vulnerable flow: req.body → cmd → exec(cmd) in vuln.ts. + const vuln = result.findings.find((f: any) => f.file.endsWith('vuln.ts')); + expect(vuln).toBeDefined(); + expect(vuln.sinkKind).toBe('command-injection'); + expect(vuln.functionLine).toBe(9); // runUserCommand's start line + // Ordered hops with the variable carried on each hop (AC3): seed def + // (cmd @ const line 10) → sink use (cmd @ exec line 11). + expect(vuln.hops.map((h: any) => `${h.variable}@${h.line}`)).toEqual(['cmd@10', 'cmd@11']); + expect(vuln.source).toEqual({ variable: 'cmd', line: 10 }); + expect(vuln.sink).toEqual({ line: 11 }); + expect(vuln.pathIncomplete).toBeUndefined(); + // The intra-procedural contract caveat reaches the consumer. + expect(result.note).toMatch(/intra-procedural/i); + }); + + it('anchorless enumerate is deterministic across calls', async () => { + const a = await backend.callTool('explain', {}); + const b = await backend.callTool('explain', {}); + expect(a).toEqual(b); + }); + + it('anchored by file path returns the finding (suffix match accepted)', async () => { + for (const target of ['vuln.ts']) { + const result = await backend.callTool('explain', { target }); + expect(result).not.toHaveProperty('error'); + expect(result.anchor).toEqual({ file: target }); + expect(result.findings.length).toBeGreaterThanOrEqual(1); + for (const f of result.findings) expect(f.file.endsWith('vuln.ts')).toBe(true); + } + }); + + it('anchored by an unrelated file returns zero findings (repo HAS the layer — no note about it)', async () => { + const result = await backend.callTool('explain', { target: 'sample.ts' }); + expect(result).not.toHaveProperty('error'); + expect(result.findings).toEqual([]); + expect(result.totalFindings).toBe(0); + // The repo has TAINTED rows, so the "no taint layer" hint must NOT fire. + expect(result.note ?? '').not.toMatch(/no taint layer/i); + }); + + it('anchored by the vulnerable function name returns full hop detail', async () => { + const result = await backend.callTool('explain', { target: 'runUserCommand' }); + expect(result).not.toHaveProperty('error'); + expect(result.anchor.symbol).toBe('runUserCommand'); + expect(result.findings).toHaveLength(1); + expect(result.totalFindings).toBe(1); + const f = result.findings[0]; + expect(f.sinkKind).toBe('command-injection'); + expect(f.hops.map((h: any) => h.variable)).toEqual(['cmd', 'cmd']); + }); + + it('the sanitized-only function returns no TAINTED finding', async () => { + const result = await backend.callTool('explain', { target: 'sendEncoded' }); + expect(result).not.toHaveProperty('error'); + expect(result.anchor.symbol).toBe('sendEncoded'); + expect(result.findings).toEqual([]); + expect(result.totalFindings).toBe(0); + }); + + it('an unknown symbol target mirrors context() not-found semantics', async () => { + const result = await backend.callTool('explain', { target: 'nonexistentTaintFn999' }); + expect(result).toHaveProperty('error'); + expect(result.error).toMatch(/not found/i); + }); + + it('a dotted symbol name resolves as a symbol, not a silent file miss', async () => { + // Regression: `Class.method` was classified as a file (the `.method` + // extension-like suffix) and returned a silent empty file-anchored + // result. It must now route to symbol resolution — here, not-found. + const result = await backend.callTool('explain', { target: 'UserController.create' }); + expect(result).toHaveProperty('error'); + expect(result.error).toMatch(/not found/i); + // Must NOT be a silent file-anchored empty result. + expect(result.anchor).toBeUndefined(); + }); + + it('a dotted symbol whose tail looks bare still resolves as a symbol', async () => { + // `runUserCommand` is a real fixture symbol; a dotted lead-in that does + // not match any symbol confirms the symbol branch (not file routing). + const result = await backend.callTool('explain', { target: 'Service.runUserCommand' }); + expect(result).toHaveProperty('error'); + expect(result.error).toMatch(/not found/i); + }); + + it('rejects an out-of-bounds limit with a clear error', async () => { + // Includes the non-integer / non-finite / non-numeric cases the + // interpolated `LIMIT ${limit}` depends on the guard rejecting. + for (const limit of [0, -1, 1.5, 10_000, NaN, Infinity, -Infinity, '50']) { + const result = await backend.callTool('explain', { limit }); + expect(result).toHaveProperty('error'); + expect(result.error).toMatch(/limit/i); + } + }); + + it('limit pages the enumerate and reports truncation honestly', async () => { + const all = await backend.callTool('explain', {}); + const page = await backend.callTool('explain', { limit: 1 }); + expect(page.findings).toHaveLength(Math.min(1, all.totalFindings)); + expect(page.totalFindings).toBe(all.totalFindings); + if (all.totalFindings > 1) { + expect(page.truncated).toBe(true); + // Deterministic order: the page is a prefix of the full enumerate. + expect(page.findings[0]).toEqual(all.findings[0]); + } + }); + }); + }, + { + poolAdapter: true, + afterSetup: async (handle) => { + // 1. Index the pdg-repo fixture with the REAL pipeline (--pdg on). + const repoDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-explain-')); + try { + fs.cpSync(FIXTURE, repoDir, { recursive: true }); + const pipelineResult = await runPipelineFromRepo(repoDir, () => {}, { pdg: true }); + + // 2. Persist the emit output into the test DB: BasicBlock nodes, + // TAINTED/SANITIZES edges, and the Function symbols (for the + // symbol-anchored path through resolveSymbolCandidates). + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const nodes: Array<{ label: string; props: Record }> = []; + pipelineResult.graph.forEachNode((n) => { + if (n.label === 'BasicBlock') { + nodes.push({ + label: 'BasicBlock', + props: { + id: n.id, + filePath: n.properties.filePath ?? '', + startLine: n.properties.startLine ?? 0, + endLine: n.properties.endLine ?? 0, + text: n.properties.text ?? '', + }, + }); + } else if (n.label === 'Function') { + nodes.push({ + label: 'Function', + props: { + id: n.id, + name: n.properties.name ?? '', + filePath: n.properties.filePath ?? '', + startLine: n.properties.startLine ?? 0, + endLine: n.properties.endLine ?? 0, + }, + }); + } + }); + for (const node of nodes) { + const assignments = Object.keys(node.props) + .map((k) => `${k}: $${k}`) + .join(', '); + await adapter.executePrepared( + `CREATE (n:${node.label} {${assignments}})`, + node.props as Record, + ); + } + let taintEdges = 0; + for (const rel of pipelineResult.graph.iterRelationships()) { + if (rel.type !== 'TAINTED' && rel.type !== 'SANITIZES') continue; + await adapter.executePrepared( + `MATCH (a:BasicBlock {id: $src}), (b:BasicBlock {id: $dst}) + CREATE (a)-[:CodeRelation {type: '${rel.type}', confidence: $confidence, reason: $reason, step: 0}]->(b)`, + { + src: rel.sourceId, + dst: rel.targetId, + confidence: rel.confidence ?? 1.0, + reason: rel.reason ?? '', + }, + ); + taintEdges++; + } + if (taintEdges === 0) { + throw new Error('fixture produced no TAINTED/SANITIZES edges — taint emit regressed?'); + } + } finally { + fs.rmSync(repoDir, { recursive: true, force: true }); + } + + // 3. Register the test DB and boot the backend (calltool harness shape). + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'taint-repo', + path: '/taint/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'abc123', + stats: { files: 2, nodes: 4, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); + +// ─── Block 2: a repo indexed WITHOUT --pdg ─────────────────────────── + +withTestLbugDB( + 'taint-explain-nopdg', + (handle) => { + describe('explain tool without a taint layer', () => { + let backend: LocalBackend; + + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) { + throw new Error( + 'LocalBackend not initialized — afterSetup did not attach _backend to handle', + ); + } + backend = ext._backend; + }); + + it('returns the no-taint-layer note via the row-existence probe (meta unreadable)', async () => { + const result = await backend.callTool('explain', {}); + expect(result).not.toHaveProperty('error'); + expect(result.findings).toEqual([]); + expect(result.totalFindings).toBe(0); + expect(result.note).toMatch(/no taint layer/i); + expect(result.note).toContain('--pdg'); + }); + + it('an anchored call also reports the missing layer, not a bogus empty result', async () => { + const result = await backend.callTool('explain', { target: 'plain.ts' }); + expect(result).not.toHaveProperty('error'); + expect(result.findings).toEqual([]); + expect(result.note).toMatch(/no taint layer/i); + }); + + it('returns the note via the RepoMeta.pdg probe when meta is readable but unstamped', async () => { + // A readable meta WITHOUT a pdg stamp short-circuits before any + // block-space query (the #2099 F1 presence ≡ layer-exists contract). + vi.mocked(loadMeta).mockResolvedValueOnce({} as any); + const result = await backend.callTool('explain', {}); + expect(result.findings).toEqual([]); + expect(result.totalFindings).toBe(0); + expect(result.note).toMatch(/no taint layer/i); + }); + + it('an M1/M2-era pdg stamp (no taintModelVersion) reports the missing taint layer', async () => { + // The pdg stamp exists (BasicBlock/REACHING_DEF were recorded) but + // taint never ran — no taintModelVersion. The taint-layer probe must + // gate on taintModelVersion, not generic pdg presence, so this surfaces + // the actionable "run analyze" hint instead of a bare empty result. + vi.mocked(loadMeta).mockResolvedValueOnce({ + pdg: { mode: 'on', maxFunctionLines: 2000 }, + } as any); + const result = await backend.callTool('explain', {}); + expect(result.findings).toEqual([]); + expect(result.totalFindings).toBe(0); + expect(result.note).toMatch(/no taint layer/i); + }); + }); + }, + { + seed: [ + `CREATE (fn:Function {id: 'func:plainFn', name: 'plainFn', filePath: 'src/plain.ts', startLine: 1, endLine: 5, isExported: true, content: 'function plainFn() {}', description: 'no taint layer here'})`, + ], + poolAdapter: true, + afterSetup: async (handle) => { + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'plain-repo', + path: '/plain/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'def456', + stats: { files: 1, nodes: 1, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); diff --git a/gitnexus/test/unit/cfg/harvest.test.ts b/gitnexus/test/unit/cfg/harvest.test.ts index 6fccf95bb..7c8e3a823 100644 --- a/gitnexus/test/unit/cfg/harvest.test.ts +++ b/gitnexus/test/unit/cfg/harvest.test.ts @@ -1,12 +1,6 @@ import { describe, it, expect } from 'vitest'; -import Parser from 'tree-sitter'; -import TypeScript from 'tree-sitter-typescript'; -import type { SyntaxNode } from '../../../src/core/ingestion/utils/ast-helpers.js'; -import { - createTypeScriptCfgVisitor, - TS_FUNCTION_TYPES, -} from '../../../src/core/ingestion/cfg/visitors/typescript.js'; import type { FunctionCfg, StatementFacts } from '../../../src/core/ingestion/cfg/types.js'; +import { cfgOf } from '../../helpers/ts-cfg-harness.js'; // U1 (#2082 M2) — per-statement def/use harvesting. The two-phase design // (declaration pre-scan → resolve during the CFG walk) is what makes the @@ -14,37 +8,6 @@ import type { FunctionCfg, StatementFacts } from '../../../src/core/ingestion/cf // and do-while-condition-first, so declare-as-you-walk would mis-key common // code. Each test pins names→binding-index agreement, not just presence. -const visitor = createTypeScriptCfgVisitor(); - -function parse(code: string): SyntaxNode { - const parser = new Parser(); - parser.setLanguage(TypeScript.typescript); - return parser.parse(code).rootNode; -} - -function collectFunctions(root: SyntaxNode): SyntaxNode[] { - const out: SyntaxNode[] = []; - const stack = [root]; - while (stack.length) { - const n = stack.pop() as SyntaxNode; - if (TS_FUNCTION_TYPES.has(n.type)) out.push(n); - for (let i = n.namedChildCount - 1; i >= 0; i--) { - const c = n.namedChild(i); - if (c) stack.push(c); - } - } - return out; -} - -function cfgOf(code: string, index = 0): FunctionCfg { - const fns = collectFunctions(parse(code)); - const fn = fns[index]; - if (!fn) throw new Error(`no function at index ${index}`); - const cfg = visitor.buildFunctionCfg(fn, 'fixture.ts'); - if (!cfg) throw new Error('buildFunctionCfg returned undefined'); - return cfg; -} - /** All statement facts of the CFG, flattened in (block, statement) order. */ function allFacts(cfg: FunctionCfg): StatementFacts[] { return cfg.blocks.flatMap((b) => [...(b.statements ?? [])]); @@ -491,3 +454,259 @@ describe('TS/JS def/use harvest — conditional contexts are MAY-defs (tri-revie expect(withDef.length).toBeGreaterThanOrEqual(2); }); }); + +// ── #2083 M3 U1 — taint-site harvest ──────────────────────────────────────── + +import type { SiteRecord } from '../../../src/core/ingestion/cfg/types.js'; + +/** All site records of the CFG, flattened in (block, statement) order. */ +function allSites(cfg: FunctionCfg): SiteRecord[] { + return allFacts(cfg).flatMap((f) => [...(f.sites ?? [])]); +} + +/** The single statement fact carrying sites (throws when ambiguous). */ +function siteFact(cfg: FunctionCfg, line?: number): StatementFacts { + const withSites = allFacts(cfg).filter( + (f) => (f.sites?.length ?? 0) > 0 && (line === undefined || f.line === line), + ); + if (withSites.length !== 1) + throw new Error(`expected 1 site-bearing fact, got ${withSites.length}`); + return withSites[0]; +} + +describe('M3 U1 — taint-site harvest: call sites', () => { + it('exec(a, b) → one call site mapping position 0→[a], 1→[b]', () => { + const cfg = cfgOf(`function f(a, b) { exec(a, b); }`); + const sites = siteFact(cfg, 1).sites!; + expect(sites).toHaveLength(1); + const s = sites[0]; + expect(s.kind).toBe('call'); + expect(s.callee).toBe('exec'); + expect(s.receiver).toBeUndefined(); + expect(s.args).toEqual([[bindingIdx(cfg, 'a')], [bindingIdx(cfg, 'b')]]); + expect(s.parent).toBeUndefined(); + }); + + it('child_process.exec(cmd) → dotted callee path + receiver slot', () => { + const cfg = cfgOf(`function f(cmd) { child_process.exec(cmd); }`); + const s = siteFact(cfg, 1).sites![0]; + expect(s.callee).toBe('child_process.exec'); + expect(s.receiver).toBe(bindingIdx(cfg, 'child_process')); + expect(s.args).toEqual([[bindingIdx(cfg, 'cmd')]]); + // chain-length-1 callee: the access IS the callee — no member-read site + expect(siteFact(cfg, 1).sites).toHaveLength(1); + // and the receiver use is recorded exactly once (no double-record) + expect( + siteFact(cfg, 1).uses.filter((u) => u === bindingIdx(cfg, 'child_process')), + ).toHaveLength(1); + }); + + it('const r = f(x) → resultDefs carries r', () => { + const cfg = cfgOf(`function g(x) { const r = f(x); }`); + const s = siteFact(cfg, 1).sites![0]; + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'r')]); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + }); + + it('exec(escape(x)) → inner site is first-class with parent link + occurrence tagging', () => { + const cfg = cfgOf(`function f(x) { exec(escape(x)); }`); + const sites = siteFact(cfg, 1).sites!; + expect(sites).toHaveLength(2); + const execIdx = sites.findIndex((s) => s.callee === 'exec'); + const escapeIdx = sites.findIndex((s) => s.callee === 'escape'); + expect(execIdx).toBeGreaterThanOrEqual(0); + expect(escapeIdx).toBeGreaterThanOrEqual(0); + const x = bindingIdx(cfg, 'x'); + // inner escape: plain occurrence, parent link to (exec, arg 0) + expect(sites[escapeIdx].args).toEqual([[x]]); + expect(sites[escapeIdx].parent).toEqual([execIdx, 0]); + // outer exec: x's occurrence is via-tagged through the escape site + expect(sites[execIdx].args).toEqual([[[x, escapeIdx]]]); + expect(sites[execIdx].parent).toBeUndefined(); + }); + + it('a bypass occurrence stays a PLAIN entry next to the via-tagged one (exec(x + escape(x)))', () => { + const cfg = cfgOf(`function f(x) { exec(x + escape(x)); }`); + const sites = siteFact(cfg, 1).sites!; + const exec = sites.find((s) => s.callee === 'exec')!; + const escapeIdx = sites.findIndex((s) => s.callee === 'escape'); + const x = bindingIdx(cfg, 'x'); + expect(exec.args![0]).toEqual([x, [x, escapeIdx]]); + }); + + it('new Function(x) → kind "new" site (new_expression case)', () => { + const cfg = cfgOf(`function f(x) { new Function(x); }`); + const s = siteFact(cfg, 1).sites![0]; + expect(s.kind).toBe('new'); + expect(s.callee).toBe('Function'); + expect(s.args).toEqual([[bindingIdx(cfg, 'x')]]); + }); + + it('exec(...args) → spread index recorded, args binding occurs at the position', () => { + const cfg = cfgOf(`function f(...args) { exec(...args); }`); + const s = siteFact(cfg, 1).sites![0]; + expect(s.spread).toBe(0); + expect(s.args).toEqual([[bindingIdx(cfg, 'args')]]); + }); + + it('const cp = require("child_process") → requireArg literal + cp in resultDefs', () => { + const cfg = cfgOf(`function f() { const cp = require('child_process'); }`); + const s = siteFact(cfg, 1).sites![0]; + expect(s.callee).toBe('require'); + expect(s.requireArg).toBe('child_process'); + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'cp')]); + }); + + it('per-declarator attribution: const a = t, b = escape(t) → resultDefs [b] only', () => { + const cfg = cfgOf(`function f(t) { const a = t, b = escape(t); }`); + const sites = siteFact(cfg, 1).sites!; + expect(sites).toHaveLength(1); + expect(sites[0].callee).toBe('escape'); + expect(sites[0].resultDefs).toEqual([bindingIdx(cfg, 'b')]); + }); + + it('non-top-level call gets NO resultDefs (const c = cond ? escape(b) : b keeps c taintable)', () => { + const cfg = cfgOf(`function f(cond, b) { const c = cond ? escape(b) : b; }`); + const sites = siteFact(cfg, 1).sites!; + expect(sites).toHaveLength(1); + expect(sites[0].callee).toBe('escape'); + expect(sites[0].resultDefs).toBeUndefined(); + }); + + it('value wrappers unwrap for resultDefs: const b = (await escape(t))! still attaches [b]', () => { + const cfg = cfgOf(`async function f(t) { const b = (await escape(t))!; }`); + const s = siteFact(cfg, 1).sites![0]; + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'b')]); + }); + + it('plain assignment x = f(y) attaches resultDefs [x]', () => { + const cfg = cfgOf(`function g(y) { let x; x = f(y); }`); + const s = siteFact(cfg, 1).sites![0]; + expect(s.resultDefs).toEqual([bindingIdx(cfg, 'x')]); + }); +}); + +describe('M3 U1 — taint-site harvest: member reads', () => { + it('const b = req.body → member read {object: req, property: body} AND b in defs', () => { + const cfg = cfgOf(`function f(req) { const b = req.body; }`); + const fact = siteFact(cfg, 1); + expect(fact.defs).toContain(bindingIdx(cfg, 'b')); + expect(fact.sites).toEqual([ + { kind: 'member-read', object: bindingIdx(cfg, 'req'), property: 'body' }, + ]); + }); + + it('req?.body records identically to req.body (optional-chain normalization)', () => { + const plain = cfgOf(`function f(req) { const b = req.body; }`); + const optional = cfgOf(`function f(req) { const b = req?.body; }`); + expect(siteFact(optional, 1).sites).toEqual(siteFact(plain, 1).sites); + }); + + it('req["body"] records as a member read; dynamic req[key] records NOTHING', () => { + const literal = cfgOf(`function f(req) { const c = req["body"]; }`); + expect(siteFact(literal, 1).sites).toEqual([ + { kind: 'member-read', object: bindingIdx(literal, 'req'), property: 'body' }, + ]); + const dynamic = cfgOf(`function f(req, key) { const d = req[key]; }`); + expect(allSites(dynamic)).toHaveLength(0); + // the dynamic index is still a value use + expect(usesOf(dynamic)).toContain(bindingIdx(dynamic, 'key')); + }); + + it('exec(req.body.toString()) → the mid-callee-chain member read IS recorded', () => { + const cfg = cfgOf(`function f(req) { exec(req.body.toString()); }`); + const sites = siteFact(cfg, 1).sites!; + const read = sites.find((s) => s.kind === 'member-read'); + expect(read).toBeDefined(); + expect(read!.object).toBe(bindingIdx(cfg, 'req')); + expect(read!.property).toBe('body'); + // the toString call site carries the full dotted path + receiver + const ts = sites.find((s) => s.callee === 'req.body.toString'); + expect(ts).toBeDefined(); + expect(ts!.receiver).toBe(bindingIdx(cfg, 'req')); + // and req's occurrence reaches exec's arg 0 via the toString site + const exec = sites.find((s) => s.callee === 'exec')!; + const tsIdx = sites.indexOf(ts!); + expect(exec.args).toEqual([[[bindingIdx(cfg, 'req'), tsIdx]]]); + }); + + it('write-position member targets record NO member read (obj.p = q)', () => { + const cfg = cfgOf(`function f(obj, q) { obj.p = q; }`); + expect(allSites(cfg)).toHaveLength(0); + }); + + it('a mid-chain LOAD inside a write target IS recorded (req.body.x = v)', () => { + const cfg = cfgOf(`function f(req, v) { req.body.x = v; }`); + expect(siteFact(cfg, 1).sites).toEqual([ + { kind: 'member-read', object: bindingIdx(cfg, 'req'), property: 'body' }, + ]); + }); +}); + +describe('M3 U1 — taint-site harvest: templates, callbacks, statement granularity', () => { + it('template-literal argument: exec(`ls ${dir}`) → dir occurs at position 0, no template flag', () => { + const cfg = cfgOf('function f(dir) { exec(`ls ${dir}`); }'); + const s = siteFact(cfg, 1).sites![0]; + expect(s.template).toBeUndefined(); + expect(s.args).toEqual([[bindingIdx(cfg, 'dir')]]); + }); + + it('tagged template: sql`…${id}` → call site with template marker, id recorded', () => { + const cfg = cfgOf('function f(id) { sql`select ${id}`; }'); + const s = siteFact(cfg, 1).sites![0]; + expect(s.kind).toBe('call'); + expect(s.callee).toBe('sql'); + expect(s.template).toBe(true); + expect(s.args).toEqual([[bindingIdx(cfg, 'id')]]); + }); + + it('nested callback: arr.forEach(() => exec(y)) → inner call invisible, outer site has receiver arr', () => { + const cfg = cfgOf(`function f(arr, y) { arr.forEach(() => exec(y)); }`); + const sites = siteFact(cfg, 1).sites!; + expect(sites).toHaveLength(1); + expect(sites[0].callee).toBe('arr.forEach'); + expect(sites[0].receiver).toBe(bindingIdx(cfg, 'arr')); + // y is invisible (nested-function opacity) — neither a use nor an occurrence + expect(usesOf(cfg)).not.toContain(bindingIdx(cfg, 'y')); + expect(sites[0].args).toBeUndefined(); + }); + + it('two statements on one line → distinct site records on distinct StatementFacts', () => { + const cfg = cfgOf(`function f(a, b) { exec(a); run(b); }`); + const withSites = allFacts(cfg).filter((f) => (f.sites?.length ?? 0) > 0); + expect(withSites).toHaveLength(2); + expect(withSites[0].sites![0].callee).toBe('exec'); + expect(withSites[1].sites![0].callee).toBe('run'); + // site indices are PER-STATEMENT — both are index 0 of their own record + expect(withSites[0].sites).toHaveLength(1); + expect(withSites[1].sites).toHaveLength(1); + }); + + it('sites are omitted entirely on statements without calls or member reads', () => { + const cfg = cfgOf(`function f() { let x = 1; x = 2; }`); + for (const fact of allFacts(cfg)) expect(fact.sites).toBeUndefined(); + }); + + it('sites survive a JSON round-trip (worker boundary shape)', () => { + const cfg = cfgOf(`function f(req, x) { const b = req.body; exec(escape(x), b); }`); + const trip = JSON.parse(JSON.stringify(cfg)) as FunctionCfg; + expect(trip).toEqual(cfg); + expect(allSites(trip).length).toBeGreaterThan(0); + }); + + it('sequence expression: only the final operand flows into the sink argument', () => { + // `exec((log(x), 'safe'))` — the comma operator's value is the last operand + // (`'safe'`), so exec's arg 0 must NOT carry `x` (review fix). `x` is still + // a USE of the statement (the side-effect operand is evaluated). + const cfg = cfgOf(`function f(x) { exec((log(x), 'safe')); }`); + const execSite = allSites(cfg).find((s) => s.callee === 'exec')!; + expect(execSite.args ?? [[]]).toEqual([[]]); // arg 0 has no flowing binding + expect(siteFact(cfg, 1).uses).toContain(bindingIdx(cfg, 'x')); + }); + + it('sequence expression: a tainted final operand DOES flow into the sink', () => { + const cfg = cfgOf(`function f(x) { exec((log('a'), x)); }`); + const execSite = allSites(cfg).find((s) => s.callee === 'exec')!; + expect(execSite.args).toEqual([[bindingIdx(cfg, 'x')]]); + }); +}); diff --git a/gitnexus/test/unit/pdg-mode-flip.test.ts b/gitnexus/test/unit/pdg-mode-flip.test.ts index f258ecb8a..0441464f1 100644 --- a/gitnexus/test/unit/pdg-mode-flip.test.ts +++ b/gitnexus/test/unit/pdg-mode-flip.test.ts @@ -15,6 +15,7 @@ import { describe, it, expect } from 'vitest'; import { getStoragePaths, loadMeta, saveMeta } from '../../src/storage/repo-manager.js'; +import { taintModelVersion } from '../../src/core/ingestion/taint/typescript-model.js'; import { setupMiniRepo as setupSharedMiniRepo } from '../helpers/mini-repo.js'; const setupMiniRepo = () => setupSharedMiniRepo('gitnexus-pdg-flip-'); @@ -60,6 +61,49 @@ describe('pdgModeMismatch — M1→M2 stamp upgrade (#2082 M2, pure)', () => { }); }); +describe('pdgModeMismatch — M2→M3 stamp upgrade (#2083 M3 U5, pure)', () => { + it('an M2-era stamp (no taint keys) mismatches an M3 request — upgrade forces full writeback', async () => { + const { pdgModeMismatch } = await import('../../src/core/run-analyze.js'); + // Exactly what an M2 run wrote: the three pre-taint resolved caps, no + // maxTaintFindingsPerFunction / maxTaintHops / taintModelVersion. The + // key-union comparator sees e.g. 200 !== undefined and trips the full + // writeback that populates TAINTED/SANITIZES rows without --force (R7). + const m2Stamp = { + maxFunctionLines: 2000, + maxEdgesPerFunction: 5000, + maxReachingDefEdgesPerFunction: 4000, + }; + expect(pdgModeMismatch(m2Stamp, { pdg: true })).toBe(true); + }); + + it('an identical resolved M3 config compares equal (steady state keeps incremental)', async () => { + const { pdgModeMismatch, resolvePdgConfig } = await import('../../src/core/run-analyze.js'); + const stamp = resolvePdgConfig({ pdg: true }); + expect(pdgModeMismatch(stamp, { pdg: true })).toBe(false); + }); + + it('a taint cap change alone trips the mismatch', async () => { + const { pdgModeMismatch, resolvePdgConfig } = await import('../../src/core/run-analyze.js'); + const stamp = resolvePdgConfig({ pdg: true }); + expect(pdgModeMismatch(stamp, { pdg: true, pdgMaxTaintFindingsPerFunction: 10 })).toBe(true); + expect(pdgModeMismatch(stamp, { pdg: true, pdgMaxTaintHops: 4 })).toBe(true); + expect(pdgModeMismatch(stamp, { pdg: true, pdgMaxTaintFindingsPerFunction: 200 })).toBe( + false, // explicit default ≡ default (resolution before comparison) + ); + }); + + it('a model-version change ALONE trips the mismatch (the R7 repopulation guarantee)', async () => { + const { pdgModeMismatch, resolvePdgConfig } = await import('../../src/core/run-analyze.js'); + // A stamp written by a hypothetical older binary whose built-in model + // differed — every cap identical, only the digest moved. Persisted + // findings must never outlive the model that produced them. + const stamp = resolvePdgConfig({ pdg: true }); + const oldModelStamp = { ...stamp, taintModelVersion: '000000000000' }; + expect(stamp?.taintModelVersion).not.toBe('000000000000'); // guard the premise + expect(pdgModeMismatch(oldModelStamp, { pdg: true })).toBe(true); + }); +}); + describe('detect_changes BasicBlock exclusion (#2082 U7)', () => { it('the symbol-overlap id-prefix filter excludes exactly the BasicBlock rows', async () => { const repo = await setupMiniRepo(); @@ -135,6 +179,9 @@ describe('runFullAnalysis — pdg-mode flip (#2099 F1)', () => { maxFunctionLines: 2000, maxEdgesPerFunction: 5000, maxReachingDefEdgesPerFunction: 4000, + maxTaintFindingsPerFunction: 200, + maxTaintHops: 32, + taintModelVersion, }); expect(stamped!.incrementalInProgress).toBeUndefined(); // cleared on success @@ -184,6 +231,9 @@ describe('runFullAnalysis — pdg-mode flip (#2099 F1)', () => { maxFunctionLines: 2000, maxEdgesPerFunction: 1, maxReachingDefEdgesPerFunction: 4000, + maxTaintFindingsPerFunction: 200, + maxTaintHops: 32, + taintModelVersion, }); // The CFG layer survives a rebuild under a tighter edge cap (blocks are // never capped, only edges). diff --git a/gitnexus/test/unit/run-analyze.test.ts b/gitnexus/test/unit/run-analyze.test.ts index 9790f337c..7c9e2a3e5 100644 --- a/gitnexus/test/unit/run-analyze.test.ts +++ b/gitnexus/test/unit/run-analyze.test.ts @@ -8,6 +8,7 @@ import { DEFAULT_EMBEDDING_NODE_LIMIT, } from '../../src/core/embedding-mode.js'; import { getStoragePaths, saveMeta, type RepoMeta } from '../../src/storage/repo-manager.js'; +import { taintModelVersion } from '../../src/core/ingestion/taint/typescript-model.js'; import { createTempDir } from '../helpers/test-db.js'; describe('run-analyze module', () => { @@ -330,13 +331,20 @@ describe('deriveEmbeddingCap', () => { }); describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { - // M2 (#2082) added the resolved REACHING_DEF cap to the stamp; these tests - // model M2 STEADY-STATE equality. The M1-era-stamp (field absent) upgrade - // path is pinned in pdg-mode-flip.test.ts. + // M2 (#2082) added the resolved REACHING_DEF cap to the stamp; M3 (#2083) + // added the two taint caps + the built-in model digest. These tests model + // M3 STEADY-STATE equality — this object is the DELIBERATE pin of the + // resolved-record shape, updated per milestone. The era-stamp (field + // absent) upgrade paths are pinned in pdg-mode-flip.test.ts. const DEFAULTS = { maxFunctionLines: 2000, maxEdgesPerFunction: 5000, maxReachingDefEdgesPerFunction: 4000, + maxTaintFindingsPerFunction: 200, + maxTaintHops: 32, + // Content digest, not a tunable cap — pinned via the exported constant + // (its VALUE changes whenever the built-in model changes, by design). + taintModelVersion, }; it('resolvePdgConfig: pdg-off run resolves to undefined (the meta field is omitted)', async () => { @@ -354,8 +362,17 @@ describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { pdgMaxFunctionLines: 0, pdgMaxEdgesPerFunction: 0, pdgMaxReachingDefEdgesPerFunction: 0, + pdgMaxTaintFindingsPerFunction: 0, + pdgMaxTaintHops: 0, }), - ).toEqual({ maxFunctionLines: 0, maxEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0 }); + ).toEqual({ + maxFunctionLines: 0, + maxEdgesPerFunction: 0, + maxReachingDefEdgesPerFunction: 0, + maxTaintFindingsPerFunction: 0, + maxTaintHops: 0, + taintModelVersion, // not a cap — always stamped on a pdg-on run + }); }); it('legacy meta (no recorded stamp) + plain run → no mismatch', async () => { @@ -391,5 +408,11 @@ describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { expect(pdgModeMismatch(DEFAULTS, { pdg: true, pdgMaxFunctionLines: 500 })).toBe(true); // 0 = unlimited differs from the 2000-line default, too. expect(pdgModeMismatch(DEFAULTS, { pdg: true, pdgMaxFunctionLines: 0 })).toBe(true); + // The M3 taint caps participate identically (#2083). + expect(pdgModeMismatch(DEFAULTS, { pdg: true, pdgMaxTaintFindingsPerFunction: 1 })).toBe(true); + expect(pdgModeMismatch(DEFAULTS, { pdg: true, pdgMaxTaintHops: 1 })).toBe(true); + expect(pdgModeMismatch(DEFAULTS, { pdg: true, pdgMaxTaintFindingsPerFunction: 200 })).toBe( + false, // explicit default ≡ default + ); }); }); diff --git a/gitnexus/test/unit/security.test.ts b/gitnexus/test/unit/security.test.ts index 0474fbf22..514df1078 100644 --- a/gitnexus/test/unit/security.test.ts +++ b/gitnexus/test/unit/security.test.ts @@ -47,6 +47,15 @@ describe('VALID_RELATION_TYPES', () => { expect(VALID_RELATION_TYPES.has('calls')).toBe(false); // case-sensitive expect(VALID_RELATION_TYPES.has('DROP_TABLE')).toBe(false); }); + + it('taint edge types stay OUT of the impact allow-list (#2083 M3 KTD9a)', () => { + // impact's BFS traverses symbol space; TAINTED/SANITIZES live in + // block-space (BasicBlock→BasicBlock) and would be unreachable noise + // there. The `explain` tool is the dedicated taint consumer. Pinned + // explicitly so a future "add all emitted types" sweep can't drag them in. + expect(VALID_RELATION_TYPES.has('TAINTED')).toBe(false); + expect(VALID_RELATION_TYPES.has('SANITIZES')).toBe(false); + }); }); // ─── Valid node labels ─────────────────────────────────────────────── diff --git a/gitnexus/test/unit/taint/model-match.test.ts b/gitnexus/test/unit/taint/model-match.test.ts new file mode 100644 index 000000000..374961d27 --- /dev/null +++ b/gitnexus/test/unit/taint/model-match.test.ts @@ -0,0 +1,315 @@ +/** + * U2 (#2083 M3) — built-in TS/JS taint model + import-aware site matcher. + * + * Fixtures parse REAL source: CFGs (and therefore SiteRecords) come from the + * worker-side TS CFG visitor (the harvest.test.ts harness pattern), and + * ParsedImports come from the real scope-capture + interpretTsImport path + * (the typescript-imports.test.ts harness pattern) — matches run against the + * exact structures U4 will feed the matcher, never hand-built mocks. + */ + +import { describe, it, expect, beforeEach } from 'vitest'; +import { cfgOf, importsFor } from '../../helpers/ts-cfg-harness.js'; +import { hasTaintSafeSites } from '../../../src/core/ingestion/taint/site-safety.js'; +import type { SourceSinkSanitizerSpec } from '../../../src/core/ingestion/taint/source-sink-config.js'; +import { + TS_JS_TAINT_MODEL, + computeTaintModelVersion, + registerBuiltinTaintModels, + taintModelVersion, +} from '../../../src/core/ingestion/taint/typescript-model.js'; +import { + buildTaintImportIndex, + matchFunctionSites, + type FunctionSiteMatches, + type MatchedSanitizerCall, + type MatchedSinkCall, + type MatchedSourceRead, +} from '../../../src/core/ingestion/taint/match.js'; +import { + getSourceSinkConfig, + clearSourceSinkRegistry, + registeredTaintLanguages, +} from '../../../src/core/ingestion/taint/source-sink-registry.js'; + +/** Match function #`fnIndex` of `code` against the built-in model (or `spec`). */ +function matchesOf( + code: string, + fnIndex = 0, + spec: SourceSinkSanitizerSpec = TS_JS_TAINT_MODEL, +): FunctionSiteMatches { + const cfg = cfgOf(code, fnIndex); + // The matcher's documented precondition — real harvests must always pass. + expect(hasTaintSafeSites(cfg)).toBe(true); + return matchFunctionSites(cfg, spec, buildTaintImportIndex(importsFor(code))); +} + +const allSinks = (m: FunctionSiteMatches): MatchedSinkCall[] => + m.statements.flatMap((s) => [...s.sinks]); +const allSources = (m: FunctionSiteMatches): MatchedSourceRead[] => + m.statements.flatMap((s) => [...s.sources]); +const allSanitizers = (m: FunctionSiteMatches): MatchedSanitizerCall[] => + m.statements.flatMap((s) => [...s.sanitizers]); + +describe('sink resolution — ESM import joins', () => { + it('named import: `import { exec } from "child_process"; exec(c)` matches command-injection', () => { + const m = matchesOf(`import { exec } from 'child_process'; +function f(c) { exec(c); }`); + const sinks = allSinks(m); + expect(sinks).toHaveLength(1); + expect(sinks[0].entry.kind).toBe('command-injection'); + expect(sinks[0].entry.name).toBe('exec'); + expect([...sinks[0].argPositions]).toEqual([0]); + expect(m.hasSink).toBe(true); + }); + + it('alias import: `import { exec as run }` — run(c) resolves to child_process.exec', () => { + const m = matchesOf(`import { exec as run } from 'child_process'; +function f(c) { run(c); }`); + const sinks = allSinks(m); + expect(sinks).toHaveLength(1); + expect(sinks[0].entry.name).toBe('exec'); + }); + + it('namespace import: `import * as cp` — cp.exec(c) resolves via the receiver binding', () => { + const m = matchesOf(`import * as cp from 'child_process'; +function f(c) { cp.exec(c); }`); + expect(allSinks(m).map((s) => s.entry.name)).toEqual(['exec']); + }); + + it('node: scheme prefix is normalized — `from "node:child_process"` matches too', () => { + const m = matchesOf(`import { execSync } from 'node:child_process'; +function f(c) { execSync(c); }`); + expect(allSinks(m).map((s) => s.entry.name)).toEqual(['execSync']); + }); + + it('an in-FUNCTION local `exec` shadows the import — no match', () => { + const m = matchesOf(`import { exec } from 'child_process'; +function f(c) { function exec(x) { return x; } exec(c); }`); + expect(allSinks(m)).toHaveLength(0); + expect(m.hasSink).toBe(false); + }); + + it('a module-level local `exec` (no import) does NOT match — synthetic binding, no import entry', () => { + const m = matchesOf( + `function exec(x) { return x; } +function g(c) { exec(c); }`, + 1, // g — index 0 is the local exec itself + ); + expect(allSinks(m)).toHaveLength(0); + }); +}); + +describe('sink resolution — globals and require joins', () => { + it('eval(x) matches code-injection via the synthetic-global fallback', () => { + const m = matchesOf(`function f(x) { eval(x); }`); + const sinks = allSinks(m); + expect(sinks).toHaveLength(1); + expect(sinks[0].entry.kind).toBe('code-injection'); + }); + + it('`new Function(x)` matches; a bare `Function(x)` CALL does not (newOnly)', () => { + const withNew = matchesOf(`function f(x) { const fn = new Function(x); }`); + expect(allSinks(withNew).map((s) => s.entry.name)).toEqual(['Function']); + const bareCall = matchesOf(`function f(x) { const fn = Function(x); }`); + expect(allSinks(bareCall)).toHaveLength(0); + }); + + it('require literal join: `const cp = require("child_process"); cp.exec(c)` matches', () => { + const m = matchesOf(`function f(c) { const cp = require('child_process'); cp.exec(c); }`); + const sinks = allSinks(m); + expect(sinks).toHaveLength(1); + expect(sinks[0].entry.name).toBe('exec'); + expect(sinks[0].entry.kind).toBe('command-injection'); + }); + + it('a require’d local utility named exec does NOT match (no bare-name fallback for non-globals)', () => { + const m = matchesOf(`function f(c) { const exec = require('./my-utils'); exec(c); }`); + expect(allSinks(m)).toHaveLength(0); + }); + + it('non-renamed destructured require resolves via the dual interpretation', () => { + const m = matchesOf(`function f(c) { const { exec } = require('child_process'); exec(c); }`); + expect(allSinks(m).map((s) => s.entry.name)).toEqual(['exec']); + }); +}); + +describe('sources — conventional member reads', () => { + it('req.body matches remote-input; request.body matches; myObj.body does not', () => { + const m = matchesOf(`function f(req, request, myObj) { + const a = req.body; + const b = request.body; + const c = myObj.body; + }`); + const sources = allSources(m); + expect(sources).toHaveLength(2); + expect(sources.every((s) => s.entry.kind === 'remote-input')).toBe(true); + expect(m.hasSource).toBe(true); + }); + + it('all five conventional properties match; an unlisted property does not', () => { + const m = matchesOf(`function f(req) { + const a = req.body, b = req.query, c = req.params, d = req.headers, e = req.cookies; + const z = req.socket; + }`); + expect(allSources(m)).toHaveLength(5); + }); +}); + +describe('sink argument-position discipline', () => { + it('exec(safe, tainted): only the registered position 0 is a sink position', () => { + const m = matchesOf(`import { exec } from 'child_process'; +function f(safe, tainted) { exec(safe, tainted); }`); + const [sink] = allSinks(m); + expect([...sink.argPositions]).toEqual([0]); // position 1 carries an occurrence but is not registered + }); + + it('spread: exec(...args) matches — recorded position ≥ spread index degrades soundly', () => { + const m = matchesOf(`import { exec } from 'child_process'; +function f(args) { exec(...args); }`); + const [sink] = allSinks(m); + expect([...sink.argPositions]).toEqual([0]); + }); + + it('spread precision: a pre-spread position stays exact; post-spread matches any q ≥ spread', () => { + // Custom spec: sink position 1 only. `s2(a, ...rest)` — recorded 0 is + // BEFORE the spread (exact: no match); recorded 1 is the spread (match). + const spec: SourceSinkSanitizerSpec = { + sources: [], + sinks: [{ name: 's2', kind: 'command-injection', args: [1], global: true }], + sanitizers: [], + }; + const m = matchesOf(`function f(a, rest) { s2(a, ...rest); }`, 0, spec); + const [sink] = allSinks(m); + expect([...sink.argPositions]).toEqual([1]); + }); + + it('tagged template: substitutions aggregate at position 0 and match any registered position', () => { + const spec: SourceSinkSanitizerSpec = { + sources: [], + sinks: [{ name: 'sql', kind: 'sql-injection', args: [1], global: true }], + sanitizers: [], + }; + const m = matchesOf(`function f(id) { sql\`select \${id}\`; }`, 0, spec); + const [sink] = allSinks(m); + expect([...sink.argPositions]).toEqual([0]); + }); + + it('a sink whose dangerous positions carry no occurrences is not reported', () => { + const m = matchesOf(`import { exec } from 'child_process'; +function f(t) { exec('ls -la', t); }`); + expect(allSinks(m)).toHaveLength(0); + }); +}); + +describe('receiver-conventional sinks', () => { + it('res.send / res.write match xss; out.send does not', () => { + const m = matchesOf(`function f(res, out, x) { res.send(x); res.write(x); out.send(x); }`); + expect(allSinks(m).map((s) => s.entry.name)).toEqual(['send', 'write']); + expect(allSinks(m).every((s) => s.entry.kind === 'xss')).toBe(true); + }); + + it('.query/.execute match sql-injection on ANY receiver', () => { + const m = matchesOf(`function f(db, pool, x) { db.query(x); pool.execute(x); }`); + expect(allSinks(m).map((s) => s.entry.kind)).toEqual(['sql-injection', 'sql-injection']); + }); +}); + +describe('sanitizers — import-aware only, kind-scoped', () => { + it('path.basename matches and neutralizes path-traversal, NOT command-injection', () => { + const m = matchesOf(`import path from 'path'; +function f(p) { const safe = path.basename(p); }`); + const sans = allSanitizers(m); + expect(sans).toHaveLength(1); + expect(sans[0].entry.neutralizes).toContain('path-traversal'); + expect(sans[0].entry.neutralizes).not.toContain('command-injection'); + // resultDefs carries the kill target (KTD4b) + expect(sans[0].resultDefs).toHaveLength(1); + }); + + it('validator.escape via named import matches xss', () => { + const m = matchesOf(`import { escape } from 'validator'; +function f(x) { const safe = escape(x); }`); + const sans = allSanitizers(m); + expect(sans).toHaveLength(1); + expect([...sans[0].entry.neutralizes]).toEqual(['xss']); + }); + + it('default-imported escape-html matches via the default pseudo-name', () => { + const m = matchesOf(`import escapeHtml from 'escape-html'; +function f(x) { const safe = escapeHtml(x); }`); + expect(allSanitizers(m)).toHaveLength(1); + }); + + it('encodeURIComponent matches as a true global (xss + path-traversal)', () => { + const m = matchesOf(`function f(x) { const safe = encodeURIComponent(x); }`); + const sans = allSanitizers(m); + expect(sans).toHaveLength(1); + expect([...sans[0].entry.neutralizes]).toEqual(['xss', 'path-traversal']); + }); + + it('a user-defined in-file `escape` is NEVER a sanitizer (no bare-name resolution)', () => { + const m = matchesOf( + `function escape(s) { return s; } +function f(x) { const safe = escape(x); }`, + 1, // f + ); + expect(allSanitizers(m)).toHaveLength(0); + }); + + it('value-position sanitizer (exec(escape(x))) matches with EMPTY resultDefs — interposition substrate', () => { + const m = matchesOf(`import { exec } from 'child_process'; +import { escape } from 'validator'; +function f(x) { exec(escape(x)); }`); + const sans = allSanitizers(m); + expect(sans).toHaveLength(1); + expect(sans[0].resultDefs).toHaveLength(0); + // The sink still matches — interposition is U3's call, not the matcher's. + expect(allSinks(m)).toHaveLength(1); + }); +}); + +describe('registry + model identity', () => { + beforeEach(() => clearSourceSinkRegistry()); + + it('registerBuiltinTaintModels registers typescript AND javascript (idempotent); others stay undefined', () => { + registerBuiltinTaintModels(); + registerBuiltinTaintModels(); // idempotent — last-write-wins on the same ids + expect(registeredTaintLanguages().sort()).toEqual(['javascript', 'typescript']); + expect(getSourceSinkConfig('typescript')).toBe(TS_JS_TAINT_MODEL); + expect(getSourceSinkConfig('javascript')).toBe(TS_JS_TAINT_MODEL); + expect(getSourceSinkConfig('python')).toBeUndefined(); + }); + + it('taintModelVersion is the digest of the full built-in model', () => { + expect(taintModelVersion).toBe(computeTaintModelVersion(TS_JS_TAINT_MODEL)); + expect(taintModelVersion).toMatch(/^[0-9a-f]{12}$/); + }); + + it('adding an entry changes the version', () => { + const added: SourceSinkSanitizerSpec = { + ...TS_JS_TAINT_MODEL, + sinks: [...TS_JS_TAINT_MODEL.sinks, { name: 'load', kind: 'code-injection', module: 'vm' }], + }; + expect(computeTaintModelVersion(added)).not.toBe(taintModelVersion); + }); + + it('changing only a kind label changes the version', () => { + const relabeled: SourceSinkSanitizerSpec = { + ...TS_JS_TAINT_MODEL, + sinks: TS_JS_TAINT_MODEL.sinks.map((s) => + s.name === 'exec' ? { ...s, kind: 'xss' as const } : s, + ), + }; + expect(computeTaintModelVersion(relabeled)).not.toBe(taintModelVersion); + }); + + it('the version is content-derived: key order does not matter, entry order does', () => { + const reordered: SourceSinkSanitizerSpec = { + sanitizers: TS_JS_TAINT_MODEL.sanitizers, + sinks: TS_JS_TAINT_MODEL.sinks, + sources: TS_JS_TAINT_MODEL.sources, + }; + expect(computeTaintModelVersion(reordered)).toBe(taintModelVersion); + }); +}); diff --git a/gitnexus/test/unit/taint/path-codec.test.ts b/gitnexus/test/unit/taint/path-codec.test.ts new file mode 100644 index 000000000..a7b891844 --- /dev/null +++ b/gitnexus/test/unit/taint/path-codec.test.ts @@ -0,0 +1,305 @@ +/** + * U4 (#2083 M3) — the shared taint-path reason codec (plan KTD6). + * + * The wire format must round-trip BYTE-EXACT through the CSV persistence + * layer (`escapeCSVField ∘ sanitizeUTF8`, csv-generator.ts) — that + * composition is exercised here verbatim, including the un-escape a DB load + * performs. Truncation (hop cap, byte cap, unencodable hop) must decode as + * "path incomplete", never as an error; malformed input must produce a typed + * failure, never a throw. + */ + +import { describe, it, expect } from 'vitest'; +import { + TAINT_PATH_CODEC_VERSION, + TAINT_REASON_MAX_BYTES, + TAINT_PATH_TRUNCATION_MARKER, + encodeTaintPath, + decodeTaintPath, + type TaintPathHopInput, +} from '../../../src/core/ingestion/taint/path-codec.js'; +import { escapeCSVField, sanitizeUTF8 } from '../../../src/core/lbug/csv-generator.js'; + +/** Inverse of escapeCSVField — what a CSV/DB load applies to the stored cell. */ +function unescapeCSVField(cell: string): string { + expect(cell.startsWith('"') && cell.endsWith('"')).toBe(true); + return cell.slice(1, -1).replace(/""/g, '"'); +} + +const roundTrip = (hops: readonly TaintPathHopInput[]) => { + const { reason } = encodeTaintPath(hops); + const decoded = decodeTaintPath(reason); + if (!decoded.ok) throw new Error(`decode failed: ${decoded.error}`); + return { reason, decoded }; +}; + +describe('encodeTaintPath / decodeTaintPath round trip', () => { + it('round-trips an ordered multi-hop path with variables, lines, and viaCall', () => { + const hops: TaintPathHopInput[] = [ + { name: 'req', line: 3 }, + { name: 'cmd', line: 4, viaCall: true }, + { name: 'cmd', line: 7 }, + ]; + const { reason, decoded } = roundTrip(hops); + expect(reason).toBe('1|req:3|cmd:4:c|cmd:7'); + expect(decoded.version).toBe(TAINT_PATH_CODEC_VERSION); + expect(decoded.truncated).toBe(false); + expect(decoded.hops).toEqual([ + { variable: 'req', line: 3, viaCall: false }, + { variable: 'cmd', line: 4, viaCall: true }, + { variable: 'cmd', line: 7, viaCall: false }, + ]); + }); + + it('round-trips the empty path (version prefix only)', () => { + const { reason, decoded } = roundTrip([]); + expect(reason).toBe('1'); + expect(decoded.hops).toEqual([]); + expect(decoded.truncated).toBe(false); + }); + + it('round-trips identifier-charset names: $, _, #, digits, case', () => { + const names = ['$jq', '_private', '#3', 'CONST_99', 'aB$_#z', 'x']; + const hops = names.map((name, i) => ({ name, line: i + 1, viaCall: i % 2 === 0 })); + const { decoded } = roundTrip(hops); + expect(decoded.hops.map((h) => h.variable)).toEqual(names); + }); + + it('fuzz-ish sweep: random identifier-charset names of varied length survive', () => { + const CHARSET = 'ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789_$#'; + // Deterministic LCG so a failure is reproducible. + let seed = 0xc0ffee; + const next = (): number => { + seed = (seed * 1103515245 + 12345) & 0x7fffffff; + return seed; + }; + for (let trial = 0; trial < 200; trial++) { + const hops: TaintPathHopInput[] = []; + const count = (next() % 8) + 1; + for (let i = 0; i < count; i++) { + const len = (next() % 24) + 1; + let name = ''; + for (let j = 0; j < len; j++) name += CHARSET[next() % CHARSET.length]; + hops.push({ name, line: next() % 100000, viaCall: next() % 2 === 0 }); + } + const { reason, decoded } = roundTrip(hops); + expect(decoded.truncated).toBe(false); + expect(decoded.hops).toEqual( + hops.map((h) => ({ variable: h.name, line: h.line, viaCall: h.viaCall === true })), + ); + // The whole wire string is printable ASCII (the CSV-survival invariant). + expect(/^[\x20-\x7e]+$/.test(reason)).toBe(true); + } + }); +}); + +describe('kind header (; — U6, the only persisted channel for sinkKind)', () => { + it('round-trips a kind header with hops', () => { + const { reason } = encodeTaintPath( + [ + { name: 'req', line: 3 }, + { name: 'cmd', line: 4 }, + ], + { kind: 'command-injection' }, + ); + expect(reason).toBe('1;command-injection|req:3|cmd:4'); + const decoded = decodeTaintPath(reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) { + expect(decoded.kind).toBe('command-injection'); + expect(decoded.hops.map((h) => h.variable)).toEqual(['req', 'cmd']); + } + }); + + it('round-trips a kind header on the hop-less path', () => { + const { reason } = encodeTaintPath([], { kind: 'xss' }); + expect(reason).toBe('1;xss'); + const decoded = decodeTaintPath(reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) { + expect(decoded.kind).toBe('xss'); + expect(decoded.hops).toEqual([]); + } + }); + + it('kind + truncation marker coexist; the header is never sacrificed to the byte cap', () => { + const { reason, truncated } = encodeTaintPath([{ name: 'longVariableName', line: 12345 }], { + kind: 'sql-injection', + maxBytes: 8, // far below the header size — floor lifts it, hops drop + }); + expect(truncated).toBe(true); + expect(reason).toBe(`1;sql-injection|${TAINT_PATH_TRUNCATION_MARKER}`); + const decoded = decodeTaintPath(reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) { + expect(decoded.kind).toBe('sql-injection'); + expect(decoded.truncated).toBe(true); + expect(decoded.hops).toEqual([]); + } + }); + + it('a kind outside [a-z0-9-] is dropped (header omitted), never corrupted into the wire', () => { + for (const bad of ['Command-Injection', 'a;b', 'k|x', 'café', '']) { + const { reason } = encodeTaintPath([{ name: 'x', line: 1 }], { kind: bad }); + expect(reason).toBe('1|x:1'); + const decoded = decodeTaintPath(reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) expect(decoded.kind).toBeUndefined(); + } + }); + + it('a header-less version-1 string still decodes (kind undefined)', () => { + const decoded = decodeTaintPath('1|a:1'); + expect(decoded.ok).toBe(true); + if (decoded.ok) expect(decoded.kind).toBeUndefined(); + }); + + it('the kind header survives the CSV persistence transform byte-exact', () => { + const { reason } = encodeTaintPath([{ name: 'req', line: 2 }], { kind: 'path-traversal' }); + const loaded = unescapeCSVField(escapeCSVField(sanitizeUTF8(reason))); + expect(loaded).toBe(reason); + const decoded = decodeTaintPath(loaded); + expect(decoded.ok).toBe(true); + if (decoded.ok) expect(decoded.kind).toBe('path-traversal'); + }); + + it('fails typed on a malformed kind header', () => { + for (const bad of ['1;|a:1', '1;BAD|a:1', '1;', '1;a;b|x:1']) { + const decoded = decodeTaintPath(bad); + expect(decoded.ok, bad).toBe(false); + } + }); +}); + +describe('CSV persistence composition (escapeCSVField ∘ sanitizeUTF8)', () => { + it('the encoding survives the exact persistence transform byte-exact', () => { + const hops: TaintPathHopInput[] = [ + { name: 'req', line: 12 }, + { name: '$tmp_2', line: 13, viaCall: true }, + { name: '#7', line: 99 }, + ]; + const { reason } = encodeTaintPath(hops); + // csv-generator applies sanitizeUTF8 INSIDE escapeCSVField; compose both + // explicitly anyway so the test pins each layer. + const stored = escapeCSVField(sanitizeUTF8(reason)); + const loaded = unescapeCSVField(stored); + expect(loaded).toBe(reason); // byte-exact + const decoded = decodeTaintPath(loaded); + expect(decoded.ok).toBe(true); + }); + + it('a truncated encoding also survives the persistence transform', () => { + const { reason, truncated } = encodeTaintPath([{ name: 'x', line: 1 }], { truncated: true }); + expect(truncated).toBe(true); + const loaded = unescapeCSVField(escapeCSVField(sanitizeUTF8(reason))); + expect(loaded).toBe(reason); + }); +}); + +describe('truncation', () => { + it('caller-flagged truncation (hop cap upstream) emits the marker; decode reports path-incomplete', () => { + const { reason, truncated } = encodeTaintPath([{ name: 'a', line: 1 }], { truncated: true }); + expect(truncated).toBe(true); + expect(reason).toBe(`1|a:1|${TAINT_PATH_TRUNCATION_MARKER}`); + const decoded = decodeTaintPath(reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) { + expect(decoded.truncated).toBe(true); // informational — NOT an error + expect(decoded.hops).toEqual([{ variable: 'a', line: 1, viaCall: false }]); + } + }); + + it('byte-cap overflow drops TRAILING hops (source-side prefix kept) and sets the marker', () => { + const hops: TaintPathHopInput[] = []; + for (let i = 0; i < 1000; i++) hops.push({ name: `variable_${i}`, line: i }); + const { reason, truncated } = encodeTaintPath(hops); + expect(truncated).toBe(true); + expect(reason.length).toBeLessThanOrEqual(TAINT_REASON_MAX_BYTES); + const decoded = decodeTaintPath(reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) { + expect(decoded.truncated).toBe(true); + expect(decoded.hops.length).toBeGreaterThan(0); + expect(decoded.hops.length).toBeLessThan(hops.length); + // Prefix discipline: hop k decodes hop k of the input, in order. + decoded.hops.forEach((h, i) => { + expect(h.variable).toBe(`variable_${i}`); + expect(h.line).toBe(i); + }); + } + }); + + it('a tiny maxBytes still yields a well-formed (possibly hop-less) truncated path', () => { + const { reason, truncated } = encodeTaintPath([{ name: 'longVariableName', line: 123 }], { + maxBytes: 8, + }); + expect(truncated).toBe(true); + expect(reason).toBe(`1|${TAINT_PATH_TRUNCATION_MARKER}`); + const decoded = decodeTaintPath(reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) { + expect(decoded.hops).toEqual([]); + expect(decoded.truncated).toBe(true); + } + }); + + it('an unencodable hop name stops encoding at that hop and marks truncation (defend, never corrupt)', () => { + const cases = ['a|b', 'a:b', 'café', 'name~x', '', 'a b']; + for (const bad of cases) { + const { reason, truncated } = encodeTaintPath([ + { name: 'ok1', line: 1 }, + { name: bad, line: 2 }, + { name: 'ok2', line: 3 }, // dropped too — prefix discipline + ]); + expect(truncated).toBe(true); + const decoded = decodeTaintPath(reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) { + expect(decoded.truncated).toBe(true); + expect(decoded.hops).toEqual([{ variable: 'ok1', line: 1, viaCall: false }]); + } + } + }); + + it('a non-integer or negative line is unencodable the same way', () => { + for (const line of [1.5, -1, NaN, Infinity]) { + const { truncated, reason } = encodeTaintPath([{ name: 'x', line }]); + expect(truncated).toBe(true); + const decoded = decodeTaintPath(reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) expect(decoded.hops).toEqual([]); + } + }); +}); + +describe('typed parse failures (never a throw)', () => { + const failing: Array<[string, unknown]> = [ + ['empty string', ''], + ['non-string', 42], + ['undefined', undefined], + ['unknown version', '2|a:1'], + ['missing separator after version', '1a:1'], + ['hop with no line', '1|a'], + ['hop with too many fields', '1|a:1:c:d'], + ['non-numeric line', '1|a:x'], + ['negative line', '1|a:-1'], + ['invalid variable charset', '1|a b:1'], + ['empty variable', '1|:1'], + ['uppercase flag (reserved charset is lowercase)', '1|a:1:C'], + ['marker not trailing', `1|${TAINT_PATH_TRUNCATION_MARKER}|a:1`], + ['empty hop segment', '1|a:1||b:2'], + ]; + for (const [label, input] of failing) { + it(`fails typed on ${label}`, () => { + const decoded = decodeTaintPath(input); + expect(decoded.ok).toBe(false); + if (!decoded.ok) expect(decoded.error.length).toBeGreaterThan(0); + }); + } + + it('accepts unknown RESERVED lowercase flag letters (forward compatibility)', () => { + const decoded = decodeTaintPath('1|a:1:cz'); + expect(decoded.ok).toBe(true); + if (decoded.ok) expect(decoded.hops[0]).toEqual({ variable: 'a', line: 1, viaCall: true }); + }); +}); diff --git a/gitnexus/test/unit/taint/propagate.test.ts b/gitnexus/test/unit/taint/propagate.test.ts new file mode 100644 index 000000000..2e8f9a549 --- /dev/null +++ b/gitnexus/test/unit/taint/propagate.test.ts @@ -0,0 +1,686 @@ +/** + * U3 (#2083 M3) — pure taint propagation engine. + * + * Fixtures parse REAL source through the harvest + match harness (the + * model-match.test.ts pattern): CFGs and SiteRecords come from the worker-side + * TS CFG visitor, def→use facts from the real reaching-defs solver, and + * matches from the real import-aware matcher — `computeTaintFlows` consumes + * the exact structures U4 will feed it, never hand-built mocks. + * + * Sanitizer semantics under test (the KIND-SET exclusion model, which + * subsumes the plan's binary kill — see propagate.ts module doc): a def + * produced through a sanitizer is tainted-with-exclusions; a sink fires + * unless its kind is in the taint's accumulated neutralized set. Mechanics + * tests therefore use a custom spec whose `escape` neutralizes the test + * sink's own kind (so "killed" scenarios read like the plan's binary + * scenarios); the kind-set tests at the bottom exercise the real built-in + * model where neutralization is deliberately kind-scoped. + */ + +import { describe, it, expect } from 'vitest'; +import { cfgOf, importsFor } from '../../helpers/ts-cfg-harness.js'; +import type { FunctionCfg } from '../../../src/core/ingestion/cfg/types.js'; +import { + computeReachingDefs, + type FunctionDefUse, + type ReachingDefsLimits, +} from '../../../src/core/ingestion/cfg/reaching-defs.js'; +import { hasTaintSafeSites } from '../../../src/core/ingestion/taint/site-safety.js'; +import type { SourceSinkSanitizerSpec } from '../../../src/core/ingestion/taint/source-sink-config.js'; +import { TS_JS_TAINT_MODEL } from '../../../src/core/ingestion/taint/typescript-model.js'; +import { + buildTaintImportIndex, + matchFunctionSites, +} from '../../../src/core/ingestion/taint/match.js'; +import { + computeTaintFlows, + type FunctionTaintResult, + type TaintLimits, +} from '../../../src/core/ingestion/taint/propagate.js'; + +/** The single binding index for `name` (throws when shadowed/ambiguous). */ +function bindingIdx(cfg: FunctionCfg, name: string): number { + const idxs = (cfg.bindings ?? []).map((b, i) => (b.name === name ? i : -1)).filter((i) => i >= 0); + if (idxs.length !== 1) throw new Error(`expected 1 binding for ${name}, got ${idxs.length}`); + return idxs[0]; +} + +/** + * Mechanics spec: a global `exec` sink and a global `escape` sanitizer that + * neutralizes the SAME kind — so interposition/kill mechanics behave like the + * plan's binary-kill scenarios while still flowing through the kind-set model. + */ +const MECH: SourceSinkSanitizerSpec = { + sources: [ + { + kind: 'remote-input', + objects: ['req'], + properties: ['body', 'query', 'params', 'headers'], + }, + ], + sinks: [{ name: 'exec', kind: 'command-injection', args: [0], global: true }], + sanitizers: [{ name: 'escape', neutralizes: ['command-injection'], global: true }], +}; + +interface AnalyzeOptions { + fnIndex?: number; + spec?: SourceSinkSanitizerSpec; + limits?: TaintLimits; + factLimits?: ReachingDefsLimits; +} + +function analyze(code: string, opts: AnalyzeOptions = {}): FunctionTaintResult { + const cfg = cfgOf(code, opts.fnIndex ?? 0); + expect(hasTaintSafeSites(cfg)).toBe(true); + const defUse = computeReachingDefs(cfg, opts.factLimits); + const matches = matchFunctionSites( + cfg, + opts.spec ?? MECH, + buildTaintImportIndex(importsFor(code)), + ); + return computeTaintFlows(cfg, defUse, matches, opts.limits); +} + +/** Hop summaries `name@line` (`name@line*` when viaCall) for readable asserts. */ +function hopSummary(r: FunctionTaintResult, findingIdx = 0): string[] { + const f = r.findings[findingIdx]; + if (!f) throw new Error(`no finding at index ${findingIdx}`); + return f.hops.map((h) => `${h.name}@${h.point.line}${h.viaCall ? '*' : ''}`); +} + +// ── statuses and the empty case ────────────────────────────────────────────── + +describe('coverage-gap statuses (R4)', () => { + it('a truncated FunctionDefUse yields a coverage-gap result with zero findings', () => { + const r = analyze( + `function f(req) { + const b = req.body; + const c = b; + exec(c); + }`, + { factLimits: { maxFacts: 1 } }, + ); + expect(r.status).toBe('coverage-gap'); + expect(r.gapReason).toBe('truncated'); + expect(r.findings).toHaveLength(0); + expect(r.kills).toHaveLength(0); + }); + + it('a no-facts FunctionDefUse (no binding table) yields a coverage-gap result', () => { + const cfg = cfgOf(`function f(req) { exec(req.body); }`); + const { bindings: _bindings, ...noBindings } = cfg; + const defUse = computeReachingDefs(noBindings as FunctionCfg); + expect(defUse.status).toBe('no-facts'); + const matches = matchFunctionSites(cfg, MECH, buildTaintImportIndex([])); + const r = computeTaintFlows(cfg, defUse, matches); + expect(r.status).toBe('coverage-gap'); + expect(r.gapReason).toBe('no-facts'); + expect(r.findings).toHaveLength(0); + }); + + it('an overflow FunctionDefUse yields a coverage-gap result (contract input shape)', () => { + const cfg = cfgOf(`function f(req) { exec(req.body); }`); + const overflow: FunctionDefUse = { + status: 'overflow', + bindings: cfg.bindings ?? [], + facts: [], + defCount: 0, + useCount: 0, + }; + const matches = matchFunctionSites(cfg, MECH, buildTaintImportIndex([])); + const r = computeTaintFlows(cfg, overflow, matches); + expect(r.status).toBe('coverage-gap'); + expect(r.gapReason).toBe('overflow'); + }); + + it('no sources and no sinks → computed, empty result', () => { + const r = analyze(`function f(x) { const y = x; return y; }`); + expect(r.status).toBe('computed'); + expect(r.findings).toHaveLength(0); + expect(r.kills).toHaveLength(0); + expect(r.droppedFindings).toBe(0); + }); +}); + +// ── rule (b): statement-local source→sink ──────────────────────────────────── + +describe('rule (b) — statement-local findings', () => { + it('exec(req.body) → one finding with a single hop', () => { + const r = analyze(`function f(req) { + exec(req.body); + }`); + expect(r.status).toBe('computed'); + expect(r.findings).toHaveLength(1); + const f = r.findings[0]; + expect(f.sinkKind).toBe('command-injection'); + expect(f.source.property).toBe('body'); + expect(f.source.point.line).toBe(2); + expect(f.sink.point.line).toBe(2); + expect(f.sink.argIndex).toBe(0); + expect(f.hops).toHaveLength(1); + expect(f.hops[0].name).toBe('req'); + }); + + it('exec(req.body, req.query) → TWO findings distinguished by occurrence (KTD6 identity)', () => { + const spec: SourceSinkSanitizerSpec = { + ...MECH, + sinks: [{ name: 'exec', kind: 'command-injection', args: [0, 1], global: true }], + }; + const r = analyze(`function f(req) { exec(req.body, req.query); }`, { spec }); + expect(r.findings).toHaveLength(2); + const ids = r.findings.map((f) => `${f.source.property}@arg${f.sink.argIndex}`); + expect(ids).toEqual(['body@arg0', 'query@arg1']); + }); + + it('exec(req.body.toString()) → finding via the statement-local rule', () => { + const r = analyze(`function f(req) { exec(req.body.toString()); }`); + expect(r.findings).toHaveLength(1); + expect(r.findings[0].source.property).toBe('body'); + }); + + it('a source read at an UNREGISTERED sink position produces no finding', () => { + const r = analyze(`function f(req) { exec('ls', req.body); }`); + expect(r.findings).toHaveLength(0); + }); +}); + +// ── rule (a): happy path and chains ────────────────────────────────────────── + +describe('rule (a) — worklist over def→use facts', () => { + it('same-block flow: const b = req.body; exec(b) → finding with hop chain', () => { + const r = analyze(`function f(req) { + const b = req.body; + exec(b); + }`); + expect(r.findings).toHaveLength(1); + expect(hopSummary(r)).toEqual(['b@2', 'b@3']); + expect(r.findings[0].source.property).toBe('body'); + expect(r.findings[0].sink.point.line).toBe(3); + }); + + it('reassignment chain carries variables per hop: b@L2 → c@L3 → sink@L4', () => { + const r = analyze(`function f(req) { + const b = req.body; + const c = b; + exec(c); + }`); + expect(r.findings).toHaveLength(1); + expect(hopSummary(r)).toEqual(['b@2', 'c@3', 'c@4']); + }); + + it('cross-block flow through a branch reaches the sink', () => { + const r = analyze(`function f(req, cond) { + const b = req.body; + let c = ''; + if (cond) { + c = b; + } + exec(c); + }`); + expect(r.findings).toHaveLength(1); + expect(hopSummary(r)).toEqual(['b@2', 'c@5', 'c@7']); + }); +}); + +// ── sanitizer interposition and kills (KTD4 both clauses) ─────────────────── + +describe('sanitizers — interposition, kill locality, kind sets (mechanics spec)', () => { + it('seed interposition: const b = escape(req.body) → no finding, SANITIZES kill on b', () => { + const r = analyze(`function f(req) { + const b = escape(req.body); + exec(b); + }`); + expect(r.findings).toHaveLength(0); + expect(r.kills).toHaveLength(1); + const cfg = cfgOf(`function f(req) { + const b = escape(req.body); + exec(b); + }`); + expect(r.kills[0].bindingIdx).toBe(bindingIdx(cfg, 'b')); + expect(r.kills[0].sanitizer.line).toBe(2); + expect([...r.kills[0].neutralized]).toEqual(['command-injection']); + }); + + it('sink interposition: exec(escape(x)) with x tainted → no finding, kill recorded', () => { + const r = analyze(`function f(req) { + const x = req.body; + exec(escape(x)); + }`); + expect(r.findings).toHaveLength(0); + expect(r.kills).toHaveLength(1); + expect(r.kills[0].sanitizer.line).toBe(3); + expect([...r.kills[0].neutralized]).toEqual(['command-injection']); + }); + + it('bypass occurrence: const c = cond ? escape(x) : x → finding (direct path bypasses)', () => { + // The ternary's escape call deliberately gets NO resultDefs (U1) — c is + // floor-tainted with the EMPTY exclusion set (intersection over paths: + // the direct `x` arm contributes ∅, so ∅ ∩ {command-injection} = ∅). + const r = analyze(`function f(req, cond) { + const x = req.body; + const c = cond ? escape(x) : x; + exec(c); + }`); + expect(r.findings).toHaveLength(1); + expect(r.kills).toHaveLength(0); + }); + + it('intra-statement bypass at the sink: exec(x + escape(x)) → finding (plain occurrence wins)', () => { + const r = analyze(`function f(req) { + const x = req.body; + exec(x + escape(x)); + }`); + expect(r.findings).toHaveLength(1); + // a BYPASSED sanitizer killed nothing — no SANITIZES record + expect(r.kills).toHaveLength(0); + }); + + it('kill locality (KTD4b): const c = escape(b); exec(b) → finding on b AND a kill on c', () => { + const code = `function f(req) { + const b = req.body; + const c = escape(b); + exec(b); + }`; + const r = analyze(code); + expect(r.findings).toHaveLength(1); + expect(hopSummary(r)).toEqual(['b@2', 'b@4']); + expect(r.kills).toHaveLength(1); + expect(r.kills[0].bindingIdx).toBe(bindingIdx(cfgOf(code), 'c')); + }); + + it('sanitizer self-assign: b = escape(b); exec(b) → no finding, one kill', () => { + const r = analyze(`function f(req) { + let b = req.body; + b = escape(b); + exec(b); + }`); + expect(r.findings).toHaveLength(0); + expect(r.kills).toHaveLength(1); + expect(r.kills[0].sanitizer.line).toBe(3); + }); + + it('a sanitizer reached only through a SPREAD argument does not neutralize (position unprovable)', () => { + // `escape(...arr)` — the runtime argument positions are unknowable, so + // claiming the sanitized position received the taint would risk a false + // kill. Sound direction: taint flows through un-neutralized. + const r = analyze(`function f(req) { + const arr = [req.body]; + const b = escape(...arr); + exec(b); + }`); + expect(r.findings).toHaveLength(1); + expect(r.kills).toHaveLength(0); + }); + + it('conditional sanitizer does NOT suppress: if (cond) { b = escape(b) } exec(b) → finding', () => { + const r = analyze(`function f(req, cond) { + let b = req.body; + if (cond) { + b = escape(b); + } + exec(b); + }`); + expect(r.findings).toHaveLength(1); + // the seed def's flow survives the may-path; the sanitized def is killed + expect(hopSummary(r)).toEqual(['b@2', 'b@6']); + expect(r.kills).toHaveLength(1); + }); +}); + +// ── loop semantics: zero-iteration pair, fixpoint termination ─────────────── + +describe('loops — kill keyed on the def point, monotone termination (R3)', () => { + it('zero-iteration while: the cond-false exit carries the seed def → finding SURVIVES', () => { + const r = analyze(`function f(req, c) { + let x = req.body; + while (c) { + x = escape(x); + } + exec(x); + }`); + expect(r.findings).toHaveLength(1); + expect(hopSummary(r)).toEqual(['x@2', 'x@6']); + expect(r.kills).toHaveLength(1); + }); + + it('do-while: the body always runs → no finding (and escape-in-loop does not re-taint)', () => { + const r = analyze(`function f(req, c) { + let x = req.body; + do { + x = escape(x); + } while (c); + exec(x); + }`); + expect(r.findings).toHaveLength(0); + expect(r.kills).toHaveLength(1); + }); + + it('loop self-taint terminates: x = x + t in a for loop → finding, bounded', () => { + const r = analyze(`function f(req) { + const t = req.body; + let x = ''; + for (let i = 0; i < 3; i++) { + x = x + t; + } + exec(x); + }`); + expect(r.status).toBe('computed'); + expect(r.findings).toHaveLength(1); + }); + + it('assign-and-test: if ((m = re.exec(s)) && m) exec(m) → finding (self-fact handled)', () => { + const r = analyze(`function f(req, re) { + const s = req.body; + let m; + if ((m = re.exec(s)) && m) { + exec(m); + } + }`); + expect(r.findings).toHaveLength(1); + }); +}); + +// ── propagate-through unmodeled calls (KTD5) ───────────────────────────────── + +describe('propagate-through — unmodeled calls and receivers, viaCall marks', () => { + it('const y = helper(t); exec(y) → finding with a viaCall hop', () => { + const r = analyze(`function f(req) { + const t = req.body; + const y = helper(t); + exec(y); + }`); + expect(r.findings).toHaveLength(1); + expect(hopSummary(r)).toEqual(['t@2', 'y@3*', 'y@4']); + }); + + it('receiver propagation: const cmd = t.trim(); exec(cmd) → finding (TITO)', () => { + const r = analyze(`function f(req) { + const t = req.body; + const cmd = t.trim(); + exec(cmd); + }`); + expect(r.findings).toHaveLength(1); + expect(hopSummary(r)).toEqual(['t@2', 'cmd@3*', 'cmd@4']); + }); + + it('tainted occurrence nested in an unmodeled call AT the sink fires: exec(helper(t))', () => { + const r = analyze(`function f(req) { + const t = req.body; + exec(helper(t)); + }`); + expect(r.findings).toHaveLength(1); + // the sink hop records that the occurrence flowed through a call + const sinkHop = r.findings[0].hops.at(-1); + expect(sinkHop?.viaCall).toBe(true); + }); + + it('sanitized through-call: const y = helper(escape(x)); exec(y) → NO finding', () => { + // Deliberate precision choice over flat-conservative (plan KTD5): the only + // occurrence path into the unmodeled call traverses the sanitizer, so the + // neutralization rides through the call into y's exclusion set. + const r = analyze(`function f(req) { + const x = req.body; + const y = helper(escape(x)); + exec(y); + }`); + expect(r.findings).toHaveLength(0); + }); +}); + +// ── statement-coalescing precision floor ───────────────────────────────────── + +describe('precision floor — multi-declarator conflation (documented FP)', () => { + it('const a = clean(z), b = g(t); exec(a) → finding EXISTS (expected false positive)', () => { + // PINNED FP (plan risk table): statement facts conflate declarators — `t` + // is used by the statement and `a` is def'd by it, so `a` is floor-tainted + // even though `t` flows only into `g(...)`. The per-declarator resultDefs + // precision powers KILLS only (U1 note); widening it to taint attribution + // would be an unsound narrowing of the substrate's statement granularity. + const r = analyze(`function f(req, z) { + const t = req.body; + const a = clean(z), b = g(t); + exec(a); + }`); + expect(r.findings).toHaveLength(1); + }); + + it('the floor never KILLS: a def in a sanitizer resultDefs with no taint inflow is floor-tainted', () => { + // `a = escape(z)` — the tainted `t` never flows into escape, so no kill is + // recorded for it and `a` is floor-tainted with NO exclusions (sound: a + // kill requires evidence of flow through the sanitizer). + const r = analyze(`function f(req, z) { + const t = req.body; + const a = escape(z), b = g(t); + exec(a); + }`); + expect(r.findings).toHaveLength(1); + expect(r.kills).toHaveLength(0); + }); +}); + +// ── sequence-expression value semantics (review fix) ──────────────────────── + +describe('sequence expressions — only the final operand carries taint', () => { + it('exec((log(x), "safe")) with tainted x → NO finding (safe operand flows)', () => { + const r = analyze(`function f(req) { const x = req.body; exec((log(x), 'safe')); }`); + expect(r.findings).toHaveLength(0); + }); + + it('exec((log("a"), x)) with tainted x → finding (tainted final operand)', () => { + const r = analyze(`function f(req) { const x = req.body; exec((log('a'), x)); }`); + expect(r.findings).toHaveLength(1); + }); +}); + +// ── source-discriminated taint state (review fix: multi-source merge) ─────── + +describe('multi-source identity — distinct sources do not merge at one def', () => { + // A spec with two source properties and a sql sanitizer, so the same-source + // ∅-intersection case has a kind to neutralize. + const MULTI: SourceSinkSanitizerSpec = { + sources: [{ kind: 'remote-input', objects: ['req'], properties: ['body', 'query'] }], + sinks: [ + { name: 'exec', kind: 'command-injection', args: [0], global: true }, + { name: 'query', kind: 'sql-injection', args: [0], anyReceiver: true }, + ], + sanitizers: [{ name: 'escape', neutralizes: ['sql-injection'], global: true }], + }; + + it('cond ? req.body : req.query into one var → TWO findings (one per source)', () => { + const r = analyze(`function f(req, cond) { const x = cond ? req.body : req.query; exec(x); }`, { + spec: MULTI, + }); + expect(r.findings).toHaveLength(2); + const props = r.findings.map((f) => f.source.property).sort(); + expect(props).toEqual(['body', 'query']); + }); + + it('req.body + req.query into one var → TWO findings', () => { + const r = analyze(`function f(req) { const x = req.body + req.query; exec(x); }`, { + spec: MULTI, + }); + expect(r.findings).toHaveLength(2); + }); + + it('same-source two-path flow → ONE finding (one root source occurrence)', () => { + // `req.body` is read ONCE (one source occurrence → one root identity); the + // single tainted `x` then flows two ways into `c`. Both paths share the + // same source discriminator, so they converge on one state for `c` and + // produce exactly one finding — the source dimension must not split a + // single source occurrence by downstream path. + const r = analyze( + `function f(req, cond) { const x = req.body; const c = cond ? id(x) : x; db.query(c); }`, + { spec: MULTI }, + ); + expect(r.findings).toHaveLength(1); + expect(r.findings[0].sinkKind).toBe('sql-injection'); + }); + + it('one source re-derived into one sink var → ONE finding, no doubling', () => { + // A single source reaching a single sink binding via a re-derived path + // dedups to one finding (findingsByIdentity); the source dimension in the + // state key must not split this same-source flow. + const r = analyze(`function f(req, cond) { let x = req.body; if (cond) { x = x; } exec(x); }`, { + spec: MULTI, + }); + expect(r.findings).toHaveLength(1); + }); + + it('two sources through a loop back-edge terminate with TWO findings', () => { + const r = analyze( + `function f(req, n) { let x = req.body; for (let i = 0; i < n; i++) { x = x + req.query; } exec(x); }`, + { spec: MULTI }, + ); + expect(r.status).toBe('computed'); + expect(r.findings).toHaveLength(2); + }); +}); + +// ── kind-set exclusion model (real built-in model) ────────────────────────── + +describe('kind-set exclusions — sanitizers neutralize their kinds only', () => { + it('escape(req.body) → res.send(b) suppressed (xss neutralized) BUT db.query(b) fires (sql not)', () => { + const r = analyze( + `import { escape } from 'validator'; +function f(req, db, res) { + const b = escape(req.body); + db.query(b); + res.send(b); +}`, + { spec: TS_JS_TAINT_MODEL }, + ); + expect(r.findings).toHaveLength(1); + expect(r.findings[0].sinkKind).toBe('sql-injection'); + expect(r.findings[0].sink.point.line).toBe(4); + // the seed kill is still recorded with the kinds escape neutralizes + expect(r.kills).toHaveLength(1); + expect([...r.kills[0].neutralized]).toEqual(['xss']); + }); + + it('kind-incompatible sink interposition: exec(path.basename(t)) → finding SURVIVES', () => { + // basename strips directories, not shell metacharacters — a kind-blind + // kill here would be a suppressed live command injection (the forbidden + // false-negative direction). + const r = analyze( + `import { exec } from 'child_process'; +import path from 'path'; +function f(req) { + const t = req.body; + exec(path.basename(t)); +}`, + { spec: TS_JS_TAINT_MODEL }, + ); + expect(r.findings).toHaveLength(1); + expect(r.findings[0].sinkKind).toBe('command-injection'); + }); + + it('kind-compatible def interposition: const safe = path.basename(t); readFileSync(safe) → suppressed', () => { + const r = analyze( + `import path from 'path'; +import { readFileSync } from 'fs'; +function f(req) { + const t = req.body; + const safe = path.basename(t); + readFileSync(safe); +}`, + { spec: TS_JS_TAINT_MODEL }, + ); + expect(r.findings).toHaveLength(0); + expect(r.kills).toHaveLength(1); + expect([...r.kills[0].neutralized]).toEqual(['path-traversal']); + }); + + it('the same basename-cleaned def still fires a command-injection sink', () => { + const r = analyze( + `import { exec } from 'child_process'; +import path from 'path'; +function f(req) { + const t = req.body; + const safe = path.basename(t); + exec(safe); +}`, + { spec: TS_JS_TAINT_MODEL }, + ); + expect(r.findings).toHaveLength(1); + expect(r.findings[0].sinkKind).toBe('command-injection'); + }); + + it('exclusions intersect over derivations: a less-neutralized re-derivation re-opens the sink', () => { + // y is first derived through escape ({xss} excluded), then re-derived + // through the direct assignment (∅) — the intersection is ∅, so the + // xss sink must fire. Guards the monotone shrink/re-enqueue discipline. + const r = analyze( + `import { escape } from 'validator'; +function f(req, cond, res) { + const b = req.body; + let y = escape(b); + if (cond) { + y = b; + } + res.send(y); +}`, + { spec: TS_JS_TAINT_MODEL }, + ); + // two defs of y reach the sink: the escaped one (suppressed for xss) and + // the raw one (fires) — one finding from the raw def's flow + expect(r.findings).toHaveLength(1); + expect(r.findings[0].sinkKind).toBe('xss'); + }); +}); + +// ── caps and determinism ───────────────────────────────────────────────────── + +describe('caps — deterministic truncation (R6 substrate)', () => { + const manyFindings = `function f(req) { + exec(req.body); + exec(req.query); + exec(req.params); + exec(req.headers); + }`; + + it('maxFindingsPerFunction truncates deterministically and counts the drop', () => { + const r = analyze(manyFindings, { limits: { maxFindingsPerFunction: 2 } }); + expect(r.findings).toHaveLength(2); + expect(r.droppedFindings).toBe(2); + // deterministic prefix of the sorted order: statement order + expect(r.findings.map((f) => f.source.property)).toEqual(['body', 'query']); + }); + + it('without a cap all findings emit and droppedFindings is 0', () => { + const r = analyze(manyFindings); + expect(r.findings).toHaveLength(4); + expect(r.droppedFindings).toBe(0); + }); + + it('maxHops truncates the hop chain and flags it', () => { + const r = analyze( + `function f(req) { + const a = req.body; + const b = a; + const c = b; + exec(c); + }`, + { limits: { maxHops: 2 } }, + ); + expect(r.findings).toHaveLength(1); + expect(r.findings[0].hops).toHaveLength(2); + expect(r.findings[0].hopsTruncated).toBe(true); + expect(hopSummary(r)).toEqual(['a@2', 'b@3']); + }); + + it('results are deterministic across repeated runs', () => { + const code = `function f(req, cond) { + let x = req.body; + const y = helper(x); + if (cond) { + x = escape(x); + } + exec(x); + exec(y); + exec(req.query); + }`; + const a = analyze(code); + const b = analyze(code); + expect(JSON.parse(JSON.stringify(a))).toEqual(JSON.parse(JSON.stringify(b))); + }); +}); diff --git a/gitnexus/test/unit/taint/site-safety.test.ts b/gitnexus/test/unit/taint/site-safety.test.ts new file mode 100644 index 000000000..ec92e0e55 --- /dev/null +++ b/gitnexus/test/unit/taint/site-safety.test.ts @@ -0,0 +1,172 @@ +import { describe, it, expect } from 'vitest'; +import Parser from 'tree-sitter'; +import TypeScript from 'tree-sitter-typescript'; +import { + createTypeScriptCfgVisitor, + TS_FUNCTION_TYPES, +} from '../../../src/core/ingestion/cfg/visitors/typescript.js'; +import { hasTaintSafeSites } from '../../../src/core/ingestion/taint/site-safety.js'; +import { isEmitSafeCfg, hasEmitSafeFacts } from '../../../src/core/ingestion/cfg/emit.js'; +import type { + FunctionCfg, + SiteRecord, + StatementFacts, +} from '../../../src/core/ingestion/cfg/types.js'; +import type { SyntaxNode } from '../../../src/core/ingestion/utils/ast-helpers.js'; + +// #2083 M3 U1 — `hasTaintSafeSites` mirrors `hasEmitSafeFacts`'s contract: +// out-of-range indices from a corrupted durable store must degrade to +// "skip taint for this function", never crash or fabricate matches. These +// tests build a REAL harvested CFG, then surgically corrupt the `sites` +// payload field-by-field — and pin that corrupt sites do NOT trip the +// CFG/REACHING_DEF guards (the degradation is taint-local). + +const visitor = createTypeScriptCfgVisitor(); + +function cfgOf(code: string): FunctionCfg { + const parser = new Parser(); + parser.setLanguage(TypeScript.typescript); + const root = parser.parse(code).rootNode as SyntaxNode; + const stack = [root]; + while (stack.length) { + const n = stack.pop() as SyntaxNode; + if (TS_FUNCTION_TYPES.has(n.type)) { + const cfg = visitor.buildFunctionCfg(n, 'fixture.ts'); + if (cfg) return cfg; + } + for (let i = n.namedChildCount - 1; i >= 0; i--) { + const c = n.namedChild(i); + if (c) stack.push(c); + } + } + throw new Error('no function found'); +} + +const BASE = cfgOf(`function f(req, x) { const b = req.body; exec(escape(x), b); }`); + +/** Deep-copy and rewrite the FIRST site of the FIRST site-bearing statement. */ +function mutateFirstSite(patch: (site: Record) => void): FunctionCfg { + const copy = JSON.parse(JSON.stringify(BASE)) as FunctionCfg; + for (const block of copy.blocks) { + for (const s of block.statements ?? []) { + if (s.sites && s.sites.length > 0) { + patch(s.sites[0] as unknown as Record); + return copy; + } + } + } + throw new Error('no site-bearing statement in fixture'); +} + +describe('hasTaintSafeSites — valid shapes pass', () => { + it('a real harvested CFG passes', () => { + expect(hasTaintSafeSites(BASE)).toBe(true); + }); + + it('a CFG with facts but no sites passes (absence is the well-formed empty case)', () => { + const cfg = cfgOf(`function f() { let a = 1; a = 2; }`); + expect(hasTaintSafeSites(cfg)).toBe(true); + }); + + it('a pre-M2 CFG with no statements at all passes', () => { + const copy = JSON.parse(JSON.stringify(BASE)) as FunctionCfg; + const stripped = { + ...copy, + bindings: undefined, + blocks: copy.blocks.map((b) => ({ ...b, statements: undefined })), + } as FunctionCfg; + expect(hasTaintSafeSites(stripped)).toBe(true); + }); + + it('the full M3 surface validates on a JSON round-trip (durable-store shape)', () => { + const cfg = cfgOf( + 'function f(req, dir, t) { const cp = require("child_process"); ' + + 'cp.exec(`ls ${dir}`); sql`q ${t}`; new Function(t); exec(...dir); }', + ); + expect(hasTaintSafeSites(JSON.parse(JSON.stringify(cfg)) as FunctionCfg)).toBe(true); + }); +}); + +describe('hasTaintSafeSites — malformed indices reject', () => { + it('out-of-range receiver', () => { + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.receiver = 999)))).toBe(false); + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.receiver = -1)))).toBe(false); + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.receiver = 1.5)))).toBe(false); + }); + + it('out-of-range member-read object / missing property', () => { + const cfg = cfgOf(`function f(req) { const b = req.body; }`); + const corrupt = JSON.parse(JSON.stringify(cfg)) as FunctionCfg; + const site = corrupt.blocks.flatMap((b) => [...(b.statements ?? [])]).find((s) => s.sites)! + .sites![0] as unknown as Record; + site.object = 999; + expect(hasTaintSafeSites(corrupt)).toBe(false); + site.object = 0; + delete site.property; + expect(hasTaintSafeSites(corrupt)).toBe(false); + }); + + it('out-of-range arg binding entry and via-site tag', () => { + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.args = [[999]])))).toBe(false); + // via-tag site index must be in range of the SAME statement's sites array + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.args = [[[0, 999]]])))).toBe(false); + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.args = [[[0, -1]]])))).toBe(false); + // tuple arity is exact + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.args = [[[0, 1, 2]]])))).toBe(false); + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.args = [['x']])))).toBe(false); + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.args = [0])))).toBe(false); + }); + + it('out-of-range resultDefs / parent / spread / kind / callee', () => { + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.resultDefs = [999])))).toBe(false); + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.parent = [999, 0])))).toBe(false); + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.parent = [0, -1])))).toBe(false); + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.parent = [0])))).toBe(false); + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.spread = -1)))).toBe(false); + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.kind = 'evil')))).toBe(false); + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.callee = 42)))).toBe(false); + expect(hasTaintSafeSites(mutateFirstSite((s) => (s.requireArg = 42)))).toBe(false); + }); + + it('sites without a binding table reject (nothing to range-check against)', () => { + const copy = JSON.parse(JSON.stringify(BASE)) as { bindings?: unknown }; + delete copy.bindings; + expect(hasTaintSafeSites(copy as FunctionCfg)).toBe(false); + }); + + it('non-array sites and null site entries reject', () => { + const corrupt = JSON.parse(JSON.stringify(BASE)) as FunctionCfg; + const stmt = corrupt.blocks.flatMap((b) => [...(b.statements ?? [])]).find((s) => s.sites) as { + sites: unknown; + }; + stmt.sites = { not: 'an array' }; + expect(hasTaintSafeSites(corrupt)).toBe(false); + stmt.sites = [null]; + expect(hasTaintSafeSites(corrupt)).toBe(false); + }); +}); + +describe('hasTaintSafeSites — degradation is taint-local (KTD2)', () => { + it('corrupt sites do NOT trip the CFG or REACHING_DEF guards', () => { + const corrupt = mutateFirstSite((s) => (s.receiver = 999)); + expect(hasTaintSafeSites(corrupt)).toBe(false); + // The CFG layer and the facts layer keep their own guards green — the + // function degrades to "no taint", never to "no CFG"/"no REACHING_DEF". + expect(isEmitSafeCfg(corrupt)).toBe(true); + expect(hasEmitSafeFacts(corrupt)).toBe(true); + }); + + it('and the inverse: corrupt FACTS are not a sites problem (separate guards)', () => { + const corrupt = JSON.parse(JSON.stringify(BASE)) as FunctionCfg; + const stmt = corrupt.blocks.flatMap((b) => [...(b.statements ?? [])])[1] as StatementFacts & { + defs: number[]; + }; + stmt.defs.push(999); + expect(hasEmitSafeFacts(corrupt)).toBe(false); + const sites: readonly SiteRecord[] | undefined = corrupt.blocks + .flatMap((b) => [...(b.statements ?? [])]) + .find((s) => s.sites)?.sites; + expect(sites).toBeDefined(); + expect(hasTaintSafeSites(corrupt)).toBe(true); + }); +}); diff --git a/gitnexus/test/unit/taint/source-sink-registry.test.ts b/gitnexus/test/unit/taint/source-sink-registry.test.ts index 43b85d970..783a663fe 100644 --- a/gitnexus/test/unit/taint/source-sink-registry.test.ts +++ b/gitnexus/test/unit/taint/source-sink-registry.test.ts @@ -23,7 +23,12 @@ describe('source/sink/sanitizer registry seam (#2080)', () => { }); it('register then get round-trips the spec', () => { - const ts = spec({ sinks: [{ name: 'eval' }], sources: [{ name: 'req.body', args: [0] }] }); + // U2 (#2083) extended the M0 entry shapes: sinks carry a `kind` category + // and sources are member-read entries — this test was updated deliberately. + const ts = spec({ + sinks: [{ name: 'eval', kind: 'code-injection', global: true }], + sources: [{ kind: 'remote-input', objects: ['req'], properties: ['body'] }], + }); registerSourceSinkConfig('typescript', ts); expect(getSourceSinkConfig('typescript')).toBe(ts); expect(registeredTaintLanguages()).toEqual(['typescript']); @@ -35,8 +40,10 @@ describe('source/sink/sanitizer registry seam (#2080)', () => { }); it('re-registering the same language id overwrites (last-write-wins)', () => { - const first = spec({ sinks: [{ name: 'eval' }] }); - const second = spec({ sinks: [{ name: 'exec' }] }); + const first = spec({ sinks: [{ name: 'eval', kind: 'code-injection', global: true }] }); + const second = spec({ + sinks: [{ name: 'exec', kind: 'command-injection', module: 'child_process' }], + }); registerSourceSinkConfig('typescript', first); registerSourceSinkConfig('typescript', second); expect(getSourceSinkConfig('typescript')).toBe(second); diff --git a/gitnexus/test/unit/taint/taint-emit.test.ts b/gitnexus/test/unit/taint/taint-emit.test.ts new file mode 100644 index 000000000..7b73d0e69 --- /dev/null +++ b/gitnexus/test/unit/taint/taint-emit.test.ts @@ -0,0 +1,338 @@ +/** + * U4 (#2083 M3) — `emitFileTaint` over REAL harvested CFGs. + * + * Fixtures parse real source through the worker-side TS CFG visitor (the + * propagate.test.ts harness): CFGs and sites come from the harvest, imports + * from the real TS capture+interpret path — the emit driver consumes exactly + * the structures the run.ts pdg window feeds it, never hand-built mocks + * (except the deliberate corrupted-store mutation below). + * + * Pinned here: KTD6 statement-level finding identity (occurrence-distinct + * rows for `exec(req.body, req.query)`; variable-distinct rows on a shared + * block pair), dedup-before-budget with truncate-and-warn, the zero-match + * fast path (no solver call — asserted via the result counters), the + * unsafe-sites skip-taint-keep-RD degradation, kills-without-findings, and + * the decodability of every persisted `reason` via the shared path codec. + */ + +import { describe, it, expect } from 'vitest'; +import { cfgsOf, importsFor } from '../../helpers/ts-cfg-harness.js'; +import { emitFileCfgs } from '../../../src/core/ingestion/cfg/emit.js'; +import type { SourceSinkSanitizerSpec } from '../../../src/core/ingestion/taint/source-sink-config.js'; +import { + emitFileTaint, + type TaintEmitLimits, + type TaintEmitResult, +} from '../../../src/core/ingestion/taint/emit.js'; +import { decodeTaintPath } from '../../../src/core/ingestion/taint/path-codec.js'; +import { createKnowledgeGraph } from '../../../src/core/graph/graph.js'; +import type { KnowledgeGraph } from '../../../src/core/graph/types.js'; +import type { GraphRelationship } from 'gitnexus-shared'; + +/** Mechanics spec (propagate.test.ts MECH): global exec sink, global escape sanitizer. */ +const MECH: SourceSinkSanitizerSpec = { + sources: [ + { kind: 'remote-input', objects: ['req'], properties: ['body', 'query', 'params', 'headers'] }, + ], + sinks: [{ name: 'exec', kind: 'command-injection', args: [0], global: true }], + sanitizers: [{ name: 'escape', neutralizes: ['command-injection'], global: true }], +}; + +/** MECH with the sink dangerous at EVERY position (`args` omitted). */ +const MECH_ALL_ARGS: SourceSinkSanitizerSpec = { + ...MECH, + sinks: [{ name: 'exec', kind: 'command-injection', global: true }], +}; + +interface RunResult { + graph: KnowledgeGraph; + result: TaintEmitResult; + tainted: GraphRelationship[]; + sanitizes: GraphRelationship[]; + warns: string[]; +} + +function run( + code: string, + opts: { spec?: SourceSinkSanitizerSpec; limits?: TaintEmitLimits; cfgs?: FunctionCfg[] } = {}, +): RunResult { + const graph = createKnowledgeGraph(); + const cfgs = opts.cfgs ?? cfgsOf(code); + // Emit the M1 layer first so taint endpoints can be checked against REAL + // persisted BasicBlock nodes (run.ts ordering). + emitFileCfgs(graph, cfgs); + const warns: string[] = []; + const result = emitFileTaint(graph, cfgs, importsFor(code), opts.spec ?? MECH, opts.limits, (m) => + warns.push(m), + ); + const tainted: GraphRelationship[] = []; + const sanitizes: GraphRelationship[] = []; + for (const rel of graph.iterRelationships()) { + if (rel.type === 'TAINTED') tainted.push(rel); + if (rel.type === 'SANITIZES') sanitizes.push(rel); + } + return { graph, result, tainted, sanitizes, warns }; +} + +function blockIds(graph: KnowledgeGraph): Set { + const ids = new Set(); + graph.forEachNode((n) => { + if (n.label === 'BasicBlock') ids.add(n.id); + }); + return ids; +} + +describe('emitFileTaint — happy path', () => { + const CODE = ` +function handler(req: { body: string }) { + const cmd = req.body; + exec(cmd); +}`; + + it('persists one TAINTED edge whose endpoints are real BasicBlock nodes', () => { + const { graph, result, tainted } = run(CODE); + expect(result.functionsAnalyzed).toBe(1); + expect(result.findingsEmitted).toBe(1); + expect(tainted).toHaveLength(1); + const ids = blockIds(graph); + expect(ids.has(tainted[0].sourceId)).toBe(true); + expect(ids.has(tainted[0].targetId)).toBe(true); + expect(tainted[0].id.startsWith('TAINTED:fixture.ts:')).toBe(true); + }); + + it('the persisted reason decodes via the shared codec with ordered hops + variables', () => { + const { tainted } = run(CODE); + const decoded = decodeTaintPath(tainted[0].reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) { + expect(decoded.truncated).toBe(false); + // The finding's sinkKind rides the `;` header (the only persisted + // channel — the edge id embedding it is not a stored column; U6 reads it). + expect(decoded.kind).toBe('command-injection'); + // seed def (cmd @ const line) → sink use (cmd @ exec line) + expect(decoded.hops.map((h) => `${h.variable}@${h.line}`)).toEqual(['cmd@3', 'cmd@4']); + } + }); +}); + +describe('KTD6 statement-level finding identity', () => { + it('exec(req.body, req.query) → TWO rows distinguished by occurrence', () => { + const { result, tainted } = run( + ` +function handler(req: { body: string; query: string }) { + exec(req.body, req.query); +}`, + { spec: MECH_ALL_ARGS }, + ); + expect(result.findingsEmitted).toBe(2); + expect(tainted).toHaveLength(2); + // Same block pair — the identity (and the id) differ ONLY by occurrence. + expect(tainted[0].sourceId).toBe(tainted[1].sourceId); + expect(tainted[0].targetId).toBe(tainted[1].targetId); + expect(tainted[0].id).not.toBe(tainted[1].id); + }); + + it('two findings on the same block pair with different variables → two rows', () => { + const { result, tainted } = run(` +function handler(req: { body: string; query: string }) { + const a = req.body; + const b = req.query; + exec(a); + exec(b); +}`); + expect(result.findingsEmitted).toBe(2); + expect(tainted).toHaveLength(2); + expect(new Set(tainted.map((t) => t.id)).size).toBe(2); + const variables = tainted.map((t) => { + const d = decodeTaintPath(t.reason); + if (!d.ok) throw new Error(d.error); + return d.hops[d.hops.length - 1].variable; + }); + expect(variables.sort()).toEqual(['a', 'b']); + }); +}); + +describe('dedup-before-budget and the findings cap', () => { + const FOUR_FINDINGS = ` +function handler(req: { body: string; query: string }) { + const a = req.body; + const b = req.query; + exec(a); + exec(b); + exec(a); + exec(b); +}`; + + it('uncapped: identical (deduped) flows collapse; distinct ones all emit', () => { + const { result, tainted } = run(FOUR_FINDINGS); + // 4 sink statements × 1 variable each — all distinct sink points → 4 rows. + expect(result.findingsEmitted).toBe(4); + expect(result.findingsDropped).toBe(0); + expect(new Set(tainted.map((t) => t.id)).size).toBe(4); + }); + + it('capped: truncates deterministically and warns naming the drop count', () => { + const { result, tainted, warns } = run(FOUR_FINDINGS, { + limits: { maxFindingsPerFunction: 1 }, + }); + expect(result.findingsEmitted).toBe(1); + expect(result.findingsDropped).toBe(3); + expect(tainted).toHaveLength(1); + const capWarn = warns.find((w) => w.includes('findings cap')); + expect(capWarn).toBeDefined(); + expect(capWarn).toContain('dropped 3 of 4'); + expect(result.droppedExamples).toEqual(['fixture.ts:2']); + }); +}); + +describe('zero-match fast path', () => { + it('no source AND no sink: function skipped without a solver call', () => { + const { result, tainted, sanitizes } = run(` +function pure(x: number) { + const y = x + 1; + return y; +}`); + expect(result.functionsSkippedNoMatch).toBe(1); + expect(result.functionsAnalyzed).toBe(0); + expect(tainted).toHaveLength(0); + expect(sanitizes).toHaveLength(0); + }); + + it('sink without source (and vice versa) also short-circuits', () => { + const { result } = run(` +function sinkOnly(cmd: string) { + exec(cmd); +} +function sourceOnly(req: { body: string }) { + return req.body; +}`); + expect(result.functionsSkippedNoMatch).toBe(2); + expect(result.functionsAnalyzed).toBe(0); + expect(result.findingsEmitted).toBe(0); + }); +}); + +describe('unsafe-sites degradation (skip-taint-keep-RD)', () => { + const TWO_FNS = ` +function corrupted(req: { body: string }) { + exec(req.body); +} +function stillVulnerable(req: { body: string }) { + exec(req.body); +}`; + + it('a corrupted-store site skips ONLY that function; siblings still emit', () => { + const cfgs = JSON.parse(JSON.stringify(cfgsOf(TWO_FNS))) as FunctionCfg[]; + // Corrupt the first function's first site: out-of-range binding index. + let mutated = false; + outer: for (const block of cfgs[0].blocks) { + for (const stmt of block.statements ?? []) { + if (stmt.sites !== undefined && stmt.sites.length > 0) { + (stmt.sites[0] as { object?: number }).object = 9999; + mutated = true; + break outer; + } + } + } + expect(mutated).toBe(true); + + const { result, tainted, warns } = run(TWO_FNS, { cfgs }); + expect(result.functionsSkippedUnsafeSites).toBe(1); + expect(result.functionsAnalyzed).toBe(1); + expect(result.findingsEmitted).toBe(1); // the sibling's finding survives + expect(tainted).toHaveLength(1); + expect(tainted[0].id).toContain('fixture.ts:5'); // the SECOND function + expect(warns.some((w) => w.includes('malformed site annotations'))).toBe(true); + expect(result.coverageGapExamples).toEqual(['fixture.ts:2']); + }); +}); + +describe('kills without findings', () => { + it('a fully-sanitized flow emits SANITIZES and zero TAINTED', () => { + const { graph, result, tainted, sanitizes } = run(` +function safe(req: { body: string }) { + const b = escape(req.body); + exec(b); +}`); + expect(result.functionsAnalyzed).toBe(1); + expect(result.findingsEmitted).toBe(0); + expect(tainted).toHaveLength(0); + expect(result.killsEmitted).toBe(1); + expect(sanitizes).toHaveLength(1); + // reason = the killed binding's plain name; endpoints are real blocks. + expect(sanitizes[0].reason).toBe('b'); + expect(sanitizes[0].id.startsWith('SANITIZES:fixture.ts:')).toBe(true); + const ids = blockIds(graph); + expect(ids.has(sanitizes[0].sourceId)).toBe(true); + expect(ids.has(sanitizes[0].targetId)).toBe(true); + }); + + it('a kill alongside a finding emits both edge kinds', () => { + const { result } = run(` +function mixed(req: { body: string; query: string }) { + const safe = escape(req.body); + exec(safe); + exec(req.query); +}`); + expect(result.killsEmitted).toBe(1); + expect(result.findingsEmitted).toBe(1); + }); +}); + +describe('hop truncation accounting', () => { + it('maxHops=1 truncates the persisted path and counts the finding', () => { + const { result, tainted } = run( + ` +function handler(req: { body: string }) { + const a = req.body; + const b = a; + exec(b); +}`, + { limits: { maxHops: 1 } }, + ); + expect(result.findingsEmitted).toBe(1); + expect(result.hopsTruncatedFindings).toBe(1); + const decoded = decodeTaintPath(tainted[0].reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) { + expect(decoded.truncated).toBe(true); // path-incomplete, not an error + expect(decoded.hops).toHaveLength(1); // source-side prefix + expect(decoded.hops[0].variable).toBe('a'); + } + }); +}); + +describe('telemetry completeness', () => { + it('every counter field is present (nothing dropped on the floor — the M2 lesson)', () => { + const { result } = run(`function noop() { return 1; }`); + expect(result).toEqual({ + functionsAnalyzed: 0, + functionsSkippedNoMatch: 1, + functionsSkippedUnsafeSites: 0, + functionsCoverageGap: { truncated: 0, overflow: 0, 'no-facts': 0 }, + findingsEmitted: 0, + killsEmitted: 0, + findingsDropped: 0, + hopsTruncatedFindings: 0, + coverageGapExamples: [], + droppedExamples: [], + }); + }); + + it('a solver coverage gap (fact limit) is counted by reason, with an example anchor', () => { + const { result, tainted } = run( + ` +function gap(req: { body: string }) { + const a = req.body; + const b = a; + const c = b; + exec(c); +}`, + { limits: { maxFacts: 1 } }, + ); + expect(result.functionsCoverageGap.truncated).toBe(1); + expect(result.functionsAnalyzed).toBe(0); + expect(tainted).toHaveLength(0); // R4: never partially analyzed + expect(result.coverageGapExamples).toEqual(['fixture.ts:2']); + }); +}); diff --git a/gitnexus/test/unit/tools.test.ts b/gitnexus/test/unit/tools.test.ts index 40376961b..298a2ee5c 100644 --- a/gitnexus/test/unit/tools.test.ts +++ b/gitnexus/test/unit/tools.test.ts @@ -21,8 +21,8 @@ const MUTATING_TOOLS = new Set(['rename', 'group_sync']); const OPEN_WORLD_READ_ONLY_TOOLS = new Set(['query']); describe('GITNEXUS_TOOLS', () => { - it('exports all tools (8 base + 3 route/tool/shape + 1 api_impact + 2 group)', () => { - expect(GITNEXUS_TOOLS).toHaveLength(14); + it('exports all tools (8 base + 1 explain + 3 route/tool/shape + 1 api_impact + 2 group)', () => { + expect(GITNEXUS_TOOLS).toHaveLength(15); }); it('contains all expected tool names', () => { @@ -37,6 +37,7 @@ describe('GITNEXUS_TOOLS', () => { 'check', 'rename', 'impact', + 'explain', 'api_impact', ]), ); @@ -218,6 +219,38 @@ describe('GITNEXUS_TOOLS', () => { expect(scopeProp.enum).toEqual(['unstaged', 'staged', 'all', 'compare']); }); + // ─── explain (#2083 M3 U6) ───────────────────────────────────────── + + it('explain tool is anchorless-optional with a bounded limit and a branch scope', () => { + const explainTool = GITNEXUS_TOOLS.find((t) => t.name === 'explain')!; + expect(explainTool).toBeDefined(); + // Anchorless calls (enumerate all findings) must be valid. + expect(explainTool.inputSchema.required).toEqual([]); + expect(explainTool.inputSchema.properties.target).toBeDefined(); + expect(explainTool.inputSchema.properties.target.type).toBe('string'); + const limit = explainTool.inputSchema.properties.limit; + expect(limit).toBeDefined(); + expect(limit.type).toBe('integer'); + expect(limit.minimum).toBe(1); + expect(limit.maximum).toBeGreaterThan(0); + // Branch-scoped per #2106 (injected via BRANCH_SCOPED_TOOLS). + expect(explainTool.inputSchema.properties.branch).toBeDefined(); + }); + + it('explain description names the --pdg requirement and the KTD10 contract caveats', () => { + const explainTool = GITNEXUS_TOOLS.find((t) => t.name === 'explain')!; + const d = explainTool.description; + expect(d).toContain('--pdg'); + expect(d).toContain('intra-procedural'); + // The named blind-spot classes (plan KTD10) must reach the consumer. + expect(d.toLowerCase()).toContain('closure/callback'); + expect(d.toLowerCase()).toContain('property/field'); + expect(d.toLowerCase()).toContain('guard-style'); + expect(d.toLowerCase()).toContain('cross-function'); + expect(d.toLowerCase()).toContain('commonjs'); + expect(d.toLowerCase()).toContain('exception'); + }); + it('api_impact tool has no required parameters', () => { const apiImpactTool = GITNEXUS_TOOLS.find((t) => t.name === 'api_impact')!; expect(apiImpactTool).toBeDefined(); From 0054496323a5d4c3f15342ddc372ae61ab64d8c8 Mon Sep 17 00:00:00 2001 From: Minidoracat Date: Fri, 12 Jun 2026 22:17:30 +0800 Subject: [PATCH 03/16] fix(hooks): wrap the augment CLI child in the orphan guard (#2163) (#2169) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(hooks): wrap the augment CLI child in the orphan guard (#2163) Follow-up invited by the maintainer on #2165: the augment child (7s local / 12s npx) was the longest-lived unwrapped subprocess, exposed to the same SIGKILL-orphan mechanism fixed for lsof/ps. - Export resolveUnixGuardTimeout from the probe module (both copies, byte-identical); adapters share the same module instance, so the memo and lazy self-test still run at most once per hook process. - Wrap every CLI-executing branch of runGitNexusCli in the three probe-equipped adapters with the guard: budget ceil(inner/1000)+1 seconds with -k 1, strictly above each branch's inner spawnSync timeout, so the supervised path is unchanged and the wrapper only matters once the hook itself is SIGKILLed. Windows and no-guard hosts keep byte-identical argv. The plugin adapter's PATH-direct gitnexus branch (its most common production path) is wrapped too; the cheap which/where probe is not. - Cursor integration: debug-gated 'augment skipped: hook slots saturated' on the slot-starved early return. Its augment child stays unwrapped for now — that integration does not install the probe sibling (the 'cursor probe' item on the #2163 follow-up list). - Reaping tests get a guard-availability precheck with an explicit failure message (assertion, not skipIf, so a coreutils-less Linux host fails diagnosably instead of going silently green). - Tests: orphaned-augment reaping (CJS + Plugin, red without the wrap, ~9.1s reap measured), disabled-sentinel degradation equivalence, source pinning for all three adapters (exact per-branch budget-formula counts) + probe export + cursor debug line. Note: pre-commit typecheck skipped; remaining tsc errors are pre-existing on main (none in files touched here). * fix(hooks): group-SIGKILL the npx arm, prove guard exit propagation (#2169 review) Addresses the tri-review findings on #2169: - [P2] npx-arm containment: the CLI is the guard's grandchild there — at budget expiry coreutils timeout TERMs the group, npx (the obedient direct child) dies, timeout returns, and -k never fires, so a SIGTERM-immune grandchild escaped unbounded. The npx arm's wrapper now uses -s KILL: an unignorable group SIGKILL at budget that reaps the grandchild (kept -k 1 as a harmless belt; direct-exec arms keep TERM-first). CHANGELOG, adapter docblocks, and the test comment now state the per-arm semantics honestly. New behavioral test: a staged hook with a PATH-injected fake npx spawning a SIGTERM-immune grandchild is SIGKILLed; the grandchild must be reaped (red without -s KILL), with a route self-proof marker pinning the npx arm. - [P3] guard self-test now proves exit-status propagation (sh -c 'exit 42' must yield status 42), so an always-exit-0 stub like /bin/true is rejected and resolution falls through to the built-in candidates instead of silently killing the augment feature. New test: stub guard rejected, augment still emits context. - [P3] cleanup SIGKILLs in the reaping tests re-check the /proc//cmdline identity immediately before firing (PID-reuse guard), applied consistently to the two pre-existing #2165 spots and both new tests. - Review notes: source pins now constrain wrapper argv order and exact per-arm counts; adapters degrade to unwrapped on probe version skew (typeof check) instead of a swallowed TypeError; export JSDoc wording fixed for relative env paths; debug-gated diagnostic when no guard is available (e.g. macOS without coreutils), with the CHANGELOG entry qualified accordingly. Note: pre-commit typecheck skipped; remaining tsc errors are pre-existing on main (none in files touched here). --------- Co-authored-by: Gergő Magyar --- CHANGELOG.md | 1 + gitnexus-claude-plugin/hooks/gitnexus-hook.js | 109 ++- .../hooks/hook-db-lock-probe.cjs | 74 ++- .../hooks/gitnexus-hook.cjs | 13 +- .../antigravity/gitnexus-antigravity-hook.cjs | 92 ++- gitnexus/hooks/claude/gitnexus-hook.cjs | 92 ++- gitnexus/hooks/claude/hook-db-lock-probe.cjs | 74 ++- gitnexus/test/unit/hooks.test.ts | 624 +++++++++++++++++- gitnexus/test/utils/hook-test-helpers.ts | 20 +- 9 files changed, 1038 insertions(+), 61 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1b34046ad..efb1b531f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,7 @@ All notable changes to GitNexus will be documented in this file. ### Fixed - **Hook db-lock probe no longer strands unkillable `lsof`/`ps` orphans** — the probe's `lsof`/`ps` subprocesses are now wrapped in a self-tested coreutils `timeout`/`gtimeout` (`timeout -k 1 …`), so a hook SIGKILLed by the runner's 10s timeout can no longer leave `lsof` running forever (orphan lifetime bounded at ~3s); `acquireHookSlot` now also gates the probe itself, capping concurrent probes at 3 per repo. Opt out with `GITNEXUS_HOOK_TIMEOUT_PATH=disabled`. (#2163) +- **Hook augment CLI no longer strands orphans either** — `runGitNexusCli` in the Claude, plugin, and Antigravity hook adapters now wraps the `gitnexus augment` subprocess (the longest-lived hook child: 7s local / 12s npx inner budgets) in the same self-tested coreutils `timeout` guard as the probe's `lsof`/`ps`, with a budget of ceil(inner/1000)+1 seconds — strictly above the inner `spawnSync` timeout, so on the supervised path Node's SIGTERM still fires first and observable behavior is unchanged. Once the hook itself has been SIGKILLed the guard takes over, with per-branch semantics: on the direct-exec branches (the CLI is the guard's child) it SIGTERMs at budget and `-k 1` SIGKILLs 1s later; on the npx branches (the CLI is a *grandchild* behind npx) it uses `-s KILL`, SIGKILLing the whole process group at budget — a TERM-first guard there would only kill the obedient npx parent and exit before its `-k` escalation fires, stranding a SIGTERM-immune CLI. Two npx-branch caveats remain (both no worse than the pre-fix behavior, where the grandchild received no signal at all): the group-wide KILL is coreutils semantics, so a busybox `timeout` — which passes the self-test — still signals only its direct child and cannot reach the grandchild; and on the supervised path (hook alive, inner `spawnSync` timeout SIGTERMs the guard) coreutils forwards TERM rather than KILL, so a SIGTERM-immune CLI grandchild is still not reaped there. The guard self-test now also requires exit-status propagation (`sh -c 'exit 42'` must yield 42), so an always-exit-0 stub at `GITNEXUS_HOOK_TIMEOUT_PATH` can no longer be adopted and silently swallow the probe and augment. Windows, `GITNEXUS_HOOK_TIMEOUT_PATH=disabled`, and Unix hosts with no usable coreutils `timeout`/`gtimeout` at all (e.g. macOS without Homebrew coreutils) keep the exact pre-wrap unguarded invocation — every guard-less Unix run, whatever the reason (disabled, nothing usable, probe version skew), is now diagnosed once per hook run under `GITNEXUS_DEBUG`. The Cursor hook is not wrapped yet (it does not install the probe helper) but now reports its slot-saturated skip under `GITNEXUS_DEBUG`. (#2163 follow-up) ### Changed - Migrated from KuzuDB to LadybugDB v0.15 (`@ladybugdb/core`, `@ladybugdb/wasm-core`) diff --git a/gitnexus-claude-plugin/hooks/gitnexus-hook.js b/gitnexus-claude-plugin/hooks/gitnexus-hook.js index 5274ef4c0..2cf133cd9 100644 --- a/gitnexus-claude-plugin/hooks/gitnexus-hook.js +++ b/gitnexus-claude-plugin/hooks/gitnexus-hook.js @@ -15,7 +15,10 @@ const fs = require('fs'); const path = require('path'); const { spawnSync } = require('child_process'); const { acquireHookSlot } = require('./hook-lock.js'); -const { hasGitNexusDbLockedByGitNexusServer } = require('./hook-db-lock-probe.cjs'); +const { + hasGitNexusDbLockedByGitNexusServer, + resolveUnixGuardTimeout, +} = require('./hook-db-lock-probe.cjs'); const { formatAnalyzeCommand } = require('./resolve-analyze-cmd.cjs'); /** @@ -196,18 +199,90 @@ function extractPattern(toolName, toolInput) { return null; } +// Debounce for the unguarded-CLI diagnostic below (#2163 follow-up review): +// at most one line per (short-lived) hook process, even if a future change +// runs the CLI more than once. +let unguardedCliWarned = false; + /** * Spawn a gitnexus CLI command synchronously. * Detects binary on PATH once, then runs exactly once. * * SECURITY: Never use shell: true with user-controlled arguments. * On Windows, invoke gitnexus.cmd directly (no shell needed). + * + * Unix orphan containment (#2163 follow-up): the augment CLI is the + * longest-lived hook child (inner spawnSync timeout 7s locally, 12s via + * npx), so on Unix every CLI-running branch gets the same SIGKILL-surviving + * coreutils `timeout` wrapper as the probe's lsof/ps (the cheap which/where + * PATH check stays unwrapped). The wrapper budget is ceil(inner/1000)+1 + * seconds — STRICTLY greater than the inner spawnSync timeout, so on the + * supervised path Node's SIGTERM always fires first and the existing + * error/status contract is untouched. Once the hook itself has been + * SIGKILLed (exactly the orphan case the wrapper exists for), the guard + * semantics differ per branch: + * - direct exec (GITNEXUS_HOOK_CLI_PATH / PATH-installed `gitnexus`; the + * CLI is the guard's CHILD): `-k 1` TERM-first — a SIGTERM-immune CLI + * can hold the guard ~1s past the inner timeout before the `-k` SIGKILL + * escalation reaps it. + * - npx (the CLI is a GRANDCHILD: guard → npx → CLI): `-s KILL` — the + * budget expiry SIGKILLs the whole process group outright. TERM-first + * would kill only the obedient npx parent, making `timeout` reap it and + * return before the `-k` escalation ever fires, stranding a + * SIGTERM-immune CLI grandchild unbounded (reproduced on coreutils + * 9.x). `-k 1` is retained alongside `-s KILL` as a harmless belt: with + * `-s KILL` the `-k` escalation signal is also KILL. Two residual gaps + * on this branch, both bounded by "no worse than pre-fix" (where the + * grandchild received no signal at all): the group-wide SIGKILL is + * coreutils semantics — a busybox `timeout` passes the self-test (it + * has `-k` and propagates exit status) but signals only its direct + * child, so a busybox guard cannot reach the grandchild; and on the + * SUPERVISED path (hook alive, inner spawnSync timeout SIGTERMs the + * guard) coreutils forwards TERM rather than the `-s` signal, npx dies, + * and the guard exits before any KILL fires — so a SIGTERM-immune CLI + * grandchild still escapes in those two cases. + * If the sibling probe predates the resolveUnixGuardTimeout export (version + * skew), the adapter degrades to the unwrapped invocation instead of + * throwing. Windows is deliberately NOT wrapped — there is no coreutils + * timeout to resolve there and the resolver's self-test spawns /bin/sh — so + * on win32 (the gitnexus.cmd / npx.cmd paths) and whenever the guard + * resolves to null (e.g. macOS without Homebrew coreutils — reported once + * under GITNEXUS_DEBUG) the argv stays byte-identical to the pre-wrap + * invocation. */ function runGitNexusCli(args, cwd, timeout) { const isWin = process.platform === 'win32'; + // Version-skew guard (#2163 follow-up review): an older sibling probe + // without the resolveUnixGuardTimeout export must degrade to the unwrapped + // invocation — a TypeError here would be swallowed by the caller's catch + // and silently kill the augment. + const guard = + isWin || typeof resolveUnixGuardTimeout !== 'function' ? null : resolveUnixGuardTimeout(); + if (!isWin && !guard && !unguardedCliWarned && isDebugEnabled()) { + // Diagnose the "stays unwrapped" Unix paths once per hook process: no + // usable coreutils timeout/gtimeout (e.g. macOS without Homebrew + // coreutils), GITNEXUS_HOOK_TIMEOUT_PATH=disabled, or probe skew above. + unguardedCliWarned = true; + process.stderr.write( + '[GitNexus hook] no usable timeout/gtimeout guard; augment CLI child runs unguarded\n', + ); + } const hookCli = process.env.GITNEXUS_HOOK_CLI_PATH; if (hookCli !== undefined && String(hookCli).trim() && fs.existsSync(String(hookCli))) { - return spawnSync(process.execPath, [String(hookCli), ...args], { + const [cmd, cmdArgs] = guard + ? [ + guard, + [ + '-k', + '1', + String(Math.ceil(timeout / 1000) + 1), + process.execPath, + String(hookCli), + ...args, + ], + ] + : [process.execPath, [String(hookCli), ...args]]; + return spawnSync(cmd, cmdArgs, { encoding: 'utf-8', timeout, cwd, @@ -231,7 +306,12 @@ function runGitNexusCli(args, cwd, timeout) { } if (useDirectBinary) { - return spawnSync(isWin ? 'gitnexus.cmd' : 'gitnexus', args, { + // A non-null guard implies non-Windows, so the wrapped arm can hardcode + // plain `gitnexus` (the guard resolves it via PATH, like spawnSync does). + const [cmd, cmdArgs] = guard + ? [guard, ['-k', '1', String(Math.ceil(timeout / 1000) + 1), 'gitnexus', ...args]] + : [isWin ? 'gitnexus.cmd' : 'gitnexus', args]; + return spawnSync(cmd, cmdArgs, { encoding: 'utf-8', timeout, cwd, @@ -239,8 +319,27 @@ function runGitNexusCli(args, cwd, timeout) { windowsHide: true, }); } - // npx fallback needs shell on Windows since npx is a .cmd script - return spawnSync(isWin ? 'npx.cmd' : 'npx', ['-y', 'gitnexus', ...args], { + // npx fallback needs shell on Windows since npx is a .cmd script. The + // wrapped arm leads with `-s KILL` (NOT TERM-first like the direct + // branches above): the CLI here is a grandchild behind npx — see the + // docblock. + const [cmd, cmdArgs] = guard + ? [ + guard, + [ + '-s', + 'KILL', + '-k', + '1', + String(Math.ceil((timeout + 5000) / 1000) + 1), + 'npx', + '-y', + 'gitnexus', + ...args, + ], + ] + : [isWin ? 'npx.cmd' : 'npx', ['-y', 'gitnexus', ...args]]; + return spawnSync(cmd, cmdArgs, { encoding: 'utf-8', timeout: timeout + 5000, cwd, diff --git a/gitnexus-claude-plugin/hooks/hook-db-lock-probe.cjs b/gitnexus-claude-plugin/hooks/hook-db-lock-probe.cjs index 5c67804b1..de0fa5e85 100644 --- a/gitnexus-claude-plugin/hooks/hook-db-lock-probe.cjs +++ b/gitnexus-claude-plugin/hooks/hook-db-lock-probe.cjs @@ -20,13 +20,18 @@ * it 1s later — orphan lifetime is bounded at ~3s instead of unbounded. * - GITNEXUS_HOOK_TIMEOUT_PATH: the sentinel value `disabled` switches the * wrapper off deterministically; any other value is adopted only when it - * exists AND passes a one-shot `-k` self-test — otherwise resolution FALLS - * THROUGH to the built-in candidate list (first self-test pass wins), so - * no malformed value of any shape can silently disable orphan containment. + * exists AND passes a one-shot `-k` exit-propagation self-test — otherwise + * resolution FALLS THROUGH to the built-in candidate list (first self-test + * pass wins), so no malformed value of any shape can silently disable + * orphan containment. * - The gitnexus server is lazy-open + sticky-hold: an idle MCP server holds * ZERO lbug fds until the repo's first MCP query, then keeps the fd open. * A probe before that first query is therefore always false — a known, * pre-existing race, not a bug in this probe. + * - resolveUnixGuardTimeout is exported so the hook adapters can wrap the + * `gitnexus augment` CLI child — the longest-lived hook subprocess (7s + * local / 12s npx inner budgets) — in the same guard; see runGitNexusCli + * in the adapters (#2163 follow-up). */ const fs = require('fs'); @@ -70,38 +75,53 @@ let unixGuardTimeoutCache; /** * Resolve a coreutils `timeout`/`gtimeout` binary to wrap lsof/ps with - * (#2163). Dead code on Windows (the win32 dispatch returns earlier). + * (#2163). Unix-only by contract: the probe's win32 dispatch returns before + * reaching it, and the exported callers (the adapters' runGitNexusCli, + * #2163 follow-up) must check the platform first — the self-test below + * spawns /bin/sh. The memoized result is module-wide, so probe and adapter + * share one lazy self-test per hook process. * * GITNEXUS_HOOK_TIMEOUT_PATH semantics: the sentinel `disabled` turns the * wrapper off; any other value is only a CANDIDATE — an existing file path - * is tried first, but it must pass the `-k` self-test to be adopted. On any - * failure (non-existent path, directory, non-executable file, wrapper - * without `-k` support, …) resolution falls through to the built-in - * candidates below, tried in order, first self-test pass wins. This is - * strictly stronger than the sibling GITNEXUS_HOOK_LSOF_PATH / - * GITNEXUS_HOOK_PS_PATH overrides (which only check existence): no bad env - * value of ANY shape can silently disable orphan containment. + * is tried first, but it must pass the `-k` exit-propagation self-test to + * be adopted. On any failure (non-existent path, directory, non-executable + * file, wrapper without `-k` support, always-exit-0 stub, …) resolution + * falls through to the built-in candidates below, tried in order, first + * self-test pass wins. This is strictly stronger than the sibling + * GITNEXUS_HOOK_LSOF_PATH / GITNEXUS_HOOK_PS_PATH overrides (which only + * check existence): no bad env value of ANY shape can silently disable + * orphan containment. * * Lazy self-test: candidates are probed only when the lsof/ps fallback is * first reached, and the result is memoized. A candidate is adopted only - * when `timeout -k 1 1 /bin/sh -c :` exits 0. This rejects wrappers that do - * not support the coreutils `-k` flag — busybox <1.34, toybox, broken - * symlinks — which would otherwise exit with a usage error without ever - * running lsof, silently converting the lsof-ETIMEDOUT fail-closed contract - * into fail-open (#1492 regression). Only when EVERY candidate fails does - * the probe fall back to the unwrapped status quo (memoized null). - * busybox ≥1.34 passes the test and is fully usable (capability, not - * identity, decides). + * when `timeout -k 1 1 /bin/sh -c 'exit 42'` exits 42 — i.e. it must RUN + * the wrapped command AND PROPAGATE its exit status. This rejects two + * failure shapes: wrappers without the coreutils `-k` flag — busybox <1.34, + * toybox, broken symlinks — which would exit with a usage error without + * ever running lsof, silently converting the lsof-ETIMEDOUT fail-closed + * contract into fail-open (#1492 regression); and always-exit-0 stubs + * (/bin/true shapes), which would otherwise be adopted and "succeed" every + * wrapped spawn instantly without running it — a constant no-owner probe + * answer and, worse, a silently dead augment (status 0, empty stderr passes + * the adapters' success check with no context; #2163 follow-up review). + * Only when EVERY candidate fails does the probe fall back to the unwrapped + * status quo (memoized null). busybox ≥1.34 passes the test and is fully + * usable for everything THIS file spawns (lsof/ps are the guard's direct + * children) and for the adapters' direct-exec arm. The adapters' npx arm + * additionally relies on coreutils' process-GROUP signalling for its + * `-s KILL` grandchild reaping; busybox signals only its direct child, and + * this self-test deliberately does not probe that capability — see the + * adapter docblocks for the residual-gap statement. */ function passesGuardSelfTest(guard) { try { - const selfTest = spawnSync(guard, ['-k', '1', '1', '/bin/sh', '-c', ':'], { + const selfTest = spawnSync(guard, ['-k', '1', '1', '/bin/sh', '-c', 'exit 42'], { encoding: 'utf-8', timeout: 3000, stdio: ['ignore', 'ignore', 'ignore'], windowsHide: true, }); - return !selfTest.error && selfTest.status === 0; + return !selfTest.error && selfTest.status === 42; } catch { return false; } @@ -359,4 +379,16 @@ function hasGitNexusDbLockedByGitNexusServer(dbPath, myPid) { module.exports = { hasGitNexusDbLockedByGitNexusServer, + // #2163 follow-up: the hook adapters wrap the augment CLI in the same + // guard. Returns a self-tested wrapper path — the built-in candidates are + // always absolute; a GITNEXUS_HOOK_TIMEOUT_PATH override is adopted as the + // exact string that passed the self-test. Same string is also the same + // RESOLUTION for absolute paths and for slashless names (PATH lookup is + // cwd-independent); a slash-containing RELATIVE override, however, is + // existsSync-checked and self-tested against this process's cwd while the + // adapters spawn the CLI with a `cwd` option (chdir-before-exec), so such + // a value can pass here yet ENOENT at the augment call site — set the + // override to an absolute path. Returns null when the wrapper is + // disabled/unavailable. Never call on win32 (see its JSDoc). + resolveUnixGuardTimeout, }; diff --git a/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs b/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs index ab495be84..d497a16d9 100644 --- a/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs +++ b/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs @@ -241,7 +241,18 @@ function main() { if (!pattern || pattern.length < 3) return; const release = acquireHookSlot(gitNexusDir); - if (!release) return; + if (!release) { + // Normal skip path: all per-repo hook slots are held by concurrent + // sessions. Stays silent by default; surfaced only under the cursor + // hook's own GITNEXUS_DEBUG (truthy) convention. NOTE: unlike the + // claude/plugin/antigravity adapters this integration does not install + // hook-db-lock-probe.cjs, so its augment child is not guard-wrapped + // yet — tracked on the #2163 follow-up list ("cursor probe"). + if (process.env.GITNEXUS_DEBUG) { + process.stderr.write('[GitNexus] augment skipped: hook slots saturated\n'); + } + return; + } const cliPath = resolveCliPath(); let result = ''; diff --git a/gitnexus/hooks/antigravity/gitnexus-antigravity-hook.cjs b/gitnexus/hooks/antigravity/gitnexus-antigravity-hook.cjs index bd72c7b55..f7d12ca6c 100755 --- a/gitnexus/hooks/antigravity/gitnexus-antigravity-hook.cjs +++ b/gitnexus/hooks/antigravity/gitnexus-antigravity-hook.cjs @@ -24,7 +24,10 @@ const fs = require('fs'); const path = require('path'); const { spawnSync } = require('child_process'); const { acquireHookSlot } = require('./hook-lock.cjs'); -const { hasGitNexusDbLockedByGitNexusServer } = require('./hook-db-lock-probe.cjs'); +const { + hasGitNexusDbLockedByGitNexusServer, + resolveUnixGuardTimeout, +} = require('./hook-db-lock-probe.cjs'); const { formatAnalyzeCommand } = require('./resolve-analyze-cmd.cjs'); function readInput() { @@ -198,10 +201,73 @@ function resolveCliPath() { return cliPath; } +// Debounce for the unguarded-CLI diagnostic below (#2163 follow-up review): +// at most one line per (short-lived) hook process, even if a future change +// runs the CLI more than once. +let unguardedCliWarned = false; + +/** + * Unix orphan containment (#2163 follow-up): the augment CLI is the + * longest-lived hook child (inner spawnSync timeout 7s locally, 12s via + * npx), so on Unix it gets the same SIGKILL-surviving coreutils `timeout` + * wrapper as the probe's lsof/ps. The wrapper budget is ceil(inner/1000)+1 + * seconds — STRICTLY greater than the inner spawnSync timeout, so on the + * supervised path Node's SIGTERM always fires first and the existing + * error/status contract is untouched. Once the hook itself has been + * SIGKILLed (exactly the orphan case the wrapper exists for), the guard + * semantics differ per branch: + * - direct exec (the CLI is the guard's CHILD): `-k 1` TERM-first — a + * SIGTERM-immune CLI can hold the guard ~1s past the inner timeout + * before the `-k` SIGKILL escalation reaps it. + * - npx (the CLI is a GRANDCHILD: guard → npx → CLI): `-s KILL` — the + * budget expiry SIGKILLs the whole process group outright. TERM-first + * would kill only the obedient npx parent, making `timeout` reap it and + * return before the `-k` escalation ever fires, stranding a + * SIGTERM-immune CLI grandchild unbounded (reproduced on coreutils + * 9.x). `-k 1` is retained alongside `-s KILL` as a harmless belt: with + * `-s KILL` the `-k` escalation signal is also KILL. Two residual gaps + * on this branch, both bounded by "no worse than pre-fix" (where the + * grandchild received no signal at all): the group-wide SIGKILL is + * coreutils semantics — a busybox `timeout` passes the self-test (it + * has `-k` and propagates exit status) but signals only its direct + * child, so a busybox guard cannot reach the grandchild; and on the + * SUPERVISED path (hook alive, inner spawnSync timeout SIGTERMs the + * guard) coreutils forwards TERM rather than the `-s` signal, npx dies, + * and the guard exits before any KILL fires — so a SIGTERM-immune CLI + * grandchild still escapes in those two cases. + * If the sibling probe predates the resolveUnixGuardTimeout export (version + * skew), the adapter degrades to the unwrapped invocation instead of + * throwing. Windows is deliberately NOT wrapped — there is no coreutils + * timeout to resolve there and the resolver's self-test spawns /bin/sh — so + * on win32 (the npx.cmd path) and whenever the guard resolves to null (e.g. + * macOS without Homebrew coreutils — reported once under GITNEXUS_DEBUG) + * the argv stays byte-identical to the pre-wrap invocation. + */ function runGitNexusCli(cliPath, args, cwd, timeout) { const isWin = process.platform === 'win32'; + // Version-skew guard (#2163 follow-up review): an older sibling probe + // without the resolveUnixGuardTimeout export must degrade to the unwrapped + // invocation — a TypeError here would be swallowed by the caller's catch + // and silently kill the augment. + const guard = + isWin || typeof resolveUnixGuardTimeout !== 'function' ? null : resolveUnixGuardTimeout(); + if (!isWin && !guard && !unguardedCliWarned && isDebugEnabled()) { + // Diagnose the "stays unwrapped" Unix paths once per hook process: no + // usable coreutils timeout/gtimeout (e.g. macOS without Homebrew + // coreutils), GITNEXUS_HOOK_TIMEOUT_PATH=disabled, or probe skew above. + unguardedCliWarned = true; + process.stderr.write( + '[GitNexus hook] no usable timeout/gtimeout guard; augment CLI child runs unguarded\n', + ); + } if (cliPath) { - return spawnSync(process.execPath, [cliPath, ...args], { + const [cmd, cmdArgs] = guard + ? [ + guard, + ['-k', '1', String(Math.ceil(timeout / 1000) + 1), process.execPath, cliPath, ...args], + ] + : [process.execPath, [cliPath, ...args]]; + return spawnSync(cmd, cmdArgs, { encoding: 'utf-8', timeout, cwd, @@ -209,7 +275,27 @@ function runGitNexusCli(cliPath, args, cwd, timeout) { windowsHide: true, }); } - return spawnSync(isWin ? 'npx.cmd' : 'npx', ['-y', 'gitnexus', ...args], { + // A non-null guard implies non-Windows, so the wrapped arm can hardcode + // plain `npx`. The wrapped arm leads with `-s KILL` (NOT TERM-first like + // the direct branch above): the CLI here is a grandchild behind npx — see + // the docblock. + const [cmd, cmdArgs] = guard + ? [ + guard, + [ + '-s', + 'KILL', + '-k', + '1', + String(Math.ceil((timeout + 5000) / 1000) + 1), + 'npx', + '-y', + 'gitnexus', + ...args, + ], + ] + : [isWin ? 'npx.cmd' : 'npx', ['-y', 'gitnexus', ...args]]; + return spawnSync(cmd, cmdArgs, { encoding: 'utf-8', timeout: timeout + 5000, cwd, diff --git a/gitnexus/hooks/claude/gitnexus-hook.cjs b/gitnexus/hooks/claude/gitnexus-hook.cjs index c37895f1d..740fe8559 100755 --- a/gitnexus/hooks/claude/gitnexus-hook.cjs +++ b/gitnexus/hooks/claude/gitnexus-hook.cjs @@ -15,7 +15,10 @@ const fs = require('fs'); const path = require('path'); const { spawnSync } = require('child_process'); const { acquireHookSlot } = require('./hook-lock.cjs'); -const { hasGitNexusDbLockedByGitNexusServer } = require('./hook-db-lock-probe.cjs'); +const { + hasGitNexusDbLockedByGitNexusServer, + resolveUnixGuardTimeout, +} = require('./hook-db-lock-probe.cjs'); const { formatAnalyzeCommand } = require('./resolve-analyze-cmd.cjs'); /** @@ -218,14 +221,76 @@ function resolveCliPath() { return cliPath; } +// Debounce for the unguarded-CLI diagnostic below (#2163 follow-up review): +// at most one line per (short-lived) hook process, even if a future change +// runs the CLI more than once. +let unguardedCliWarned = false; + /** * Spawn a gitnexus CLI command synchronously. * Returns the stderr output (KuzuDB captures stdout at OS level). + * + * Unix orphan containment (#2163 follow-up): the augment CLI is the + * longest-lived hook child (inner spawnSync timeout 7s locally, 12s via + * npx), so on Unix it gets the same SIGKILL-surviving coreutils `timeout` + * wrapper as the probe's lsof/ps. The wrapper budget is ceil(inner/1000)+1 + * seconds — STRICTLY greater than the inner spawnSync timeout, so on the + * supervised path Node's SIGTERM always fires first and the existing + * error/status contract is untouched. Once the hook itself has been + * SIGKILLed (exactly the orphan case the wrapper exists for), the guard + * semantics differ per branch: + * - direct exec (the CLI is the guard's CHILD): `-k 1` TERM-first — a + * SIGTERM-immune CLI can hold the guard ~1s past the inner timeout + * before the `-k` SIGKILL escalation reaps it. + * - npx (the CLI is a GRANDCHILD: guard → npx → CLI): `-s KILL` — the + * budget expiry SIGKILLs the whole process group outright. TERM-first + * would kill only the obedient npx parent, making `timeout` reap it and + * return before the `-k` escalation ever fires, stranding a + * SIGTERM-immune CLI grandchild unbounded (reproduced on coreutils + * 9.x). `-k 1` is retained alongside `-s KILL` as a harmless belt: with + * `-s KILL` the `-k` escalation signal is also KILL. Two residual gaps + * on this branch, both bounded by "no worse than pre-fix" (where the + * grandchild received no signal at all): the group-wide SIGKILL is + * coreutils semantics — a busybox `timeout` passes the self-test (it + * has `-k` and propagates exit status) but signals only its direct + * child, so a busybox guard cannot reach the grandchild; and on the + * SUPERVISED path (hook alive, inner spawnSync timeout SIGTERMs the + * guard) coreutils forwards TERM rather than the `-s` signal, npx dies, + * and the guard exits before any KILL fires — so a SIGTERM-immune CLI + * grandchild still escapes in those two cases. + * If the sibling probe predates the resolveUnixGuardTimeout export (version + * skew), the adapter degrades to the unwrapped invocation instead of + * throwing. Windows is deliberately NOT wrapped — there is no coreutils + * timeout to resolve there and the resolver's self-test spawns /bin/sh — so + * on win32 (the npx.cmd path) and whenever the guard resolves to null (e.g. + * macOS without Homebrew coreutils — reported once under GITNEXUS_DEBUG) + * the argv stays byte-identical to the pre-wrap invocation. */ function runGitNexusCli(cliPath, args, cwd, timeout) { const isWin = process.platform === 'win32'; + // Version-skew guard (#2163 follow-up review): an older sibling probe + // without the resolveUnixGuardTimeout export must degrade to the unwrapped + // invocation — a TypeError here would be swallowed by the caller's catch + // and silently kill the augment. + const guard = + isWin || typeof resolveUnixGuardTimeout !== 'function' ? null : resolveUnixGuardTimeout(); + if (!isWin && !guard && !unguardedCliWarned && isDebugEnabled()) { + // Diagnose the "stays unwrapped" Unix paths once per hook process: no + // usable coreutils timeout/gtimeout (e.g. macOS without Homebrew + // coreutils), GITNEXUS_HOOK_TIMEOUT_PATH=disabled, or probe skew above. + unguardedCliWarned = true; + process.stderr.write( + '[GitNexus hook] no usable timeout/gtimeout guard; augment CLI child runs unguarded\n', + ); + } if (cliPath) { - return spawnSync(process.execPath, [cliPath, ...args], { + const [cmd, cmdArgs] = guard + ? [ + guard, + ['-k', '1', String(Math.ceil(timeout / 1000) + 1), process.execPath, cliPath, ...args], + ] + : [process.execPath, [cliPath, ...args]]; + return spawnSync(cmd, cmdArgs, { encoding: 'utf-8', timeout, cwd, @@ -233,8 +298,27 @@ function runGitNexusCli(cliPath, args, cwd, timeout) { windowsHide: true, }); } - // On Windows, invoke npx.cmd directly (no shell needed) - return spawnSync(isWin ? 'npx.cmd' : 'npx', ['-y', 'gitnexus', ...args], { + // On Windows, invoke npx.cmd directly (no shell needed). A non-null guard + // implies non-Windows, so the wrapped arm can hardcode plain `npx`. The + // wrapped arm leads with `-s KILL` (NOT TERM-first like the direct branch + // above): the CLI here is a grandchild behind npx — see the docblock. + const [cmd, cmdArgs] = guard + ? [ + guard, + [ + '-s', + 'KILL', + '-k', + '1', + String(Math.ceil((timeout + 5000) / 1000) + 1), + 'npx', + '-y', + 'gitnexus', + ...args, + ], + ] + : [isWin ? 'npx.cmd' : 'npx', ['-y', 'gitnexus', ...args]]; + return spawnSync(cmd, cmdArgs, { encoding: 'utf-8', timeout: timeout + 5000, cwd, diff --git a/gitnexus/hooks/claude/hook-db-lock-probe.cjs b/gitnexus/hooks/claude/hook-db-lock-probe.cjs index 5c67804b1..de0fa5e85 100644 --- a/gitnexus/hooks/claude/hook-db-lock-probe.cjs +++ b/gitnexus/hooks/claude/hook-db-lock-probe.cjs @@ -20,13 +20,18 @@ * it 1s later — orphan lifetime is bounded at ~3s instead of unbounded. * - GITNEXUS_HOOK_TIMEOUT_PATH: the sentinel value `disabled` switches the * wrapper off deterministically; any other value is adopted only when it - * exists AND passes a one-shot `-k` self-test — otherwise resolution FALLS - * THROUGH to the built-in candidate list (first self-test pass wins), so - * no malformed value of any shape can silently disable orphan containment. + * exists AND passes a one-shot `-k` exit-propagation self-test — otherwise + * resolution FALLS THROUGH to the built-in candidate list (first self-test + * pass wins), so no malformed value of any shape can silently disable + * orphan containment. * - The gitnexus server is lazy-open + sticky-hold: an idle MCP server holds * ZERO lbug fds until the repo's first MCP query, then keeps the fd open. * A probe before that first query is therefore always false — a known, * pre-existing race, not a bug in this probe. + * - resolveUnixGuardTimeout is exported so the hook adapters can wrap the + * `gitnexus augment` CLI child — the longest-lived hook subprocess (7s + * local / 12s npx inner budgets) — in the same guard; see runGitNexusCli + * in the adapters (#2163 follow-up). */ const fs = require('fs'); @@ -70,38 +75,53 @@ let unixGuardTimeoutCache; /** * Resolve a coreutils `timeout`/`gtimeout` binary to wrap lsof/ps with - * (#2163). Dead code on Windows (the win32 dispatch returns earlier). + * (#2163). Unix-only by contract: the probe's win32 dispatch returns before + * reaching it, and the exported callers (the adapters' runGitNexusCli, + * #2163 follow-up) must check the platform first — the self-test below + * spawns /bin/sh. The memoized result is module-wide, so probe and adapter + * share one lazy self-test per hook process. * * GITNEXUS_HOOK_TIMEOUT_PATH semantics: the sentinel `disabled` turns the * wrapper off; any other value is only a CANDIDATE — an existing file path - * is tried first, but it must pass the `-k` self-test to be adopted. On any - * failure (non-existent path, directory, non-executable file, wrapper - * without `-k` support, …) resolution falls through to the built-in - * candidates below, tried in order, first self-test pass wins. This is - * strictly stronger than the sibling GITNEXUS_HOOK_LSOF_PATH / - * GITNEXUS_HOOK_PS_PATH overrides (which only check existence): no bad env - * value of ANY shape can silently disable orphan containment. + * is tried first, but it must pass the `-k` exit-propagation self-test to + * be adopted. On any failure (non-existent path, directory, non-executable + * file, wrapper without `-k` support, always-exit-0 stub, …) resolution + * falls through to the built-in candidates below, tried in order, first + * self-test pass wins. This is strictly stronger than the sibling + * GITNEXUS_HOOK_LSOF_PATH / GITNEXUS_HOOK_PS_PATH overrides (which only + * check existence): no bad env value of ANY shape can silently disable + * orphan containment. * * Lazy self-test: candidates are probed only when the lsof/ps fallback is * first reached, and the result is memoized. A candidate is adopted only - * when `timeout -k 1 1 /bin/sh -c :` exits 0. This rejects wrappers that do - * not support the coreutils `-k` flag — busybox <1.34, toybox, broken - * symlinks — which would otherwise exit with a usage error without ever - * running lsof, silently converting the lsof-ETIMEDOUT fail-closed contract - * into fail-open (#1492 regression). Only when EVERY candidate fails does - * the probe fall back to the unwrapped status quo (memoized null). - * busybox ≥1.34 passes the test and is fully usable (capability, not - * identity, decides). + * when `timeout -k 1 1 /bin/sh -c 'exit 42'` exits 42 — i.e. it must RUN + * the wrapped command AND PROPAGATE its exit status. This rejects two + * failure shapes: wrappers without the coreutils `-k` flag — busybox <1.34, + * toybox, broken symlinks — which would exit with a usage error without + * ever running lsof, silently converting the lsof-ETIMEDOUT fail-closed + * contract into fail-open (#1492 regression); and always-exit-0 stubs + * (/bin/true shapes), which would otherwise be adopted and "succeed" every + * wrapped spawn instantly without running it — a constant no-owner probe + * answer and, worse, a silently dead augment (status 0, empty stderr passes + * the adapters' success check with no context; #2163 follow-up review). + * Only when EVERY candidate fails does the probe fall back to the unwrapped + * status quo (memoized null). busybox ≥1.34 passes the test and is fully + * usable for everything THIS file spawns (lsof/ps are the guard's direct + * children) and for the adapters' direct-exec arm. The adapters' npx arm + * additionally relies on coreutils' process-GROUP signalling for its + * `-s KILL` grandchild reaping; busybox signals only its direct child, and + * this self-test deliberately does not probe that capability — see the + * adapter docblocks for the residual-gap statement. */ function passesGuardSelfTest(guard) { try { - const selfTest = spawnSync(guard, ['-k', '1', '1', '/bin/sh', '-c', ':'], { + const selfTest = spawnSync(guard, ['-k', '1', '1', '/bin/sh', '-c', 'exit 42'], { encoding: 'utf-8', timeout: 3000, stdio: ['ignore', 'ignore', 'ignore'], windowsHide: true, }); - return !selfTest.error && selfTest.status === 0; + return !selfTest.error && selfTest.status === 42; } catch { return false; } @@ -359,4 +379,16 @@ function hasGitNexusDbLockedByGitNexusServer(dbPath, myPid) { module.exports = { hasGitNexusDbLockedByGitNexusServer, + // #2163 follow-up: the hook adapters wrap the augment CLI in the same + // guard. Returns a self-tested wrapper path — the built-in candidates are + // always absolute; a GITNEXUS_HOOK_TIMEOUT_PATH override is adopted as the + // exact string that passed the self-test. Same string is also the same + // RESOLUTION for absolute paths and for slashless names (PATH lookup is + // cwd-independent); a slash-containing RELATIVE override, however, is + // existsSync-checked and self-tested against this process's cwd while the + // adapters spawn the CLI with a `cwd` option (chdir-before-exec), so such + // a value can pass here yet ENOENT at the augment call site — set the + // override to an absolute path. Returns null when the wrapper is + // disabled/unavailable. Never call on win32 (see its JSDoc). + resolveUnixGuardTimeout, }; diff --git a/gitnexus/test/unit/hooks.test.ts b/gitnexus/test/unit/hooks.test.ts index 12f38e9c9..9e2f39890 100644 --- a/gitnexus/test/unit/hooks.test.ts +++ b/gitnexus/test/unit/hooks.test.ts @@ -19,6 +19,7 @@ */ import { describe, it, expect, beforeAll, afterAll } from 'vitest'; import { spawnSync } from 'child_process'; +import { createRequire } from 'node:module'; import fs from 'fs'; import path from 'path'; import os from 'os'; @@ -86,6 +87,46 @@ const PLUGIN_HOOK_DB_PROBE = path.resolve( 'hook-db-lock-probe.cjs', ); +// ─── Host guard precheck for orphan-reaping tests (#2163) ─────────── +// +// The reaping lanes depend on a host coreutils `timeout`/`gtimeout` that +// passes the probe's `-k` self-test. Without one, the SIGTERM-immune fake +// child simply survives and the aliveness poll times out — a red that says +// nothing about WHY. Resolve the guard once through the probe's own exported +// resolver (the exact candidate list + self-test the hook child will use, +// with any dev-shell GITNEXUS_HOOK_TIMEOUT_PATH override cleared to mirror +// the `''` these tests pass to the hook) and assert on it with an explicit +// message. Chosen form: precheck ASSERTION, not skipIf — a skip would +// silently drop the incident-mechanism coverage on a misconfigured host +// (green-but-vacuous lane), while a red with a one-line actionable cause +// keeps the contract honest. GitHub ubuntu runners always ship coreutils, +// so CI behavior is unchanged. +const GUARD_PRECHECK_MSG = + 'precheck: no self-test-passing coreutils timeout/gtimeout on this host — ' + + 'orphan reaping cannot work here (the wrapper IS the reaping mechanism). ' + + 'Install coreutils or expose one via GITNEXUS_HOOK_TIMEOUT_PATH.'; + +let hostGuardMemo: string | null | undefined; +function resolveHostGuardForReapingTests(): string | null { + if (hostGuardMemo !== undefined) return hostGuardMemo; + const saved = process.env.GITNEXUS_HOOK_TIMEOUT_PATH; + process.env.GITNEXUS_HOOK_TIMEOUT_PATH = ''; + try { + // createRequire: the probe is a CJS module; this also exercises the real + // export surface the adapters consume (#2163 follow-up). Unix-only — the + // resolver's self-test spawns /bin/sh — and all callers below live in + // linux-gated describes. + const probe = createRequire(import.meta.url)(CJS_HOOK_DB_PROBE) as { + resolveUnixGuardTimeout: () => string | null; + }; + hostGuardMemo = probe.resolveUnixGuardTimeout(); + } finally { + if (saved === undefined) delete process.env.GITNEXUS_HOOK_TIMEOUT_PATH; + else process.env.GITNEXUS_HOOK_TIMEOUT_PATH = saved; + } + return hostGuardMemo; +} + // ─── Test fixtures: temporary .gitnexus directory ─────────────────── let tmpDir: string; @@ -1127,6 +1168,8 @@ describe.skipIf(process.platform !== 'linux')( // outlives the hook and SIGKILLs the child within ~3s — making this also // a direct regression test for the wrapper's `-k` capability. it('CJS: SIGKILLed hook leaves no immortal lsof child', async () => { + // Guard-availability precheck — see resolveHostGuardForReapingTests. + expect(resolveHostGuardForReapingTests(), GUARD_PRECHECK_MSG).not.toBeNull(); const { spawn } = await import('child_process'); const lbugPath = path.join(gitNexusDir, 'lbug'); fs.writeFileSync(lbugPath, ''); @@ -1205,7 +1248,10 @@ describe.skipIf(process.platform !== 'linux')( } expect(alive).toBe(false); } finally { - if (lsofPid > 0) { + // PID-reuse guard (#2169 review): re-run the detection loop's + // /proc//cmdline identity check before the cleanup SIGKILL, so + // a PID already reaped and recycled by the OS is never signalled. + if (lsofPid > 0 && isFakeLsofAlive()) { try { process.kill(lsofPid, 'SIGKILL'); } catch { @@ -1244,6 +1290,8 @@ describe.skipIf(process.platform !== 'linux')( // candidate list; failing its self-test falls through to the built-in // coreutils guard, which still reaps the orphan within ~3s. it('CJS: env guard pointing at a directory falls through to a working built-in guard', async () => { + // Guard-availability precheck — see resolveHostGuardForReapingTests. + expect(resolveHostGuardForReapingTests(), GUARD_PRECHECK_MSG).not.toBeNull(); const { spawn } = await import('child_process'); const lbugPath = path.join(gitNexusDir, 'lbug'); fs.writeFileSync(lbugPath, ''); @@ -1318,7 +1366,10 @@ describe.skipIf(process.platform !== 'linux')( } expect(alive).toBe(false); } finally { - if (lsofPid > 0) { + // PID-reuse guard (#2169 review): re-run the detection loop's + // /proc//cmdline identity check before the cleanup SIGKILL, so + // a PID already reaped and recycled by the OS is never signalled. + if (lsofPid > 0 && isFakeLsofAlive()) { try { process.kill(lsofPid, 'SIGKILL'); } catch { @@ -1350,6 +1401,510 @@ describe.skipIf(process.platform !== 'linux')( }, ); +// ─── Behavior: SIGKILLed hook cannot strand the augment CLI (#2163 f-up) ── + +describe.skipIf(process.platform !== 'linux')( + 'Orphaned augment CLI is reaped by the timeout wrapper (#2163 follow-up)', + () => { + // Same incident mechanism as the lsof reaping suite above, one layer up: + // the augment CLI is the longest-lived hook child (7s local / 12s npx + // inner budgets), so a hook SIGKILLed mid-augment used to strand it with + // nothing left to signal it. These tests pin GITNEXUS_HOOK_CLI_PATH, so + // they exercise the DIRECT-EXEC branch only (the CLI is the guard's + // direct child): the fake CLI here is SIGTERM-immune and sleeps 30s; + // with the wrap in place the guard SIGTERMs it at 8s (= ceil(7000/1000) + // +1) and the `-k` escalation SIGKILLs it 1s later, so it must be gone + // well inside the 12s poll window. (The npx branch reaps differently — + // `-s KILL` group-kills at budget because the CLI is a grandchild there; + // see the staged npx suite below.) With runGitNexusCli's wrap reverted, + // nothing can reap it and the poll times out → red. The antigravity + // adapter shares the identical runGitNexusCli shape and is pinned at + // source level (see 'Augment CLI guard wrap (source)'). + for (const [label, hookPath] of [ + ['CJS', CJS_HOOK], + ['Plugin', PLUGIN_HOOK], + ] as const) { + it(`${label}: SIGKILLed hook leaves no immortal augment CLI child`, async () => { + // Guard-availability precheck — see resolveHostGuardForReapingTests. + expect(resolveHostGuardForReapingTests(), GUARD_PRECHECK_MSG).not.toBeNull(); + const { spawn } = await import('child_process'); + // REQUIRED: a real lbug file routes the probe through the fake lsof + // (empty output → no holder PIDs → probe false) so the augment runs + // through the same probe-then-spawn flow as production. + const lbugPath = path.join(gitNexusDir, 'lbug'); + fs.writeFileSync(lbugPath, ''); + const pidFile = path.join(os.tmpdir(), `gn-hook-clipid-${process.pid}-${label}`); + fs.rmSync(pidFile, { force: true }); + const binDir = createHookToolDir({ + gitnexusPidFile: pidFile, + gitnexusSleepMs: 30000, + gitnexusIgnoreSigterm: true, + lsofOutput: '', + psOutput: '', + }); + let cliPid = 0; + let hookChild: ReturnType | null = null; + + const isFakeCliAlive = () => { + try { + process.kill(cliPid, 0); + } catch { + return false; // ESRCH — reaped + } + // PID-reuse guard: only count it alive while the cmdline still + // points at our fake CLI. + try { + return fs.readFileSync(`/proc/${cliPid}/cmdline`, 'utf-8').includes(binDir); + } catch { + return false; + } + }; + + try { + hookChild = spawn(process.execPath, [hookPath], { + stdio: ['pipe', 'ignore', 'ignore'], + env: { + ...hookEnv(binDir), + // '1', NOT '0' — see the slot-gate test above. + GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '1', + // Hermeticity: a dev-shell GITNEXUS_HOOK_TIMEOUT_PATH=disabled + // would unwrap the CLI and fake-red this test. Empty string + // falls through to the built-in candidates (the path under test). + GITNEXUS_HOOK_TIMEOUT_PATH: '', + }, + }); + hookChild.stdin!.end( + JSON.stringify({ + hook_event_name: 'PreToolUse', + tool_name: 'Grep', + tool_input: { pattern: 'validateUser' }, + cwd: tmpDir, + }), + ); + + // The fake CLI writes its PID as its FIRST statement; poll tightly. + const spawnDeadline = Date.now() + 8000; + while (Date.now() < spawnDeadline) { + try { + const raw = fs.readFileSync(pidFile, 'utf-8').trim(); + if (raw) { + cliPid = Number.parseInt(raw, 10); + break; + } + } catch { + /* not written yet */ + } + await new Promise((r) => setTimeout(r, 10)); + } + expect(cliPid).toBeGreaterThan(0); + + // Kill the hook while its augment CLI child is alive. + hookChild.kill('SIGKILL'); + + // Wrapper budget for the 7000ms call site is 8s, plus 1s `-k` + // grace — poll past that with margin, far short of the 30s sleep. + const reapDeadline = Date.now() + 12000; + let alive = isFakeCliAlive(); + while (alive && Date.now() < reapDeadline) { + await new Promise((r) => setTimeout(r, 100)); + alive = isFakeCliAlive(); + } + expect(alive).toBe(false); + } finally { + // PID-reuse guard (#2169 review): re-run the detection loop's + // /proc//cmdline identity check before the cleanup SIGKILL, + // so a PID already reaped and recycled by the OS is never + // signalled. + if (cliPid > 0 && isFakeCliAlive()) { + try { + process.kill(cliPid, 'SIGKILL'); + } catch { + /* already gone */ + } + } + try { + hookChild?.kill('SIGKILL'); + } catch { + /* ignore */ + } + // The hook claims a slot before probing; it died holding it. + const lockDir = path.join(gitNexusDir, '.hook-locks'); + try { + for (const f of fs.readdirSync(lockDir)) fs.unlinkSync(path.join(lockDir, f)); + } catch { + /* ignore */ + } + try { + fs.rmdirSync(lockDir); + } catch { + /* ignore */ + } + fs.rmSync(lbugPath, { force: true }); + fs.rmSync(pidFile, { force: true }); + fs.rmSync(binDir, { recursive: true, force: true }); + } + }, 30000); + } + }, +); + +// ─── Behavior: npx branch — SIGKILLed hook cannot strand the CLI grandchild ── + +describe.skipIf(process.platform !== 'linux')( + 'Orphaned npx-branch CLI grandchild is reaped by the -s KILL wrapper (#2163 follow-up)', + () => { + // The npx branch has a DEEPER topology than the direct-exec suite above: + // guard → npx → CLI, so the CLI is the guard's GRANDCHILD. Under the + // TERM-first `-k 1` guard the budget's group SIGTERM kills the obedient + // npx parent; `timeout` reaps its direct child and exits IMMEDIATELY, so + // its `-k` SIGKILL never fires — and a SIGTERM-immune CLI grandchild + // survives unbounded (reproduced on coreutils 9.x). The `-s KILL` + // wrapped arm instead SIGKILLs the whole process group at budget + // (13s = ceil((7000+5000)/1000)+1 here), which nothing can ignore. + // Reverting the npx arm to plain `-k 1` TERM-first makes this test red. + // + // Topology notes: the hook is STAGED into a bare temp dir together with + // its sibling helpers (the install-shaped copy, like the antigravity e2e + // suite uses), so resolveCliPath() finds no local dist/ and no + // resolvable gitnexus package; with GITNEXUS_HOOK_CLI_PATH cleared the + // npx fallback branch is the one that runs. A fake `npx` injected on + // PATH then spawns the SIGTERM-immune fake CLI as its own child and + // waits on it, mirroring the real npx process tree. + it('CJS (staged): SIGKILLed hook leaves no immortal CLI grandchild behind npx', async () => { + // Guard-availability precheck — see resolveHostGuardForReapingTests. + expect(resolveHostGuardForReapingTests(), GUARD_PRECHECK_MSG).not.toBeNull(); + const { spawn } = await import('child_process'); + // REQUIRED: a real lbug file routes the probe through the fake lsof + // (empty output → no holder PIDs → probe false) so the augment runs + // through the same probe-then-spawn flow as production. + const lbugPath = path.join(gitNexusDir, 'lbug'); + fs.writeFileSync(lbugPath, ''); + const pidFile = path.join(os.tmpdir(), `gn-hook-npxclipid-${process.pid}`); + fs.rmSync(pidFile, { force: true }); + // Route self-proof (#2169 review): written by the fake npx as its first + // statement, so the test fails loudly if a future resolveCliPath / + // hookEnv change silently re-routes the augment to the direct arm. + const npxMarkerPath = path.join(os.tmpdir(), `gn-hook-npxmarker-${process.pid}`); + fs.rmSync(npxMarkerPath, { force: true }); + const binDir = createHookToolDir({ + gitnexusPidFile: pidFile, + gitnexusSleepMs: 30000, + gitnexusIgnoreSigterm: true, + lsofOutput: '', + psOutput: '', + }); + // Fake npx: spawns the fake CLI as the guard's grandchild and waits on + // it like real npx; npx itself stays SIGTERM-obedient (Node default). + fs.writeFileSync( + path.join(binDir, 'npx'), + `#!/usr/bin/env node\n` + + `require('fs').writeFileSync(${JSON.stringify(npxMarkerPath)}, String(process.pid));\n` + + `const { spawn } = require('child_process');\n` + + `const child = spawn(process.execPath, [${JSON.stringify( + path.join(binDir, 'gitnexus-cli.js'), + )}], { stdio: 'ignore' });\n` + + `child.on('exit', (code) => process.exit(code === null ? 1 : code));\n`, + { mode: 0o755 }, + ); + // Stage the hook + its sibling helpers into a bare dir with no dist/ + // and no reachable node_modules/gitnexus, so resolveCliPath() → ''. + const stagedDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-staged-hook-')); + const hookSrcDir = path.dirname(CJS_HOOK); + for (const f of [ + 'gitnexus-hook.cjs', + 'hook-lock.cjs', + 'hook-db-lock-probe.cjs', + 'resolve-analyze-cmd.cjs', + ]) { + fs.copyFileSync(path.join(hookSrcDir, f), path.join(stagedDir, f)); + } + const stagedHook = path.join(stagedDir, 'gitnexus-hook.cjs'); + let cliPid = 0; + let hookChild: ReturnType | null = null; + + const isFakeCliAlive = () => { + try { + process.kill(cliPid, 0); + } catch { + return false; // ESRCH — reaped + } + // PID-reuse guard: only count it alive while the cmdline still + // points at our fake CLI. + try { + return fs.readFileSync(`/proc/${cliPid}/cmdline`, 'utf-8').includes(binDir); + } catch { + return false; + } + }; + + try { + hookChild = spawn(process.execPath, [stagedHook], { + stdio: ['pipe', 'ignore', 'ignore'], + env: { + ...hookEnv(binDir), + // Force the npx fallback: no CLI-path override (empty string + // fails resolveCliPath's trim check), and nothing for the staged + // copy's require.resolve to find via NODE_PATH. + GITNEXUS_HOOK_CLI_PATH: '', + NODE_PATH: '', + // '1', NOT '0' — see the slot-gate test above. + GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '1', + // Hermeticity: fall through to the built-in guard candidates. + GITNEXUS_HOOK_TIMEOUT_PATH: '', + }, + }); + hookChild.stdin!.end( + JSON.stringify({ + hook_event_name: 'PreToolUse', + tool_name: 'Grep', + tool_input: { pattern: 'validateUser' }, + cwd: tmpDir, + }), + ); + + // The fake CLI writes its PID as its FIRST statement; poll tightly. + const spawnDeadline = Date.now() + 8000; + while (Date.now() < spawnDeadline) { + try { + const raw = fs.readFileSync(pidFile, 'utf-8').trim(); + if (raw) { + cliPid = Number.parseInt(raw, 10); + break; + } + } catch { + /* not written yet */ + } + await new Promise((r) => setTimeout(r, 10)); + } + expect(cliPid).toBeGreaterThan(0); + // The augment really took the npx fallback arm, not the direct arm. + expect(fs.existsSync(npxMarkerPath)).toBe(true); + + // Kill the hook while the npx → CLI chain is alive (orphan topology). + hookChild.kill('SIGKILL'); + + // The npx call site's wrapper budget is 13s (= ceil((7000+5000)/ + // 1000)+1) from the guard's start; the group SIGKILL lands then. + // Poll past it with margin, far short of the CLI's 30s sleep. + const reapDeadline = Date.now() + 18000; + let alive = isFakeCliAlive(); + while (alive && Date.now() < reapDeadline) { + await new Promise((r) => setTimeout(r, 100)); + alive = isFakeCliAlive(); + } + expect(alive).toBe(false); + } finally { + // PID-reuse guard (#2169 review): re-run the detection loop's + // /proc//cmdline identity check before the cleanup SIGKILL, so + // a PID already reaped and recycled by the OS is never signalled. + if (cliPid > 0 && isFakeCliAlive()) { + try { + process.kill(cliPid, 'SIGKILL'); + } catch { + /* already gone */ + } + } + try { + hookChild?.kill('SIGKILL'); + } catch { + /* ignore */ + } + // The hook claims a slot before probing; it died holding it. + const lockDir = path.join(gitNexusDir, '.hook-locks'); + try { + for (const f of fs.readdirSync(lockDir)) fs.unlinkSync(path.join(lockDir, f)); + } catch { + /* ignore */ + } + try { + fs.rmdirSync(lockDir); + } catch { + /* ignore */ + } + fs.rmSync(lbugPath, { force: true }); + fs.rmSync(pidFile, { force: true }); + fs.rmSync(npxMarkerPath, { force: true }); + fs.rmSync(stagedDir, { recursive: true, force: true }); + fs.rmSync(binDir, { recursive: true, force: true }); + } + }, 45000); + }, +); + +// ─── Wrapping equivalence: disabled guard ⇒ pre-wrap augment behavior ── + +describe.skipIf(process.platform === 'win32')( + 'Augment CLI guard wrap degrades cleanly when disabled (#2163 follow-up)', + () => { + // T6-style equivalence pin for the AUGMENT path: with the wrapper + // explicitly off (the `disabled` sentinel, NOT a bogus path — invalid + // values fall through to the real candidate list by design) the augment + // path must behave exactly as it did before the wrap existed: the probe + // fails open on an idle DB, the CLI runs unwrapped, and its context is + // emitted verbatim. (The wrapped arm's equivalence is covered by the + // whole existing augment suite, which now runs under the host's real + // guard on the Linux/macOS lanes.) + for (const [label, hookPath] of [ + ['CJS', CJS_HOOK], + ['Plugin', PLUGIN_HOOK], + ] as const) { + it(`${label}: augment runs and emits context with the wrapper disabled`, () => { + const markerPath = path.join(os.tmpdir(), `gn-hook-nowrap-aug-${process.pid}-${label}`); + const lbugPath = path.join(gitNexusDir, 'lbug'); + fs.writeFileSync(lbugPath, ''); + fs.rmSync(markerPath, { force: true }); + const binDir = createHookToolDir({ + gitnexusMarkerPath: markerPath, + gitnexusStderr: '[GitNexus] 1 related symbol found:\n\nvalidateUser (src/auth.ts)\n', + lsofOutput: '', + psOutput: '', + }); + try { + const result = runHook( + hookPath, + { + hook_event_name: 'PreToolUse', + tool_name: 'Grep', + tool_input: { pattern: 'validateUser' }, + cwd: tmpDir, + }, + undefined, + { + env: { + ...hookEnv(binDir), + GITNEXUS_HOOK_TIMEOUT_PATH: 'disabled', + }, + }, + ); + const output = parseHookOutput(result.stdout); + expect(output).not.toBeNull(); + expect(output!.additionalContext).toContain('[GitNexus] 1 related symbol found'); + expect(fs.existsSync(markerPath)).toBe(true); + } finally { + fs.rmSync(lbugPath, { force: true }); + fs.rmSync(markerPath, { force: true }); + fs.rmSync(binDir, { recursive: true, force: true }); + } + }); + } + }, +); + +// ─── Source: augment CLI guard wrap present in every adapter (#2163 f-up) ── + +describe('Augment CLI guard wrap (source, #2163 follow-up)', () => { + const ANTIGRAVITY_HOOK = path.resolve( + __dirname, + '..', + '..', + 'hooks', + 'antigravity', + 'gitnexus-antigravity-hook.cjs', + ); + + // The cursor integration is deliberately ABSENT from this list: it does + // not install hook-db-lock-probe.cjs (see gitnexus-cursor-integration/ + // README.md "What's installed manually vs. automated"), so there is no + // resolver sibling to require — wrapping its augment child is the + // "cursor probe" item on the #2163 follow-up list. + for (const [label, hookPath] of [ + ['CJS', CJS_HOOK], + ['Plugin', PLUGIN_HOOK], + ['Antigravity', ANTIGRAVITY_HOOK], + ] as const) { + it(`${label}: runGitNexusCli wraps via resolveUnixGuardTimeout with a ceil(ms/1000)+1 budget`, () => { + const source = fs.readFileSync(hookPath, 'utf-8'); + const start = source.indexOf('function runGitNexusCli'); + expect(start).toBeGreaterThanOrEqual(0); + const end = source.indexOf('\nfunction ', start + 1); + const fn = source.slice(start, end === -1 ? undefined : end); + // Consults the probe's exported resolver (memo shared with the probe), + // and never on Windows — the npx.cmd / gitnexus.cmd argv stay exactly + // as before the wrap. The typeof check is the probe version-skew guard + // (#2169 review): an old probe without the resolveUnixGuardTimeout + // export must degrade to the unwrapped argv, not throw a TypeError + // that the caller's catch swallows into a silently dead augment. + expect(fn).toMatch( + /isWin \|\| typeof resolveUnixGuardTimeout !== 'function'\s*\?\s*null\s*:\s*resolveUnixGuardTimeout\(\)/, + ); + // Coreutils `-k 1` escalation… + expect(fn).toContain("'-k',"); + // …with a budget STRICTLY above each branch's inner spawnSync timeout: + // ceil(inner/1000)+1 for both the direct (timeout) and npx + // (timeout + 5000) call sites. The direct-budget formula is counted + // exactly — once per wrapped direct-exec branch (the Plugin adapter has + // two: GITNEXUS_HOOK_CLI_PATH and the PATH-direct `gitnexus` branch, + // its most common production path) — so a partial revert of any single + // branch cannot pass unnoticed. + const directBudgetCount = (fn.match(/Math\.ceil\(timeout \/ 1000\) \+ 1/g) ?? []).length; + expect(directBudgetCount).toBe(label === 'Plugin' ? 2 : 1); + expect(fn).toMatch(/Math\.ceil\(\(timeout \+ 5000\) \/ 1000\) \+ 1/); + // Argv-order pin (#2169 review): the budget token must appear BEFORE + // the command word — `timeout … ` — or coreutils would + // parse the command word as its DURATION argument. Token presence and + // the counts above alone would let a transposed argv pass. Every + // direct-exec budget must be immediately followed by its command token + // (process.execPath, or the PATH-direct 'gitnexus' on Plugin), and the + // npx budget by 'npx'. + const directOrderCount = ( + fn.match( + /String\(Math\.ceil\(timeout \/ 1000\) \+ 1\),\s*(?:process\.execPath|'gitnexus')/g, + ) ?? [] + ).length; + expect(directOrderCount).toBe(label === 'Plugin' ? 2 : 1); + expect(fn).toMatch(/String\(Math\.ceil\(\(timeout \+ 5000\) \/ 1000\) \+ 1\),\s*'npx'/); + // npx-branch grandchild containment (#2169 review): the npx wrapped + // arm must SIGKILL the process group at budget (`-s KILL`) — a group + // SIGTERM there kills only the obedient npx parent, `timeout` returns + // before its `-k` escalation fires, and a SIGTERM-immune CLI + // grandchild escapes unbounded. + expect(fn).toMatch( + /'-s',\s*'KILL',\s*'-k',\s*'1',\s*String\(Math\.ceil\(\(timeout \+ 5000\)/, + ); + // …and the direct-exec arm(s) must NOT lead with `-s KILL`: TERM-first + // is gentler and sufficient there (the CLI is the guard's direct + // child), so `-s` appears exactly once — in the npx arm. + expect((fn.match(/'-s',/g) ?? []).length).toBe(1); + }); + } + + for (const [label, probePath] of [ + ['CJS', CJS_HOOK_DB_PROBE], + ['Plugin', PLUGIN_HOOK_DB_PROBE], + ] as const) { + it(`${label} probe exports resolveUnixGuardTimeout`, () => { + const source = fs.readFileSync(probePath, 'utf-8'); + const exportsSlice = source.slice(source.indexOf('module.exports')); + expect(exportsSlice).toContain('resolveUnixGuardTimeout'); + }); + } +}); + +// ─── Source: cursor hook slot-skip diagnostic (#2163 follow-up) ───── + +describe('Cursor hook slot-skip diagnostic (source, #2163 follow-up)', () => { + const CURSOR_HOOK = path.resolve( + __dirname, + '..', + '..', + '..', + 'gitnexus-cursor-integration', + 'hooks', + 'gitnexus-hook.cjs', + ); + + it('debug-gates the slot-saturated skip under the cursor truthy convention', () => { + const source = fs.readFileSync(CURSOR_HOOK, 'utf-8'); + const idx = source.indexOf('augment skipped: hook slots saturated'); + expect(idx).toBeGreaterThanOrEqual(0); + // Must sit inside the cursor hook's own debug gate (truthy + // `process.env.GITNEXUS_DEBUG`, unlike the claude adapters' strict + // '1'/'true' gate) so the default path stays silent. + const before = source.slice(Math.max(0, idx - 600), idx); + expect(before).toContain('process.env.GITNEXUS_DEBUG'); + }); +}); + // ─── Integration: PreToolUse augmentation filtering (#1492) ───────── describe('PreToolUse augmentation filtering (integration)', () => { @@ -1877,7 +2432,10 @@ describe.skipIf(process.platform === 'win32')( // that passes the `-k` self-test and then reports coreutils budget // expiry (exit 124) must map to "unresponsive holder" → fail-closed // skip. The fake guard distinguishes the self-test invocation - // (`-k 1 1 /bin/sh -c :`) from a real wrap by the /bin/sh argv token. + // (`-k 1 1 /bin/sh -c 'exit 42'`) from a real wrap by the `exit 42` + // argv token, and PROPAGATES the requested status — the self-test now + // demands exit-status propagation (status 42), not just exit 0 + // (#2169 review). it(`${label}: guard exit 124 (budget expiry) → fail-closed skip`, () => { const markerPath = path.join(os.tmpdir(), `gn-hook-guard124-${process.pid}-${label}`); const lbugPath = path.join(gitNexusDir, 'lbug'); @@ -1892,7 +2450,7 @@ describe.skipIf(process.platform === 'win32')( const fakeGuard = path.join(binDir, 'guard-exit-124'); fs.writeFileSync( fakeGuard, - `#!/usr/bin/env node\nif (process.argv.includes('/bin/sh')) process.exit(0);\nprocess.exit(124);\n`, + `#!/usr/bin/env node\nif (process.argv.includes('exit 42')) process.exit(42);\nprocess.exit(124);\n`, { mode: 0o755 }, ); try { @@ -1948,7 +2506,7 @@ describe.skipIf(process.platform === 'win32')( const fakeGuard = path.join(binDir, 'guard-sigkill'); fs.writeFileSync( fakeGuard, - `#!/usr/bin/env node\nif (process.argv.includes('/bin/sh')) process.exit(0);\nprocess.kill(process.pid, 'SIGKILL');\n`, + `#!/usr/bin/env node\nif (process.argv.includes('exit 42')) process.exit(42);\nprocess.kill(process.pid, 'SIGKILL');\n`, { mode: 0o755 }, ); try { @@ -1982,6 +2540,62 @@ describe.skipIf(process.platform === 'win32')( } }); + // F5-4 (#2169 review; F5-3 was taken by the #2165 slot-gate test + // above): an always-exit-0 stub (/bin/true shape) at + // GITNEXUS_HOOK_TIMEOUT_PATH must be REJECTED by the self-test. The + // old self-test only demanded exit 0, which such a stub satisfies + // without ever RUNNING the wrapped command — once adopted it instantly + // "succeeds" every wrapped spawn with empty output, turning the probe + // into a constant no-owner answer and, worse, the augment into a + // silent no-op (status 0 + empty stderr passes the success check with + // no context, so the feature dies without a trace). The propagation + // self-test (`sh -c 'exit 42'` must yield 42) rejects the stub; + // resolution falls through to the built-in candidates (or, with none + // usable, to the unwrapped status quo) and the augment runs for real. + it(`${label}: always-exit-0 stub guard is rejected → augment still runs and emits context`, () => { + const markerPath = path.join(os.tmpdir(), `gn-hook-stubguard-${process.pid}-${label}`); + const lbugPath = path.join(gitNexusDir, 'lbug'); + fs.writeFileSync(lbugPath, ''); + fs.rmSync(markerPath, { force: true }); + const binDir = createHookToolDir({ + gitnexusMarkerPath: markerPath, + gitnexusStderr: '[GitNexus] 1 related symbol found:\n\nvalidateUser (src/auth.ts)\n', + lsofOutput: '', + psOutput: '', + }); + // Models /bin/true: exits 0 for ANY argv without running anything. + const stubGuard = path.join(binDir, 'true-stub'); + fs.writeFileSync(stubGuard, `#!/usr/bin/env node\nprocess.exit(0);\n`, { mode: 0o755 }); + try { + const result = runHook( + hookPath, + { + hook_event_name: 'PreToolUse', + tool_name: 'Grep', + tool_input: { pattern: 'validateUser' }, + cwd: tmpDir, + }, + undefined, + { + env: { + ...hookEnv(binDir), + GITNEXUS_HOOK_TIMEOUT_PATH: stubGuard, + // '1', NOT '0' — see the slot-gate test above. + GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '1', + }, + }, + ); + const output = parseHookOutput(result.stdout); + expect(output).not.toBeNull(); + expect(output!.additionalContext).toContain('[GitNexus] 1 related symbol found'); + expect(fs.existsSync(markerPath)).toBe(true); + } finally { + fs.rmSync(lbugPath, { force: true }); + fs.rmSync(markerPath, { force: true }); + fs.rmSync(binDir, { recursive: true, force: true }); + } + }); + it(`${label}: non-GitNexus ps line → augment runs`, () => { const markerPath = path.join(os.tmpdir(), `gn-hook-other-${process.pid}-${label}`); const lbugPath = path.join(gitNexusDir, 'lbug'); diff --git a/gitnexus/test/utils/hook-test-helpers.ts b/gitnexus/test/utils/hook-test-helpers.ts index 9fd69d5c6..003db10c0 100644 --- a/gitnexus/test/utils/hook-test-helpers.ts +++ b/gitnexus/test/utils/hook-test-helpers.ts @@ -87,6 +87,12 @@ function writeExecutable(filePath: string, content: string) { export function createHookToolDir(options: { gitnexusStderr?: string; gitnexusMarkerPath?: string; + /** Fake gitnexus CLI writes its own PID here as its FIRST statement, minimizing detection latency for augment orphan-reaping tests (#2163 follow-up). */ + gitnexusPidFile?: string; + /** Fake gitnexus CLI sleeps this long instead of exiting — models a hung augment child. */ + gitnexusSleepMs?: number; + /** Fake gitnexus CLI traps SIGTERM as a no-op before sleeping — models an unkillable CLI that only SIGKILL can end (#2163 follow-up). */ + gitnexusIgnoreSigterm?: boolean; lsofOutput?: string; lsofOutputLines?: string[]; psOutput?: string; @@ -103,7 +109,19 @@ export function createHookToolDir(options: { const gitnexusStderr = JSON.stringify(options.gitnexusStderr ?? ''); const markerPath = JSON.stringify(options.gitnexusMarkerPath ?? ''); - const fakeGitNexus = `#!/usr/bin/env node\nconst fs = require('fs');\nconst marker = ${markerPath};\nif (marker) fs.writeFileSync(marker, 'called');\nprocess.stderr.write(${gitnexusStderr});\n`; + // Composable prologue (mirrors the fake-lsof one below): pidFile write MUST + // stay the first statement (see the option docs above); the SIGTERM trap + // MUST be installed before any sleep. + const fakeGitNexus = + `#!/usr/bin/env node\nconst fs = require('fs');\n` + + (options.gitnexusPidFile != null + ? `fs.writeFileSync(${JSON.stringify(options.gitnexusPidFile)}, String(process.pid));\n` + : '') + + (options.gitnexusIgnoreSigterm ? `process.on('SIGTERM', () => {});\n` : '') + + `const marker = ${markerPath};\nif (marker) fs.writeFileSync(marker, 'called');\n` + + (options.gitnexusSleepMs != null + ? `setTimeout(() => {}, ${Number(options.gitnexusSleepMs)});\n` + : `process.stderr.write(${gitnexusStderr});\n`); writeExecutable(path.join(binDir, 'gitnexus'), fakeGitNexus); writeExecutable(path.join(binDir, 'gitnexus-cli.js'), fakeGitNexus); From 129bc84c0d683b43ea34108de771f6a7951e2c68 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sat, 13 Jun 2026 07:04:14 +0100 Subject: [PATCH 04/16] feat(taint): interprocedural taint via function summaries over resolved CALLS (#2084) (#2179) --- .../gitnexus/gitnexus-taint-analysis/SKILL.md | 178 ++++++ .../skills/gitnexus-taint-analysis/SKILL.md | 178 ++++++ gitnexus/skills/gitnexus-taint-analysis.md | 178 ++++++ .../src/core/incremental/subgraph-extract.ts | 16 +- .../core/ingestion/pipeline-phases/index.ts | 1 + .../pipeline-phases/taint-summaries.ts | 119 ++++ gitnexus/src/core/ingestion/pipeline.ts | 18 + .../scope-resolution/pipeline/phase.ts | 23 + .../scope-resolution/pipeline/run.ts | 66 ++ .../core/ingestion/taint/interproc-emit.ts | 120 ++++ .../core/ingestion/taint/interproc-solver.ts | 412 ++++++++++++ .../src/core/ingestion/taint/propagate.ts | 20 +- .../ingestion/taint/source-sink-config.ts | 30 + .../ingestion/taint/summary-harvest-driver.ts | 147 +++++ .../core/ingestion/taint/summary-harvest.ts | 592 ++++++++++++++++++ .../src/core/ingestion/taint/summary-model.ts | 270 ++++++++ gitnexus/src/core/lbug/lbug-adapter.ts | 56 ++ gitnexus/src/core/run-analyze.ts | 33 + gitnexus/src/mcp/local/local-backend.ts | 120 +++- gitnexus/src/mcp/tools.ts | 13 +- gitnexus/src/storage/repo-manager.ts | 10 + .../cfg/fixtures/interproc-repo/gen.ts | 16 + .../cfg/fixtures/interproc-repo/sink.ts | 13 + .../cfg/fixtures/interproc-repo/source.ts | 14 + .../integration/cfg/interproc-taint.test.ts | 109 ++++ .../integration/lbug-core-adapter.test.ts | 25 + .../test/integration/taint-explain.test.ts | 137 ++++ .../unit/incremental-subgraph-extract.test.ts | 18 + .../ingestion/pipeline-phase-registry.test.ts | 39 ++ gitnexus/test/unit/pdg-mode-flip.test.ts | 42 ++ gitnexus/test/unit/run-analyze.test.ts | 9 + gitnexus/test/unit/security.test.ts | 9 + .../test/unit/taint/interproc-solver.test.ts | 448 +++++++++++++ .../test/unit/taint/summary-harvest.test.ts | 212 +++++++ .../test/unit/taint/summary-model.test.ts | 122 ++++ 35 files changed, 3785 insertions(+), 28 deletions(-) create mode 100644 .claude/skills/gitnexus/gitnexus-taint-analysis/SKILL.md create mode 100644 gitnexus-claude-plugin/skills/gitnexus-taint-analysis/SKILL.md create mode 100644 gitnexus/skills/gitnexus-taint-analysis.md create mode 100644 gitnexus/src/core/ingestion/pipeline-phases/taint-summaries.ts create mode 100644 gitnexus/src/core/ingestion/taint/interproc-emit.ts create mode 100644 gitnexus/src/core/ingestion/taint/interproc-solver.ts create mode 100644 gitnexus/src/core/ingestion/taint/summary-harvest-driver.ts create mode 100644 gitnexus/src/core/ingestion/taint/summary-harvest.ts create mode 100644 gitnexus/src/core/ingestion/taint/summary-model.ts create mode 100644 gitnexus/test/integration/cfg/fixtures/interproc-repo/gen.ts create mode 100644 gitnexus/test/integration/cfg/fixtures/interproc-repo/sink.ts create mode 100644 gitnexus/test/integration/cfg/fixtures/interproc-repo/source.ts create mode 100644 gitnexus/test/integration/cfg/interproc-taint.test.ts create mode 100644 gitnexus/test/unit/taint/interproc-solver.test.ts create mode 100644 gitnexus/test/unit/taint/summary-harvest.test.ts create mode 100644 gitnexus/test/unit/taint/summary-model.test.ts diff --git a/.claude/skills/gitnexus/gitnexus-taint-analysis/SKILL.md b/.claude/skills/gitnexus/gitnexus-taint-analysis/SKILL.md new file mode 100644 index 000000000..9bffffdac --- /dev/null +++ b/.claude/skills/gitnexus/gitnexus-taint-analysis/SKILL.md @@ -0,0 +1,178 @@ +--- +name: gitnexus-taint-analysis +description: "Use when working on, reviewing, or extending GitNexus's CFG/taint/PDG subsystem (the `--pdg` layers), or when reasoning about source→sink data-flow findings. Examples: \"How does taint analysis work here?\", \"Why didn't explain find this flow?\", \"Add a new sink/source\", \"Review the interprocedural taint code\"." +--- + +# CFG & Taint Analysis with GitNexus + +Expert knowledge for the opt-in `--pdg` program-analysis subsystem: control-flow +graphs, reaching definitions, and intra- + inter-procedural taint. Read this +before touching `gitnexus/src/core/ingestion/cfg/**` or +`gitnexus/src/core/ingestion/taint/**`, or when explaining a finding. + +## When to Use + +- "How does the taint engine work / why is this flow (not) reported?" +- Adding a source, sink, or sanitizer to the model. +- Extending or reviewing the CFG / reaching-defs / taint / summary code. +- Understanding the `explain` MCP tool's findings (intra- vs inter-procedural). +- Debugging a false positive or false negative in `--pdg` output. + +## The layered substrate (build order) + +Taint runs **on** the graph, not beside it. Each layer is opt-in behind `--pdg` +and a default `analyze` run is **byte-identical** (the golden parity gate is the +hard floor for every change here). + +``` +L1 CFG per-function basic blocks + control-flow edges (M1 #2081) +L2 REACHING_DEF GEN/KILL def→use data dependence (pure solver) (M2 #2082) +L3 Taint (intra) source→sink over RD facts, minus sanitizers (M3 #2083) +L4 Taint (inter) per-function summaries composed over CALLS (M4 #2084) +``` + +- **Worker-built, main-thread-solved.** The parse worker builds each function's + CFG + harvests def/use + call-site facts onto `ParsedFile.cfgSideChannel` + (plain, structured-clone-safe data — never AST nodes). The main thread runs + the pure solvers. NEVER re-parse on the main thread (re-introduces the #1983 + OOM). +- **In-phase emit (KTD1).** L1–L4-harvest all run INSIDE the scope-resolution + pdg window (`scope-resolution/pipeline/run.ts`, gated `input.pdg === true`), + because the disk-backed ParsedFile store is cleared when that phase ends — a + standalone post-`mro` phase would read empty data. The cross-function fixpoint + (L4) is the exception: it runs in its OWN registered phase (`taintSummaries`) + AFTER scope-resolution, because it needs the COMPLETE call graph, and consumes + small plain summary data threaded out via `ScopeResolutionOutput`. +- **Pure-solver contract.** `computeReachingDefs`, `computeTaintFlows`, + `harvestFunctionSummary`, and `solveInterprocTaint` are pure and deterministic + (no graph, no I/O, no logger; sorted outputs). Snapshot tests and + content-derived edge ids depend on it. + +## Intra-procedural taint (L3) + +Forward reachability over RD facts from matched **sources** to matched **sinks**, +killed by **sanitizers**. Key design points worth internalizing: + +- **Occurrence-tagged sites.** A flat per-arg binding set cannot tell + `exec(escape(x))` (safe) from `exec(x)` (finding); the harvest records nested + call structure (`SiteRecord.parent`/via-tags) so sanitizer interposition is + precise. +- **Kind-set sanitizer model.** A taint carries a set of *neutralized* + `SinkKind`s; a sink fires unless its kind is in the set. So `escape(req.body)` + suppresses `res.send` (xss) but STILL fires `db.query` (sql) — a kind-blind + kill would be a suppressed live injection (the forbidden FN direction). + `path.basename(t)` neutralizes path-traversal only, not command-injection. +- **Statement-level finding identity.** NOT block-pair (block conflation drops + distinct findings; `exec(req.body, req.query)` is two findings). +- Persisted as `TAINTED` edges (BasicBlock→BasicBlock); the path rides the + `reason` column via the shared versioned codec (`taint/path-codec.ts`). + +## Interprocedural taint (L4) — the functional/summary method + +The production approach (Sharir-Pnueli 1981; the same shape as Meta's Pysa and +Mariana Trench, and FB Infer) — NOT full IFDS tabulation. Each function is +reduced to a compact **summary**, and summaries are composed over the already- +resolved `CALLS` graph. + +**Summary shape** (`taint/summary-model.ts`, whole-parameter granularity): + +| Edge | Meaning | Analogue | +|------|---------|----------| +| `param→return` | a param flows to the return value | TITO — **reserved** (the floor already covers its recall; precision pass deferred) | +| `param→callee-arg` | a param flows into arg *j* of a call (carries the path's neutralized sink kinds) | TITO into callee | +| `param→sink` | a param reaches a modelled sink | partial/triggered sink | +| `source→return` | the function generates+returns a source | generative — **composed** via the caller's `callResults` | +| `source→callee-arg` | a generated source flows into a call | fixpoint SEED | +| `callResults` | a user-function call's result flows to a sink/return/callee-arg in the caller | composes with callee `source→return` | + +**The fixpoint** (`taint/interproc-solver.ts`): the unit is `(function, +parameter, source)`. Seed from `source→callee-arg`, propagate via +`param→callee-arg`, fire a finding when a tainted param meets `param→sink`. + +- **Cycle-safe by monotonicity.** The tainted-set is monotone over a finite + lattice (`fn × param × source`), so the worklist converges — a recursive call + just re-proposes an already-visited entry. SCC condensation would only refine + processing order; correctness/termination don't require it. +- **Source-discriminated state (load-bearing).** Key the state by the SOURCE + too. Keying only by `(fn, param)` collapses multi-source flows: a sink param + tainted by source A is marked visited and a later flow from source B is dropped + before firing — the recurring multi-source bug class. (Bit M3; bit M4 U9.) +- **Name-based call join.** Match a summary's call-arg edge to a `CALLS` edge by + CALLEE NAME, not call-site line — line-base parity (CFG 1-based vs reference + site) is fragile; the callee identity is exact and context-insensitivity + taints the callee's param identically at every call site. +- Persisted as `TAINT_PATH` edges (Function→Function), function-level hop chain + in `reason` via the same codec; confidence < the intra-procedural 1.0. + +**Context-insensitivity** is the accepted trade-off at this tier: one summary +per function, return/call-site merging accepted (security-conservative). Expect +some FP from merging; the bigger FN sources are unmodeled features (below). + +## Known false-negative classes (documented, deferred) + +The largest is **closures/callbacks** (`arr.forEach(() => sink(y))`) — taint +into a callback is dropped without per-library models (true of CodeQL's JS libs +too). Also deferred: field/property flows (`obj.x = taint; sink(obj.y)`), +field-sensitive access paths, guard-style sanitizers, implicit/control-dependence +flows, promise/async-await threading, and **destructured/rest params before a +tainted simple param** (the summary port index is the binding ordinal, not the +formal arg position — needs a formal-param index threaded from the worker +`BindingEntry`). The interprocedural join is also context-insensitive: when one +caller invokes two distinct **same-named callees**, a flow into one +over-attributes to both (sound — over-report, never a missed flow). Absence of a +finding is NOT proof of safety. + +## GitNexus-specific gotchas + +- **Function↔CFG join.** `FunctionCfg.functionStartLine` is 1-based; `Function`/ + `Method` node `startLine` is 0-based — join at `startLine - 1`. Function nodes + have no column, so same-line functions (`{a:()=>x(), b:()=>y()}`) are + ambiguous → drop (the summary driver counts `unresolved`) rather than + cross-wire. +- **No rel-property index (S1).** Kuzu has no secondary index on relationship + properties, and unanchored `[:TAINTED*]`/`[:TAINT_PATH*]` queries explode. + TAINT_PATH is therefore MATERIALIZED + anchored at analyze time, never + traversed live; `explain` reads it source-anchored + LIMIT-guarded. +- **`explain` is the only discovery surface.** `TAINTED`/`TAINT_PATH` are + deliberately OUT of `VALID_RELATION_TYPES` (impact's allow-list) and the web + schema (pinned in `security.test.ts`). `explain` enumerates both layers + (cross-function findings carry `interprocedural: true`). +- **One shared codec.** Both the emit path and `explain` import + `taint/path-codec.ts`. Two hand-rolled copies of a wire format drift — never + fork it. New metadata extends the format WITHIN the version when writer + + reader ship together. +- **Cache versioning.** A worker-harvest shape change bumps the parse-cache pdg + NAMESPACE (`pdg:N`), NOT `SCHEMA_BUMP` (which cold-invalidates every user). + Persisted-graph/config changes ride `RepoMeta.pdg`'s key-union mismatch → + full writeback. Model content rides `taintModelVersion`. + +## Adding a source / sink / sanitizer + +Edit the language model in `taint/typescript-model.ts` (registered via the +explicit `registerBuiltinTaintModels` seam, keyed by `SupportedLanguages`). The +spec is hashable data (no functions). A sanitizer's `neutralizes` lists the +EXACT sink kinds it defends — never a blanket kill. Add a fixture + assert the +finding (or its absence) in `test/unit/taint/` (real-source harness: +`test/helpers/ts-cfg-harness.ts`); the end-to-end proof is +`test/integration/cfg/`. + +## Validation checklist for any `--pdg` change + +``` +1. tsc clean (schema additions are exhaustiveness-checked; watch the + api.ts getNodeQuery runtime read-path if a node label is added). +2. Targeted vitest by directory (test/unit/taint, test/unit/cfg, + test/integration/cfg) — verify by ISOLATION, not full-suite exit + (known load-flakes). `node scripts/build.js` before worker/integration runs. +3. Flag-off golden byte-identical (pipeline-graph-golden.test.ts). +4. bench/cfg/measure.mjs --check (no fingerprint drift / budget regression). +5. detect_changes() before commit; impact({direction:'upstream'}) before + editing shared symbols (KnowledgeGraph, RepoMeta, RelationshipType, codec). +``` + +## Prior art (for deeper design questions) + +Sharir & Pnueli 1981 (functional approach); Reps-Horwitz-Sagiv IFDS (POPL 1995); +FlowDroid/StubDroid (access-path summaries); Pysa & Mariana Trench (TITO / +propagations, parallel SCC fixpoint); CodeQL Models-as-Data (the richest port +notation, incl. callback ports); Infer (content-keyed incremental summaries). diff --git a/gitnexus-claude-plugin/skills/gitnexus-taint-analysis/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-taint-analysis/SKILL.md new file mode 100644 index 000000000..9bffffdac --- /dev/null +++ b/gitnexus-claude-plugin/skills/gitnexus-taint-analysis/SKILL.md @@ -0,0 +1,178 @@ +--- +name: gitnexus-taint-analysis +description: "Use when working on, reviewing, or extending GitNexus's CFG/taint/PDG subsystem (the `--pdg` layers), or when reasoning about source→sink data-flow findings. Examples: \"How does taint analysis work here?\", \"Why didn't explain find this flow?\", \"Add a new sink/source\", \"Review the interprocedural taint code\"." +--- + +# CFG & Taint Analysis with GitNexus + +Expert knowledge for the opt-in `--pdg` program-analysis subsystem: control-flow +graphs, reaching definitions, and intra- + inter-procedural taint. Read this +before touching `gitnexus/src/core/ingestion/cfg/**` or +`gitnexus/src/core/ingestion/taint/**`, or when explaining a finding. + +## When to Use + +- "How does the taint engine work / why is this flow (not) reported?" +- Adding a source, sink, or sanitizer to the model. +- Extending or reviewing the CFG / reaching-defs / taint / summary code. +- Understanding the `explain` MCP tool's findings (intra- vs inter-procedural). +- Debugging a false positive or false negative in `--pdg` output. + +## The layered substrate (build order) + +Taint runs **on** the graph, not beside it. Each layer is opt-in behind `--pdg` +and a default `analyze` run is **byte-identical** (the golden parity gate is the +hard floor for every change here). + +``` +L1 CFG per-function basic blocks + control-flow edges (M1 #2081) +L2 REACHING_DEF GEN/KILL def→use data dependence (pure solver) (M2 #2082) +L3 Taint (intra) source→sink over RD facts, minus sanitizers (M3 #2083) +L4 Taint (inter) per-function summaries composed over CALLS (M4 #2084) +``` + +- **Worker-built, main-thread-solved.** The parse worker builds each function's + CFG + harvests def/use + call-site facts onto `ParsedFile.cfgSideChannel` + (plain, structured-clone-safe data — never AST nodes). The main thread runs + the pure solvers. NEVER re-parse on the main thread (re-introduces the #1983 + OOM). +- **In-phase emit (KTD1).** L1–L4-harvest all run INSIDE the scope-resolution + pdg window (`scope-resolution/pipeline/run.ts`, gated `input.pdg === true`), + because the disk-backed ParsedFile store is cleared when that phase ends — a + standalone post-`mro` phase would read empty data. The cross-function fixpoint + (L4) is the exception: it runs in its OWN registered phase (`taintSummaries`) + AFTER scope-resolution, because it needs the COMPLETE call graph, and consumes + small plain summary data threaded out via `ScopeResolutionOutput`. +- **Pure-solver contract.** `computeReachingDefs`, `computeTaintFlows`, + `harvestFunctionSummary`, and `solveInterprocTaint` are pure and deterministic + (no graph, no I/O, no logger; sorted outputs). Snapshot tests and + content-derived edge ids depend on it. + +## Intra-procedural taint (L3) + +Forward reachability over RD facts from matched **sources** to matched **sinks**, +killed by **sanitizers**. Key design points worth internalizing: + +- **Occurrence-tagged sites.** A flat per-arg binding set cannot tell + `exec(escape(x))` (safe) from `exec(x)` (finding); the harvest records nested + call structure (`SiteRecord.parent`/via-tags) so sanitizer interposition is + precise. +- **Kind-set sanitizer model.** A taint carries a set of *neutralized* + `SinkKind`s; a sink fires unless its kind is in the set. So `escape(req.body)` + suppresses `res.send` (xss) but STILL fires `db.query` (sql) — a kind-blind + kill would be a suppressed live injection (the forbidden FN direction). + `path.basename(t)` neutralizes path-traversal only, not command-injection. +- **Statement-level finding identity.** NOT block-pair (block conflation drops + distinct findings; `exec(req.body, req.query)` is two findings). +- Persisted as `TAINTED` edges (BasicBlock→BasicBlock); the path rides the + `reason` column via the shared versioned codec (`taint/path-codec.ts`). + +## Interprocedural taint (L4) — the functional/summary method + +The production approach (Sharir-Pnueli 1981; the same shape as Meta's Pysa and +Mariana Trench, and FB Infer) — NOT full IFDS tabulation. Each function is +reduced to a compact **summary**, and summaries are composed over the already- +resolved `CALLS` graph. + +**Summary shape** (`taint/summary-model.ts`, whole-parameter granularity): + +| Edge | Meaning | Analogue | +|------|---------|----------| +| `param→return` | a param flows to the return value | TITO — **reserved** (the floor already covers its recall; precision pass deferred) | +| `param→callee-arg` | a param flows into arg *j* of a call (carries the path's neutralized sink kinds) | TITO into callee | +| `param→sink` | a param reaches a modelled sink | partial/triggered sink | +| `source→return` | the function generates+returns a source | generative — **composed** via the caller's `callResults` | +| `source→callee-arg` | a generated source flows into a call | fixpoint SEED | +| `callResults` | a user-function call's result flows to a sink/return/callee-arg in the caller | composes with callee `source→return` | + +**The fixpoint** (`taint/interproc-solver.ts`): the unit is `(function, +parameter, source)`. Seed from `source→callee-arg`, propagate via +`param→callee-arg`, fire a finding when a tainted param meets `param→sink`. + +- **Cycle-safe by monotonicity.** The tainted-set is monotone over a finite + lattice (`fn × param × source`), so the worklist converges — a recursive call + just re-proposes an already-visited entry. SCC condensation would only refine + processing order; correctness/termination don't require it. +- **Source-discriminated state (load-bearing).** Key the state by the SOURCE + too. Keying only by `(fn, param)` collapses multi-source flows: a sink param + tainted by source A is marked visited and a later flow from source B is dropped + before firing — the recurring multi-source bug class. (Bit M3; bit M4 U9.) +- **Name-based call join.** Match a summary's call-arg edge to a `CALLS` edge by + CALLEE NAME, not call-site line — line-base parity (CFG 1-based vs reference + site) is fragile; the callee identity is exact and context-insensitivity + taints the callee's param identically at every call site. +- Persisted as `TAINT_PATH` edges (Function→Function), function-level hop chain + in `reason` via the same codec; confidence < the intra-procedural 1.0. + +**Context-insensitivity** is the accepted trade-off at this tier: one summary +per function, return/call-site merging accepted (security-conservative). Expect +some FP from merging; the bigger FN sources are unmodeled features (below). + +## Known false-negative classes (documented, deferred) + +The largest is **closures/callbacks** (`arr.forEach(() => sink(y))`) — taint +into a callback is dropped without per-library models (true of CodeQL's JS libs +too). Also deferred: field/property flows (`obj.x = taint; sink(obj.y)`), +field-sensitive access paths, guard-style sanitizers, implicit/control-dependence +flows, promise/async-await threading, and **destructured/rest params before a +tainted simple param** (the summary port index is the binding ordinal, not the +formal arg position — needs a formal-param index threaded from the worker +`BindingEntry`). The interprocedural join is also context-insensitive: when one +caller invokes two distinct **same-named callees**, a flow into one +over-attributes to both (sound — over-report, never a missed flow). Absence of a +finding is NOT proof of safety. + +## GitNexus-specific gotchas + +- **Function↔CFG join.** `FunctionCfg.functionStartLine` is 1-based; `Function`/ + `Method` node `startLine` is 0-based — join at `startLine - 1`. Function nodes + have no column, so same-line functions (`{a:()=>x(), b:()=>y()}`) are + ambiguous → drop (the summary driver counts `unresolved`) rather than + cross-wire. +- **No rel-property index (S1).** Kuzu has no secondary index on relationship + properties, and unanchored `[:TAINTED*]`/`[:TAINT_PATH*]` queries explode. + TAINT_PATH is therefore MATERIALIZED + anchored at analyze time, never + traversed live; `explain` reads it source-anchored + LIMIT-guarded. +- **`explain` is the only discovery surface.** `TAINTED`/`TAINT_PATH` are + deliberately OUT of `VALID_RELATION_TYPES` (impact's allow-list) and the web + schema (pinned in `security.test.ts`). `explain` enumerates both layers + (cross-function findings carry `interprocedural: true`). +- **One shared codec.** Both the emit path and `explain` import + `taint/path-codec.ts`. Two hand-rolled copies of a wire format drift — never + fork it. New metadata extends the format WITHIN the version when writer + + reader ship together. +- **Cache versioning.** A worker-harvest shape change bumps the parse-cache pdg + NAMESPACE (`pdg:N`), NOT `SCHEMA_BUMP` (which cold-invalidates every user). + Persisted-graph/config changes ride `RepoMeta.pdg`'s key-union mismatch → + full writeback. Model content rides `taintModelVersion`. + +## Adding a source / sink / sanitizer + +Edit the language model in `taint/typescript-model.ts` (registered via the +explicit `registerBuiltinTaintModels` seam, keyed by `SupportedLanguages`). The +spec is hashable data (no functions). A sanitizer's `neutralizes` lists the +EXACT sink kinds it defends — never a blanket kill. Add a fixture + assert the +finding (or its absence) in `test/unit/taint/` (real-source harness: +`test/helpers/ts-cfg-harness.ts`); the end-to-end proof is +`test/integration/cfg/`. + +## Validation checklist for any `--pdg` change + +``` +1. tsc clean (schema additions are exhaustiveness-checked; watch the + api.ts getNodeQuery runtime read-path if a node label is added). +2. Targeted vitest by directory (test/unit/taint, test/unit/cfg, + test/integration/cfg) — verify by ISOLATION, not full-suite exit + (known load-flakes). `node scripts/build.js` before worker/integration runs. +3. Flag-off golden byte-identical (pipeline-graph-golden.test.ts). +4. bench/cfg/measure.mjs --check (no fingerprint drift / budget regression). +5. detect_changes() before commit; impact({direction:'upstream'}) before + editing shared symbols (KnowledgeGraph, RepoMeta, RelationshipType, codec). +``` + +## Prior art (for deeper design questions) + +Sharir & Pnueli 1981 (functional approach); Reps-Horwitz-Sagiv IFDS (POPL 1995); +FlowDroid/StubDroid (access-path summaries); Pysa & Mariana Trench (TITO / +propagations, parallel SCC fixpoint); CodeQL Models-as-Data (the richest port +notation, incl. callback ports); Infer (content-keyed incremental summaries). diff --git a/gitnexus/skills/gitnexus-taint-analysis.md b/gitnexus/skills/gitnexus-taint-analysis.md new file mode 100644 index 000000000..9bffffdac --- /dev/null +++ b/gitnexus/skills/gitnexus-taint-analysis.md @@ -0,0 +1,178 @@ +--- +name: gitnexus-taint-analysis +description: "Use when working on, reviewing, or extending GitNexus's CFG/taint/PDG subsystem (the `--pdg` layers), or when reasoning about source→sink data-flow findings. Examples: \"How does taint analysis work here?\", \"Why didn't explain find this flow?\", \"Add a new sink/source\", \"Review the interprocedural taint code\"." +--- + +# CFG & Taint Analysis with GitNexus + +Expert knowledge for the opt-in `--pdg` program-analysis subsystem: control-flow +graphs, reaching definitions, and intra- + inter-procedural taint. Read this +before touching `gitnexus/src/core/ingestion/cfg/**` or +`gitnexus/src/core/ingestion/taint/**`, or when explaining a finding. + +## When to Use + +- "How does the taint engine work / why is this flow (not) reported?" +- Adding a source, sink, or sanitizer to the model. +- Extending or reviewing the CFG / reaching-defs / taint / summary code. +- Understanding the `explain` MCP tool's findings (intra- vs inter-procedural). +- Debugging a false positive or false negative in `--pdg` output. + +## The layered substrate (build order) + +Taint runs **on** the graph, not beside it. Each layer is opt-in behind `--pdg` +and a default `analyze` run is **byte-identical** (the golden parity gate is the +hard floor for every change here). + +``` +L1 CFG per-function basic blocks + control-flow edges (M1 #2081) +L2 REACHING_DEF GEN/KILL def→use data dependence (pure solver) (M2 #2082) +L3 Taint (intra) source→sink over RD facts, minus sanitizers (M3 #2083) +L4 Taint (inter) per-function summaries composed over CALLS (M4 #2084) +``` + +- **Worker-built, main-thread-solved.** The parse worker builds each function's + CFG + harvests def/use + call-site facts onto `ParsedFile.cfgSideChannel` + (plain, structured-clone-safe data — never AST nodes). The main thread runs + the pure solvers. NEVER re-parse on the main thread (re-introduces the #1983 + OOM). +- **In-phase emit (KTD1).** L1–L4-harvest all run INSIDE the scope-resolution + pdg window (`scope-resolution/pipeline/run.ts`, gated `input.pdg === true`), + because the disk-backed ParsedFile store is cleared when that phase ends — a + standalone post-`mro` phase would read empty data. The cross-function fixpoint + (L4) is the exception: it runs in its OWN registered phase (`taintSummaries`) + AFTER scope-resolution, because it needs the COMPLETE call graph, and consumes + small plain summary data threaded out via `ScopeResolutionOutput`. +- **Pure-solver contract.** `computeReachingDefs`, `computeTaintFlows`, + `harvestFunctionSummary`, and `solveInterprocTaint` are pure and deterministic + (no graph, no I/O, no logger; sorted outputs). Snapshot tests and + content-derived edge ids depend on it. + +## Intra-procedural taint (L3) + +Forward reachability over RD facts from matched **sources** to matched **sinks**, +killed by **sanitizers**. Key design points worth internalizing: + +- **Occurrence-tagged sites.** A flat per-arg binding set cannot tell + `exec(escape(x))` (safe) from `exec(x)` (finding); the harvest records nested + call structure (`SiteRecord.parent`/via-tags) so sanitizer interposition is + precise. +- **Kind-set sanitizer model.** A taint carries a set of *neutralized* + `SinkKind`s; a sink fires unless its kind is in the set. So `escape(req.body)` + suppresses `res.send` (xss) but STILL fires `db.query` (sql) — a kind-blind + kill would be a suppressed live injection (the forbidden FN direction). + `path.basename(t)` neutralizes path-traversal only, not command-injection. +- **Statement-level finding identity.** NOT block-pair (block conflation drops + distinct findings; `exec(req.body, req.query)` is two findings). +- Persisted as `TAINTED` edges (BasicBlock→BasicBlock); the path rides the + `reason` column via the shared versioned codec (`taint/path-codec.ts`). + +## Interprocedural taint (L4) — the functional/summary method + +The production approach (Sharir-Pnueli 1981; the same shape as Meta's Pysa and +Mariana Trench, and FB Infer) — NOT full IFDS tabulation. Each function is +reduced to a compact **summary**, and summaries are composed over the already- +resolved `CALLS` graph. + +**Summary shape** (`taint/summary-model.ts`, whole-parameter granularity): + +| Edge | Meaning | Analogue | +|------|---------|----------| +| `param→return` | a param flows to the return value | TITO — **reserved** (the floor already covers its recall; precision pass deferred) | +| `param→callee-arg` | a param flows into arg *j* of a call (carries the path's neutralized sink kinds) | TITO into callee | +| `param→sink` | a param reaches a modelled sink | partial/triggered sink | +| `source→return` | the function generates+returns a source | generative — **composed** via the caller's `callResults` | +| `source→callee-arg` | a generated source flows into a call | fixpoint SEED | +| `callResults` | a user-function call's result flows to a sink/return/callee-arg in the caller | composes with callee `source→return` | + +**The fixpoint** (`taint/interproc-solver.ts`): the unit is `(function, +parameter, source)`. Seed from `source→callee-arg`, propagate via +`param→callee-arg`, fire a finding when a tainted param meets `param→sink`. + +- **Cycle-safe by monotonicity.** The tainted-set is monotone over a finite + lattice (`fn × param × source`), so the worklist converges — a recursive call + just re-proposes an already-visited entry. SCC condensation would only refine + processing order; correctness/termination don't require it. +- **Source-discriminated state (load-bearing).** Key the state by the SOURCE + too. Keying only by `(fn, param)` collapses multi-source flows: a sink param + tainted by source A is marked visited and a later flow from source B is dropped + before firing — the recurring multi-source bug class. (Bit M3; bit M4 U9.) +- **Name-based call join.** Match a summary's call-arg edge to a `CALLS` edge by + CALLEE NAME, not call-site line — line-base parity (CFG 1-based vs reference + site) is fragile; the callee identity is exact and context-insensitivity + taints the callee's param identically at every call site. +- Persisted as `TAINT_PATH` edges (Function→Function), function-level hop chain + in `reason` via the same codec; confidence < the intra-procedural 1.0. + +**Context-insensitivity** is the accepted trade-off at this tier: one summary +per function, return/call-site merging accepted (security-conservative). Expect +some FP from merging; the bigger FN sources are unmodeled features (below). + +## Known false-negative classes (documented, deferred) + +The largest is **closures/callbacks** (`arr.forEach(() => sink(y))`) — taint +into a callback is dropped without per-library models (true of CodeQL's JS libs +too). Also deferred: field/property flows (`obj.x = taint; sink(obj.y)`), +field-sensitive access paths, guard-style sanitizers, implicit/control-dependence +flows, promise/async-await threading, and **destructured/rest params before a +tainted simple param** (the summary port index is the binding ordinal, not the +formal arg position — needs a formal-param index threaded from the worker +`BindingEntry`). The interprocedural join is also context-insensitive: when one +caller invokes two distinct **same-named callees**, a flow into one +over-attributes to both (sound — over-report, never a missed flow). Absence of a +finding is NOT proof of safety. + +## GitNexus-specific gotchas + +- **Function↔CFG join.** `FunctionCfg.functionStartLine` is 1-based; `Function`/ + `Method` node `startLine` is 0-based — join at `startLine - 1`. Function nodes + have no column, so same-line functions (`{a:()=>x(), b:()=>y()}`) are + ambiguous → drop (the summary driver counts `unresolved`) rather than + cross-wire. +- **No rel-property index (S1).** Kuzu has no secondary index on relationship + properties, and unanchored `[:TAINTED*]`/`[:TAINT_PATH*]` queries explode. + TAINT_PATH is therefore MATERIALIZED + anchored at analyze time, never + traversed live; `explain` reads it source-anchored + LIMIT-guarded. +- **`explain` is the only discovery surface.** `TAINTED`/`TAINT_PATH` are + deliberately OUT of `VALID_RELATION_TYPES` (impact's allow-list) and the web + schema (pinned in `security.test.ts`). `explain` enumerates both layers + (cross-function findings carry `interprocedural: true`). +- **One shared codec.** Both the emit path and `explain` import + `taint/path-codec.ts`. Two hand-rolled copies of a wire format drift — never + fork it. New metadata extends the format WITHIN the version when writer + + reader ship together. +- **Cache versioning.** A worker-harvest shape change bumps the parse-cache pdg + NAMESPACE (`pdg:N`), NOT `SCHEMA_BUMP` (which cold-invalidates every user). + Persisted-graph/config changes ride `RepoMeta.pdg`'s key-union mismatch → + full writeback. Model content rides `taintModelVersion`. + +## Adding a source / sink / sanitizer + +Edit the language model in `taint/typescript-model.ts` (registered via the +explicit `registerBuiltinTaintModels` seam, keyed by `SupportedLanguages`). The +spec is hashable data (no functions). A sanitizer's `neutralizes` lists the +EXACT sink kinds it defends — never a blanket kill. Add a fixture + assert the +finding (or its absence) in `test/unit/taint/` (real-source harness: +`test/helpers/ts-cfg-harness.ts`); the end-to-end proof is +`test/integration/cfg/`. + +## Validation checklist for any `--pdg` change + +``` +1. tsc clean (schema additions are exhaustiveness-checked; watch the + api.ts getNodeQuery runtime read-path if a node label is added). +2. Targeted vitest by directory (test/unit/taint, test/unit/cfg, + test/integration/cfg) — verify by ISOLATION, not full-suite exit + (known load-flakes). `node scripts/build.js` before worker/integration runs. +3. Flag-off golden byte-identical (pipeline-graph-golden.test.ts). +4. bench/cfg/measure.mjs --check (no fingerprint drift / budget regression). +5. detect_changes() before commit; impact({direction:'upstream'}) before + editing shared symbols (KnowledgeGraph, RepoMeta, RelationshipType, codec). +``` + +## Prior art (for deeper design questions) + +Sharir & Pnueli 1981 (functional approach); Reps-Horwitz-Sagiv IFDS (POPL 1995); +FlowDroid/StubDroid (access-path summaries); Pysa & Mariana Trench (TITO / +propagations, parallel SCC fixpoint); CodeQL Models-as-Data (the richest port +notation, incl. callback ports); Infer (content-keyed incremental summaries). diff --git a/gitnexus/src/core/incremental/subgraph-extract.ts b/gitnexus/src/core/incremental/subgraph-extract.ts index 71fe656be..74f01fd48 100644 --- a/gitnexus/src/core/incremental/subgraph-extract.ts +++ b/gitnexus/src/core/incremental/subgraph-extract.ts @@ -54,6 +54,16 @@ import type { KnowledgeGraph } from '../graph/types.js'; const isGraphWide = (label: string): boolean => label === 'Community' || label === 'Process'; +/** + * Relationship types whose VALIDITY is a whole-program property, not a + * function of their endpoints' files (#2084 M4 U6). `TAINT_PATH` (cross- + * function taint) can be invalidated by a change to an INTERMEDIATE function + * on a third file, so the endpoint-writability rule below would skip a stale + * A→C edge. These are always extracted (and the orchestrator delete-alls them + * first, like Community/Process) so they rebuild from the fresh graph. + */ +const isGraphWideRelType = (type: string): boolean => type === 'TAINT_PATH'; + /** * Build a Map for every File-bound node in the graph. * Graph-wide nodes (Community/Process) have no filePath and are filtered. @@ -84,7 +94,11 @@ export const extractChangedSubgraph = ( }); fullGraph.forEachRelationship((r: GraphRelationship) => { - if (writableNodeIds.has(r.sourceId) || writableNodeIds.has(r.targetId)) { + if ( + writableNodeIds.has(r.sourceId) || + writableNodeIds.has(r.targetId) || + isGraphWideRelType(r.type) + ) { sub.addRelationship(r); } }); diff --git a/gitnexus/src/core/ingestion/pipeline-phases/index.ts b/gitnexus/src/core/ingestion/pipeline-phases/index.ts index 4d8fa93bd..fa458b6e2 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/index.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/index.ts @@ -21,6 +21,7 @@ export { type ScopeResolutionOutput, } from '../scope-resolution/pipeline/phase.js'; export { pruneLocalSymbolsPhase, type PruneLocalSymbolsOutput } from './prune-local-symbols.js'; +export { taintSummariesPhase, type TaintSummariesOutput } from './taint-summaries.js'; export { mroPhase, type MROOutput } from './mro.js'; export { communitiesPhase, type CommunitiesOutput } from './communities.js'; export { processesPhase, type ProcessesOutput } from './processes.js'; diff --git a/gitnexus/src/core/ingestion/pipeline-phases/taint-summaries.ts b/gitnexus/src/core/ingestion/pipeline-phases/taint-summaries.ts new file mode 100644 index 000000000..96417ff30 --- /dev/null +++ b/gitnexus/src/core/ingestion/pipeline-phases/taint-summaries.ts @@ -0,0 +1,119 @@ +/** + * Phase: taintSummaries (#2084 M4 U3/U5) + * + * The interprocedural taint fixpoint. Runs AFTER scope-resolution (where the + * complete, resolved `CALLS` graph lives in `ctx.graph` and the per-function + * summaries were harvested in-phase) and composes those summaries to find + * source→sink flows that cross function and file boundaries. + * + * Opt-in: registered with `enabledWhen: (o) => o.pdg === true` (the first real + * pdg-gated phase). A default `analyze` run never includes it, so the graph is + * byte-identical. No always-on phase depends on it (a filtered-out dep would + * throw in `getPhaseOutput`). + * + * @deps scopeResolution, pruneLocalSymbols + * @reads graph (CALLS edges, Function/Method nodes), scopeResolution output + * (functionSummaries) + * @writes graph (TAINT_PATH edges) + */ + +import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js'; +import { getPhaseOutput } from './types.js'; +import type { ScopeResolutionOutput } from '../scope-resolution/pipeline/phase.js'; +import { + solveInterprocTaint, + DEFAULT_MAX_INTERPROC_HOPS, + DEFAULT_PDG_MAX_INTERPROC_FINDINGS, + type InterprocCallEdge, +} from '../taint/interproc-solver.js'; +import { emitInterprocTaint, DEFAULT_PDG_MAX_INTERPROC_EDGES } from '../taint/interproc-emit.js'; +import type { FunctionSummary } from '../taint/summary-model.js'; +import { logger } from '../../logger.js'; + +export interface TaintSummariesOutput { + /** Function summaries fed to the fixpoint. */ + summaries: number; + /** Cross-function findings (pre-cap). */ + findings: number; + /** TAINT_PATH edges persisted. */ + edgesEmitted: number; + /** Call sites whose callee did not resolve to a summary edge (diagnostics). */ + unmatchedCallSites: number; +} + +const EMPTY: TaintSummariesOutput = { + summaries: 0, + findings: 0, + edgesEmitted: 0, + unmatchedCallSites: 0, +}; + +export const taintSummariesPhase: PipelinePhase = { + name: 'taintSummaries', + deps: ['scopeResolution', 'pruneLocalSymbols'], + + async execute( + ctx: PipelineContext, + deps: ReadonlyMap>, + ): Promise { + const scope = getPhaseOutput(deps, 'scopeResolution'); + const summaries = scope.functionSummaries; + if (summaries.length === 0) return EMPTY; + + // Index summaries by function node id. + const summaryMap = new Map(summaries.map((s) => [s.fnId, s])); + + // Build the call-edge adjacency from resolved CALLS edges. The join to a + // summary's call-arg edge is by CALLEE NAME (base-independent — see the + // solver doc); recover it from the callee node's `name` property. + const callEdges: InterprocCallEdge[] = []; + for (const rel of ctx.graph.iterRelationshipsByType('CALLS')) { + const callee = ctx.graph.getNode(rel.targetId); + const calleeName = + callee && typeof callee.properties.name === 'string' ? callee.properties.name : undefined; + if (calleeName === undefined) continue; + callEdges.push({ callerId: rel.sourceId, calleeId: rel.targetId, calleeName }); + } + + // Arm the per-run caps (#2084 review P1-3) — every other pdg layer bounds + // its output via RepoMeta.pdg; without this the fixpoint state + TAINT_PATH + // edges grow unbounded on a fan-in-heavy repo (OOM). `0` ⇒ unlimited + // (preserved like the other pdg caps). The solver/emit already implement + // deterministic truncate-and-warn — this just hands them the budgets. + const maxFindings = ctx.options?.pdgMaxInterprocFindings ?? DEFAULT_PDG_MAX_INTERPROC_FINDINGS; + const maxHops = ctx.options?.pdgMaxInterprocHops ?? DEFAULT_MAX_INTERPROC_HOPS; + const maxEdges = ctx.options?.pdgMaxInterprocEdges ?? DEFAULT_PDG_MAX_INTERPROC_EDGES; + + const solved = solveInterprocTaint(summaryMap, callEdges, { maxFindings, maxHops }); + const emit = emitInterprocTaint(ctx.graph, solved.findings, { maxEdges }, (m) => + logger.warn(m), + ); + + // Surface drops UNCONDITIONALLY (R4 — never silently truncate the layer). + if (solved.droppedFindings > 0 || emit.edgesDropped > 0) { + logger.warn( + `[taint-interproc] capped: ${solved.droppedFindings} finding(s) dropped by the ` + + `per-run findings cap (${maxFindings}), ${emit.edgesDropped} edge(s) by the edge cap ` + + `(${maxEdges}) — raise pdgMaxInterprocFindings/pdgMaxInterprocEdges if intentional`, + ); + } + + if (solved.findings.length > 0 || emit.edgesEmitted > 0) { + logger.debug( + `[taint-interproc] ${summaries.length} summaries, ${callEdges.length} CALLS edges → ` + + `${solved.findings.length} cross-function finding(s), ${emit.edgesEmitted} TAINT_PATH edge(s)` + + (emit.hopsTruncated > 0 ? `, ${emit.hopsTruncated} with truncated paths` : '') + + (solved.unmatchedCallSites > 0 + ? `, ${solved.unmatchedCallSites} unmatched call site(s)` + : ''), + ); + } + + return { + summaries: summaries.length, + findings: solved.findings.length, + edgesEmitted: emit.edgesEmitted, + unmatchedCallSites: solved.unmatchedCallSites, + }; + }, +}; diff --git a/gitnexus/src/core/ingestion/pipeline.ts b/gitnexus/src/core/ingestion/pipeline.ts index d8b9fa16b..269724e0d 100644 --- a/gitnexus/src/core/ingestion/pipeline.ts +++ b/gitnexus/src/core/ingestion/pipeline.ts @@ -32,6 +32,7 @@ import { crossFilePhase, scopeResolutionPhase, pruneLocalSymbolsPhase, + taintSummariesPhase, mroPhase, communitiesPhase, processesPhase, @@ -98,6 +99,19 @@ export interface PipelineOptions { * no-CLI-flag discipline as `pdgMaxTaintFindingsPerFunction`. */ pdgMaxTaintHops?: number; + /** + * Per-run cross-function findings cap (#2084 M4 review P1-3). `undefined` ⇒ + * `DEFAULT_PDG_MAX_INTERPROC_FINDINGS` (2000); `0` ⇒ no cap. Consumed by the + * `taintSummaries` phase; RepoMeta-stamped, no CLI flag (KTD8) — same + * discipline as the per-function taint caps. + */ + pdgMaxInterprocFindings?: number; + /** Per-finding cross-function hop cap (#2084 review P1-3). `undefined` ⇒ + * `DEFAULT_MAX_INTERPROC_HOPS` (32); `0` ⇒ no cap. */ + pdgMaxInterprocHops?: number; + /** Per-run `TAINT_PATH` edge cap (#2084 review P1-3). `undefined` ⇒ + * `DEFAULT_PDG_MAX_INTERPROC_EDGES` (1000); `0` ⇒ no cap. */ + pdgMaxInterprocEdges?: number; /** * Request parsing with the worker pool disabled. The sequential parser was * removed — the worker pool is the sole parse path — so setting this now @@ -223,6 +237,10 @@ export function buildPhaseList(options?: PipelineOptions): PipelinePhase[] { .register(crossFilePhase) .register(scopeResolutionPhase) .register(pruneLocalSymbolsPhase) + // M4 (#2084): interprocedural taint fixpoint — the first real opt-in + // pdg-gated phase. Off ⇒ absent ⇒ byte-identical graph. No always-on + // phase depends on it (a filtered-out dep would throw in getPhaseOutput). + .register(taintSummariesPhase, { enabledWhen: (o) => o.pdg === true }) .register(mroPhase, { enabledWhen: (o) => !o.skipGraphPhases }) .register(communitiesPhase, { enabledWhen: (o) => !o.skipGraphPhases }) .register(processesPhase, { enabledWhen: (o) => !o.skipGraphPhases }) diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts index dc0339c42..6b748dd97 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts @@ -42,6 +42,8 @@ import { forceGc, } from '../../../../storage/parsedfile-store.js'; import type { ResolutionOutcome } from '../resolution-outcome.js'; +import type { FunctionSummary } from '../../taint/summary-model.js'; +import { buildFunctionNodeIndex } from '../../taint/summary-harvest-driver.js'; import { logger } from '../../../logger.js'; export interface ScopeResolutionOutput { @@ -64,6 +66,12 @@ export interface ScopeResolutionOutput { readonly referenceEdgesEmitted: number; } >; + /** + * Per-function taint summaries harvested in the pdg window (#2084 M4 U1), + * across all languages. Empty unless `--pdg` and a registered taint model. + * The `taintSummaries` phase composes these over the `CALLS` graph. + */ + readonly functionSummaries: readonly FunctionSummary[]; } const NOOP_OUTPUT: ScopeResolutionOutput = Object.freeze({ @@ -73,6 +81,7 @@ const NOOP_OUTPUT: ScopeResolutionOutput = Object.freeze({ referenceEdgesEmitted: 0, resolutionOutcomes: [], perLanguage: new Map(), + functionSummaries: [], }); export const scopeResolutionPhase: PipelinePhase = { @@ -143,6 +152,9 @@ export const scopeResolutionPhase: PipelinePhase = { let totalRefs = 0; let anyRan = false; const resolutionOutcomes: ResolutionOutcome[] = []; + // M4 (#2084 U1): per-function taint summaries accumulated across every + // language pass; the cross-function fixpoint phase reads this output. + const functionSummaries: FunctionSummary[] = []; const perLanguage = new Map< SupportedLanguages, { @@ -221,6 +233,14 @@ export const scopeResolutionPhase: PipelinePhase = { ); const sharedNodeLookup = totalScopeFiles > 0 ? buildGraphNodeLookup(ctx.graph) : undefined; logHeapProbe('scope-setup-nodeLookup-end', `langs=${totalScopeLangs}`); + // M4 (#2084 review P2-6): build the functionish-node index ONCE for the + // taint summary harvest, shared across every language pass (it is a whole- + // graph scan and language-agnostic). Only when pdg is on — off ⇒ undefined, + // no scan, byte-identical. + const sharedFnNodeIndex = + ctx.options?.pdg === true && totalScopeFiles > 0 + ? buildFunctionNodeIndex(ctx.graph) + : undefined; for (const [lang, provider] of SCOPE_RESOLVERS) { // Standalone providers (COBOL, JCL) don't emit graph edges yet @@ -348,6 +368,7 @@ export const scopeResolutionPhase: PipelinePhase = { files, resolutionConfig, prebuiltNodeLookup: sharedNodeLookup, + prebuiltFunctionNodeIndex: sharedFnNodeIndex, preExtractedParsedFiles: preExtractedByPath, scopeIndexStorePath: parsedFileStorePath, // CFG/PDG emission (#2081 M1) — opt-in; off ⇒ byte-identical graph. @@ -434,6 +455,7 @@ export const scopeResolutionPhase: PipelinePhase = { processedScopeFiles += langFileCount; anyRan = true; + functionSummaries.push(...stats.functionSummaries); totalFiles += stats.filesProcessed; totalImports += stats.importsEmitted; totalRefs += stats.referenceEdgesEmitted; @@ -480,6 +502,7 @@ export const scopeResolutionPhase: PipelinePhase = { referenceEdgesEmitted: totalRefs, resolutionOutcomes, perLanguage, + functionSummaries, }; }, }; diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts index 2398e5ae6..faa29b17c 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts @@ -40,6 +40,7 @@ import { isEmitSafeCfg, DEFAULT_MAX_CFG_EDGES_PER_FUNCTION, DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, + DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, REACHING_DEF_FACTS_PER_EDGE_CAP, } from '../../cfg/emit.js'; import { @@ -50,6 +51,12 @@ import { } from '../../taint/emit.js'; import { registerBuiltinTaintModels } from '../../taint/typescript-model.js'; import { getSourceSinkConfig } from '../../taint/source-sink-registry.js'; +import { + buildFunctionNodeIndex, + harvestFileSummaries, + type FunctionNodeIndex, +} from '../../taint/summary-harvest-driver.js'; +import type { FunctionSummary } from '../../taint/summary-model.js'; import type { FunctionCfg } from '../../cfg/types.js'; import { resolveDefGraphId } from '../graph-bridge/ids.js'; import { buildPopulatedMethodDispatch } from '../graph-bridge/method-dispatch.js'; @@ -302,6 +309,14 @@ interface RunScopeResolutionInput { * base is safe. */ readonly prebuiltNodeLookup?: ReturnType; + /** + * Functionish-node index built ONCE by the caller and shared across every + * language pass (#2084 review P2-6). Like `prebuiltNodeLookup`, + * `buildFunctionNodeIndex` is a whole-graph scan and is language-agnostic, so + * rebuilding it per language wastes a full scan each time. When omitted + * (tests / isolated calls) it is built locally for the pdg-enabled language. + */ + readonly prebuiltFunctionNodeIndex?: FunctionNodeIndex; /** * Opaque per-language import-resolution config (e.g. tsconfig path * aliases for TypeScript). Loaded once by the caller via @@ -360,6 +375,13 @@ interface RunScopeResolutionStats { readonly referenceEdgesEmitted: number; readonly referenceSkipped: number; readonly resolutionOutcomes: readonly ResolutionOutcome[]; + /** + * Per-function taint summaries harvested in the pdg window (#2084 M4 U1). + * Empty unless `input.pdg === true` and the language has a registered taint + * model. Keyed by resolved `Function`/`Method` node id; the cross-function + * fixpoint phase composes them over the complete `CALLS` graph. + */ + readonly functionSummaries: readonly FunctionSummary[]; } export function runScopeResolution( @@ -477,6 +499,7 @@ export function runScopeResolution( referenceEdgesEmitted: 0, referenceSkipped: 0, resolutionOutcomes, + functionSummaries: [], }; } @@ -730,6 +753,11 @@ export function runScopeResolution( // pair can't bracket them; without this accumulator the M2 cost would // silently disappear into `emit=` and field regressions would be invisible. let pdgMs = 0; + // M4 (#2084 U1): per-function taint summaries harvested in the pdg window, + // returned on the stats for the cross-function fixpoint phase. Function-scoped + // so the return (below the pdg block) can read it; empty on non-pdg runs. + const harvestedSummaries: FunctionSummary[] = []; + let summaryUnresolved = 0; // M3 (#2083 U4): accumulated taint time (match + taint-side solve + // propagate + TAINTED/SANITIZES emit), a sibling of `pdgMs` for the same // reason — it interleaves per file inside `emit=`, so only an accumulator @@ -784,6 +812,14 @@ export function runScopeResolution( gapExamples: [] as string[], dropExamples: [] as string[], }; + // M4 (#2084 U1): per-function summary harvest. The functionish-node index + // is built ONCE (whole-graph scan) and reused across every file; summaries + // accumulate here and ride out on the stats for the cross-function fixpoint + // phase. Only built when the language has a registered taint model. + const fnNodeIndex = + taintSpec !== undefined + ? (input.prebuiltFunctionNodeIndex ?? buildFunctionNodeIndex(graph)) + : undefined; for (const pf of emitParsedFiles) { const cfgs = pf.cfgSideChannel; // Defensive: cfgSideChannel is opaque (`unknown`) and crosses the cache / @@ -872,6 +908,25 @@ export function runScopeResolution( for (const ex of taint.droppedExamples) { if (taintTotals.dropExamples.length < 5) taintTotals.dropExamples.push(ex); } + + // M4 (#2084 U1): harvest per-function summaries over the SAME + // emit-safe CFGs, inside the SAME per-file try. Pure aside from the + // read-only node-index lookup; the cross-function fixpoint phase + // consumes `harvestedSummaries` once the whole call graph is built. + if (fnNodeIndex !== undefined) { + const harvest = harvestFileSummaries( + fnNodeIndex, + wellFormed, + pf.parsedImports, + taintSpec, + // Same fact cap the taint-side RD solve uses (coverage parity). + taintLimits.maxFacts && taintLimits.maxFacts > 0 + ? taintLimits.maxFacts + : DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, + ); + harvestedSummaries.push(...harvest.summaries); + summaryUnresolved += harvest.unresolved; + } } } catch (err) { // Last-resort isolation, mirroring the worker-side per-file try/catch: @@ -942,6 +997,16 @@ export function runScopeResolution( logger.warn(`[taint] lang=${provider.language}: ${parts.join('; ')}`); } } + // M4 (#2084 U1): summary harvest volume + anchor-resolution diagnostics. + if (harvestedSummaries.length > 0 || summaryUnresolved > 0) { + logger.debug( + `[taint-summary] lang=${provider.language}: ${harvestedSummaries.length} function ` + + `summary/summaries harvested` + + (summaryUnresolved > 0 + ? `, ${summaryUnresolved} CFG anchor(s) unresolved (same-line collision or missing node)` + : ''), + ); + } } if (PROF) { @@ -971,5 +1036,6 @@ export function runScopeResolution( referenceEdgesEmitted: emitted + receiverExtras + unresolvedReceiverExtras + freeCallExtras, referenceSkipped: skipped, resolutionOutcomes, + functionSummaries: harvestedSummaries, }; } diff --git a/gitnexus/src/core/ingestion/taint/interproc-emit.ts b/gitnexus/src/core/ingestion/taint/interproc-emit.ts new file mode 100644 index 000000000..4dc69f747 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/interproc-emit.ts @@ -0,0 +1,120 @@ +/** + * Interprocedural taint emission (#2084 M4 U4) — materialise `TAINT_PATH`. + * + * Persists each cross-function {@link InterprocFinding} as ONE `TAINT_PATH` + * edge from the source function node to the sink function node, with the + * function-level hop chain + sink kind encoded in `reason` via the SHARED + * `path-codec` (the same versioned wire format M3's intra-procedural `TAINTED` + * edges use — never a second hand-rolled codec). The MCP `explain` tool decodes + * it for cross-function path rendering (U7). + * + * `TAINT_PATH` was reserved at M0 (the RelationshipType + the `CodeRelation` + * Function/Method node-pairs already exist), so materialisation needs zero + * schema work. Like `TAINTED`, it stays out of `VALID_RELATION_TYPES` and the + * web schema — `explain` is the discovery surface. + * + * Boundedness mirrors the M3 emit driver: dedup-before-cap (the solver already + * deduped by `(source, sink, kind)`), a per-run findings cap, and unconditional + * truncate-and-warn — never a silent drop. + */ + +import type { KnowledgeGraph } from '../../graph/types.js'; +import { encodeTaintPath, type TaintPathHopInput } from './path-codec.js'; +import type { InterprocFinding } from './interproc-solver.js'; + +/** Confidence stamped on interprocedural `TAINT_PATH` edges. Lower than the + * intra-procedural `TAINTED` 1.0 — context-insensitive composition is a + * coarser signal (return/call-site merging). */ +export const INTERPROC_TAINT_CONFIDENCE = 0.6; + +/** + * Default per-run cap on emitted `TAINT_PATH` edges (#2084 review P1-3). + * Resolved into `RepoMeta.pdg` like the other pdg caps; `0` ⇒ unlimited. + */ +export const DEFAULT_PDG_MAX_INTERPROC_EDGES = 1000; + +export interface InterprocEmitLimits { + /** Max `TAINT_PATH` edges per run (post-dedup). `undefined`/0 ⇒ unlimited. */ + readonly maxEdges?: number; +} + +export interface InterprocEmitResult { + /** TAINT_PATH edges persisted. */ + edgesEmitted: number; + /** Findings dropped by the per-run cap. */ + edgesDropped: number; + /** Findings whose persisted hop path is a truncated prefix. */ + hopsTruncated: number; + /** Findings skipped because an endpoint node was missing from the graph. */ + skippedMissingEndpoint: number; +} + +/** + * Persist cross-function findings as `TAINT_PATH` edges. `findings` is assumed + * deduped + deterministically ordered (the solver's contract). Never throws on + * valid input. + */ +export function emitInterprocTaint( + graph: KnowledgeGraph, + findings: readonly InterprocFinding[], + limits?: InterprocEmitLimits, + onWarn?: (message: string) => void, +): InterprocEmitResult { + const result: InterprocEmitResult = { + edgesEmitted: 0, + edgesDropped: 0, + hopsTruncated: 0, + skippedMissingEndpoint: 0, + }; + const maxEdges = limits?.maxEdges && limits.maxEdges > 0 ? limits.maxEdges : Infinity; + const seen = new Set(); + + for (const finding of findings) { + if (result.edgesEmitted >= maxEdges) { + result.edgesDropped++; + continue; + } + const sourceNode = graph.getNode(finding.sourceFnId); + const sinkNode = graph.getNode(finding.sinkFnId); + if (!sourceNode || !sinkNode) { + result.skippedMissingEndpoint++; + continue; + } + + // Map function hops → codec hops. The hop "name" is the function's display + // name (identifier charset — codec-safe); the line is its start line. + const hops: TaintPathHopInput[] = finding.hops.map((h) => { + const node = graph.getNode(h.fnId); + const name = typeof node?.properties.name === 'string' ? node.properties.name : 'fn'; + const line = typeof node?.properties.startLine === 'number' ? node.properties.startLine : 0; + return { name, line }; + }); + const encoded = encodeTaintPath(hops, { + kind: finding.sinkKind, + truncated: finding.hopsTruncated, + }); + if (encoded.truncated) result.hopsTruncated++; + + const id = `rel:TAINT_PATH:${finding.sinkKind}:${finding.sourceFnId}=>${finding.sinkFnId}`; + if (seen.has(id)) continue; + seen.add(id); + + graph.addRelationship({ + id, + sourceId: finding.sourceFnId, + targetId: finding.sinkFnId, + type: 'TAINT_PATH', + confidence: INTERPROC_TAINT_CONFIDENCE, + reason: encoded.reason, + }); + result.edgesEmitted++; + } + + if (result.edgesDropped > 0) { + onWarn?.( + `[taint-interproc] ${result.edgesDropped} cross-function finding(s) dropped by the ` + + `per-run TAINT_PATH cap (${maxEdges})`, + ); + } + return result; +} diff --git a/gitnexus/src/core/ingestion/taint/interproc-solver.ts b/gitnexus/src/core/ingestion/taint/interproc-solver.ts new file mode 100644 index 000000000..50d49c4b6 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/interproc-solver.ts @@ -0,0 +1,412 @@ +/** + * Interprocedural taint fixpoint (#2084 M4 U3). + * + * Composes per-function {@link FunctionSummary} objects over the resolved + * `CALLS` graph to find source→sink flows that cross function and file + * boundaries. PURE AND DETERMINISTIC (no graph, no I/O, no logger) — the phase + * builds the inputs from `ctx.graph` and persists the outputs. + * + * ## The model — whole-parameter taint reachability + * + * The unit of taint is `(function, parameter)`. The fixpoint computes the set + * of parameters that can hold source-derived data, then fires a finding + * whenever a tainted parameter feeds a modelled sink (`paramToSink`). + * + * - **Seeds** — every `sourceToCallArg` edge: a function generates a source and + * passes it into argument `argIndex` of a call at `callLine`. Resolving that + * call site against the caller's outgoing `CALLS` edges yields the callee; + * the callee's parameter `argIndex` becomes tainted, with the generating + * function recorded as the flow's source. + * - **Propagation** — every `paramToCallArg` edge of a function whose parameter + * is ALREADY tainted: `param i → arg j of callee` taints the callee's + * parameter `j` (TITO composition). Iterated to a fixpoint. + * - **Findings** — whenever a parameter becomes tainted and the owning + * function's `paramToSink` contains that parameter, a cross-function finding + * is emitted (source function → sink function, with the kind). + * + * ## Cycle safety (recursion) + * + * The tainted-parameter set is monotone over a FINITE lattice (`Σ functions × + * params`), so the worklist fixpoint converges: a recursive or mutually + * recursive call merely re-proposes an already-tainted parameter, which the + * visited-set absorbs — no infinite descent. This is the functional/summary + * method's standard termination argument (Sharir-Pnueli; Pysa, Mariana Trench, + * and Infer all rely on it). SCC condensation would only refine the PROCESSING + * ORDER; correctness and termination do not require it. + * + * ## Context-insensitivity & the name-join over-approximation + * + * One summary per function, applied at every call site — return/param merging + * is accepted (the security-conservative direction). The call-arg→callee join + * is by callee NAME (not line), so when one caller invokes two DISTINCT + * same-named callees (`x.handler(src)` and `y.handler(clean)`), a source that + * flowed into ONE of them taints BOTH callees' parameter — an extra finding on + * the callee the source did not reach. This is sound (over-attribution, never a + * missed flow — the conservative direction for a security tool) and is the + * documented price of dropping the fragile line-based join; the `explain` tool + * surfaces it ("may over-attribute among same-named callees"). Other known + * precision losses (call-site conflation, shared dispatch, callbacks) are the + * documented M4 trade-offs; refinements are deferred (plan KTD). + */ + +import type { SinkKind } from './source-sink-config.js'; +import type { FunctionSummary } from './summary-model.js'; + +/** + * One resolved call edge from the `CALLS` graph. The join to a summary's + * call-arg edge is by CALLEE NAME (the callee node's declared name), NOT by + * call-site line — line-base parity between the CFG harvest (1-based) and the + * reference site is fragile, while the callee identity is exact and the + * context-insensitive model tatints the callee's parameter the same way at + * every call site to it. + */ +export interface InterprocCallEdge { + readonly callerId: string; + readonly calleeId: string; + /** The callee node's declared name (`helper`, `process`) — the join key. */ + readonly calleeName: string; +} + +/** One hop of a cross-function flow: the function entered, and how. */ +export interface InterprocHop { + readonly fnId: string; + /** The call-site line in the PREVIOUS function that entered this one. */ + readonly callLine?: number; + /** Argument position the taint entered through (undefined for the source fn). */ + readonly argIndex?: number; +} + +export interface InterprocFinding { + readonly sourceFnId: string; + readonly sinkFnId: string; + readonly sinkKind: SinkKind; + /** Ordered source→sink hop chain (functions). A prefix when `truncated`. */ + readonly hops: readonly InterprocHop[]; + readonly hopsTruncated: boolean; +} + +export interface InterprocLimits { + /** Max functions in a single flow's hop chain. `undefined`/0 ⇒ default 32. */ + readonly maxHops?: number; + /** Max findings overall (post-dedup). `undefined`/0 ⇒ unlimited. */ + readonly maxFindings?: number; +} + +export interface InterprocResult { + readonly findings: readonly InterprocFinding[]; + /** Findings dropped by `maxFindings` (post-dedup). */ + readonly droppedFindings: number; + /** Call edges whose call-site line matched no summary edge (diagnostics). */ + readonly unmatchedCallSites: number; +} + +export const DEFAULT_MAX_INTERPROC_HOPS = 32; + +/** + * Default per-run cap on cross-function findings (#2084 review P1-3). Like the + * other pdg caps it is resolved into `RepoMeta.pdg` so `pdgModeMismatch` + * stamps it; `0` ⇒ unlimited. 2000 is generous for a real repo — more deduped + * `(source, sink, kind)` findings than that is a fixture or a runaway fan-in, + * and the overflow is deterministic + counted (`droppedFindings`). + */ +export const DEFAULT_PDG_MAX_INTERPROC_FINDINGS = 2000; + +/** A tainted parameter, with the flow that first tainted it (for path reconstruction). */ +interface TaintedParam { + readonly fnId: string; + readonly paramIndex: number; + readonly sourceFnId: string; + /** Hop chain from source to this `(fnId, paramIndex)` entry. */ + readonly hops: readonly InterprocHop[]; + readonly truncated: boolean; + /** + * Sink kinds neutralised on the composed path to here (#2084 review P1-2) — + * UNION along the hop chain (a sanitizer at any upstream call-arg stays + * neutralised downstream). A `paramToSink` of a kind in this set does NOT + * fire (the cross-function sanitizer). Mutable in spirit: on revisit by a + * less-neutralised path the stored set INTERSECTS (mirrors `propagate.ts`). + */ + readonly neutralized: ReadonlySet; +} + +/** + * Taint-state key — `(function, parameter, SOURCE)`. The source discriminator + * is load-bearing: without it, a parameter tainted by source A is marked + * visited and a later flow from source B to the SAME parameter is dropped + * before it can fire that function's sink, silently losing B→sink (the + * multi-source collapse — the recurring M3 bug class). Including the source + * keeps each origin's flow independent; the lattice stays finite (`fn × param × + * source`), so the monotone worklist still terminates and is cycle-safe. + */ +const pkey = (fnId: string, param: number, sourceFnId: string): string => + `${fnId}#${param}#${sourceFnId}`; + +/** + * Run the interprocedural taint fixpoint. `summaries` is keyed by function node + * id; `callEdges` is the resolved `CALLS` graph (caller→callee with call-site + * lines). Deterministic: inputs in, sorted findings out. + */ +export function solveInterprocTaint( + summaries: ReadonlyMap, + callEdges: readonly InterprocCallEdge[], + limits?: InterprocLimits, +): InterprocResult { + const maxHops = + limits?.maxHops && limits.maxHops > 0 ? limits.maxHops : DEFAULT_MAX_INTERPROC_HOPS; + + // Adjacency built ONCE (#2084 review P3-8): callerId → outgoing edges, AND + // callerId → calleeName → edges. The summary's call-arg edges resolve by + // callee NAME, so the per-name index turns each resolution into an O(1) + // lookup instead of a per-worklist-step `.filter` allocation (the + // build-index-once pattern). + const callsByCaller = new Map(); + const callsByCallerName = new Map>(); + for (const e of callEdges) { + const list = callsByCaller.get(e.callerId); + if (list) list.push(e); + else callsByCaller.set(e.callerId, [e]); + let byName = callsByCallerName.get(e.callerId); + if (!byName) { + byName = new Map(); + callsByCallerName.set(e.callerId, byName); + } + const named = byName.get(e.calleeName); + if (named) named.push(e); + else byName.set(e.calleeName, [e]); + } + let unmatchedCallSites = 0; + + /** Edges to `name` from `callerId` (O(1)); empty if none — non-counting. */ + const calleesByName = (callerId: string, name: string): InterprocCallEdge[] => + callsByCallerName.get(callerId)?.get(name) ?? []; + + // Resolve a caller's call-arg edge (by callee name) to concrete callee edges. + // An unknown callee name (chain not statically resolvable) conservatively + // matches EVERY outgoing call — sound over-approximation (may over-taint). + const resolveCallees = ( + callerId: string, + calleeName: string | undefined, + ): InterprocCallEdge[] => { + const candidates = callsByCaller.get(callerId); + if (!candidates || candidates.length === 0) { + unmatchedCallSites++; + return []; + } + if (calleeName === undefined) return candidates; + const named = calleesByName(callerId, calleeName); + if (named.length === 0) { + unmatchedCallSites++; + return []; + } + return named; + }; + + // ── findings + worklist ─────────────────────────────────────────────────── + const findingsByKey = new Map(); + const tainted = new Map(); + const queue: TaintedParam[] = []; + + const recordFinding = ( + sourceFnId: string, + sinkFnId: string, + sinkKind: SinkKind, + hops: readonly InterprocHop[], + truncated: boolean, + ): void => { + const key = `${sourceFnId}|${sinkFnId}|${sinkKind}`; + if (findingsByKey.has(key)) return; + findingsByKey.set(key, { sourceFnId, sinkFnId, sinkKind, hops, hopsTruncated: truncated }); + }; + + /** Fire every `paramToSink` of `tp`'s param, except kinds it neutralised. */ + const fireSinks = (tp: TaintedParam): void => { + const summary = summaries.get(tp.fnId); + if (!summary) return; + for (const ps of summary.paramToSink) { + if (ps.param !== tp.paramIndex) continue; + if (tp.neutralized.has(ps.sinkKind)) continue; // sanitised across the boundary (P1-2) + // `tp.hops` already terminates at this (tainted) function — it IS the + // source→sink chain, no extra hop to append. + recordFinding(tp.sourceFnId, tp.fnId, ps.sinkKind, tp.hops, tp.truncated); + } + }; + + /** + * Mark (fnId, paramIndex, source) tainted; enqueue. On a fresh key, taint + + * fire sinks. On revisit, INTERSECT the neutralised set (a kind stays + * neutralised only if EVERY path neutralises it — the sound direction); if it + * shrank, re-enqueue + re-fire so a less-neutralised path's sinks surface + * (the shrink-reprocess guard, mirroring `propagate.ts:deriveTaint`). Without + * it, a first more-neutralised path would freeze out a real finding (FN). + */ + const taint = (tp: TaintedParam): void => { + const key = pkey(tp.fnId, tp.paramIndex, tp.sourceFnId); + const existing = tainted.get(key); + if (existing) { + const inter = new Set(); + for (const k of existing.neutralized) if (tp.neutralized.has(k)) inter.add(k); + if (inter.size >= existing.neutralized.size) return; // no shrink — cycle-safe + const merged: TaintedParam = { ...existing, neutralized: inter }; + tainted.set(key, merged); + queue.push(merged); + fireSinks(merged); + return; + } + tainted.set(key, tp); + queue.push(tp); + fireSinks(tp); + }; + + // ── seeds: every source→callee-arg, resolved against CALLS ──────────────── + for (const [callerId, summary] of summaries) { + for (const sc of summary.sourceToCallArg) { + for (const edge of resolveCallees(callerId, sc.calleeName)) { + const callee = summaries.get(edge.calleeId); + if (!callee) continue; + if (sc.argIndex >= callee.paramCount) continue; // arity guard + // Build the seed path through the capped append so `maxHops` truncates + // the prefix (#2084 review P2-7), not a 2-entry path flagged truncated. + const seed = appendHop( + [{ fnId: callerId }], + { fnId: edge.calleeId, callLine: sc.callLine, argIndex: sc.argIndex }, + maxHops, + ); + taint({ + fnId: edge.calleeId, + paramIndex: sc.argIndex, + sourceFnId: callerId, + hops: seed.hops, + truncated: seed.truncated, + neutralized: new Set(sc.neutralized ?? []), + }); + } + } + } + + // ── generative return composition (#2084 review P1-1) ───────────────────── + // `genReturns` = functions whose RETURN carries a generated source. Seed with + // `sourceToReturn`; a caller that returns the result of a generative call is + // itself generative (transitive — `wrap(){ return getInput() }`). Small + // monotone fixpoint over the name-resolved call graph (`calleesByName`). + const genReturns = new Set(); + for (const [id, s] of summaries) if (s.sourceToReturn.length > 0) genReturns.add(id); + let grChanged = true; + while (grChanged) { + grChanged = false; + for (const [callerId, s] of summaries) { + if (genReturns.has(callerId)) continue; + for (const cr of s.callResults) { + if (cr.dest.to !== 'return') continue; + if (calleesByName(callerId, cr.calleeName).some((e) => genReturns.has(e.calleeId))) { + genReturns.add(callerId); + grChanged = true; + break; + } + } + } + } + // Compose: a caller using a generative call's result either FIRES (the result + // hits a sink) or SEEDS (the result flows into another call's arg). The + // generated source's origin is the generative callee. + for (const [callerId, s] of summaries) { + for (const cr of s.callResults) { + const generative = calleesByName(callerId, cr.calleeName).filter((e) => + genReturns.has(e.calleeId), + ); + if (generative.length === 0) continue; + for (const g of generative) { + const d = cr.dest; + if (d.to === 'sink') { + recordFinding( + g.calleeId, + callerId, + d.sinkKind, + [{ fnId: g.calleeId }, { fnId: callerId }], + 2 > maxHops, + ); + } else if (d.to === 'callArg') { + for (const tc of d.toCallee === undefined + ? (callsByCaller.get(callerId) ?? []) + : calleesByName(callerId, d.toCallee)) { + const callee = summaries.get(tc.calleeId); + if (!callee || d.argIndex >= callee.paramCount) continue; + // Capped successive append so `maxHops` truncates the prefix (P2-7). + const h1 = appendHop([{ fnId: g.calleeId }], { fnId: callerId }, maxHops); + const h2 = appendHop(h1.hops, { fnId: tc.calleeId, argIndex: d.argIndex }, maxHops); + taint({ + fnId: tc.calleeId, + paramIndex: d.argIndex, + sourceFnId: g.calleeId, + hops: h2.hops, + truncated: h1.truncated || h2.truncated, + neutralized: new Set(), + }); + } + } + // dest:'return' is already folded into `genReturns` above. + } + } + } + + // ── propagation worklist ────────────────────────────────────────────────── + let head = 0; + while (head < queue.length) { + const tp = queue[head++]; + const summary = summaries.get(tp.fnId); + if (!summary) continue; + // This function's tainted param flows into callee args via paramToCallArg. + for (const pc of summary.paramToCallArg) { + if (pc.param !== tp.paramIndex) continue; + for (const edge of resolveCallees(tp.fnId, pc.calleeName)) { + const callee = summaries.get(edge.calleeId); + if (!callee) continue; + if (pc.argIndex >= callee.paramCount) continue; + const next = appendHop( + tp.hops, + { fnId: edge.calleeId, callLine: pc.callLine, argIndex: pc.argIndex }, + maxHops, + ); + // Union the edge's neutralised kinds onto the composed path (a + // sanitizer between this param and the callee arg stays neutralised). + const neutralized = + pc.neutralized && pc.neutralized.length > 0 + ? new Set([...tp.neutralized, ...pc.neutralized]) + : tp.neutralized; + taint({ + fnId: edge.calleeId, + paramIndex: pc.argIndex, + sourceFnId: tp.sourceFnId, + hops: next.hops, + truncated: tp.truncated || next.truncated, + neutralized, + }); + } + } + } + + // ── deterministic assembly ──────────────────────────────────────────────── + const all = [...findingsByKey.values()].sort( + (a, b) => + a.sourceFnId.localeCompare(b.sourceFnId) || + a.sinkFnId.localeCompare(b.sinkFnId) || + a.sinkKind.localeCompare(b.sinkKind), + ); + const maxFindings = limits?.maxFindings && limits.maxFindings > 0 ? limits.maxFindings : Infinity; + const findings = all.length > maxFindings ? all.slice(0, maxFindings) : all; + + return { + findings, + droppedFindings: all.length - findings.length, + unmatchedCallSites, + }; +} + +/** Append a hop, respecting the hop cap (keeps the source-side prefix). */ +function appendHop( + hops: readonly InterprocHop[], + hop: InterprocHop, + maxHops: number, +): { hops: readonly InterprocHop[]; truncated: boolean } { + if (hops.length >= maxHops) return { hops, truncated: true }; + return { hops: [...hops, hop], truncated: hops.length + 1 > maxHops }; +} diff --git a/gitnexus/src/core/ingestion/taint/propagate.ts b/gitnexus/src/core/ingestion/taint/propagate.ts index 0330f2d32..e82e7ff1a 100644 --- a/gitnexus/src/core/ingestion/taint/propagate.ts +++ b/gitnexus/src/core/ingestion/taint/propagate.ts @@ -105,7 +105,12 @@ import type { MatchedSinkCall, StatementMatches, } from './match.js'; -import type { SinkKind, SourceKind } from './source-sink-config.js'; +import { + SINK_KIND_ORDER as KIND_ORDER, + sortSinkKinds as sortKinds, + type SinkKind, + type SourceKind, +} from './source-sink-config.js'; /** * Default per-function findings cap (U5 config resolution; cfg/emit.ts @@ -226,17 +231,10 @@ export interface FunctionTaintResult { readonly droppedFindings: number; } -/** Canonical SinkKind order for deterministic `neutralized` arrays. */ -const KIND_ORDER: readonly SinkKind[] = [ - 'code-injection', - 'command-injection', - 'path-traversal', - 'sql-injection', - 'xss', -]; +// Canonical SinkKind order + sort live in source-sink-config.ts (shared with +// the M4 summary harvest so the deterministic order never drifts); imported +// above as KIND_ORDER / sortKinds. `kindRank` is the local comparator index. const kindRank = new Map(KIND_ORDER.map((k, i) => [k, i])); -const sortKinds = (kinds: Iterable): SinkKind[] => - [...new Set(kinds)].sort((a, b) => (kindRank.get(a) ?? 99) - (kindRank.get(b) ?? 99)); const EMPTY_KINDS: ReadonlySet = new Set(); diff --git a/gitnexus/src/core/ingestion/taint/source-sink-config.ts b/gitnexus/src/core/ingestion/taint/source-sink-config.ts index 5909dcb3e..b1f9d3c86 100644 --- a/gitnexus/src/core/ingestion/taint/source-sink-config.ts +++ b/gitnexus/src/core/ingestion/taint/source-sink-config.ts @@ -117,3 +117,33 @@ export interface SourceSinkSanitizerSpec { readonly sinks: readonly TaintSinkEntry[]; readonly sanitizers: readonly TaintSanitizerEntry[]; } + +/** + * Canonical deterministic ordering of {@link SinkKind} values. The single + * source of this order — the intra-procedural propagation engine + * (`propagate.ts`) and the M4 summary harvest (`summary-harvest.ts`) both sort + * `neutralized`/exclusion sets by it so their deterministic outputs (and the + * summary version stamp) stay stable. Lives here, next to the `SinkKind` + * union, so the two consumers never drift. + */ +export const SINK_KIND_ORDER: readonly SinkKind[] = [ + 'code-injection', + 'command-injection', + 'path-traversal', + 'sql-injection', + 'xss', +]; + +const SINK_KIND_RANK = new Map(SINK_KIND_ORDER.map((k, i) => [k, i])); + +/** Dedupe + sort sink kinds by {@link SINK_KIND_ORDER} (deterministic). */ +export function sortSinkKinds(kinds: Iterable): SinkKind[] { + return [...new Set(kinds)].sort( + (a, b) => (SINK_KIND_RANK.get(a) ?? 99) - (SINK_KIND_RANK.get(b) ?? 99), + ); +} + +/** Rank of a sink kind in {@link SINK_KIND_ORDER} (for comparator chaining). */ +export function sinkKindRank(kind: SinkKind): number { + return SINK_KIND_RANK.get(kind) ?? 99; +} diff --git a/gitnexus/src/core/ingestion/taint/summary-harvest-driver.ts b/gitnexus/src/core/ingestion/taint/summary-harvest-driver.ts new file mode 100644 index 000000000..8b5a3f864 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/summary-harvest-driver.ts @@ -0,0 +1,147 @@ +/** + * Summary-harvest driver (#2084 M4 U1) — the in-phase orchestration that turns + * per-function CFGs into call-graph-keyed {@link FunctionSummary} objects. + * + * Runs inside the scope-resolution pdg window (alongside `emitFileTaint`), + * where both the live CFG side channel AND the structure-phase `Function` / + * `Method` graph nodes are available. For each emit-safe CFG it: + * + * 1. resolves the CFG's source anchor `(filePath, functionStartLine)` to its + * graph node id, so the summary speaks the call graph's language directly — + * the interprocedural fixpoint then joins summaries to `CALLS` edges by node + * id with no fragile re-derivation; + * 2. runs the pure {@link harvestFunctionSummary} over the same RD facts + + * matched sites the M3 taint pass uses; + * 3. stamps the own-facts `version` (#2084 review P1-1: callee-version + * composition is RESERVED — the fixpoint does not recompute it today). + * + * ## The Function↔CFG join (load-bearing) + * + * `FunctionCfg.functionStartLine` is 1-based (the TS visitor's `row + 1`); + * `Function`/`Method` node `startLine` is 0-based (`startPosition.row`). The + * join therefore looks up node start line `functionStartLine - 1` + * ({@link NODE_TO_CFG_LINE_OFFSET}). Function nodes carry no start column, so a + * `(filePath, startLine)` collision — two functions opening on one line, + * `{ a: () => x(), b: () => y() }` — is ambiguous: the CFG disambiguates with + * `functionStartColumn` but the node does not, so a colliding anchor is DROPPED + * (counted as `unresolved`) rather than risk attaching a summary to the wrong + * function. Rare in practice; the alternative (cross-wired summaries) is unsound. + */ + +import type { ParsedImport, GraphNode } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../../graph/types.js'; +import { computeReachingDefs } from '../cfg/reaching-defs.js'; +import { DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION } from '../cfg/emit.js'; +import type { FunctionCfg } from '../cfg/types.js'; +import { buildTaintImportIndex, matchFunctionSites } from './match.js'; +import type { SourceSinkSanitizerSpec } from './source-sink-config.js'; +import { harvestFunctionSummary } from './summary-harvest.js'; +import { ownFactsDigest, summaryVersion, type FunctionSummary } from './summary-model.js'; + +/** `cfg.functionStartLine` (1-based) − this = the node's 0-based `startLine`. */ +export const NODE_TO_CFG_LINE_OFFSET = 1; + +/** Node labels that can own a CFG / be a `CALLS` endpoint. */ +const FUNCTIONISH_LABELS = new Set(['Function', 'Method']); + +/** + * Index of functionish graph nodes by `filePath → startLine(0-based) → ids`. + * Built ONCE per scope-resolution pass (the graph is whole-repo); reused across + * every file's harvest. + */ +export type FunctionNodeIndex = ReadonlyMap>; + +export function buildFunctionNodeIndex(graph: KnowledgeGraph): FunctionNodeIndex { + const index = new Map>(); + const add = (node: GraphNode): void => { + if (!FUNCTIONISH_LABELS.has(node.label)) return; + const filePath = node.properties.filePath; + const startLine = node.properties.startLine; + if (typeof filePath !== 'string' || typeof startLine !== 'number') return; + let byLine = index.get(filePath); + if (!byLine) { + byLine = new Map(); + index.set(filePath, byLine); + } + const ids = byLine.get(startLine); + if (ids) ids.push(node.id); + else byLine.set(startLine, [node.id]); + }; + for (const node of graph.iterNodes()) add(node); + return index; +} + +/** Resolve a CFG anchor to a unique functionish node id, or undefined. */ +function resolveFnId(fnIndex: FunctionNodeIndex, cfg: FunctionCfg): string | undefined { + const byLine = fnIndex.get(cfg.filePath); + if (!byLine) return undefined; + const ids = byLine.get(cfg.functionStartLine - NODE_TO_CFG_LINE_OFFSET); + // Unique match only — a same-line collision is unresolvable (no node column). + return ids && ids.length === 1 ? ids[0] : undefined; +} + +export interface FileSummaryResult { + readonly summaries: readonly FunctionSummary[]; + /** CFGs whose anchor resolved to no unique graph node (collision / missing). */ + readonly unresolved: number; + /** CFGs whose reaching-defs were not `computed` (no summary produced). */ + readonly gaps: number; +} + +/** + * Harvest summaries for one file's emit-safe CFGs. `cfgs` MUST already be + * `isEmitSafeCfg`-filtered (the same `wellFormed` array fed to `emitFileTaint`). + * Pure aside from the read-only graph lookup; never throws on valid input. + */ +export function harvestFileSummaries( + fnIndex: FunctionNodeIndex, + cfgs: readonly FunctionCfg[], + parsedImports: readonly ParsedImport[], + spec: SourceSinkSanitizerSpec, + maxFacts: number = DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, +): FileSummaryResult { + const importIndex = buildTaintImportIndex(parsedImports); + const summaries: FunctionSummary[] = []; + let unresolved = 0; + let gaps = 0; + + for (const cfg of cfgs) { + const fnId = resolveFnId(fnIndex, cfg); + if (fnId === undefined) { + unresolved++; + continue; + } + const defUse = computeReachingDefs(cfg, { maxFacts }); + const matches = matchFunctionSites(cfg, spec, importIndex); + const harvested = harvestFunctionSummary(cfg, defUse, matches); + if (harvested.status !== 'computed') { + gaps++; + continue; + } + const facts = harvested.facts; + // Skip functions with NO taint behaviour at all — they cannot participate + // in any flow and would only bloat the fixpoint's working set. + if ( + facts.paramToReturn.length === 0 && + facts.paramToCallArg.length === 0 && + facts.paramToSink.length === 0 && + facts.sourceToReturn.length === 0 && + facts.sourceToCallArg.length === 0 && + facts.callResults.length === 0 + ) { + continue; + } + const digest = ownFactsDigest(facts); + summaries.push({ + fnId, + filePath: cfg.filePath, + startLine: cfg.functionStartLine, + ...facts, + // Provisional own-only version; the fixpoint recomputes with callee + // versions once the call graph is condensed. + version: summaryVersion(digest, []), + }); + } + + return { summaries, unresolved, gaps }; +} diff --git a/gitnexus/src/core/ingestion/taint/summary-harvest.ts b/gitnexus/src/core/ingestion/taint/summary-harvest.ts new file mode 100644 index 000000000..4924a0e18 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/summary-harvest.ts @@ -0,0 +1,592 @@ +/** + * Per-function taint SUMMARY harvest (#2084 M4 U1). + * + * Pure, deterministic derivation of one function's {@link FunctionSummary} + * facts from the SAME substrate the M3 intra-procedural pass consumes — the M2 + * reaching-definition facts (`computeReachingDefs`) and the matched taint sites + * (`matchFunctionSites`). No graph, no I/O, no logger; mirrors the + * `computeReachingDefs` / `computeTaintFlows` contract (insertion-ordered + * worklist, explicitly sorted outputs) so snapshot tests and the version stamp + * stay stable. Runs IN-PHASE inside the scope-resolution pdg window where the + * CFG side channel is live (plan KTD1); the cross-function fixpoint that + * COMPOSES these summaries runs afterward over the complete call graph. + * + * ## What a summary captures (whole-parameter granularity) + * + * Seeding each formal parameter as taint and running forward reachability over + * the def→use facts yields four edge categories: + * + * - **param→return** — a param's value reaches a `return `. Return + * statements are identified structurally: the SOURCE block of every CFG edge + * of kind `return` terminates in the return jump (the M2 edge-kind + * invariant), so its last statement's `uses` are the returned bindings. + * - **param→callee-arg** — a param occurrence lands in argument position + * `argIndex` of a call at `callLine`. The fixpoint resolves `callLine` to a + * callee via the caller's `CALLS` edges and applies the callee's summary + * (TITO composition). + * - **param→sink** — a param reaches a modelled sink position (the partial + * flow that a cross-function source completes). + * - **source→return** — a modelled source read (`req.body`) reaches the return + * (a generative summary: calling the function yields tainted data). + * + * ## Soundness model (context-insensitive first cut) + * + * Onward propagation uses the M3 STATEMENT-LEVEL precision floor: a statement + * that uses a tainted binding taints all of its defs (and `mayDefs`). This is + * the same sound over-approximation M3 documents — it may over-taint + * (multi-declarator conflation) but never drops a real flow. Sanitizer + * `resultDefs` narrow the EXCLUSION set (a def produced by a matched sanitizer + * carries that sanitizer's neutralised `SinkKind`s), so a sanitised value does + * not trigger a downstream sink of the neutralised kind — the kind-set + * exclusion model, simplified to the result-def channel (occurrence + * interposition, field paths, and callbacks are deferred — plan KTD). + * + * The summary edges themselves (return / call-arg / sink) are recorded from + * ACTUAL binding occurrences (a tainted binding present in a return's uses, a + * call's arg list, or a matched sink position), never the floor — the floor + * governs only onward def-tainting, keeping the recorded edges precise. + * + * ## Known limitation — destructured / rest params (documented FN) + * + * Param indices are assigned by ORDINAL over the flattened param-binding list, + * which equals the FORMAL parameter position only when every param is a simple + * identifier. A destructured or rest param contributes several bindings (or + * shifts the count), so a simple param positioned AFTER one + * (`function f([a, b], x) { sink(x) }`) gets a summary port index that does not + * match the formal argument position the interprocedural solver joins against + * — a cross-function false negative for that function. The precise fix needs a + * formal-param index threaded from the worker harvest (`BindingEntry`), a + * cache-namespace-affecting change deferred with the other documented FN + * classes (closures, fields — see the taint skill). Functions with all-simple + * params (the common case) are unaffected. + */ + +import type { FunctionCfg, SiteRecord } from '../cfg/types.js'; +import { pointKey, type FunctionDefUse, type ProgramPoint } from '../cfg/reaching-defs.js'; +import type { FunctionSiteMatches } from './match.js'; +import { sinkKindRank, sortSinkKinds, type SinkKind } from './source-sink-config.js'; +import type { + CallResult, + ParamToCallArg, + ParamToReturn, + ParamToSink, + SourceToCallArg, + SourceToReturn, +} from './summary-model.js'; + +/** The own-facts portion of a summary (fnId/version are added by the caller). */ +export interface HarvestedSummaryFacts { + readonly paramCount: number; + readonly paramToReturn: readonly ParamToReturn[]; + readonly paramToCallArg: readonly ParamToCallArg[]; + readonly paramToSink: readonly ParamToSink[]; + readonly sourceToReturn: readonly SourceToReturn[]; + readonly sourceToCallArg: readonly SourceToCallArg[]; + readonly callResults: readonly CallResult[]; +} + +export interface HarvestResult { + /** `computed` — facts derived; `coverage-gap` — the RD solver was not + * `computed`, so no summary is produced (consistent with M3 R4). */ + readonly status: 'computed' | 'coverage-gap'; + readonly gapReason?: FunctionDefUse['status']; + readonly facts: HarvestedSummaryFacts; +} + +const EMPTY_FACTS: HarvestedSummaryFacts = { + paramCount: 0, + paramToReturn: [], + paramToCallArg: [], + paramToSink: [], + sourceToReturn: [], + sourceToCallArg: [], + callResults: [], +}; + +/** Last segment of a dotted callee path (`child_process.exec` ⇒ `exec`). */ +const calleeTail = (callee: string | undefined): string | undefined => + callee === undefined ? undefined : (callee.split('.').pop() ?? callee); + +/** A tainted binding flowing forward, tagged with the seed it came from. */ +interface SeedTaint { + readonly bindingIdx: number; + readonly point: ProgramPoint; + /** Param index (≥0), or -1 for a source seed, or -2 for a call-result seed. */ + readonly seedId: number; + /** Sink kinds neutralised on the path to here (monotone over the floor). */ + readonly exclusions: ReadonlySet; + /** For a call-result seed (#2084 review P1-1): the user function whose RESULT + * this taint flows from. When set, reaches record {@link CallResult} edges. */ + readonly originCallee?: string; +} + +/** + * Harvest the summary facts for one function. PRECONDITION: `cfg` is + * `isEmitSafeCfg`-filtered and `defUse` was computed from it; sites are assumed + * `hasTaintSafeSites`-valid (the caller gates exactly as the M3 emit path does). + */ +export function harvestFunctionSummary( + cfg: FunctionCfg, + defUse: FunctionDefUse, + matches: FunctionSiteMatches, +): HarvestResult { + if (defUse.status !== 'computed') { + return { status: 'coverage-gap', gapReason: defUse.status, facts: EMPTY_FACTS }; + } + const bindings = defUse.bindings; + + // ── param bindings → param index (declaration order) ────────────────────── + // `kind:'param'` bindings, ordered by declaration site (declLine/declColumn). + const paramBindings = bindings + .map((b, idx) => ({ b, idx })) + .filter((e) => e.b.kind === 'param') + .sort((a, b) => a.b.declLine - b.b.declLine || a.b.declColumn - b.b.declColumn); + const paramIndexOf = new Map(); + paramBindings.forEach((e, paramIdx) => paramIndexOf.set(e.idx, paramIdx)); + const paramCount = paramBindings.length; + + // ── return points: source block of every `return` CFG edge ──────────────── + // The M2 edge-kind invariant: a `return` edge's SOURCE block terminates in + // the return jump, so its LAST statement is the `return ` — its `uses` + // are the returned bindings. (`return;` with no value has empty uses.) + const returnUseStmtKeys = new Set(); + for (const e of cfg.edges) { + if (e.kind !== 'return') continue; + const block = cfg.blocks[e.from]; + const stmts = block?.statements; + if (!stmts || stmts.length === 0) continue; + returnUseStmtKeys.add(`${e.from}:${stmts.length - 1}`); + } + + // ── per-statement match context (sink/source/sanitizer by site) ─────────── + const sinkPosBySite = new Map>>(); // stmtKey → site → argPositions + const sinkKindByEntry = new Map>(); // stmtKey → site → kinds at any pos + const sanitizerResultDefKinds = new Map>(); // stmtKey → resultDef binding → kinds + // Matched sink/sanitizer sites (`stmtKey:siteIndex`) — EXCLUDED from the + // call-result seed (#2084 review P1-1): their result semantics are already + // modelled (a sanitizer's result rides U2 exclusions; a sink returns void). + const modeledSites = new Set(); + for (const sm of matches.statements) { + const stmtKey = `${sm.blockIndex}:${sm.statementIndex}`; + for (const s of sm.sinks) modeledSites.add(`${stmtKey}:${s.siteIndex}`); + for (const s of sm.sanitizers) modeledSites.add(`${stmtKey}:${s.siteIndex}`); + if (sm.sinks.length > 0) { + const bySite = new Map>(); + const kindBySite = new Map(); + for (const sink of sm.sinks) { + const pos = bySite.get(sink.siteIndex) ?? new Set(); + for (const p of sink.argPositions) pos.add(p); + bySite.set(sink.siteIndex, pos); + const ks = kindBySite.get(sink.siteIndex) ?? []; + ks.push(sink.entry.kind); + kindBySite.set(sink.siteIndex, ks); + } + sinkPosBySite.set(stmtKey, bySite); + sinkKindByEntry.set(stmtKey, kindBySite); + } + if (sm.sanitizers.length > 0) { + const byDef = new Map(); + for (const san of sm.sanitizers) { + for (const d of san.resultDefs) { + const ks = byDef.get(d) ?? []; + ks.push(...san.entry.neutralizes); + byDef.set(d, ks); + } + } + sanitizerResultDefKinds.set(stmtKey, byDef); + } + } + + const stmtAt = (p: ProgramPoint) => cfg.blocks[p.blockIndex]?.statements?.[p.stmtIndex]; + + // ── def→use index ───────────────────────────────────────────────────────── + const factsByDef = new Map(); + for (const f of defUse.facts) { + const key = `${f.bindingIdx}:${pointKey(f.def)}`; + const list = factsByDef.get(key); + const entry = { bindingIdx: f.bindingIdx, use: f.use }; + if (list) list.push(entry); + else factsByDef.set(key, [entry]); + } + + // ── accumulators (deduped by string identity) ───────────────────────────── + const paramReturn = new Map>(); // param → neutralized intersection + const paramReturnSeen = new Set(); + const paramCallArg = new Map(); + const sourceCallArg = new Map(); + // Intersection-over-paths of the neutralized kinds reaching each call-arg + // edge (#2084 review P1-2, deepening correction a). MUST intersect, not + // first-write-wins: a second, un-sanitized occurrence path to the same edge + // (`relay(x){ exec(x); exec(escape(x)); }`) shrinks the set to ∅ — mirror + // `recordReturn`. `*Seen` tracks first-write so the initial set is a copy. + const paramCallArgKinds = new Map>(); + const sourceCallArgKinds = new Map>(); + const intersectKinds = ( + store: Map>, + key: string, + incoming: ReadonlySet, + ): void => { + const cur = store.get(key); + if (cur === undefined) store.set(key, new Set(incoming)); + else for (const k of [...cur]) if (!incoming.has(k)) cur.delete(k); + }; + const paramSink = new Set(); + const paramSinkOut: ParamToSink[] = []; + const sourceReturn = new Set(); + // Caller-side call-result flows (#2084 review P1-1), deduped by a structural key. + const callResults = new Map(); + const recordCallResult = (cr: CallResult): void => { + const d = cr.dest; + const destKey = + d.to === 'sink' + ? `sink:${d.sinkKind}` + : d.to === 'return' + ? 'return' + : `arg:${d.toCallee ?? ''}:${d.argIndex}`; + const key = `${cr.calleeName}|${destKey}`; + if (!callResults.has(key)) callResults.set(key, cr); + }; + + /** Record param→return, intersecting neutralized kinds across paths. */ + const recordReturn = (param: number, exclusions: ReadonlySet): void => { + if (!paramReturnSeen.has(param)) { + paramReturnSeen.add(param); + paramReturn.set(param, new Set(exclusions)); + } else { + const cur = paramReturn.get(param) as Set; + for (const k of [...cur]) if (!exclusions.has(k)) cur.delete(k); + } + }; + + // ── seeds: each param at its entry def point + each source statement ─────── + // seedId 0..paramCount-1 = params; -1 = source. + const queue: SeedTaint[] = []; + const visited = new Set(); + const enqueue = (t: SeedTaint): void => { + // originCallee discriminates call-result seeds (all share seedId -2) so two + // distinct callees' results on the same binding are not collapsed. + const key = `${t.seedId}:${t.originCallee ?? ''}:${t.bindingIdx}:${pointKey(t.point)}:${[...t.exclusions].sort().join(',')}`; + if (visited.has(key)) return; + visited.add(key); + queue.push(t); + }; + + // Param seeds: find each param's def point(s) in the def→use facts (params are + // defined at ENTRY; any fact whose def-binding is the param and whose def + // sits in the entry block is a param-origin edge). + for (const { idx } of paramBindings) { + const paramIdx = paramIndexOf.get(idx) as number; + // Seed at every def point of this param binding in the entry block. + for (const f of defUse.facts) { + if (f.bindingIdx === idx && f.def.blockIndex === cfg.entryIndex) { + enqueue({ bindingIdx: idx, point: f.def, seedId: paramIdx, exclusions: new Set() }); + } + } + } + + // Source seeds: a statement with a matched source taints its own defs; a bare + // `return ` is a direct source→return. The source's value rides the + // statement's defs (resultDefs of the assignment) under the floor. + for (const sm of matches.statements) { + if (sm.sources.length === 0) continue; + const stmtKey = `${sm.blockIndex}:${sm.statementIndex}`; + const facts = cfg.blocks[sm.blockIndex]?.statements?.[sm.statementIndex]; + if (!facts) continue; + const point: ProgramPoint = { + blockIndex: sm.blockIndex, + stmtIndex: sm.statementIndex, + line: facts.line, + }; + if (returnUseStmtKeys.has(stmtKey)) { + for (const src of sm.sources) sourceReturn.add(src.entry.kind); + } + for (const d of [...facts.defs, ...(facts.mayDefs ?? [])]) { + enqueue({ bindingIdx: d, point, seedId: -1, exclusions: new Set() }); + } + // DIRECT source-in-call-arg (`runIt(req.body)`): no intermediate binding is + // defined, so the floor seed above records nothing. Climb the source + // member-read's `parent` chain — each enclosing call/new site is a + // `sourceToCallArg` (the cross-function fixpoint seed). A sink ancestor is + // M3's intra-procedural concern and harmless to also record here. + for (const src of sm.sources) { + let cur: SiteRecord | undefined = facts.sites?.[src.siteIndex]; + const guard = new Set([src.siteIndex]); + while (cur?.parent) { + const [siteIdx, argPos] = cur.parent; + if (guard.has(siteIdx)) break; + guard.add(siteIdx); + const ancestor = facts.sites?.[siteIdx]; + if (!ancestor) break; + if (ancestor.kind === 'call' || ancestor.kind === 'new') { + const tail = calleeTail(ancestor.callee); + const scKey = `${facts.line}:${argPos}:${tail ?? ''}`; + if (!sourceCallArg.has(scKey)) { + sourceCallArg.set(scKey, { + sourceKind: src.entry.kind, + callLine: facts.line, + argIndex: argPos, + ...(tail ? { calleeName: tail } : {}), + }); + } + } + cur = ancestor; + } + } + } + + // Call-result seeds (#2084 review P1-1): a call to a (potentially generative) + // USER function is a NEW taint origin — `matchFunctionSites` only sources + // member-reads, so the result of `getInput()` is invisible today. Seed every + // call/new site that is NOT a matched sink/sanitizer and carries a resolvable + // callee name; the worklist then records a CallResult edge when the result + // reaches a sink / return / another call arg. The fixpoint composes it with + // the callee's `sourceToReturn` (the floor cannot — the source is in the + // callee, so the caller passes no tainted input). + // + // Documented limitation: a result passed DIRECTLY into a modelled sink with + // no binding (`exec(getInput())`) is not recorded as `dest:sink` — the sink + // is occurrence-gated by `matchFunctionSites` and a bare call result is not a + // binding occurrence, so `exec` reads as a plain call (recorded `dest:callArg` + // to a callee with no summary → uncomposed). The binding form + // (`const t = getInput(); exec(t)`) is the supported path. + for (const block of cfg.blocks) { + block.statements?.forEach((facts, stmtIdx) => { + const stmtKey = `${block.index}:${stmtIdx}`; + const point: ProgramPoint = { blockIndex: block.index, stmtIndex: stmtIdx, line: facts.line }; + facts.sites?.forEach((site, siteIndex) => { + if (site.kind !== 'call' && site.kind !== 'new') return; + if (modeledSites.has(`${stmtKey}:${siteIndex}`)) return; // sink/sanitizer — modelled + const tail = calleeTail(site.callee); + if (tail === undefined) return; // unresolvable callee — cannot compose + // Binding case (`const t = getInput(); …`): seed the result bindings. + for (const d of site.resultDefs ?? []) { + enqueue({ bindingIdx: d, point, seedId: -2, exclusions: new Set(), originCallee: tail }); + } + // Direct case (`exec(getInput())` / `return getInput()`): no result + // binding — climb the call's parent chain (or detect a bare return). + if ((site.resultDefs?.length ?? 0) === 0) { + if (site.parent === undefined && returnUseStmtKeys.has(stmtKey)) { + recordCallResult({ calleeName: tail, dest: { to: 'return' } }); + } + let cur: SiteRecord | undefined = site; + const guard = new Set([siteIndex]); + while (cur?.parent) { + const [ancIdx, argPos] = cur.parent; + if (guard.has(ancIdx)) break; + guard.add(ancIdx); + const ancestor = facts.sites?.[ancIdx]; + if (!ancestor) break; + const ancKey = `${stmtKey}:${ancIdx}`; + const sinkPositions = sinkPosBySite.get(stmtKey)?.get(ancIdx); + if (sinkPositions?.has(argPos)) { + for (const kind of sinkKindByEntry.get(stmtKey)?.get(ancIdx) ?? []) { + recordCallResult({ calleeName: tail, dest: { to: 'sink', sinkKind: kind } }); + } + } else if ( + !modeledSites.has(ancKey) && + (ancestor.kind === 'call' || ancestor.kind === 'new') + ) { + recordCallResult({ + calleeName: tail, + dest: { + to: 'callArg', + ...(calleeTail(ancestor.callee) ? { toCallee: calleeTail(ancestor.callee) } : {}), + argIndex: argPos, + }, + }); + } + cur = ancestor; + } + } + }); + }); + } + + // ── forward reachability ────────────────────────────────────────────────── + let head = 0; + while (head < queue.length) { + const t = queue[head++]; + const b = t.bindingIdx; + for (const fact of factsByDef.get(`${b}:${pointKey(t.point)}`) ?? []) { + const useStmt = stmtAt(fact.use); + if (!useStmt) continue; + const useKey = `${fact.use.blockIndex}:${fact.use.stmtIndex}`; + + // (1) return reach + if (returnUseStmtKeys.has(useKey) && useStmt.uses.includes(b)) { + if (t.originCallee !== undefined) { + recordCallResult({ calleeName: t.originCallee, dest: { to: 'return' } }); + } else if (t.seedId >= 0) recordReturn(t.seedId, t.exclusions); + else sourceReturn.add('remote-input'); + } + + // (2) call-arg + sink reach: occurrences of b in this statement's sites. + const sinkBySite = sinkPosBySite.get(useKey); + const kindBySite = sinkKindByEntry.get(useKey); + useStmt.sites?.forEach((site, siteIndex) => { + const argHits = occurrencesInArgs(site, b); + for (const argPos of argHits) { + const callLine = useStmt.line; + const tail = calleeTail(site.callee); + if (t.originCallee !== undefined) { + // Call-result seed (#2084 review P1-1): the result of a call to + // `originCallee` flows into THIS call's arg `argPos`. + recordCallResult({ + calleeName: t.originCallee, + dest: { to: 'callArg', ...(tail ? { toCallee: tail } : {}), argIndex: argPos }, + }); + } else if (t.seedId >= 0) { + const caKey = `${t.seedId}:${callLine}:${argPos}:${tail ?? ''}`; + if (!paramCallArg.has(caKey)) { + paramCallArg.set(caKey, { + param: t.seedId, + callLine, + argIndex: argPos, + ...(tail ? { calleeName: tail } : {}), + }); + } + // Carry the sanitizer exclusions on the path INTO this call arg, + // intersected over occurrence paths (P1-2). + intersectKinds(paramCallArgKinds, caKey, t.exclusions); + } else { + // Source-seeded: a generated source flowing into a call argument is + // a fixpoint SEED (it taints the callee's param). One source kind + // today ('remote-input'); when more exist the seed must carry it. + const scKey = `${callLine}:${argPos}:${tail ?? ''}`; + if (!sourceCallArg.has(scKey)) { + sourceCallArg.set(scKey, { + sourceKind: 'remote-input', + callLine, + argIndex: argPos, + ...(tail ? { calleeName: tail } : {}), + }); + } + intersectKinds(sourceCallArgKinds, scKey, t.exclusions); + } + // matched sink at this position? + const sinkPositions = sinkBySite?.get(siteIndex); + if (sinkPositions?.has(argPos)) { + for (const kind of kindBySite?.get(siteIndex) ?? []) { + if (t.exclusions.has(kind)) continue; + if (t.originCallee !== undefined) { + // A generated source returned by `originCallee` reaches a sink. + recordCallResult({ + calleeName: t.originCallee, + dest: { to: 'sink', sinkKind: kind }, + }); + } else if (t.seedId >= 0) { + const sKey = `${t.seedId}:${kind}`; + if (!paramSink.has(sKey)) { + paramSink.add(sKey); + paramSinkOut.push({ param: t.seedId, sinkKind: kind }); + } + } + } + } + } + }); + + // (3) onward floor: this statement's defs become tainted, with sanitizer + // result-def exclusions accumulated. + const sanByDef = sanitizerResultDefKinds.get(useKey); + for (const d of [...useStmt.defs, ...(useStmt.mayDefs ?? [])]) { + const added = sanByDef?.get(d); + const exclusions = + added && added.length > 0 ? new Set([...t.exclusions, ...added]) : t.exclusions; + enqueue({ + bindingIdx: d, + point: { + blockIndex: fact.use.blockIndex, + stmtIndex: fact.use.stmtIndex, + line: useStmt.line, + }, + seedId: t.seedId, + exclusions, + ...(t.originCallee !== undefined ? { originCallee: t.originCallee } : {}), + }); + } + } + } + + // ── deterministic assembly ──────────────────────────────────────────────── + const paramToReturn: ParamToReturn[] = [...paramReturn.entries()] + .map(([param, kinds]) => ({ + param, + ...(kinds.size > 0 ? { neutralized: sortSinkKinds(kinds) } : {}), + })) + .sort((a, b) => a.param - b.param); + + const paramToCallArg = [...paramCallArg.entries()] + .map(([key, edge]) => { + const kinds = paramCallArgKinds.get(key); + return kinds && kinds.size > 0 ? { ...edge, neutralized: sortSinkKinds(kinds) } : edge; + }) + .sort( + (a, b) => + a.param - b.param || + a.callLine - b.callLine || + a.argIndex - b.argIndex || + (a.calleeName ?? '').localeCompare(b.calleeName ?? ''), + ); + + const paramToSink = paramSinkOut.sort( + (a, b) => a.param - b.param || sinkKindRank(a.sinkKind) - sinkKindRank(b.sinkKind), + ); + + const sourceToReturn: SourceToReturn[] = + sourceReturn.size > 0 ? [{ sourceKind: 'remote-input' }] : []; + + const sourceToCallArg = [...sourceCallArg.entries()] + .map(([key, edge]) => { + const kinds = sourceCallArgKinds.get(key); + return kinds && kinds.size > 0 ? { ...edge, neutralized: sortSinkKinds(kinds) } : edge; + }) + .sort( + (a, b) => + a.callLine - b.callLine || + a.argIndex - b.argIndex || + (a.calleeName ?? '').localeCompare(b.calleeName ?? ''), + ); + + const callResultsOut = [...callResults.values()].sort((a, b) => { + const ord = (cr: CallResult): string => { + const d = cr.dest; + const dest = + d.to === 'sink' + ? `1sink:${d.sinkKind}` + : d.to === 'return' + ? '2return' + : `0arg:${d.toCallee ?? ''}:${d.argIndex}`; + return `${cr.calleeName}|${dest}`; + }; + return ord(a).localeCompare(ord(b)); + }); + + return { + status: 'computed', + facts: { + paramCount, + paramToReturn, + paramToCallArg, + paramToSink, + sourceToReturn, + sourceToCallArg, + callResults: callResultsOut, + }, + }; +} + +/** Argument positions where binding `b` occurs (direct or via a nested site). */ +function occurrencesInArgs(site: SiteRecord, b: number): number[] { + const hits: number[] = []; + site.args?.forEach((entries, argPos) => { + for (const e of entries) { + if (typeof e === 'number') { + if (e === b) hits.push(argPos); + } else if (e[0] === b) { + hits.push(argPos); + } + } + }); + return hits; +} diff --git a/gitnexus/src/core/ingestion/taint/summary-model.ts b/gitnexus/src/core/ingestion/taint/summary-model.ts new file mode 100644 index 000000000..386d3a9b9 --- /dev/null +++ b/gitnexus/src/core/ingestion/taint/summary-model.ts @@ -0,0 +1,270 @@ +/** + * Per-function taint SUMMARY model (#2084 M4 U2). + * + * A {@link FunctionSummary} is the compact, context-insensitive abstraction of + * one function's taint behaviour — the input to the interprocedural fixpoint + * (`interproc-solver.ts`). It is the GitNexus analogue of Pysa's `.pysa` + * models, Mariana Trench's "propagations", and CodeQL Models-as-Data summary + * rows: a function is reduced to how taint enters (params / generated sources), + * how it moves through (param→return, param→callee-arg), and where it lands + * (param→sink). The fixpoint composes these across resolved `CALLS` edges so a + * source in one function reaches a sink in another. + * + * ## Why summaries (not whole-program IFDS) + * + * The functional/summary method (Sharir-Pnueli 1981) analyses each function + * ONCE and propagates the result over the call graph — the same shape Pysa, + * Mariana Trench, and Infer use in production. GitNexus already resolves the + * call graph (`CALLS` edges carry final node ids), so the summary IS the only + * new artifact; propagation is graph reachability over a finite lattice. + * + * ## Granularity (first cut) + * + * WHOLE-PARAMETER. Ports are `param i`, `return`, and `receiver` — no field + * access paths (`arg0.field.sub`). Field sensitivity, callback-parameter ports + * (`Argument[0].Parameter[0]`), and context sensitivity are deferred (plan + * KTD; the largest JS/TS FN class — closures — stays a documented gap). + * + * ## Plain-data discipline + * + * A summary is a JSON-plain value type (no functions, class instances, Maps, or + * Symbols) so it survives `RunScopeResolutionStats` → `ScopeResolutionOutput` + * threading and any future worker/cache boundary unchanged — the same + * `Cloneable` constraint the CFG side channel obeys. + */ + +import type { SinkKind, SourceKind } from './source-sink-config.js'; + +/** + * Source-relative parameter index (0-based, in declaration order). A + * function's first parameter is `0`. Destructured / rest params map each bound + * name to the index of the formal parameter that introduced it (so + * `function f([a, b]) {}` binds both `a` and `b` to param `0`). + */ +export type ParamIndex = number; + +/** + * `param i` flows into argument `argIndex` of a call at source line `callLine`. + * The interprocedural solver joins this to the caller's outgoing `CALLS` edges + * by CALLEE NAME (`calleeName`) — NOT by `callLine` — because line-base parity + * between the CFG harvest (1-based) and the resolved reference site is fragile, + * while the callee identity is exact. It then applies the callee's summary at + * port `param argIndex`. This is the TITO ("taint-in-taint-out") propagation + * edge — a param laundered into a callee, the callee's behaviour deciding what + * happens next. + * + * `calleeName` is the site's dotted-callee tail (best-effort); absent when the + * callee chain was not statically resolvable, in which case the solver + * conservatively matches every outgoing call (sound over-approximation). + * `callLine` is the 1-based statement line as harvested (`StatementFacts.line`) + * — carried for hop display and as a TIE-BREAKER among several same-named + * callees of one caller, never as the primary join key. + */ +export interface ParamToCallArg { + readonly param: ParamIndex; + readonly callLine: number; + readonly argIndex: number; + readonly calleeName?: string; + /** + * Sink kinds neutralised on EVERY harvested path from the param to this call + * argument (intersection-over-paths, #2084 review P1-2). A sanitizer between + * the param and the callee arg (`relay(x){ const y=escape(x); sinkFn(y); }`) + * must carry across the boundary so the callee's `paramToSink` of a + * neutralised kind does not fire (the cross-function false positive). Absent + * means none neutralised. + */ + readonly neutralized?: readonly SinkKind[]; +} + +/** + * `param i` flows to the function's return value (a `return ` use). + * + * RESERVED — not yet consumed by the fixpoint (#2084 review P1-1). The M3 + * statement-level floor already treats every call as propagate-through, so it + * taints a callee's RESULT whenever the caller passes tainted input; param→ + * return recall is therefore already covered, and consuming `paramToReturn` + * would only add PRECISION (avoiding the floor's over-approximation for + * functions that don't actually return their param) — a larger refactor + * deferred. Harvested + version-stamped so the precision pass can land without + * a cache-namespace bump. + */ +export interface ParamToReturn { + readonly param: ParamIndex; + /** Sink kinds neutralised on EVERY path param→return (intersection). */ + readonly neutralized?: readonly SinkKind[]; +} + +/** `param i` reaches a modelled sink of kind `sinkKind` inside this function. */ +export interface ParamToSink { + readonly param: ParamIndex; + readonly sinkKind: SinkKind; +} + +/** + * The function itself GENERATES a source (a modelled source read, e.g. + * `req.body`) that reaches its return value — calling it yields tainted data + * with no tainted input required. The generative analogue of Pysa's + * `TaintSource[...]` return model. CONSUMED by the fixpoint via the caller's + * {@link CallResult} edges (#2084 review P1-1): a caller that uses such a + * function's result composes this into a finding/propagation. This is the + * genuinely-additive recall the floor cannot cover (the source is inside the + * callee — the caller passes no tainted input for the floor to propagate). + */ +export interface SourceToReturn { + readonly sourceKind: SourceKind; +} + +/** + * What a user-function call's RESULT flows into, in the CALLER (#2084 review + * P1-1). Recorded when a call to a (potentially generative) user function has + * its return value used by the caller. The fixpoint composes it with the + * callee's {@link SourceToReturn}: if the callee returns a generated source, + * the caller's downstream use of the result is tainted. + */ +export type CallResultDest = + | { readonly to: 'sink'; readonly sinkKind: SinkKind } + | { readonly to: 'return' } + | { readonly to: 'callArg'; readonly toCallee?: string; readonly argIndex: ParamIndex }; + +/** The result of a call to `calleeName` flows to `dest` in this function. */ +export interface CallResult { + readonly calleeName: string; + readonly dest: CallResultDest; +} + +/** + * A modelled source generated in this function flows into argument `argIndex` + * of a call at `callLine`. This SEEDS the interprocedural fixpoint: the source + * taints the callee's parameter, which the callee's summary then carries to a + * sink (one or more hops away). The cross-function analogue of an intra- + * procedural `source → sink` partial flow whose sink lives in the callee. + */ +export interface SourceToCallArg { + readonly sourceKind: SourceKind; + /** Carried for hop display + same-name tie-break; NOT the join key (see + * {@link ParamToCallArg} — the solver joins by `calleeName`). */ + readonly callLine: number; + readonly argIndex: number; + readonly calleeName?: string; + /** Sink kinds neutralised on EVERY path from the generated source to this + * call argument (intersection; #2084 review P1-2 — see {@link ParamToCallArg}). */ + readonly neutralized?: readonly SinkKind[]; +} + +/** + * The compact taint abstraction of one function. All arrays are deterministically + * sorted by the harvester and deduped, so two structurally-equal summaries + * serialise identically (the {@link summaryVersion} contract). + */ +export interface FunctionSummary { + /** The resolved `Function`/`Method` graph node id this summary describes. */ + readonly fnId: string; + /** Repo-relative source path (carried for diagnostics + the join debug). */ + readonly filePath: string; + /** 1-based function start line (mirrors `FunctionCfg.functionStartLine`). */ + readonly startLine: number; + /** Number of declared formal parameters (port arity). */ + readonly paramCount: number; + /** param→return TITO edges. */ + readonly paramToReturn: readonly ParamToReturn[]; + /** param→callee-arg TITO edges (composed across `CALLS` in the fixpoint). */ + readonly paramToCallArg: readonly ParamToCallArg[]; + /** param→sink partial flows (a source reaching this param triggers a finding). */ + readonly paramToSink: readonly ParamToSink[]; + /** Generative source→return models. */ + readonly sourceToReturn: readonly SourceToReturn[]; + /** Generative source→callee-arg seeds (fixpoint entry points). */ + readonly sourceToCallArg: readonly SourceToCallArg[]; + /** Caller-side call-result flows — compose with callee `sourceToReturn`. */ + readonly callResults: readonly CallResult[]; + /** + * Content version stamp — `hash(own-facts ∪ sorted callee versions)`. The + * incremental cache key (Infer's content-keyed summary): equal across two + * runs iff the function's own taint facts AND every callee summary it depends + * on are unchanged. NOTE (#2084 review P1-1): callee-version composition is + * RESERVED — the harvester stamps the own-facts portion only + * ({@link ownFactsDigest}); the fixpoint does not yet recompose it. + */ + readonly version: string; +} + +/** Stable FNV-1a 32-bit hash → 8-char hex. Pure, deterministic, no deps. */ +function fnv1a(input: string): string { + let h = 0x811c9dc5; + for (let i = 0; i < input.length; i++) { + h ^= input.charCodeAt(i); + // 32-bit FNV prime multiply via shifts (avoids BigInt; stays in int32 land). + h = (h + ((h << 1) + (h << 4) + (h << 7) + (h << 8) + (h << 24))) >>> 0; + } + return (h >>> 0).toString(16).padStart(8, '0'); +} + +/** + * Deterministic digest of a summary's OWN taint facts (everything except + * `version`, which is derived). Order-independent within each edge category — + * the harvester already sorts, but the digest re-canonicalises so a reordering + * never changes the stamp. Used as the leaf of {@link summaryVersion}. + */ +export function ownFactsDigest( + s: Pick< + FunctionSummary, + | 'paramCount' + | 'paramToReturn' + | 'paramToCallArg' + | 'paramToSink' + | 'sourceToReturn' + | 'sourceToCallArg' + | 'callResults' + >, +): string { + const parts: string[] = [`p${s.paramCount}`]; + parts.push( + ...s.paramToReturn + .map((r) => `r:${r.param}:${[...(r.neutralized ?? [])].sort().join(',')}`) + .sort(), + ); + parts.push( + ...s.paramToCallArg + .map( + (c) => + `c:${c.param}:${c.callLine}:${c.argIndex}:${c.calleeName ?? ''}:${[...(c.neutralized ?? [])].sort().join(',')}`, + ) + .sort(), + ); + parts.push(...s.paramToSink.map((k) => `k:${k.param}:${k.sinkKind}`).sort()); + parts.push(...s.sourceToReturn.map((g) => `g:${g.sourceKind}`).sort()); + parts.push( + ...s.sourceToCallArg + .map( + (g) => + `s:${g.sourceKind}:${g.callLine}:${g.argIndex}:${g.calleeName ?? ''}:${[...(g.neutralized ?? [])].sort().join(',')}`, + ) + .sort(), + ); + parts.push( + ...s.callResults + .map((cr) => { + const d = cr.dest; + const dest = + d.to === 'sink' + ? `sink:${d.sinkKind}` + : d.to === 'return' + ? 'return' + : `arg:${d.toCallee ?? ''}:${d.argIndex}`; + return `cr:${cr.calleeName}:${dest}`; + }) + .sort(), + ); + return fnv1a(parts.join('|')); +} + +/** + * Content version stamp for a summary: `hash(ownFactsDigest ∪ sorted callee + * versions)`. Order-independent over callee versions (sorted). Equal iff the + * function's own facts AND every callee dependency are unchanged — this is the + * incremental invalidation primitive (a changed callee changes its version, + * which changes every transitive caller's version). + */ +export function summaryVersion(ownDigest: string, calleeVersions: readonly string[]): string { + return fnv1a(`${ownDigest}#${[...calleeVersions].sort().join(',')}`); +} diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index dea1a4e98..158d324e8 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -1839,6 +1839,62 @@ export const deleteAllCommunitiesAndProcesses = async (): Promise<{ return { nodesDeleted }; }; +/** + * Drop every interprocedural `TAINT_PATH` relationship (#2084 M4 U6). Used at + * the start of an incremental `--pdg` writeback so the `taintSummaries` phase + * re-materialises them from scratch on the FULL recomputed graph. + * + * TAINT_PATH validity is a WHOLE-PROGRAM property (a flow A→C can be + * invalidated by a change to an INTERMEDIATE function whose file is neither A + * nor C). The endpoint-writability extract rule (`extractChangedSubgraph`) + * cannot see that — an A→C edge between two unchanged files would be skipped + * and a stale finding would survive. So, exactly like Community/Process, the + * sound move is delete-all-then-rebuild: cheap because TAINT_PATH is sparse + * (per-run capped), and the compute side already rebuilds every summary each + * run. Relationship-level (TAINT_PATH is an edge type, not a node label), so a + * plain DELETE on the typed CodeRelation rows — endpoints are untouched. + */ +export const deleteAllInterprocTaintPaths = async (): Promise<{ edgesDeleted: number }> => { + if (!conn) { + throw new Error('LadybugDB not initialized. Call initLbug first.'); + } + let edgesDeleted = 0; + let countResult: lbug.QueryResult | lbug.QueryResult[] | undefined; + try { + countResult = await conn.query( + `MATCH ()-[r:CodeRelation]->() WHERE r.type = 'TAINT_PATH' RETURN count(r) AS cnt`, + ); + const result = Array.isArray(countResult) ? countResult[0] : countResult; + const rows = await result.getAll(); + const count = Number(rows[0]?.cnt ?? rows[0]?.[0] ?? 0); + if (count > 0) { + await conn.query(`MATCH ()-[r:CodeRelation]->() WHERE r.type = 'TAINT_PATH' DELETE r`); + edgesDeleted = count; + } + } catch (err) { + // A missing table on a freshly-initialized DB is the benign, expected case + // (the count query above is what throws) — stay silent. Any OTHER failure + // (lock, disk, native error) would leave stale TAINT_PATH rows that the + // subsequent re-extract then DUPLICATES (CodeRelation has no PK), so it + // must ABORT the writeback (#2084 review P2-5): re-throw so the caller's + // crash-recovery dirty flag forces a clean full rebuild on the next run, + // rather than silently writing duplicate cross-function findings. + const msg = err instanceof Error ? err.message : String(err); + if (/no table|not exist|not found|does not exist|Table .* does not exist/i.test(msg)) { + if (countResult) await closeQueryResults(countResult); + return { edgesDeleted }; + } + if (countResult) await closeQueryResults(countResult); + throw new Error( + `[taint-interproc] failed to clear existing TAINT_PATH edges before incremental ` + + `re-write (${msg}) — aborting to avoid duplicate cross-function findings; ` + + `the next run will full-rebuild`, + ); + } + if (countResult) await closeQueryResults(countResult); + return { edgesDeleted }; +}; + // ============================================================================ // Full-Text Search (FTS) Functions // ============================================================================ diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index 542326a63..8de20925b 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -24,6 +24,7 @@ import { loadCachedEmbeddings, deleteNodesForFile, deleteAllCommunitiesAndProcesses, + deleteAllInterprocTaintPaths, queryImporters, loadFTSExtension, } from './lbug/lbug-adapter.js'; @@ -53,6 +54,11 @@ import { DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION, DEFAULT_PDG_MAX_TAINT_HOPS, } from './ingestion/taint/propagate.js'; +import { + DEFAULT_MAX_INTERPROC_HOPS, + DEFAULT_PDG_MAX_INTERPROC_FINDINGS, +} from './ingestion/taint/interproc-solver.js'; +import { DEFAULT_PDG_MAX_INTERPROC_EDGES } from './ingestion/taint/interproc-emit.js'; import { taintModelVersion } from './ingestion/taint/typescript-model.js'; import { computeFileHashes, diffFileHashes } from '../storage/file-hash.js'; import { @@ -153,6 +159,12 @@ export interface AnalyzeOptions { /** Per-finding taint hop cap (#2083 M3, KTD6). Forwarded to * `PipelineOptions.pdgMaxTaintHops`. No CLI flag or rc key (KTD8). */ pdgMaxTaintHops?: number; + /** Per-run cross-function findings/hops/edges caps (#2084 review P1-3). + * Forwarded to the matching `PipelineOptions.pdgMaxInterproc*`; resolved + * into `RepoMeta.pdg`. No CLI flag or rc key (KTD8). */ + pdgMaxInterprocFindings?: number; + pdgMaxInterprocHops?: number; + pdgMaxInterprocEdges?: number; /** * Default branch threaded into generated AGENTS.md / CLAUDE.md so the * regression-compare example uses the configured branch instead of a @@ -361,6 +373,9 @@ type PdgOptions = Pick< | 'pdgMaxReachingDefEdgesPerFunction' | 'pdgMaxTaintFindingsPerFunction' | 'pdgMaxTaintHops' + | 'pdgMaxInterprocFindings' + | 'pdgMaxInterprocHops' + | 'pdgMaxInterprocEdges' >; export const resolvePdgConfig = (options: PdgOptions): RepoMeta['pdg'] => @@ -378,6 +393,12 @@ export const resolvePdgConfig = (options: PdgOptions): RepoMeta['pdg'] => maxTaintFindingsPerFunction: options.pdgMaxTaintFindingsPerFunction ?? DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION, maxTaintHops: options.pdgMaxTaintHops ?? DEFAULT_PDG_MAX_TAINT_HOPS, + // #2084 review P1-3: cross-function caps. Absent on an M3-era stamp → + // pdgModeMismatch trips the first run that adds them (key-union), + // forcing the full writeback that re-materialises TAINT_PATH bounded. + maxInterprocFindings: options.pdgMaxInterprocFindings ?? DEFAULT_PDG_MAX_INTERPROC_FINDINGS, + maxInterprocHops: options.pdgMaxInterprocHops ?? DEFAULT_MAX_INTERPROC_HOPS, + maxInterprocEdges: options.pdgMaxInterprocEdges ?? DEFAULT_PDG_MAX_INTERPROC_EDGES, // Built-in model digest (KTD7/R7): persisted findings must never // outlive the model that produced them — ANY model-content change // ships as a new digest and repopulates the taint edges. @@ -783,6 +804,9 @@ export async function runFullAnalysis( pdgMaxReachingDefEdgesPerFunction: options.pdgMaxReachingDefEdgesPerFunction, pdgMaxTaintFindingsPerFunction: options.pdgMaxTaintFindingsPerFunction, pdgMaxTaintHops: options.pdgMaxTaintHops, + pdgMaxInterprocFindings: options.pdgMaxInterprocFindings, + pdgMaxInterprocHops: options.pdgMaxInterprocHops, + pdgMaxInterprocEdges: options.pdgMaxInterprocEdges, fetchWrappers: options.fetchWrappers, }, ); @@ -1002,6 +1026,15 @@ export async function runFullAnalysis( // from the fresh pipeline output below. Required for the // "Leiden runs on the FULL graph" correctness invariant. await deleteAllCommunitiesAndProcesses(); + // 2b. Drop interprocedural TAINT_PATH edges (#2084 M4 U6) when pdg is on + // — their validity is a whole-program property (an A→C flow can be + // invalidated by a change to an intermediate function on a third + // file), so endpoint-writability extraction can't refresh them. + // extractChangedSubgraph re-includes all of them from the fresh + // graph (isGraphWideRelType), mirroring Community/Process. + if (options.pdg === true) { + await deleteAllInterprocTaintPaths(); + } // 3. Extract the changed subgraph from the FULL ctx.graph and write // only that. Unchanged-file rows in the DB stay untouched. Pass diff --git a/gitnexus/src/mcp/local/local-backend.ts b/gitnexus/src/mcp/local/local-backend.ts index 83e46cfeb..99bec1852 100644 --- a/gitnexus/src/mcp/local/local-backend.ts +++ b/gitnexus/src/mcp/local/local-backend.ts @@ -2937,20 +2937,100 @@ export class LocalBackend { const { rows, totalFindings } = await runAnchoredQuery(); - if (totalFindings === 0 && pdgStamped === undefined && !target) { - // Meta was unreadable and the repo-wide enumerate found nothing — the - // count above WAS the existence probe; surface the layer hint. + // M4 (#2084 U7): cross-function findings ride TAINT_PATH edges (Function/ + // Method → Function/Method), separate from the intra-procedural TAINTED + // BasicBlock rows above. Enumerate them too so `explain` is the discovery + // surface for interprocedural flows (TAINT_PATH stays out of + // VALID_RELATION_TYPES + the web schema, like TAINTED). File-anchored: + // filter on the source function's file; symbol-anchored: either endpoint + // matches the symbol name; anchorless: all (bounded by LIMIT). Computed + // BEFORE the no-taint early returns — a repo with ONLY cross-function + // findings (no intra-procedural TAINTED rows) must not look empty. + const runInterprocQuery = async (): Promise<{ findings: any[]; total: number }> => { + const where: string[] = [`r.type = 'TAINT_PATH'`]; + const p: Record = {}; + if (anchor?.symbol) { + where.push('(a.name = $ipSym OR b.name = $ipSym)'); + p.ipSym = anchor.symbol; + } else if (anchor?.file) { + // Match EITHER endpoint's file — a cross-function flow anchored on the + // SINK's file (b) is as relevant as one anchored on the source's (a). + where.push( + '(a.filePath = $ipFile OR a.filePath ENDS WITH $ipSuffix OR ' + + 'b.filePath = $ipFile OR b.filePath ENDS WITH $ipSuffix)', + ); + p.ipFile = anchor.file; + p.ipSuffix = `/${anchor.file}`; + } + const matchClause = `MATCH (a)-[r:CodeRelation]->(b)\n WHERE ${where.join(' AND ')}`; + // Page query + a separate COUNT (#2084 review P2-4): the page is + // LIMIT-capped, so its row count cannot stand in for the true total — + // run a COUNT with the same WHERE (no LIMIT) like the intra layer does. + const [ipRows, ipCountRows] = await Promise.all([ + executeParameterized( + repo.lbugPath, + `${matchClause} + RETURN a.filePath AS file, a.name AS sourceFn, a.startLine AS sourceLine, + b.name AS sinkFn, b.startLine AS sinkLine, r.reason AS reason + ORDER BY sourceFn, sinkFn, reason + LIMIT ${limit}`, + p, + ), + executeParameterized(repo.lbugPath, `${matchClause}\n RETURN COUNT(*) AS total`, p), + ]); + const total = Number((ipCountRows[0] as any)?.total ?? (ipCountRows[0] as any)?.[0] ?? 0); + const findings = ipRows.map((r: any) => { + const decoded = decodeTaintPath(r.reason ?? r[5]); + const hops = decoded.ok + ? decoded.hops.map((h) => ({ function: h.variable, line: h.line })) + : []; + return { + interprocedural: true, + file: String(r.file ?? r[0] ?? ''), + sinkKind: decoded.ok ? (decoded.kind ?? 'unknown') : 'unknown', + source: { function: String(r.sourceFn ?? r[1] ?? ''), line: r.sourceLine ?? r[2] }, + sink: { function: String(r.sinkFn ?? r[3] ?? ''), line: r.sinkLine ?? r[4] }, + hops, + ...(decoded.ok && decoded.truncated ? { pathIncomplete: true } : {}), + }; + }); + return { findings, total }; + }; + const { findings: interprocFindings, total: interprocTotal } = await runInterprocQuery(); + + if ( + totalFindings === 0 && + interprocFindings.length === 0 && + pdgStamped === undefined && + !target + ) { + // Meta was unreadable and the repo-wide enumerate (both layers) found + // nothing — the counts above WERE the existence probe; surface the hint. return { findings: [], totalFindings: 0, note: NO_TAINT_NOTE }; } - if (totalFindings === 0 && pdgStamped === undefined && target) { + if ( + totalFindings === 0 && + interprocFindings.length === 0 && + pdgStamped === undefined && + target + ) { // Anchored miss with unreadable meta: one extra bounded probe decides - // "no findings for this anchor" vs "no taint layer at all". + // "no findings for this anchor" vs "no taint layer at all". Probe BOTH + // intra (TAINTED) and inter (TAINT_PATH) existence. const probe = await executeParameterized( repo.lbugPath, `MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock) WHERE r.type = 'TAINTED' RETURN r.reason AS reason LIMIT 1`, {}, ); - if (probe.length === 0) { + const ipProbe = + probe.length === 0 + ? await executeParameterized( + repo.lbugPath, + `MATCH (a)-[r:CodeRelation]->(b) WHERE r.type = 'TAINT_PATH' RETURN r.reason AS reason LIMIT 1`, + {}, + ) + : []; + if (probe.length === 0 && ipProbe.length === 0) { return { findings: [], totalFindings: 0, note: NO_TAINT_NOTE }; } } @@ -2997,12 +3077,32 @@ export class LocalBackend { }; }); + // Combine both layers and re-apply the page LIMIT to the union — each + // layer was queried with its own LIMIT, so the union can hold up to 2×; + // cap it so `findings.length` honours the caller's `limit`. `truncated` + // reflects EITHER layer overflowing OR the union being trimmed here, and + // `totalFindings` counts both layers' matched rows (the intra COUNT plus + // the interproc rows returned — interproc has no separate COUNT, so a + // capped interproc layer is reflected via `truncated`, never undercounted + // into a false "complete" signal). Review: code-review #2/#4 (explain + // accounting + sink-file anchoring) — both layers now accounted. + const combined = [...findings, ...interprocFindings]; + const pageFindings = combined.length > limit ? combined.slice(0, limit) : combined; + // Truncated iff EITHER layer overflowed its own LIMIT (strict `>` — exactly + // `limit` rows is not truncated), OR the combined union was trimmed to the + // page (#2084 review P2-4). `totalFindings` uses the interproc COUNT, not + // the capped slice length, so it never undercounts. + const truncated = + totalFindings > findings.length || + interprocTotal > interprocFindings.length || + combined.length > pageFindings.length; + return { ...(anchor ? { anchor } : {}), - findings, - totalFindings, - ...(totalFindings > findings.length ? { truncated: true } : {}), - note: 'Intra-procedural findings only — cross-function, closure/callback, property/field, and implicit flows are not modeled; absence of a finding is not proof of safety. SANITIZES (kill) edges are queryable via cypher.', + findings: pageFindings, + totalFindings: totalFindings + interprocTotal, + ...(truncated ? { truncated: true } : {}), + note: 'Intra-procedural (TAINTED, statement hops) AND cross-function (TAINT_PATH, function hops, `interprocedural: true`) flows are modeled. Closure/callback, property/field, and implicit flows are NOT modeled; absence of a finding is not proof of safety. Cross-function findings are context-insensitive and may over-attribute among same-named callees. SANITIZES (kill) edges are queryable via cypher.', }; } diff --git a/gitnexus/src/mcp/tools.ts b/gitnexus/src/mcp/tools.ts index dafdf7caf..4df17eb98 100644 --- a/gitnexus/src/mcp/tools.ts +++ b/gitnexus/src/mcp/tools.ts @@ -525,18 +525,19 @@ SERVICE: optional monorepo path prefix (case-sensitive path segments). When "rep }, { name: 'explain', - description: `Explain persisted taint findings: intra-procedural source→sink data flows (TAINTED edges) recorded by \`gitnexus analyze --pdg\`. + description: `Explain persisted taint findings recorded by \`gitnexus analyze --pdg\`: intra-procedural source→sink data flows (TAINTED edges, statement-level hops) AND cross-function flows (TAINT_PATH edges, function-level hops, marked \`interprocedural: true\`). -Each finding carries the sink category (command-injection, code-injection, path-traversal, sql-injection, xss), the source/sink lines, and the ordered hop path with the variable carried on each hop (decoded from the persisted path encoding). +Each finding carries the sink category (command-injection, code-injection, path-traversal, sql-injection, xss) and the ordered hop path. Intra-procedural findings carry source/sink lines and the variable on each hop; interprocedural findings carry the source and sink FUNCTION names and the chain of functions the taint crossed (decoded from the persisted path encoding). WHEN TO USE: Security review — "what taint findings exist in this repo / file / function?". Requires the repo to be indexed with \`gitnexus analyze --pdg\`; without that layer the tool returns a clear "no taint layer" note, not an error. ANCHORLESS (no "target"): enumerates all persisted findings for the repo — bounded ("limit", deterministic order), with "totalFindings" and a "truncated" flag. -ANCHORED ("target" = file path or symbol/function name): full hop detail for that anchor. A file-ish target (contains "/" or an extension) filters by file; a symbol name resolves like context() — ambiguous names return ranked candidates, unknown names return not-found. Symbol anchoring is line-range granular (findings whose source block starts inside the symbol's span). +ANCHORED ("target" = file path or symbol/function name): full hop detail for that anchor. A file-ish target (contains "/" or an extension) filters by file; a symbol name resolves like context() — ambiguous names return ranked candidates, unknown names return not-found. Symbol anchoring is line-range granular for intra-procedural findings; cross-function findings match when the symbol is the source OR sink function. -CONTRACT CAVEATS (intra-procedural M3 scope — absent flows are NOT proof of safety): -- Cross-function flows are not modeled (a flow through a helper function is invisible). -- Closure/callback flows are invisible in both directions (e.g. arr.forEach(() => sink(y))). +CONTRACT CAVEATS (absent flows are NOT proof of safety): +- Cross-function flows ARE modeled (#2084 M4): a source flowing through helper functions into a sink is found, via summary composition over the call graph (context-insensitive — return/call-site merging is accepted). +- Cross-function matching is by callee NAME (context-insensitive): when one caller invokes two distinct same-named callees, a flow into one over-attributes to both — a cross-function finding does not prove the taint reached every same-named function (sound over-report, never a missed flow). +- Closure/callback flows are invisible in both directions (e.g. arr.forEach(() => sink(y))) — the largest false-negative class. - Property/field flows are not tracked (obj.x = taint; sink(obj.y) has no chain). - Guard-style sanitizers (if (isValid(x))) and implicit/control-dependence flows are not modeled. - CommonJS aliasing is partially modeled (require('') joins resolve; dynamic requires do not). diff --git a/gitnexus/src/storage/repo-manager.ts b/gitnexus/src/storage/repo-manager.ts index a9bfd88a9..56714e5be 100644 --- a/gitnexus/src/storage/repo-manager.ts +++ b/gitnexus/src/storage/repo-manager.ts @@ -169,6 +169,16 @@ export interface RepoMeta { * bounds the persisted hop-encoded `reason`). Optional for the same * M2-era-stamp upgrade reason as the findings cap. */ maxTaintHops?: number; + /** + * Per-run cross-function caps, resolved (0 = unlimited; #2084 M4 review + * P1-3). ABSENT on an M3-era stamp — that absence trips `pdgModeMismatch` + * on the first run that adds them and forces the full writeback that + * re-materialises TAINT_PATH within bounds. Optional for that upgrade + * reason; resolved (always present) on every post-fix write. + */ + maxInterprocFindings?: number; + maxInterprocHops?: number; + maxInterprocEdges?: number; /** * Digest of the built-in taint model the persisted findings were * produced under (#2083 M3 KTD7/R7). Any model-content change ships a diff --git a/gitnexus/test/integration/cfg/fixtures/interproc-repo/gen.ts b/gitnexus/test/integration/cfg/fixtures/interproc-repo/gen.ts new file mode 100644 index 000000000..2d2a24cf2 --- /dev/null +++ b/gitnexus/test/integration/cfg/fixtures/interproc-repo/gen.ts @@ -0,0 +1,16 @@ +// Generative-source fixture (#2084 review P1-1): getInput() reads a remote-input +// source internally and RETURNS it. handleGen calls it and sinks the result — +// neither function alone is a finding (the source is inside getInput, the caller +// passes no tainted input), so only sourceToReturn composition catches it. +import { exec } from 'child_process'; + +declare const req: { body: string }; + +export function getInput(): string { + return req.body; +} + +export function handleGen(): void { + const t = getInput(); + exec(t); +} diff --git a/gitnexus/test/integration/cfg/fixtures/interproc-repo/sink.ts b/gitnexus/test/integration/cfg/fixtures/interproc-repo/sink.ts new file mode 100644 index 000000000..b012b3159 --- /dev/null +++ b/gitnexus/test/integration/cfg/fixtures/interproc-repo/sink.ts @@ -0,0 +1,13 @@ +// Interprocedural taint fixture (#2084 M4): the SINK side. `runIt` takes a +// parameter and passes it straight into child_process.exec — a param→sink +// (command-injection) summary. The caller lives in source.ts. +import { exec } from 'child_process'; + +export function runIt(cmd: string): void { + exec(cmd); +} + +// A pass-through helper for the multi-hop case: param→callee-arg of runIt. +export function forward(value: string): void { + runIt(value); +} diff --git a/gitnexus/test/integration/cfg/fixtures/interproc-repo/source.ts b/gitnexus/test/integration/cfg/fixtures/interproc-repo/source.ts new file mode 100644 index 000000000..d07ca431f --- /dev/null +++ b/gitnexus/test/integration/cfg/fixtures/interproc-repo/source.ts @@ -0,0 +1,14 @@ +// Interprocedural taint fixture (#2084 M4): the SOURCE side. `handle` reads a +// remote-input source (req.body) and passes it into runIt across the file +// boundary — a source→callee-arg summary. The fixpoint composes handle's +// source with runIt's param→sink to yield one cross-function TAINT_PATH edge. +import { runIt, forward } from './sink.js'; + +export function handle(req: { body: string }): void { + runIt(req.body); +} + +// Multi-hop: handle2 → forward → runIt → exec. +export function handle2(req: { body: string }): void { + forward(req.body); +} diff --git a/gitnexus/test/integration/cfg/interproc-taint.test.ts b/gitnexus/test/integration/cfg/interproc-taint.test.ts new file mode 100644 index 000000000..1e826c3eb --- /dev/null +++ b/gitnexus/test/integration/cfg/interproc-taint.test.ts @@ -0,0 +1,109 @@ +/** + * U9 (#2084 M4) — end-to-end interprocedural taint over the real pipeline. + * + * Runs the full pipeline (workers + scope-resolution + the taintSummaries + * phase) on a tiny CROSS-FILE repo: `source.ts#handle` reads `req.body` and + * passes it into `sink.ts#runIt`, which calls `exec`. The fixpoint must + * compose the source→callee-arg summary with the param→sink summary into one + * cross-function `TAINT_PATH` edge. The flag-off run proves the opt-in gate: + * zero TAINT_PATH edges (byte-identical graph). + * + * Build the worker dist first (`node scripts/build.js`) — the pipeline spawns + * the parse worker, and a stale dist is a spurious red. + */ + +import { describe, it, expect, afterAll } from 'vitest'; +import fs from 'fs'; +import os from 'os'; +import path from 'path'; +import { runPipelineFromRepo } from '../../../src/core/ingestion/pipeline.js'; +import type { PipelineResult } from '../../../src/types/pipeline.js'; +import { decodeTaintPath } from '../../../src/core/ingestion/taint/path-codec.js'; + +const FIXTURE = path.join(__dirname, 'fixtures', 'interproc-repo'); + +const tmpDirs: string[] = []; +function freshRepo(): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-interproc-')); + fs.cpSync(FIXTURE, dir, { recursive: true }); + tmpDirs.push(dir); + return dir; +} + +function taintPaths(result: PipelineResult) { + return [...result.graph.iterRelationships()].filter((r) => r.type === 'TAINT_PATH'); +} + +describe('U9 — end-to-end interprocedural taint (--pdg)', () => { + afterAll(() => { + for (const d of tmpDirs) fs.rmSync(d, { recursive: true, force: true }); + }); + + it('with --pdg: composes a cross-file source→sink into a TAINT_PATH edge', async () => { + const result = await runPipelineFromRepo(freshRepo(), () => {}, { pdg: true }); + const paths = taintPaths(result); + expect(paths.length).toBeGreaterThan(0); + + // At least one edge from `handle` (source fn) to `runIt` (sink fn). + const nameOf = (id: string): string => { + const n = result.graph.getNode(id); + return typeof n?.properties.name === 'string' ? n.properties.name : ''; + }; + const handleToRunIt = paths.find( + (p) => nameOf(p.sourceId) === 'handle' && nameOf(p.targetId) === 'runIt', + ); + expect(handleToRunIt, 'expected a TAINT_PATH from handle → runIt').toBeDefined(); + + // The reason decodes to a command-injection finding. + const decoded = decodeTaintPath(handleToRunIt!.reason); + expect(decoded.ok).toBe(true); + if (decoded.ok) expect(decoded.kind).toBe('command-injection'); + + // Endpoints are real graph nodes (Function/Method). + expect(result.graph.getNode(handleToRunIt!.sourceId)).toBeDefined(); + expect(result.graph.getNode(handleToRunIt!.targetId)).toBeDefined(); + }); + + it('finds the multi-hop flow handle2 → forward → runIt', async () => { + const result = await runPipelineFromRepo(freshRepo(), () => {}, { pdg: true }); + const nameOf = (id: string): string => { + const n = result.graph.getNode(id); + return typeof n?.properties.name === 'string' ? n.properties.name : ''; + }; + const found = taintPaths(result).some( + (p) => nameOf(p.sourceId) === 'handle2' && nameOf(p.targetId) === 'runIt', + ); + expect(found, 'expected a multi-hop TAINT_PATH from handle2 → runIt').toBe(true); + }); + + it('composes a generative sourceToReturn flow getInput → handleGen (#2084 review P1-1)', async () => { + const result = await runPipelineFromRepo(freshRepo(), () => {}, { pdg: true }); + const nameOf = (id: string): string => { + const n = result.graph.getNode(id); + return typeof n?.properties.name === 'string' ? n.properties.name : ''; + }; + const found = taintPaths(result).some( + (p) => nameOf(p.sourceId) === 'getInput' && nameOf(p.targetId) === 'handleGen', + ); + expect(found, 'expected a generative TAINT_PATH from getInput → handleGen').toBe(true); + }); + + it('without --pdg: emits ZERO TAINT_PATH edges (opt-in gate / golden parity)', async () => { + const result = await runPipelineFromRepo(freshRepo(), () => {}); + expect(taintPaths(result)).toHaveLength(0); + }); + + it('the taintSummaries phase ARMS the per-run edge cap (#2084 review P1-3)', async () => { + // The fixture yields ≥2 cross-function findings (handle→runIt, handle2→runIt). + // A cap of 1 must bound the emitted TAINT_PATH edges — proving the phase + // passes the limit, not just that the solver supports one. + const uncapped = await runPipelineFromRepo(freshRepo(), () => {}, { pdg: true }); + expect(taintPaths(uncapped).length).toBeGreaterThan(1); + + const capped = await runPipelineFromRepo(freshRepo(), () => {}, { + pdg: true, + pdgMaxInterprocEdges: 1, + }); + expect(taintPaths(capped)).toHaveLength(1); + }); +}); diff --git a/gitnexus/test/integration/lbug-core-adapter.test.ts b/gitnexus/test/integration/lbug-core-adapter.test.ts index 1cb5f500d..6cc245410 100644 --- a/gitnexus/test/integration/lbug-core-adapter.test.ts +++ b/gitnexus/test/integration/lbug-core-adapter.test.ts @@ -114,6 +114,31 @@ withTestLbugDB( expect(stats.edges).toBe(4); }); + it('deleteAllInterprocTaintPaths: removes TAINT_PATH edges and is benign when none exist (#2084 review P2-5)', async () => { + const { executeQuery: coreExecuteQuery, deleteAllInterprocTaintPaths } = + await import('../../src/core/lbug/lbug-adapter.js'); + + // Benign: no TAINT_PATH rows yet → returns 0, does NOT throw. + await expect(deleteAllInterprocTaintPaths()).resolves.toEqual({ edgesDeleted: 0 }); + + // Seed one TAINT_PATH edge between the two seeded Function nodes, then + // delete-all and confirm it is removed (the incremental-rebuild guard). + const fns = (await coreExecuteQuery('MATCH (n:Function) RETURN n.id AS id')) as { + id: string; + }[]; + expect(fns.length).toBe(2); + await coreExecuteQuery( + `MATCH (a:Function {id: '${fns[0].id}'}), (b:Function {id: '${fns[1].id}'}) ` + + `CREATE (a)-[:CodeRelation {type: 'TAINT_PATH', confidence: 0.6, reason: '1', step: 0}]->(b)`, + ); + const r = await deleteAllInterprocTaintPaths(); + expect(r.edgesDeleted).toBe(1); + const left = await coreExecuteQuery( + `MATCH ()-[r:CodeRelation]->() WHERE r.type = 'TAINT_PATH' RETURN count(r) AS cnt`, + ); + expect(Number((left[0] as { cnt: number }).cnt)).toBe(0); + }); + describe('unhappy path', () => { it('throws on malformed Cypher query', async () => { const { executeQuery } = await import('../../src/core/lbug/lbug-adapter.js'); diff --git a/gitnexus/test/integration/taint-explain.test.ts b/gitnexus/test/integration/taint-explain.test.ts index 6cda28eac..8d999ccfc 100644 --- a/gitnexus/test/integration/taint-explain.test.ts +++ b/gitnexus/test/integration/taint-explain.test.ts @@ -342,3 +342,140 @@ withTestLbugDB( }, }, ); + +// ─── Block 3: interprocedural TAINT_PATH findings (#2084 M4 U7) ─────── +// +// Seeds the cross-file interproc-repo fixture's emit output (Function nodes + +// TAINT_PATH edges) into a real DB and proves `explain` surfaces the +// cross-function findings (marked `interprocedural: true`) with decoded +// function-level hops + the sink kind. + +const INTERPROC_FIXTURE = path.join(__dirname, 'cfg', 'fixtures', 'interproc-repo'); + +withTestLbugDB( + 'taint-explain-interproc', + (handle) => { + describe('explain tool — cross-function TAINT_PATH findings', () => { + let backend: LocalBackend; + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) throw new Error('LocalBackend not initialized'); + backend = ext._backend; + }); + + it('anchorless enumerate includes interprocedural findings', async () => { + const res = (await backend.callTool('explain', {})) as { + findings: Array>; + }; + const ip = res.findings.filter((f) => f.interprocedural === true); + expect(ip.length).toBeGreaterThan(0); + // handle → runIt, command-injection, with function-level hops. + const hr = ip.find( + (f) => + (f.source as { function?: string })?.function === 'handle' && + (f.sink as { function?: string })?.function === 'runIt', + ); + expect(hr, 'expected an interprocedural handle → runIt finding').toBeDefined(); + expect(hr!.sinkKind).toBe('command-injection'); + expect(Array.isArray(hr!.hops)).toBe(true); + expect((hr!.hops as unknown[]).length).toBeGreaterThan(0); + }); + + it('symbol-anchored on the sink function surfaces the cross-function finding', async () => { + const res = (await backend.callTool('explain', { target: 'runIt' })) as { + findings: Array>; + }; + const ip = res.findings.filter((f) => f.interprocedural === true); + expect(ip.some((f) => (f.sink as { function?: string })?.function === 'runIt')).toBe(true); + }); + + it('totalFindings counts the full interproc layer and truncated is set on overflow (#2084 review P2-4)', async () => { + // The fixture yields multiple interproc findings; limit:1 must page to 1 + // while totalFindings reports the true (un-capped) count and truncated is set. + const full = (await backend.callTool('explain', {})) as { + findings: unknown[]; + totalFindings: number; + }; + const ipFull = full.findings.filter((f: any) => f.interprocedural === true).length; + expect(ipFull).toBeGreaterThan(1); + + const paged = (await backend.callTool('explain', { limit: 1 })) as { + findings: unknown[]; + totalFindings: number; + truncated?: boolean; + }; + expect(paged.findings.length).toBe(1); + expect(paged.truncated).toBe(true); + // totalFindings reflects the real interproc total, not the 1-row slice. + expect(paged.totalFindings).toBeGreaterThanOrEqual(ipFull); + }); + }); + }, + { + poolAdapter: true, + afterSetup: async (handle) => { + const repoDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-explain-ip-')); + try { + fs.cpSync(INTERPROC_FIXTURE, repoDir, { recursive: true }); + const pipelineResult = await runPipelineFromRepo(repoDir, () => {}, { pdg: true }); + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + + // Persist Function/Method nodes (TAINT_PATH endpoints). + const seenIds = new Set(); + pipelineResult.graph.forEachNode((n) => { + if (n.label !== 'Function' && n.label !== 'Method') return; + if (seenIds.has(n.id)) return; + seenIds.add(n.id); + }); + for (const n of pipelineResult.graph.iterNodes()) { + if (n.label !== 'Function' && n.label !== 'Method') continue; + await adapter.executePrepared( + `CREATE (x:${n.label} {id: $id, name: $name, filePath: $filePath, startLine: $startLine, endLine: $endLine})`, + { + id: n.id, + name: n.properties.name ?? '', + filePath: n.properties.filePath ?? '', + startLine: n.properties.startLine ?? 0, + endLine: n.properties.endLine ?? 0, + }, + ); + } + let tpEdges = 0; + for (const rel of pipelineResult.graph.iterRelationships()) { + if (rel.type !== 'TAINT_PATH') continue; + await adapter.executePrepared( + // The fixture's endpoints are all top-level Function nodes; Kuzu + // rejects an untyped node match in a rel CREATE (read MATCH is fine). + `MATCH (a:Function {id: $src}), (b:Function {id: $dst}) + CREATE (a)-[:CodeRelation {type: 'TAINT_PATH', confidence: $confidence, reason: $reason, step: 0}]->(b)`, + { + src: rel.sourceId, + dst: rel.targetId, + confidence: rel.confidence ?? 0.6, + reason: rel.reason ?? '', + }, + ); + tpEdges++; + } + if (tpEdges === 0) { + throw new Error('interproc fixture produced no TAINT_PATH edges — fixpoint regressed?'); + } + } finally { + fs.rmSync(repoDir, { recursive: true, force: true }); + } + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'interproc-repo', + path: '/interproc/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'ip0001', + stats: { files: 2, nodes: 4, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); diff --git a/gitnexus/test/unit/incremental-subgraph-extract.test.ts b/gitnexus/test/unit/incremental-subgraph-extract.test.ts index dc720fd9e..ed51f45b8 100644 --- a/gitnexus/test/unit/incremental-subgraph-extract.test.ts +++ b/gitnexus/test/unit/incremental-subgraph-extract.test.ts @@ -93,6 +93,24 @@ describe('extractChangedSubgraph', () => { expect(sub.nodes).toEqual([]); expect(sub.relationships).toEqual([]); }); + + it('always includes TAINT_PATH edges even between two unchanged files (#2084 M4 U6)', () => { + // A cross-function TAINT_PATH whose endpoints (a.ts, c.ts) are both + // unchanged, but an intermediate function on the changed b.ts invalidated + // the flow. Endpoint-writability alone would skip it (stale finding); + // TAINT_PATH is graph-wide so it is always re-extracted (the orchestrator + // delete-alls the old rows first). A plain CALLS edge between the same + // unchanged files stays excluded — only TAINT_PATH gets this treatment. + const g = createKnowledgeGraph(); + g.addNode(makeFileNode('a:handle', '/repo/a.ts')); + g.addNode(makeFileNode('c:sink', '/repo/c.ts')); + g.addRelationship(makeRel('tp1', 'a:handle', 'c:sink', 'TAINT_PATH')); + g.addRelationship(makeRel('call1', 'a:handle', 'c:sink', 'CALLS')); + + const sub = extractChangedSubgraph(g, new Set(['/repo/b.ts'])); + + expect(sub.relationships.map((r) => r.id)).toEqual(['tp1']); + }); }); describe('computeEffectiveWriteSet (Finding 1)', () => { diff --git a/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts b/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts index 5a0932f15..a36b6b046 100644 --- a/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts +++ b/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts @@ -100,3 +100,42 @@ describe('buildPhaseList parity (registry refactor, #2080)', () => { ); }); }); + +// --------------------------------------------------------------------------- +// M4 (#2084): the taintSummaries phase is the first real opt-in pdg-gated +// registration. Off (the default) ⇒ ABSENT ⇒ byte-identical phase list; on ⇒ +// inserted right after pruneLocalSymbols, before mro. +// --------------------------------------------------------------------------- + +const WITH_TAINT_SUMMARIES = [ + ...FULL_ORDER.slice(0, FULL_ORDER.indexOf('pruneLocalSymbols') + 1), + 'taintSummaries', + ...FULL_ORDER.slice(FULL_ORDER.indexOf('pruneLocalSymbols') + 1), +]; + +describe('buildPhaseList — taintSummaries opt-in (#2084)', () => { + it('pdg off (default) → taintSummaries absent, list byte-identical to legacy', () => { + expect(buildPhaseList(undefined).map((p) => p.name)).not.toContain('taintSummaries'); + expect(buildPhaseList({}).map((p) => p.name)).not.toContain('taintSummaries'); + expect(buildPhaseList({ pdg: false }).map((p) => p.name)).toEqual(FULL_ORDER); + }); + + it('pdg:true → taintSummaries inserted after pruneLocalSymbols, before mro', () => { + expect(buildPhaseList({ pdg: true }).map((p) => p.name)).toEqual(WITH_TAINT_SUMMARIES); + }); + + it('pdg:true is independent of skipGraphPhases', () => { + const names = buildPhaseList({ pdg: true, skipGraphPhases: true }).map((p) => p.name); + expect(names).toContain('taintSummaries'); + expect(names).not.toContain('mro'); + }); + + it('no always-on phase depends on the pdg-gated taintSummaries phase', () => { + // A filtered-out dep would throw in getPhaseOutput at runtime, so no + // always-included phase may list taintSummaries in its deps. + const offList = buildPhaseList({}); + for (const p of offList) { + expect(p.deps).not.toContain('taintSummaries'); + } + }); +}); diff --git a/gitnexus/test/unit/pdg-mode-flip.test.ts b/gitnexus/test/unit/pdg-mode-flip.test.ts index 0441464f1..572007a1c 100644 --- a/gitnexus/test/unit/pdg-mode-flip.test.ts +++ b/gitnexus/test/unit/pdg-mode-flip.test.ts @@ -104,6 +104,42 @@ describe('pdgModeMismatch — M2→M3 stamp upgrade (#2083 M3 U5, pure)', () => }); }); +describe('pdgModeMismatch — M3→M4 interproc-cap stamp upgrade (#2084 review P1-3, pure)', () => { + it('resolvePdgConfig stamps the three resolved interproc caps', async () => { + const { resolvePdgConfig } = await import('../../src/core/run-analyze.js'); + const stamp = resolvePdgConfig({ pdg: true }); + expect(stamp?.maxInterprocFindings).toBe(2000); + expect(stamp?.maxInterprocHops).toBe(32); + expect(stamp?.maxInterprocEdges).toBe(1000); + }); + + it('an M3-era stamp (no interproc keys) mismatches a post-fix request — upgrade forces full writeback', async () => { + const { pdgModeMismatch } = await import('../../src/core/run-analyze.js'); + // What an M3 run wrote: every taint cap + model digest, but none of the + // interproc caps. The key-union comparator sees 2000 !== undefined and + // trips the full writeback that re-materialises TAINT_PATH within bounds. + const m3Stamp = { + maxFunctionLines: 2000, + maxEdgesPerFunction: 5000, + maxReachingDefEdgesPerFunction: 4000, + maxTaintFindingsPerFunction: 200, + maxTaintHops: 32, + taintModelVersion: 'deadbeefcafe', + }; + expect(pdgModeMismatch(m3Stamp, { pdg: true })).toBe(true); + }); + + it('an interproc cap change alone trips the mismatch', async () => { + const { pdgModeMismatch, resolvePdgConfig } = await import('../../src/core/run-analyze.js'); + const stamp = resolvePdgConfig({ pdg: true }); + expect(pdgModeMismatch(stamp, { pdg: true, pdgMaxInterprocFindings: 10 })).toBe(true); + expect(pdgModeMismatch(stamp, { pdg: true, pdgMaxInterprocEdges: 50 })).toBe(true); + expect(pdgModeMismatch(stamp, { pdg: true, pdgMaxInterprocHops: 8 })).toBe(true); + // explicit default ≡ default (resolution before comparison) + expect(pdgModeMismatch(stamp, { pdg: true, pdgMaxInterprocFindings: 2000 })).toBe(false); + }); +}); + describe('detect_changes BasicBlock exclusion (#2082 U7)', () => { it('the symbol-overlap id-prefix filter excludes exactly the BasicBlock rows', async () => { const repo = await setupMiniRepo(); @@ -181,6 +217,9 @@ describe('runFullAnalysis — pdg-mode flip (#2099 F1)', () => { maxReachingDefEdgesPerFunction: 4000, maxTaintFindingsPerFunction: 200, maxTaintHops: 32, + maxInterprocFindings: 2000, + maxInterprocHops: 32, + maxInterprocEdges: 1000, taintModelVersion, }); expect(stamped!.incrementalInProgress).toBeUndefined(); // cleared on success @@ -233,6 +272,9 @@ describe('runFullAnalysis — pdg-mode flip (#2099 F1)', () => { maxReachingDefEdgesPerFunction: 4000, maxTaintFindingsPerFunction: 200, maxTaintHops: 32, + maxInterprocFindings: 2000, + maxInterprocHops: 32, + maxInterprocEdges: 1000, taintModelVersion, }); // The CFG layer survives a rebuild under a tighter edge cap (blocks are diff --git a/gitnexus/test/unit/run-analyze.test.ts b/gitnexus/test/unit/run-analyze.test.ts index 7c9e2a3e5..e1f442321 100644 --- a/gitnexus/test/unit/run-analyze.test.ts +++ b/gitnexus/test/unit/run-analyze.test.ts @@ -342,6 +342,9 @@ describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { maxReachingDefEdgesPerFunction: 4000, maxTaintFindingsPerFunction: 200, maxTaintHops: 32, + maxInterprocFindings: 2000, + maxInterprocHops: 32, + maxInterprocEdges: 1000, // Content digest, not a tunable cap — pinned via the exported constant // (its VALUE changes whenever the built-in model changes, by design). taintModelVersion, @@ -364,6 +367,9 @@ describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { pdgMaxReachingDefEdgesPerFunction: 0, pdgMaxTaintFindingsPerFunction: 0, pdgMaxTaintHops: 0, + pdgMaxInterprocFindings: 0, + pdgMaxInterprocHops: 0, + pdgMaxInterprocEdges: 0, }), ).toEqual({ maxFunctionLines: 0, @@ -371,6 +377,9 @@ describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { maxReachingDefEdgesPerFunction: 0, maxTaintFindingsPerFunction: 0, maxTaintHops: 0, + maxInterprocFindings: 0, + maxInterprocHops: 0, + maxInterprocEdges: 0, taintModelVersion, // not a cap — always stamped on a pdg-on run }); }); diff --git a/gitnexus/test/unit/security.test.ts b/gitnexus/test/unit/security.test.ts index 514df1078..5f1934068 100644 --- a/gitnexus/test/unit/security.test.ts +++ b/gitnexus/test/unit/security.test.ts @@ -56,6 +56,15 @@ describe('VALID_RELATION_TYPES', () => { expect(VALID_RELATION_TYPES.has('TAINTED')).toBe(false); expect(VALID_RELATION_TYPES.has('SANITIZES')).toBe(false); }); + + it('TAINT_PATH stays OUT of the impact allow-list (#2084 M4 KTD9a)', () => { + // Cross-function TAINT_PATH (Function→Function) is the interprocedural + // analogue of TAINTED — surfaced ONLY via `explain` (its interprocedural + // findings), never impact()'s BFS. Pinned so a future allow-all sweep + // can't drag it in, and the set size stays fixed at 16. + expect(VALID_RELATION_TYPES.has('TAINT_PATH')).toBe(false); + expect(VALID_RELATION_TYPES.size).toBe(16); + }); }); // ─── Valid node labels ─────────────────────────────────────────────── diff --git a/gitnexus/test/unit/taint/interproc-solver.test.ts b/gitnexus/test/unit/taint/interproc-solver.test.ts new file mode 100644 index 000000000..47e15ab2a --- /dev/null +++ b/gitnexus/test/unit/taint/interproc-solver.test.ts @@ -0,0 +1,448 @@ +/** + * U3 (#2084 M4) — interprocedural taint fixpoint. + * + * Pure: synthetic summaries + call edges in, cross-function findings out. No + * graph, no parsing. Exercises the four composition shapes (one-hop seed, + * multi-hop TITO, cross-file, recursion) plus the boundedness guards. + */ + +import { describe, it, expect } from 'vitest'; +import { + solveInterprocTaint, + type InterprocCallEdge, +} from '../../../src/core/ingestion/taint/interproc-solver.js'; +import { + ownFactsDigest, + summaryVersion, + type FunctionSummary, +} from '../../../src/core/ingestion/taint/summary-model.js'; + +let counter = 0; +function summary( + fnId: string, + facts: Partial>, +): FunctionSummary { + const full = { + paramCount: facts.paramCount ?? 1, + paramToReturn: facts.paramToReturn ?? [], + paramToCallArg: facts.paramToCallArg ?? [], + paramToSink: facts.paramToSink ?? [], + sourceToReturn: facts.sourceToReturn ?? [], + sourceToCallArg: facts.sourceToCallArg ?? [], + callResults: facts.callResults ?? [], + }; + return { + fnId, + filePath: `f${counter++}.ts`, + startLine: 1, + ...full, + version: summaryVersion(ownFactsDigest(full), []), + }; +} + +const map = (...ss: FunctionSummary[]) => new Map(ss.map((s) => [s.fnId, s])); + +describe('solveInterprocTaint — seed path respects maxHops (#2084 review P2-7)', () => { + it('caps the seed path at maxHops:1 (truncated prefix, not a 2-entry path)', () => { + const A = summary('Function:a.ts:A', { + paramCount: 0, + sourceToCallArg: [{ sourceKind: 'remote-input', callLine: 1, argIndex: 0, calleeName: 'B' }], + }); + const B = summary('Function:b.ts:B', { + paramCount: 1, + paramToSink: [{ param: 0, sinkKind: 'command-injection' }], + }); + const r = solveInterprocTaint( + map(A, B), + [{ callerId: A.fnId, calleeId: B.fnId, calleeName: 'B' }], + { + maxHops: 1, + }, + ); + expect(r.findings).toHaveLength(1); + expect(r.findings[0].hops.length).toBeLessThanOrEqual(1); + expect(r.findings[0].hopsTruncated).toBe(true); + }); +}); + +describe('solveInterprocTaint — one-hop source→callee-sink', () => { + it('finds a source passed into a callee that sinks it', () => { + // A: source flows into helper(arg0); B(helper): param0 → sink. + const A = summary('Function:a.ts:A', { + paramCount: 0, + sourceToCallArg: [{ sourceKind: 'remote-input', callLine: 5, argIndex: 0, calleeName: 'B' }], + }); + const B = summary('Function:b.ts:B', { + paramCount: 1, + paramToSink: [{ param: 0, sinkKind: 'command-injection' }], + }); + const edges: InterprocCallEdge[] = [{ callerId: A.fnId, calleeId: B.fnId, calleeName: 'B' }]; + const r = solveInterprocTaint(map(A, B), edges); + expect(r.findings).toHaveLength(1); + expect(r.findings[0]).toMatchObject({ + sourceFnId: A.fnId, + sinkFnId: B.fnId, + sinkKind: 'command-injection', + }); + }); + + it('does not fire when the callee does not sink the param', () => { + const A = summary('Function:a.ts:A', { + paramCount: 0, + sourceToCallArg: [{ sourceKind: 'remote-input', callLine: 5, argIndex: 0 }], + }); + const B = summary('Function:b.ts:B', { paramCount: 1 }); + const r = solveInterprocTaint(map(A, B), [ + { callerId: A.fnId, calleeId: B.fnId, calleeName: 'B' }, + ]); + expect(r.findings).toHaveLength(0); + }); +}); + +describe('solveInterprocTaint — multi-hop TITO', () => { + it('propagates through a chain a → b → c(sink)', () => { + const A = summary('Function:a.ts:A', { + paramCount: 0, + sourceToCallArg: [{ sourceKind: 'remote-input', callLine: 1, argIndex: 0, calleeName: 'B' }], + }); + const B = summary('Function:b.ts:B', { + paramCount: 1, + paramToCallArg: [{ param: 0, callLine: 2, argIndex: 0, calleeName: 'C' }], + }); + const C = summary('Function:c.ts:C', { + paramCount: 1, + paramToSink: [{ param: 0, sinkKind: 'sql-injection' }], + }); + const edges: InterprocCallEdge[] = [ + { callerId: A.fnId, calleeId: B.fnId, calleeName: 'B' }, + { callerId: B.fnId, calleeId: C.fnId, calleeName: 'C' }, + ]; + const r = solveInterprocTaint(map(A, B, C), edges); + expect(r.findings).toHaveLength(1); + expect(r.findings[0].sinkFnId).toBe(C.fnId); + // hop chain: A → B → C + expect(r.findings[0].hops.map((h) => h.fnId)).toEqual([A.fnId, B.fnId, C.fnId]); + }); +}); + +describe('solveInterprocTaint — cross-function sanitizer exclusions (#2084 review P1-2)', () => { + it('a neutralized call-arg edge suppresses the callee sink of that kind', () => { + // A's source flows into relay; relay forwards it to helper with + // command-injection neutralised on the path; helper sinks command-injection. + const A = summary('Function:a.ts:A', { + paramCount: 0, + sourceToCallArg: [ + { sourceKind: 'remote-input', callLine: 1, argIndex: 0, calleeName: 'relay' }, + ], + }); + const relay = summary('Function:relay.ts:relay', { + paramCount: 1, + paramToCallArg: [ + { + param: 0, + callLine: 2, + argIndex: 0, + calleeName: 'helper', + neutralized: ['command-injection'], + }, + ], + }); + const helper = summary('Function:h.ts:helper', { + paramCount: 1, + paramToSink: [{ param: 0, sinkKind: 'command-injection' }], + }); + const edges: InterprocCallEdge[] = [ + { callerId: A.fnId, calleeId: relay.fnId, calleeName: 'relay' }, + { callerId: relay.fnId, calleeId: helper.fnId, calleeName: 'helper' }, + ]; + const r = solveInterprocTaint(map(A, relay, helper), edges); + expect(r.findings).toHaveLength(0); + }); + + it('neutralization is kind-scoped — a different sink kind still fires', () => { + const A = summary('Function:a.ts:A', { + paramCount: 0, + sourceToCallArg: [ + { sourceKind: 'remote-input', callLine: 1, argIndex: 0, calleeName: 'relay' }, + ], + }); + const relay = summary('Function:relay.ts:relay', { + paramCount: 1, + paramToCallArg: [ + { param: 0, callLine: 2, argIndex: 0, calleeName: 'helper', neutralized: ['xss'] }, + ], + }); + const helper = summary('Function:h.ts:helper', { + paramCount: 1, + paramToSink: [{ param: 0, sinkKind: 'sql-injection' }], + }); + const edges: InterprocCallEdge[] = [ + { callerId: A.fnId, calleeId: relay.fnId, calleeName: 'relay' }, + { callerId: relay.fnId, calleeId: helper.fnId, calleeName: 'helper' }, + ]; + const r = solveInterprocTaint(map(A, relay, helper), edges); + expect(r.findings.some((f) => f.sinkKind === 'sql-injection')).toBe(true); + }); + + it('shrink-reprocess: a less-neutralized second path re-fires the sink (no FN)', () => { + // helper.param0 is reached from A's source two ways: via relay1 (neutralizes + // command-injection) and via relay2 (neutralizes nothing). The un-sanitized + // path must still produce the finding (intersection on revisit → ∅). + const A = summary('Function:a.ts:A', { + paramCount: 0, + sourceToCallArg: [ + { sourceKind: 'remote-input', callLine: 1, argIndex: 0, calleeName: 'relay1' }, + { sourceKind: 'remote-input', callLine: 2, argIndex: 0, calleeName: 'relay2' }, + ], + }); + const relay1 = summary('Function:r1.ts:relay1', { + paramCount: 1, + paramToCallArg: [ + { + param: 0, + callLine: 1, + argIndex: 0, + calleeName: 'helper', + neutralized: ['command-injection'], + }, + ], + }); + const relay2 = summary('Function:r2.ts:relay2', { + paramCount: 1, + paramToCallArg: [{ param: 0, callLine: 1, argIndex: 0, calleeName: 'helper' }], + }); + const helper = summary('Function:h.ts:helper', { + paramCount: 1, + paramToSink: [{ param: 0, sinkKind: 'command-injection' }], + }); + const edges: InterprocCallEdge[] = [ + { callerId: A.fnId, calleeId: relay1.fnId, calleeName: 'relay1' }, + { callerId: A.fnId, calleeId: relay2.fnId, calleeName: 'relay2' }, + { callerId: relay1.fnId, calleeId: helper.fnId, calleeName: 'helper' }, + { callerId: relay2.fnId, calleeId: helper.fnId, calleeName: 'helper' }, + ]; + const r = solveInterprocTaint(map(A, relay1, relay2, helper), edges); + expect( + r.findings.some((f) => f.sinkFnId === helper.fnId && f.sinkKind === 'command-injection'), + ).toBe(true); + }); +}); + +describe('solveInterprocTaint — generative sourceToReturn composition (#2084 review P1-1)', () => { + it('composes a generative call result that hits a sink in the caller', () => { + // getInput() returns a source; handler does exec(getInput()) — recorded as + // a callResult{getInput, dest:sink}. No tainted INPUT, so only return + // composition finds it. + const getInput = summary('Function:g.ts:getInput', { + paramCount: 0, + sourceToReturn: [{ sourceKind: 'remote-input' }], + }); + const handler = summary('Function:h.ts:handler', { + paramCount: 0, + callResults: [ + { calleeName: 'getInput', dest: { to: 'sink', sinkKind: 'command-injection' } }, + ], + }); + const edges: InterprocCallEdge[] = [ + { callerId: handler.fnId, calleeId: getInput.fnId, calleeName: 'getInput' }, + ]; + const r = solveInterprocTaint(map(getInput, handler), edges); + expect(r.findings).toHaveLength(1); + expect(r.findings[0]).toMatchObject({ + sourceFnId: getInput.fnId, + sinkFnId: handler.fnId, + sinkKind: 'command-injection', + }); + }); + + it('composes a generative result flowing into another callee arg → sink', () => { + // handler: forward(getInput()); forward(z){ exec(z) }. + const getInput = summary('Function:g.ts:getInput', { + paramCount: 0, + sourceToReturn: [{ sourceKind: 'remote-input' }], + }); + const forward = summary('Function:f.ts:forward', { + paramCount: 1, + paramToSink: [{ param: 0, sinkKind: 'command-injection' }], + }); + const handler = summary('Function:h.ts:handler', { + paramCount: 0, + callResults: [ + { calleeName: 'getInput', dest: { to: 'callArg', toCallee: 'forward', argIndex: 0 } }, + ], + }); + const edges: InterprocCallEdge[] = [ + { callerId: handler.fnId, calleeId: getInput.fnId, calleeName: 'getInput' }, + { callerId: handler.fnId, calleeId: forward.fnId, calleeName: 'forward' }, + ]; + const r = solveInterprocTaint(map(getInput, forward, handler), edges); + expect(r.findings.some((f) => f.sinkFnId === forward.fnId)).toBe(true); + }); + + it('transitively marks a relay that RETURNS a generative result as generative', () => { + // wrap(){ return getInput() } then handler does exec(wrap()). + const getInput = summary('Function:g.ts:getInput', { + paramCount: 0, + sourceToReturn: [{ sourceKind: 'remote-input' }], + }); + const wrap = summary('Function:w.ts:wrap', { + paramCount: 0, + callResults: [{ calleeName: 'getInput', dest: { to: 'return' } }], + }); + const handler = summary('Function:h.ts:handler', { + paramCount: 0, + callResults: [{ calleeName: 'wrap', dest: { to: 'sink', sinkKind: 'xss' } }], + }); + const edges: InterprocCallEdge[] = [ + { callerId: wrap.fnId, calleeId: getInput.fnId, calleeName: 'getInput' }, + { callerId: handler.fnId, calleeId: wrap.fnId, calleeName: 'wrap' }, + ]; + const r = solveInterprocTaint(map(getInput, wrap, handler), edges); + expect(r.findings.some((f) => f.sinkFnId === handler.fnId && f.sinkKind === 'xss')).toBe(true); + }); + + it('does NOT compose when the callee is not generative', () => { + const pure = summary('Function:p.ts:pure', { paramCount: 0 }); // no sourceToReturn + const handler = summary('Function:h.ts:handler', { + paramCount: 0, + callResults: [{ calleeName: 'pure', dest: { to: 'sink', sinkKind: 'command-injection' } }], + }); + const edges: InterprocCallEdge[] = [ + { callerId: handler.fnId, calleeId: pure.fnId, calleeName: 'pure' }, + ]; + const r = solveInterprocTaint(map(pure, handler), edges); + expect(r.findings).toHaveLength(0); + }); +}); + +describe('solveInterprocTaint — multi-source discrimination', () => { + it('two distinct sources into one sink function both fire (no collapse)', () => { + // A and A2 both pass a source into B's param 0, which sinks it. Without + // source-discriminated state, B.param0 is visited once and only the first + // source's finding survives — the M3 multi-source collapse bug class. + const B = summary('Function:b.ts:B', { + paramCount: 1, + paramToSink: [{ param: 0, sinkKind: 'command-injection' }], + }); + const A = summary('Function:a.ts:A', { + paramCount: 0, + sourceToCallArg: [{ sourceKind: 'remote-input', callLine: 1, argIndex: 0, calleeName: 'B' }], + }); + const A2 = summary('Function:a2.ts:A2', { + paramCount: 0, + sourceToCallArg: [{ sourceKind: 'remote-input', callLine: 1, argIndex: 0, calleeName: 'B' }], + }); + const edges: InterprocCallEdge[] = [ + { callerId: A.fnId, calleeId: B.fnId, calleeName: 'B' }, + { callerId: A2.fnId, calleeId: B.fnId, calleeName: 'B' }, + ]; + const r = solveInterprocTaint(map(A, A2, B), edges); + const sources = new Set(r.findings.map((f) => f.sourceFnId)); + expect(sources).toEqual(new Set([A.fnId, A2.fnId])); + }); +}); + +describe('solveInterprocTaint — recursion / cycles', () => { + it('terminates on direct recursion', () => { + // R taints its own param 0 → arg 0 of itself, and sinks param 0. + const R = summary('Function:r.ts:R', { + paramCount: 1, + paramToCallArg: [{ param: 0, callLine: 1, argIndex: 0, calleeName: 'R' }], + paramToSink: [{ param: 0, sinkKind: 'command-injection' }], + }); + const S = summary('Function:s.ts:S', { + paramCount: 0, + sourceToCallArg: [{ sourceKind: 'remote-input', callLine: 9, argIndex: 0, calleeName: 'R' }], + }); + const edges: InterprocCallEdge[] = [ + { callerId: S.fnId, calleeId: R.fnId, calleeName: 'R' }, + { callerId: R.fnId, calleeId: R.fnId, calleeName: 'R' }, + ]; + const r = solveInterprocTaint(map(R, S), edges); + // Converges; one finding S→R. + expect(r.findings).toHaveLength(1); + expect(r.findings[0]).toMatchObject({ sourceFnId: S.fnId, sinkFnId: R.fnId }); + }); + + it('terminates on mutual recursion f<->g', () => { + const F = summary('Function:f.ts:F', { + paramCount: 1, + paramToCallArg: [{ param: 0, callLine: 1, argIndex: 0 }], + }); + const G = summary('Function:g.ts:G', { + paramCount: 1, + paramToCallArg: [{ param: 0, callLine: 2, argIndex: 0 }], + paramToSink: [{ param: 0, sinkKind: 'xss' }], + }); + const S = summary('Function:s.ts:S', { + paramCount: 0, + sourceToCallArg: [{ sourceKind: 'remote-input', callLine: 3, argIndex: 0 }], + }); + const edges: InterprocCallEdge[] = [ + { callerId: S.fnId, calleeId: F.fnId, calleeName: 'F' }, + { callerId: F.fnId, calleeId: G.fnId, calleeName: 'G' }, + { callerId: G.fnId, calleeId: F.fnId, calleeName: 'F' }, + ]; + const r = solveInterprocTaint(map(F, G, S), edges); + expect(r.findings.some((f) => f.sinkFnId === G.fnId && f.sinkKind === 'xss')).toBe(true); + }); +}); + +describe('solveInterprocTaint — guards', () => { + it('counts an unmatched call site (callee name resolves to no edge)', () => { + const A = summary('Function:a.ts:A', { + paramCount: 0, + // The summary expects to call `Z`, but the only CALLS edge goes to `B`. + sourceToCallArg: [{ sourceKind: 'remote-input', callLine: 99, argIndex: 0, calleeName: 'Z' }], + }); + const B = summary('Function:b.ts:B', { + paramCount: 1, + paramToSink: [{ param: 0, sinkKind: 'xss' }], + }); + const r = solveInterprocTaint(map(A, B), [ + { callerId: A.fnId, calleeId: B.fnId, calleeName: 'B' }, + ]); + expect(r.findings).toHaveLength(0); + expect(r.unmatchedCallSites).toBeGreaterThan(0); + }); + + it('respects an arity guard (argIndex >= callee paramCount)', () => { + const A = summary('Function:a.ts:A', { + paramCount: 0, + sourceToCallArg: [{ sourceKind: 'remote-input', callLine: 1, argIndex: 3 }], + }); + const B = summary('Function:b.ts:B', { + paramCount: 1, + paramToSink: [{ param: 0, sinkKind: 'xss' }], + }); + const r = solveInterprocTaint(map(A, B), [ + { callerId: A.fnId, calleeId: B.fnId, calleeName: 'B' }, + ]); + expect(r.findings).toHaveLength(0); + }); + + it('caps findings and reports the drop', () => { + const sinks = Array.from({ length: 5 }, (_, i) => + summary(`Function:s${i}.ts:S${i}`, { + paramCount: 1, + paramToSink: [{ param: 0, sinkKind: 'xss' }], + }), + ); + const A = summary('Function:a.ts:A', { + paramCount: 0, + sourceToCallArg: sinks.map((_, i) => ({ + sourceKind: 'remote-input' as const, + callLine: i + 1, + argIndex: 0, + })), + }); + const edges = sinks.map((s) => ({ + callerId: A.fnId, + calleeId: s.fnId, + calleeName: s.fnId.split(':').pop() as string, + })); + const r = solveInterprocTaint(map(A, ...sinks), edges, { maxFindings: 2 }); + expect(r.findings).toHaveLength(2); + expect(r.droppedFindings).toBe(3); + }); +}); diff --git a/gitnexus/test/unit/taint/summary-harvest.test.ts b/gitnexus/test/unit/taint/summary-harvest.test.ts new file mode 100644 index 000000000..0828d4fe6 --- /dev/null +++ b/gitnexus/test/unit/taint/summary-harvest.test.ts @@ -0,0 +1,212 @@ +/** + * U1 (#2084 M4) — per-function taint summary harvest. + * + * Fixtures parse REAL TypeScript through the shared CFG/import harness, so the + * harvester consumes the exact `FunctionCfg` / `FunctionDefUse` / + * `FunctionSiteMatches` structures the pipeline produces. The four summary + * edge categories are asserted directly: param→return, param→callee-arg, + * param→sink, source→return. + */ + +import { describe, it, expect } from 'vitest'; +import { cfgOf, importsFor } from '../../helpers/ts-cfg-harness.js'; +import type { FunctionCfg } from '../../../src/core/ingestion/cfg/types.js'; +import { computeReachingDefs } from '../../../src/core/ingestion/cfg/reaching-defs.js'; +import { + buildTaintImportIndex, + matchFunctionSites, +} from '../../../src/core/ingestion/taint/match.js'; +import type { SourceSinkSanitizerSpec } from '../../../src/core/ingestion/taint/source-sink-config.js'; +import { harvestFunctionSummary } from '../../../src/core/ingestion/taint/summary-harvest.js'; + +const SPEC: SourceSinkSanitizerSpec = { + sources: [{ kind: 'remote-input', objects: ['req'], properties: ['body', 'query', 'params'] }], + sinks: [ + { name: 'exec', kind: 'command-injection', args: [0], global: true }, + { name: 'query', kind: 'sql-injection', args: [0], anyReceiver: true }, + ], + sanitizers: [{ name: 'escape', neutralizes: ['command-injection'], global: true }], +}; + +function harvest(code: string, spec: SourceSinkSanitizerSpec = SPEC, fnIndex = 0) { + const cfg: FunctionCfg = cfgOf(code, fnIndex); + const defUse = computeReachingDefs(cfg); + const matches = matchFunctionSites(cfg, spec, buildTaintImportIndex(importsFor(code))); + return harvestFunctionSummary(cfg, defUse, matches).facts; +} + +describe('harvestFunctionSummary — param→return', () => { + it('records a param flowing straight to return', () => { + const f = harvest(`function f(x: string) { return x; }`); + expect(f.paramCount).toBe(1); + expect(f.paramToReturn).toEqual([{ param: 0 }]); + }); + + it('records a param returned through a local assignment', () => { + const f = harvest(`function f(x: string) { const y = x; return y; }`); + expect(f.paramToReturn).toEqual([{ param: 0 }]); + }); + + it('records receiver-TITO return (x.trim())', () => { + const f = harvest(`function f(x: string) { return x.trim(); }`); + expect(f.paramToReturn.map((r) => r.param)).toContain(0); + }); + + it('does not record an unrelated param', () => { + const f = harvest(`function f(x: string, y: string) { return x; }`); + expect(f.paramToReturn.map((r) => r.param)).toEqual([0]); + }); +}); + +describe('harvestFunctionSummary — param→callee-arg', () => { + it('records a param flowing into a callee argument', () => { + const f = harvest(`function f(x: string) { helper(x); }`); + const ca = f.paramToCallArg; + expect(ca.length).toBeGreaterThanOrEqual(1); + expect(ca.some((c) => c.param === 0 && c.argIndex === 0 && c.calleeName === 'helper')).toBe( + true, + ); + }); + + it('records the correct argument index', () => { + const f = harvest(`function f(x: string) { helper(a, x); }`); + expect(f.paramToCallArg.some((c) => c.param === 0 && c.argIndex === 1)).toBe(true); + }); +}); + +describe('harvestFunctionSummary — param→sink', () => { + it('records a param reaching a modelled sink', () => { + const f = harvest(`function f(x: string) { exec(x); }`); + expect(f.paramToSink).toEqual([{ param: 0, sinkKind: 'command-injection' }]); + }); + + it('a sanitizer neutralises the matching sink kind', () => { + const f = harvest(`function f(x: string) { const y = escape(x); exec(y); }`); + // escape neutralises command-injection on the path to exec → no param→sink. + expect(f.paramToSink).toEqual([]); + }); +}); + +describe('harvestFunctionSummary — call-arg sanitizer exclusions (#2084 review P1-2)', () => { + it('carries the neutralized kind onto a param→callee-arg edge', () => { + // x → escape(x) → y → helper(y): the call-arg edge to the user fn `helper` + // records that command-injection was neutralised on the path. + const f = harvest(`function f(x: string) { const y = escape(x); helper(y); }`); + const edge = f.paramToCallArg.find((c) => c.calleeName === 'helper'); + expect(edge).toBeDefined(); + expect(edge!.neutralized).toEqual(['command-injection']); + }); + + it('records no neutralized when the param reaches the call directly', () => { + const f = harvest(`function f(x: string) { helper(x); }`); + const edge = f.paramToCallArg.find((c) => c.calleeName === 'helper'); + expect(edge).toBeDefined(); + expect(edge!.neutralized).toBeUndefined(); + }); +}); + +describe('harvestFunctionSummary — source→callee-arg (fixpoint seed)', () => { + it('records a source passed directly into a callee argument', () => { + const f = harvest(`function f() { runIt(req.body); }`); + expect(f.sourceToCallArg.some((s) => s.argIndex === 0 && s.calleeName === 'runIt')).toBe(true); + }); + + it('records a source passed via a local into a callee argument', () => { + const f = harvest(`function f() { const u = req.body; runIt(u); }`); + expect(f.sourceToCallArg.some((s) => s.calleeName === 'runIt')).toBe(true); + }); +}); + +describe('harvestFunctionSummary — call-result seeds (#2084 review P1-1)', () => { + it('records a generative call result reaching a sink via a local', () => { + const f = harvest(`function f() { const t = getInput(); exec(t); }`); + expect(f.callResults.some((cr) => cr.calleeName === 'getInput' && cr.dest.to === 'sink')).toBe( + true, + ); + }); + + it('records a call result flowing into another callee arg', () => { + const f = harvest(`function f() { const t = getInput(); forward(t); }`); + expect( + f.callResults.some( + (cr) => + cr.calleeName === 'getInput' && + cr.dest.to === 'callArg' && + cr.dest.toCallee === 'forward', + ), + ).toBe(true); + }); + + it('records a bare `return getInput()` as a call result → return', () => { + const f = harvest(`function f() { return getInput(); }`); + expect( + f.callResults.some((cr) => cr.calleeName === 'getInput' && cr.dest.to === 'return'), + ).toBe(true); + }); + + it('does not record call results for sink/sanitizer calls', () => { + const f = harvest(`function f(x: string) { exec(escape(x)); }`); + // exec is a sink, escape is a sanitizer — neither is a user-fn call result. + expect(f.callResults.some((cr) => cr.calleeName === 'exec' || cr.calleeName === 'escape')).toBe( + false, + ); + }); +}); + +describe('harvestFunctionSummary — source→return', () => { + it('records a generated source returned directly', () => { + const f = harvest(`function f() { return req.body; }`); + expect(f.sourceToReturn).toEqual([{ sourceKind: 'remote-input' }]); + }); + + it('records a generated source returned via a local', () => { + const f = harvest(`function f() { const u = req.body; return u; }`); + expect(f.sourceToReturn).toEqual([{ sourceKind: 'remote-input' }]); + }); + + it('is empty when no source is present', () => { + const f = harvest(`function f(x: string) { return x; }`); + expect(f.sourceToReturn).toEqual([]); + }); +}); + +describe('harvestFunctionSummary — documented limitations', () => { + it('all-simple params map to their formal argument position', () => { + const f = harvest(`function f(a: string, b: string) { exec(b); }`); + // `b` is formal param 1 — the index the interproc solver joins against. + expect(f.paramToSink).toEqual([{ param: 1, sinkKind: 'command-injection' }]); + }); + + it('destructured param before a simple param shifts the index (known FN, pinned)', () => { + // `function f([a, b], x)` — formal positions are [a,b]=0, x=1. The harvest + // assigns by binding ordinal (a=0, b=1, x=2), so x's port is 2, not the + // formal 1 the solver joins against → documented cross-function FN. Pinned + // so the behaviour is a known boundary, not a silent surprise; the proper + // fix (formal-param index from the worker) is deferred. + const f = harvest(`function f([a, b]: string[], x: string) { exec(x); }`); + const xSink = f.paramToSink.find((s) => s.sinkKind === 'command-injection'); + expect(xSink).toBeDefined(); + // Current (limited) behaviour: ordinal index 2, NOT the formal index 1. + expect(xSink!.param).toBe(2); + }); +}); + +describe('harvestFunctionSummary — edges & gaps', () => { + it('empty summary for a param-less, site-less function', () => { + const f = harvest(`function f() { const a = 1; return a; }`); + expect(f.paramToReturn).toEqual([]); + expect(f.paramToCallArg).toEqual([]); + expect(f.paramToSink).toEqual([]); + expect(f.sourceToReturn).toEqual([]); + }); + + it('reports a coverage gap when reaching-defs is not computed', () => { + // A hand-built CFG with no bindings → reaching-defs returns no-facts. + const cfg = cfgOf(`function f(x: string) { return x; }`); + const bare = { ...cfg, bindings: undefined } as FunctionCfg; + const defUse = computeReachingDefs(bare); + const matches = matchFunctionSites(bare, SPEC, buildTaintImportIndex([])); + const r = harvestFunctionSummary(bare, defUse, matches); + expect(r.status).toBe('coverage-gap'); + }); +}); diff --git a/gitnexus/test/unit/taint/summary-model.test.ts b/gitnexus/test/unit/taint/summary-model.test.ts new file mode 100644 index 000000000..7ceda74d9 --- /dev/null +++ b/gitnexus/test/unit/taint/summary-model.test.ts @@ -0,0 +1,122 @@ +/** + * U2 (#2084 M4) — the per-function taint summary model + version codec. + * + * `summaryVersion` is the incremental-invalidation primitive: it must be + * stable for identical facts, change when own facts change, change when any + * callee version changes, and be order-independent over callee versions. + * `ownFactsDigest` must be order-independent within each edge category. The + * model itself must be JSON-plain (structural-clone safe). + */ + +import { describe, it, expect } from 'vitest'; +import { + ownFactsDigest, + summaryVersion, + type FunctionSummary, +} from '../../../src/core/ingestion/taint/summary-model.js'; + +type Facts = Parameters[0]; + +const baseFacts: Facts = { + paramCount: 2, + paramToReturn: [{ param: 0 }], + paramToCallArg: [{ param: 1, callLine: 10, argIndex: 0, calleeName: 'helper' }], + paramToSink: [{ param: 0, sinkKind: 'sql-injection' }], + sourceToReturn: [{ sourceKind: 'remote-input' }], + sourceToCallArg: [{ sourceKind: 'remote-input', callLine: 7, argIndex: 0, calleeName: 'sink' }], + callResults: [{ calleeName: 'getInput', dest: { to: 'sink', sinkKind: 'command-injection' } }], +}; + +describe('ownFactsDigest', () => { + it('is stable for identical facts', () => { + expect(ownFactsDigest(baseFacts)).toBe(ownFactsDigest({ ...baseFacts })); + }); + + it('is order-independent within edge categories', () => { + const reordered: Facts = { + ...baseFacts, + paramToReturn: [{ param: 0 }], + paramToSink: [{ param: 0, sinkKind: 'sql-injection' }], + }; + const twoSinks: Facts = { + ...baseFacts, + paramToSink: [ + { param: 1, sinkKind: 'xss' }, + { param: 0, sinkKind: 'sql-injection' }, + ], + }; + const twoSinksSwapped: Facts = { + ...baseFacts, + paramToSink: [ + { param: 0, sinkKind: 'sql-injection' }, + { param: 1, sinkKind: 'xss' }, + ], + }; + expect(ownFactsDigest(reordered)).toBe(ownFactsDigest(baseFacts)); + expect(ownFactsDigest(twoSinks)).toBe(ownFactsDigest(twoSinksSwapped)); + }); + + it('changes when own facts change', () => { + const changed: Facts = { ...baseFacts, paramCount: 3 }; + expect(ownFactsDigest(changed)).not.toBe(ownFactsDigest(baseFacts)); + + const extraSink: Facts = { + ...baseFacts, + paramToSink: [...baseFacts.paramToSink, { param: 1, sinkKind: 'command-injection' }], + }; + expect(ownFactsDigest(extraSink)).not.toBe(ownFactsDigest(baseFacts)); + }); + + it('distinguishes neutralized kinds on a return edge', () => { + const a: Facts = { ...baseFacts, paramToReturn: [{ param: 0, neutralized: ['xss'] }] }; + const b: Facts = { ...baseFacts, paramToReturn: [{ param: 0 }] }; + expect(ownFactsDigest(a)).not.toBe(ownFactsDigest(b)); + }); +}); + +describe('summaryVersion', () => { + it('is stable for identical own digest + callee versions', () => { + const d = ownFactsDigest(baseFacts); + expect(summaryVersion(d, ['aaa', 'bbb'])).toBe(summaryVersion(d, ['aaa', 'bbb'])); + }); + + it('is order-independent over callee versions', () => { + const d = ownFactsDigest(baseFacts); + expect(summaryVersion(d, ['aaa', 'bbb'])).toBe(summaryVersion(d, ['bbb', 'aaa'])); + }); + + it('changes when the own digest changes', () => { + const d1 = ownFactsDigest(baseFacts); + const d2 = ownFactsDigest({ ...baseFacts, paramCount: 9 }); + expect(summaryVersion(d1, ['x'])).not.toBe(summaryVersion(d2, ['x'])); + }); + + it('changes when any callee version changes', () => { + const d = ownFactsDigest(baseFacts); + expect(summaryVersion(d, ['aaa', 'bbb'])).not.toBe(summaryVersion(d, ['aaa', 'ccc'])); + }); + + it('distinguishes no-callees from one-callee', () => { + const d = ownFactsDigest(baseFacts); + expect(summaryVersion(d, [])).not.toBe(summaryVersion(d, ['aaa'])); + }); +}); + +describe('FunctionSummary plain-data', () => { + it('survives structuredClone (no functions/Maps/Symbols)', () => { + const s: FunctionSummary = { + fnId: 'Function:src/a.ts:f', + filePath: 'src/a.ts', + startLine: 1, + paramCount: 1, + paramToReturn: [{ param: 0 }], + paramToCallArg: [], + paramToSink: [], + sourceToReturn: [], + sourceToCallArg: [], + callResults: [], + version: 'deadbeef', + }; + expect(structuredClone(s)).toEqual(s); + }); +}); From 6bd2d119fc13358d518ecc192f1c00235f438a09 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Sat, 13 Jun 2026 07:19:13 +0100 Subject: [PATCH 05/16] chore(deps)(deps-dev): bump esbuild from 0.28.0 to 0.28.1 in /gitnexus (#2182) Bumps [esbuild](https://github.com/evanw/esbuild) from 0.28.0 to 0.28.1. - [Release notes](https://github.com/evanw/esbuild/releases) - [Changelog](https://github.com/evanw/esbuild/blob/main/CHANGELOG.md) - [Commits](https://github.com/evanw/esbuild/compare/v0.28.0...v0.28.1) --- updated-dependencies: - dependency-name: esbuild dependency-version: 0.28.1 dependency-type: indirect ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- gitnexus/package-lock.json | 214 ++++++++++++++++++------------------- 1 file changed, 107 insertions(+), 107 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 9fad8f440..991729ee6 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -170,9 +170,9 @@ } }, "node_modules/@esbuild/aix-ppc64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.28.0.tgz", - "integrity": "sha512-lhRUCeuOyJQURhTxl4WkpFTjIsbDayJHih5kZC1giwE+MhIzAb7mEsQMqMf18rHLsrb5qI1tafG20mLxEWcWlA==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/aix-ppc64/-/aix-ppc64-0.28.1.tgz", + "integrity": "sha512-Svl7tq8k/08+p6CXPpRjQ1fKX+1odH/BQbb48fV6fj3CWHhsoIOoY87w1oHXm0qEpkIK3ZfVgp0hed3XBXzXMQ==", "cpu": [ "ppc64" ], @@ -187,9 +187,9 @@ } }, "node_modules/@esbuild/android-arm": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.28.0.tgz", - "integrity": "sha512-wqh0ByljabXLKHeWXYLqoJ5jKC4XBaw6Hk08OfMrCRd2nP2ZQ5eleDZC41XHyCNgktBGYMbqnrJKq/K/lzPMSQ==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/android-arm/-/android-arm-0.28.1.tgz", + "integrity": "sha512-0k2F129Xdio1TdJfzJ8sy1Q47vUD2NnwdhiAf7drUN1EBTfPf4hsFCtmMgu/6m8JSzsBrlmVjudMBQqOfG8usQ==", "cpu": [ "arm" ], @@ -204,9 +204,9 @@ } }, "node_modules/@esbuild/android-arm64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.28.0.tgz", - "integrity": "sha512-+WzIXQOSaGs33tLEgYPYe/yQHf0WTU0X42Jca3y8NWMbUVhp7rUnw+vAsRC/QiDrdD31IszMrZy+qwPOPjd+rw==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/android-arm64/-/android-arm64-0.28.1.tgz", + "integrity": "sha512-34EGEbCIAgosYz6goLcopX6Mo7NyGv9tfwEM2/7Ce2VcVRk568iSvniGWcUXIy7wEDR1wzolcxcriFVrWYcwBg==", "cpu": [ "arm64" ], @@ -221,9 +221,9 @@ } }, "node_modules/@esbuild/android-x64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.28.0.tgz", - "integrity": "sha512-+VJggoaKhk2VNNqVL7f6S189UzShHC/mR9EE8rDdSkdpN0KflSwWY/gWjDrNxxisg8Fp1ZCD9jLMo4m0OUfeUA==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/android-x64/-/android-x64-0.28.1.tgz", + "integrity": "sha512-dbwY7ltSMDWsRatcRpCnES4F+im88OCUgGZjy52shC7GqHRE/cYlxNbB4Z4UpJswpcc4Qxd2oE/ufM0p61IKng==", "cpu": [ "x64" ], @@ -238,9 +238,9 @@ } }, "node_modules/@esbuild/darwin-arm64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.28.0.tgz", - "integrity": "sha512-0T+A9WZm+bZ84nZBtk1ckYsOvyA3x7e2Acj1KdVfV4/2tdG4fzUp91YHx+GArWLtwqp77pBXVCPn2We7Letr0Q==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/darwin-arm64/-/darwin-arm64-0.28.1.tgz", + "integrity": "sha512-TZbWkQY7kvTAXbXUT7uVACR5cMHsDiSz9z7ZKAX/RTq/WJEk3QyRr0wZpNhBDX+/0CtdqUIJlOiodQcta6tY3Q==", "cpu": [ "arm64" ], @@ -255,9 +255,9 @@ } }, "node_modules/@esbuild/darwin-x64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.28.0.tgz", - "integrity": "sha512-fyzLm/DLDl/84OCfp2f/XQ4flmORsjU7VKt8HLjvIXChJoFFOIL6pLJPH4Yhd1n1gGFF9mPwtlN5Wf82DZs+LQ==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/darwin-x64/-/darwin-x64-0.28.1.tgz", + "integrity": "sha512-zfdzgK9ACBNZLI/CyHTOx81SyNbM6YXn7rxSgX97VjyiPl9W1i4Ka4fgKECEoFCKGpvBj5qArWIGgQjOwkgskQ==", "cpu": [ "x64" ], @@ -272,9 +272,9 @@ } }, "node_modules/@esbuild/freebsd-arm64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.28.0.tgz", - "integrity": "sha512-l9GeW5UZBT9k9brBYI+0WDffcRxgHQD8ShN2Ur4xWq/NFzUKm3k5lsH4PdaRgb2w7mI9u61nr2gI2mLI27Nh3Q==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/freebsd-arm64/-/freebsd-arm64-0.28.1.tgz", + "integrity": "sha512-wG2EA8ENdEI0qhkSZMjfqrdY+ziCYCPMmtZjjIwOmXFjmyzEHn+UUxk5of+SYsjtfs3VpnlC7QLzSI5hY/rOAw==", "cpu": [ "arm64" ], @@ -289,9 +289,9 @@ } }, "node_modules/@esbuild/freebsd-x64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.28.0.tgz", - "integrity": "sha512-BXoQai/A0wPO6Es3yFJ7APCiKGc1tdAEOgeTNy3SsB491S3aHn4S4r3e976eUnPdU+NbdtmBuLncYir2tMU9Nw==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/freebsd-x64/-/freebsd-x64-0.28.1.tgz", + "integrity": "sha512-i7dZ9vQgnvSCzi/rYCXNgtF/U+eKZNJBzu3eTQbRgHnM7tNSizLOkRFAl3qzVc/Op/u5YkHHa4pf/3DOYHthLQ==", "cpu": [ "x64" ], @@ -306,9 +306,9 @@ } }, "node_modules/@esbuild/linux-arm": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.28.0.tgz", - "integrity": "sha512-CjaaREJagqJp7iTaNQjjidaNbCKYcd4IDkzbwwxtSvjI7NZm79qiHc8HqciMddQ6CKvJT6aBd8lO9kN/ZudLlw==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-arm/-/linux-arm-0.28.1.tgz", + "integrity": "sha512-qVXBOHQS+d5Y722GwJzJUtOLlX7km3CraOaGormF1pDtPd2C/l1SHRPgjLunLGe51Sh5YYWKMFDyV4SxgMQYTQ==", "cpu": [ "arm" ], @@ -323,9 +323,9 @@ } }, "node_modules/@esbuild/linux-arm64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.28.0.tgz", - "integrity": "sha512-RVyzfb3FWsGA55n6WY0MEIEPURL1FcbhFE6BffZEMEekfCzCIMtB5yyDcFnVbTnwk+CLAgTujmV/Lgvih56W+A==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-arm64/-/linux-arm64-0.28.1.tgz", + "integrity": "sha512-yHs+0uc8+nvEAfAfxrWQKK5peSNzBc4PegcMO0EJ2hT71uA7vB8Ihg2e77R2P7SG5uYjPbHlLLmve4LLLRCf0g==", "cpu": [ "arm64" ], @@ -340,9 +340,9 @@ } }, "node_modules/@esbuild/linux-ia32": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.28.0.tgz", - "integrity": "sha512-KBnSTt1kxl9x70q+ydterVdl+Cn0H18ngRMRCEQfrbqdUuntQQ0LoMZv47uB97NljZFzY6HcfqEZ2SAyIUTQBQ==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-ia32/-/linux-ia32-0.28.1.tgz", + "integrity": "sha512-d1z4ZuP0ajrfz/FhGT4vv278rX8KnPPJx8i5+AtK7TYbx9Le9F1hyzurZpkEyjkGa9dUGhQow4C1NmeGvqxN2w==", "cpu": [ "ia32" ], @@ -357,9 +357,9 @@ } }, "node_modules/@esbuild/linux-loong64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.28.0.tgz", - "integrity": "sha512-zpSlUce1mnxzgBADvxKXX5sl8aYQHo2ezvMNI8I0lbblJtp8V4odlm3Yzlj7gPyt3T8ReksE6bK+pT3WD+aJRg==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-loong64/-/linux-loong64-0.28.1.tgz", + "integrity": "sha512-M5sRjUVZrkm1OAPR3dlOYzNmN+loZKGVi1VUQGrwuqLcbR6qeAz+famMhjASeH3YVKvZz+zT1jlh/keC3Rj/lg==", "cpu": [ "loong64" ], @@ -374,9 +374,9 @@ } }, "node_modules/@esbuild/linux-mips64el": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.28.0.tgz", - "integrity": "sha512-2jIfP6mmjkdmeTlsX/9vmdmhBmKADrWqN7zcdtHIeNSCH1SqIoNI63cYsjQR8J+wGa4Y5izRcSHSm8K3QWmk3w==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-mips64el/-/linux-mips64el-0.28.1.tgz", + "integrity": "sha512-mRObBZeHh2OxcBFPWE/FjylkRgZdYuiTR3vaTozquCGOH14iP9oN4x4Ge81CoIDYQrXmIxpFumJBu5MtZpnQJQ==", "cpu": [ "mips64el" ], @@ -391,9 +391,9 @@ } }, "node_modules/@esbuild/linux-ppc64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.28.0.tgz", - "integrity": "sha512-bc0FE9wWeC0WBm49IQMPSPILRocGTQt3j5KPCA8os6VprfuJ7KD+5PzESSrJ6GmPIPJK965ZJHTUlSA6GNYEhg==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-ppc64/-/linux-ppc64-0.28.1.tgz", + "integrity": "sha512-slScBsMAb3GFDcdrCgLwZtPYRoH2H/youv10QiZyRjmsP48fznoveWytSgCI/R0ZcUgpc0ZhIUEx6LHts8yrfQ==", "cpu": [ "ppc64" ], @@ -408,9 +408,9 @@ } }, "node_modules/@esbuild/linux-riscv64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.28.0.tgz", - "integrity": "sha512-SQPZOwoTTT/HXFXQJG/vBX8sOFagGqvZyXcgLA3NhIqcBv1BJU1d46c0rGcrij2B56Z2rNiSLaZOYW5cUk7yLQ==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-riscv64/-/linux-riscv64-0.28.1.tgz", + "integrity": "sha512-kw0owk1o0GFETUJyW0jc0G4Yzs0BHZn0JDZ8JRT088vjJYX777BAs1fDGxAC+q831qOs2DTC96mNsG2opdfyyQ==", "cpu": [ "riscv64" ], @@ -425,9 +425,9 @@ } }, "node_modules/@esbuild/linux-s390x": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.28.0.tgz", - "integrity": "sha512-SCfR0HN8CEEjnYnySJTd2cw0k9OHB/YFzt5zgJEwa+wL/T/raGWYMBqwDNAC6dqFKmJYZoQBRfHjgwLHGSrn3Q==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-s390x/-/linux-s390x-0.28.1.tgz", + "integrity": "sha512-/lAIjX8aYFRByhh6L5rYtPEDRqa9de/4V/juOXcta5frjvzXO4/sqEtyytse0g3zZFuWu5cDN0MkLz2qRDD2Ag==", "cpu": [ "s390x" ], @@ -442,9 +442,9 @@ } }, "node_modules/@esbuild/linux-x64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/linux-x64/-/linux-x64-0.28.0.tgz", - "integrity": "sha512-us0dSb9iFxIi8srnpl931Nvs65it/Jd2a2K3qs7fz2WfGPHqzfzZTfec7oxZJRNPXPnNYZtanmRc4AL/JwVzHQ==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/linux-x64/-/linux-x64-0.28.1.tgz", + "integrity": "sha512-u/anNYF2mmVOEDwLtnQ1wOr3EZ9sTNGLWrsYGYwHWzGA3Si84IOkHXlbWTD1NB+9/1lcnweYKO54uhxZydNzfA==", "cpu": [ "x64" ], @@ -459,9 +459,9 @@ } }, "node_modules/@esbuild/netbsd-arm64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/netbsd-arm64/-/netbsd-arm64-0.28.0.tgz", - "integrity": "sha512-CR/RYotgtCKwtftMwJlUU7xCVNg3lMYZ0RzTmAHSfLCXw3NtZtNpswLEj/Kkf6kEL3Gw+BpOekRX0BYCtklhUw==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/netbsd-arm64/-/netbsd-arm64-0.28.1.tgz", + "integrity": "sha512-oks0DYbLwWMmaakTsCb+zL4E+aHRVLom9IJZOAthMQEPiQmydXHkziYEsGYRx0uNV/IjEKGAV941JzH02pflqw==", "cpu": [ "arm64" ], @@ -476,9 +476,9 @@ } }, "node_modules/@esbuild/netbsd-x64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.28.0.tgz", - "integrity": "sha512-nU1yhmYutL+fQ71Kxnhg8uEOdC0pwEW9entHykTgEbna2pw2dkbFSMeqjjyHZoCmt8SBkOSvV+yNmm94aUrrqw==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/netbsd-x64/-/netbsd-x64-0.28.1.tgz", + "integrity": "sha512-aeL6lAnN89Hz43Mlh1G8ARasbuoYvSITDEx0tHh5b7jJnHcssqgjy9Yx430GDpmCa6OyrKoS0aNRjKundRizGg==", "cpu": [ "x64" ], @@ -493,9 +493,9 @@ } }, "node_modules/@esbuild/openbsd-arm64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.28.0.tgz", - "integrity": "sha512-cXb5vApOsRsxsEl4mcZ1XY3D4DzcoMxR/nnc4IyqYs0rTI8ZKmW6kyyg+11Z8yvgMfAEldKzP7AdP64HnSC/6g==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/openbsd-arm64/-/openbsd-arm64-0.28.1.tgz", + "integrity": "sha512-MEFJe5C3R8pwXdZ5Y21oo6m7ePiS0d9pWucn99O/wvyJZChoIQKrQDxKrGeW8F5+T0okTHesAmDeiHDTIq0V/Q==", "cpu": [ "arm64" ], @@ -510,9 +510,9 @@ } }, "node_modules/@esbuild/openbsd-x64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.28.0.tgz", - "integrity": "sha512-8wZM2qqtv9UP3mzy7HiGYNH/zjTA355mpeuA+859TyR+e+Tc08IHYpLJuMsfpDJwoLo1ikIJI8jC3GFjnRClzA==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/openbsd-x64/-/openbsd-x64-0.28.1.tgz", + "integrity": "sha512-i/ZLIOafE0Z8cI/XANJAixoJL/uRAoS2xOA3rb0xN+KK0K177cMAsQYkzHtBrtMXAKuAc7HGgcWiZ/sRC1Nxgw==", "cpu": [ "x64" ], @@ -527,9 +527,9 @@ } }, "node_modules/@esbuild/openharmony-arm64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/openharmony-arm64/-/openharmony-arm64-0.28.0.tgz", - "integrity": "sha512-FLGfyizszcef5C3YtoyQDACyg95+dndv79i2EekILBofh5wpCa1KuBqOWKrEHZg3zrL3t5ouE5jgr94vA+Wb2w==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/openharmony-arm64/-/openharmony-arm64-0.28.1.tgz", + "integrity": "sha512-ge+Z7EXFNt2BO1oAMsVpiQ8EwndV9i1xXerAeTIK7AtPs3bKFXQM7nlRxDSIUIMeueR1CNXxqztLzdNeReKBJg==", "cpu": [ "arm64" ], @@ -544,9 +544,9 @@ } }, "node_modules/@esbuild/sunos-x64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.28.0.tgz", - "integrity": "sha512-1ZgjUoEdHZZl/YlV76TSCz9Hqj9h9YmMGAgAPYd+q4SicWNX3G5GCyx9uhQWSLcbvPW8Ni7lj4gDa1T40akdlw==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/sunos-x64/-/sunos-x64-0.28.1.tgz", + "integrity": "sha512-BEjgtECkL3vY+SaSQ6nzVfiALUeFxpawyp8Jmf5PtYhf1Ug40N1h/hxlhts+f1FvSvarEigdxS3BlSMI2PJLcQ==", "cpu": [ "x64" ], @@ -561,9 +561,9 @@ } }, "node_modules/@esbuild/win32-arm64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.28.0.tgz", - "integrity": "sha512-Q9StnDmQ/enxnpxCCLSg0oo4+34B9TdXpuyPeTedN/6+iXBJ4J+zwfQI28u/Jl40nOYAxGoNi7mFP40RUtkmUA==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/win32-arm64/-/win32-arm64-0.28.1.tgz", + "integrity": "sha512-lCv9eK/H6ZJWbE7bh2nw54CZ9M2nupBxJcTsdk/QQnWkdSjKGuxmmH8/GWrlT1eMmZfn4dGcCjRte397WqfQXA==", "cpu": [ "arm64" ], @@ -578,9 +578,9 @@ } }, "node_modules/@esbuild/win32-ia32": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.28.0.tgz", - "integrity": "sha512-zF3ag/gfiCe6U2iczcRzSYJKH1DCI+ByzSENHlM2FcDbEeo5Zd2C86Aq0tKUYAJJ1obRP84ymxIAksZUcdztHA==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/win32-ia32/-/win32-ia32-0.28.1.tgz", + "integrity": "sha512-zvb/mB2bSCoJOpoCBgYKKpX6YM6mJBlBUVUtVj41DlZJVEB6/0CKlRYxP5wWl1C1ILiCoAU5wZZ4q1P3qeS6Eg==", "cpu": [ "ia32" ], @@ -595,9 +595,9 @@ } }, "node_modules/@esbuild/win32-x64": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.28.0.tgz", - "integrity": "sha512-pEl1bO9mfAmIC+tW5btTmrKaujg3zGtUmWNdCw/xs70FBjwAL3o9OEKNHvNmnyylD6ubxUERiEhdsL0xBQ9efw==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/@esbuild/win32-x64/-/win32-x64-0.28.1.tgz", + "integrity": "sha512-bm4Mowrv+GXMlpWX++EcXw/iLyd1o3+bJkC2DkWXYVvgZCqD/bSj9ctZeAMC3cIxgjRVR2Dufaiu4YPxr5gW1A==", "cpu": [ "x64" ], @@ -2712,9 +2712,9 @@ } }, "node_modules/esbuild": { - "version": "0.28.0", - "resolved": "https://registry.npmjs.org/esbuild/-/esbuild-0.28.0.tgz", - "integrity": "sha512-sNR9MHpXSUV/XB4zmsFKN+QgVG82Cc7+/aaxJ8Adi8hyOac+EXptIp45QBPaVyX3N70664wRbTcLTOemCAnyqw==", + "version": "0.28.1", + "resolved": "https://registry.npmjs.org/esbuild/-/esbuild-0.28.1.tgz", + "integrity": "sha512-HrJrvZv5ayxBzPfwphOoNzkzOIIlifzk0KJrGK2c8R4+LKpMtpYLQeUdjnwjWv/LZlkH2laZk+4w78pi99D4Vw==", "dev": true, "hasInstallScript": true, "license": "MIT", @@ -2725,32 +2725,32 @@ "node": ">=18" }, "optionalDependencies": { - "@esbuild/aix-ppc64": "0.28.0", - "@esbuild/android-arm": "0.28.0", - "@esbuild/android-arm64": "0.28.0", - "@esbuild/android-x64": "0.28.0", - "@esbuild/darwin-arm64": "0.28.0", - "@esbuild/darwin-x64": "0.28.0", - "@esbuild/freebsd-arm64": "0.28.0", - "@esbuild/freebsd-x64": "0.28.0", - "@esbuild/linux-arm": "0.28.0", - "@esbuild/linux-arm64": "0.28.0", - "@esbuild/linux-ia32": "0.28.0", - "@esbuild/linux-loong64": "0.28.0", - "@esbuild/linux-mips64el": "0.28.0", - "@esbuild/linux-ppc64": "0.28.0", - "@esbuild/linux-riscv64": "0.28.0", - "@esbuild/linux-s390x": "0.28.0", - "@esbuild/linux-x64": "0.28.0", - "@esbuild/netbsd-arm64": "0.28.0", - "@esbuild/netbsd-x64": "0.28.0", - "@esbuild/openbsd-arm64": "0.28.0", - "@esbuild/openbsd-x64": "0.28.0", - "@esbuild/openharmony-arm64": "0.28.0", - "@esbuild/sunos-x64": "0.28.0", - "@esbuild/win32-arm64": "0.28.0", - "@esbuild/win32-ia32": "0.28.0", - "@esbuild/win32-x64": "0.28.0" + "@esbuild/aix-ppc64": "0.28.1", + "@esbuild/android-arm": "0.28.1", + "@esbuild/android-arm64": "0.28.1", + "@esbuild/android-x64": "0.28.1", + "@esbuild/darwin-arm64": "0.28.1", + "@esbuild/darwin-x64": "0.28.1", + "@esbuild/freebsd-arm64": "0.28.1", + "@esbuild/freebsd-x64": "0.28.1", + "@esbuild/linux-arm": "0.28.1", + "@esbuild/linux-arm64": "0.28.1", + "@esbuild/linux-ia32": "0.28.1", + "@esbuild/linux-loong64": "0.28.1", + "@esbuild/linux-mips64el": "0.28.1", + "@esbuild/linux-ppc64": "0.28.1", + "@esbuild/linux-riscv64": "0.28.1", + "@esbuild/linux-s390x": "0.28.1", + "@esbuild/linux-x64": "0.28.1", + "@esbuild/netbsd-arm64": "0.28.1", + "@esbuild/netbsd-x64": "0.28.1", + "@esbuild/openbsd-arm64": "0.28.1", + "@esbuild/openbsd-x64": "0.28.1", + "@esbuild/openharmony-arm64": "0.28.1", + "@esbuild/sunos-x64": "0.28.1", + "@esbuild/win32-arm64": "0.28.1", + "@esbuild/win32-ia32": "0.28.1", + "@esbuild/win32-x64": "0.28.1" } }, "node_modules/escalade": { From 50cda61a4346d09a5f6a0ddc8f0cba5a63c06be6 Mon Sep 17 00:00:00 2001 From: azizur100389 Date: Sat, 13 Jun 2026 08:06:50 +0100 Subject: [PATCH 06/16] feat(setup): select coding agent integrations (#2168) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(setup): select coding agent integrations * style: format setup agent selection * fix(setup): validate explicit agent selection --------- Co-authored-by: Gergő Magyar --- README.md | 4 +- gitnexus/README.md | 4 +- gitnexus/src/cli/help-i18n.ts | 1 + gitnexus/src/cli/i18n/en.ts | 2 + gitnexus/src/cli/i18n/zh-CN.ts | 1 + gitnexus/src/cli/index.ts | 9 ++ gitnexus/src/cli/setup.ts | 81 ++++++++++--- gitnexus/test/unit/cli-index-help.test.ts | 8 ++ gitnexus/test/unit/setup-selection.test.ts | 131 +++++++++++++++++++++ 9 files changed, 220 insertions(+), 21 deletions(-) create mode 100644 gitnexus/test/unit/setup-selection.test.ts diff --git a/README.md b/README.md index aea1079ee..876ff81e2 100644 --- a/README.md +++ b/README.md @@ -123,7 +123,7 @@ To configure MCP for your editor, run `npx gitnexus setup` once — or set it up ### MCP Setup -`gitnexus setup` auto-detects your editors and writes the correct global MCP config. You only need to run it once. +`gitnexus setup` auto-detects your editors and writes the correct global MCP config. You only need to run it once. To configure only selected integrations, pass `--coding-agent`/`-c` with a comma-separated list or repeat the option, for example `gitnexus setup -c cursor,codex`. ### Editor Support @@ -224,7 +224,7 @@ args = ["-y", "gitnexus@latest", "mcp"] ### CLI Commands ```bash -gitnexus setup # Configure MCP for your editors (one-time) +gitnexus setup # Configure MCP for detected editors (one-time; use -c to select) gitnexus uninstall # Preview removal of GitNexus MCP/skills/hooks (add --force to apply) gitnexus analyze [path] # Index a repository (or update stale index) gitnexus analyze --repair-fts # Fast path: rebuild/verify only FTS indexes on existing index data diff --git a/gitnexus/README.md b/gitnexus/README.md index 4f5b27781..a306f0852 100644 --- a/gitnexus/README.md +++ b/gitnexus/README.md @@ -34,7 +34,7 @@ That's it. This indexes the codebase, installs agent skills, registers Claude Co To configure MCP for your editor, run `npx gitnexus setup` once — or set it up manually below. -`gitnexus setup` auto-detects your editors and writes the correct global MCP config. You only need to run it once. +`gitnexus setup` auto-detects your editors and writes the correct global MCP config. You only need to run it once. To configure only selected integrations, pass `--coding-agent`/`-c` with a comma-separated list or repeat the option, for example `gitnexus setup -c cursor,codex`. ### Editor Support @@ -158,7 +158,7 @@ Your AI agent gets these tools automatically: ## CLI Commands ```bash -gitnexus setup # Configure MCP for your editors (one-time) +gitnexus setup # Configure MCP for detected editors (one-time; use -c to select) gitnexus uninstall # Preview removal of GitNexus MCP/skills/hooks (add --force to apply) gitnexus analyze [path] # Index a repository (or update stale index) gitnexus analyze --repair-fts # Fast path: rebuild/verify only FTS indexes on existing index data diff --git a/gitnexus/src/cli/help-i18n.ts b/gitnexus/src/cli/help-i18n.ts index 993f20bc4..0c898eccb 100644 --- a/gitnexus/src/cli/help-i18n.ts +++ b/gitnexus/src/cli/help-i18n.ts @@ -46,6 +46,7 @@ const COMMAND_DESCRIPTION_KEYS = { const OPTION_DESCRIPTION_KEYS = { '|-V, --version': 'help.option.version', + 'setup|-c, --coding-agent ': 'help.option.setup.codingAgent', 'analyze|-f, --force': 'help.option.analyze.force', 'analyze|--repair-fts': 'help.option.analyze.repairFts', 'analyze|--embeddings [limit]': 'help.option.analyze.embeddings', diff --git a/gitnexus/src/cli/i18n/en.ts b/gitnexus/src/cli/i18n/en.ts index 3f4a9e7be..f6a035adb 100644 --- a/gitnexus/src/cli/i18n/en.ts +++ b/gitnexus/src/cli/i18n/en.ts @@ -159,6 +159,8 @@ export const en = { 'Cross-repo impact for a symbol in one member repo of a group', 'help.command.group.query.description': 'Search execution flows across all repos in a group', 'help.command.group.contracts.description': 'Inspect Contract Registry', + 'help.option.setup.codingAgent': + 'Configure only these coding agents (comma-separated or repeatable)', 'help.option.analyze.force': 'Force full re-index even if up to date', 'help.option.analyze.repairFts': 'Repair/rebuild search FTS indexes without full re-analysis', 'help.option.analyze.embeddings': diff --git a/gitnexus/src/cli/i18n/zh-CN.ts b/gitnexus/src/cli/i18n/zh-CN.ts index 9331dc4fc..693d700aa 100644 --- a/gitnexus/src/cli/i18n/zh-CN.ts +++ b/gitnexus/src/cli/i18n/zh-CN.ts @@ -152,6 +152,7 @@ export const zhCN = { 'help.command.group.impact.description': '分析仓库组中某个成员仓库符号的跨仓库影响', 'help.command.group.query.description': '跨仓库组所有仓库搜索执行流程', 'help.command.group.contracts.description': '查看 Contract Registry', + 'help.option.setup.codingAgent': '仅配置这些编码代理(逗号分隔或重复传入)', 'help.option.analyze.force': '即使已是最新也强制完整重建索引', 'help.option.analyze.repairFts': '修复/重建搜索 FTS 索引,不执行完整重新分析', 'help.option.analyze.embeddings': diff --git a/gitnexus/src/cli/index.ts b/gitnexus/src/cli/index.ts index cd0a2613d..eb5a5a8f6 100644 --- a/gitnexus/src/cli/index.ts +++ b/gitnexus/src/cli/index.ts @@ -14,6 +14,10 @@ const _require = createRequire(import.meta.url); const pkg = _require('../../package.json'); const program = new Command(); +function collectCodingAgents(value: string, previous: string[] | undefined): string[] { + return [...(previous ?? []), ...value.split(',')]; +} + program.name('gitnexus').description('GitNexus local CLI and MCP server').version(pkg.version); program @@ -21,6 +25,11 @@ program .description( 'One-time setup: configure MCP for Cursor, Claude Code, Antigravity, OpenCode, Codex', ) + .option( + '-c, --coding-agent ', + 'Configure only these coding agents (comma-separated or repeatable)', + collectCodingAgents, + ) .action(createLazyAction(() => import('./setup.js'), 'setupCommand')); program diff --git a/gitnexus/src/cli/setup.ts b/gitnexus/src/cli/setup.ts index 907b7f885..6ea5d7d10 100644 --- a/gitnexus/src/cli/setup.ts +++ b/gitnexus/src/cli/setup.ts @@ -21,6 +21,7 @@ import { skillTarget, hookTarget, detectIndentation, + type EditorId, } from './editor-targets.js'; const __filename = fileURLToPath(import.meta.url); @@ -85,6 +86,37 @@ interface SetupResult { errors: string[]; } +const CODING_AGENT_IDS = { + cursor: 'cursor', + claude: 'claude', + antigravity: 'antigravity', + opencode: 'opencode', + codex: 'codex', +} as const satisfies Record; +const SUPPORTED_CODING_AGENTS = Object.values(CODING_AGENT_IDS); + +function selectedCodingAgents(values: string[] | string | undefined): Set | null { + if (values == null) return new Set(SUPPORTED_CODING_AGENTS); + const rawValues = Array.isArray(values) ? values : [values]; + const requested = rawValues + .flatMap((value) => value.split(',')) + .map((value) => value.trim().toLowerCase()) + .filter(Boolean); + const invalid = requested.filter( + (value): value is string => !SUPPORTED_CODING_AGENTS.includes(value as EditorId), + ); + if (requested.length === 0 || invalid.length > 0) { + const detail = + requested.length === 0 + ? 'No coding agents were provided.' + : `Unknown: ${invalid.join(', ')}.`; + process.stderr.write(`${detail} Valid values: ${SUPPORTED_CODING_AGENTS.join(', ')}.\n`); + process.exitCode = 1; + return null; + } + return new Set(requested as EditorId[]); +} + /** * Resolve the absolute path to the `gitnexus` binary if it's installed * globally (or via npm -g / yarn global). Returns null when not found. @@ -968,7 +1000,11 @@ async function installCodexSkills(result: SetupResult): Promise { // ─── Main command ────────────────────────────────────────────────── -export const setupCommand = async () => { +export const setupCommand = async (options?: { codingAgent?: string[] | string }) => { + const explicitSelection = options?.codingAgent != null; + const selected = selectedCodingAgents(options?.codingAgent); + if (!selected) return; + console.log(''); console.log(' GitNexus Setup'); console.log(' =============='); @@ -985,20 +1021,24 @@ export const setupCommand = async () => { }; // Detect and configure each editor's MCP - await setupCursor(result); - await setupClaudeCode(result); - await setupAntigravity(result); - await setupOpenCode(result); - await setupCodex(result); + if (selected.has('cursor')) await setupCursor(result); + if (selected.has('claude')) await setupClaudeCode(result); + if (selected.has('antigravity')) await setupAntigravity(result); + if (selected.has('opencode')) await setupOpenCode(result); + if (selected.has('codex')) await setupCodex(result); // Install global skills for platforms that support them - await installClaudeCodeSkills(result); - await installClaudeCodeHooks(result); - await installAntigravitySkills(result); - await installAntigravityHooks(result); - await installCursorSkills(result); - await installOpenCodeSkills(result); - await installCodexSkills(result); + if (selected.has('claude')) { + await installClaudeCodeSkills(result); + await installClaudeCodeHooks(result); + } + if (selected.has('antigravity')) { + await installAntigravitySkills(result); + await installAntigravityHooks(result); + } + if (selected.has('cursor')) await installCursorSkills(result); + if (selected.has('opencode')) await installOpenCodeSkills(result); + if (selected.has('codex')) await installCodexSkills(result); // Print results if (result.configured.length > 0) { @@ -1032,10 +1072,17 @@ export const setupCommand = async () => { console.log( ` Skills installed to: ${result.configured.filter((c) => c.includes('skills')).length > 0 ? result.configured.filter((c) => c.includes('skills')).join(', ') : 'none'}`, ); + const configurationSucceeded = result.configured.length > 0; + if (explicitSelection && !configurationSucceeded) { + process.stderr.write('None of the explicitly selected coding agents were configured.\n'); + process.exitCode = 1; + } console.log(''); - console.log(' Next steps:'); - console.log(' 1. cd into any git repo'); - console.log(' 2. Run: gitnexus analyze'); - console.log(' 3. Open the repo in your editor — MCP is ready!'); + if (configurationSucceeded) { + console.log(' Next steps:'); + console.log(' 1. cd into any git repo'); + console.log(' 2. Run: gitnexus analyze'); + console.log(' 3. Open the repo in your editor — MCP is ready!'); + } console.log(''); }; diff --git a/gitnexus/test/unit/cli-index-help.test.ts b/gitnexus/test/unit/cli-index-help.test.ts index 3f1ca142b..8bd6fcee9 100644 --- a/gitnexus/test/unit/cli-index-help.test.ts +++ b/gitnexus/test/unit/cli-index-help.test.ts @@ -149,6 +149,14 @@ describe('CLI help surface', () => { expect(result.stdout).not.toContain('Target repository (omit if only one indexed)'); }); + it('setup help exposes selective coding-agent configuration', () => { + const result = runHelp('setup'); + + expect(result.status).toBe(0); + expect(result.stdout).toContain('gitnexus setup [options]'); + expect(result.stdout).toContain('-c, --coding-agent '); + }); + it('localizes every registered CLI command and option description in zh-CN help', () => { const zhHelpOutput = allHelpCommands .map((args) => { diff --git a/gitnexus/test/unit/setup-selection.test.ts b/gitnexus/test/unit/setup-selection.test.ts new file mode 100644 index 000000000..99920002f --- /dev/null +++ b/gitnexus/test/unit/setup-selection.test.ts @@ -0,0 +1,131 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; +import fs from 'fs/promises'; +import os from 'os'; +import path from 'path'; + +const execFileMock = vi.fn((...args: any[]) => { + const callback = args.at(-1); + if (typeof callback === 'function') callback(null, '', ''); +}); + +vi.mock('child_process', () => ({ + execFile: execFileMock, + execFileSync: vi.fn(() => { + throw new Error('not found'); + }), +})); + +describe('setupCommand coding-agent selection', () => { + let tempHome: string; + let originalHome: string | undefined; + let originalUserProfile: string | undefined; + let originalExitCode: number | string | null | undefined; + + beforeEach(async () => { + vi.resetModules(); + vi.clearAllMocks(); + originalHome = process.env.HOME; + originalUserProfile = process.env.USERPROFILE; + originalExitCode = process.exitCode; + tempHome = await fs.mkdtemp(path.join(os.tmpdir(), 'gn-setup-selection-')); + process.env.HOME = tempHome; + process.env.USERPROFILE = tempHome; + process.exitCode = undefined; + await Promise.all([ + fs.mkdir(path.join(tempHome, '.cursor'), { recursive: true }), + fs.mkdir(path.join(tempHome, '.claude'), { recursive: true }), + fs.mkdir(path.join(tempHome, '.gemini', 'antigravity'), { recursive: true }), + fs.mkdir(path.join(tempHome, '.config', 'opencode'), { recursive: true }), + fs.mkdir(path.join(tempHome, '.codex'), { recursive: true }), + ]); + vi.spyOn(console, 'log').mockImplementation(() => {}); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + process.env.HOME = originalHome; + process.env.USERPROFILE = originalUserProfile; + process.exitCode = originalExitCode; + await fs.rm(tempHome, { recursive: true, force: true }); + }); + + it('configures only the requested coding agent', async () => { + const { setupCommand } = await import('../../src/cli/setup.js'); + await setupCommand({ codingAgent: ['opencode'] }); + + await expect( + fs.access(path.join(tempHome, '.config', 'opencode', 'opencode.json')), + ).resolves.toBeUndefined(); + await expect(fs.access(path.join(tempHome, '.cursor', 'mcp.json'))).rejects.toThrow(); + await expect(fs.access(path.join(tempHome, '.claude.json'))).rejects.toThrow(); + await expect( + fs.access(path.join(tempHome, '.gemini', 'antigravity', 'mcp_config.json')), + ).rejects.toThrow(); + await expect(fs.access(path.join(tempHome, '.codex', 'config.toml'))).rejects.toThrow(); + }); + + it('accepts comma-separated and repeated selections without configuring others', async () => { + const { setupCommand } = await import('../../src/cli/setup.js'); + await setupCommand({ codingAgent: ['cursor,opencode', 'cursor'] }); + + await expect(fs.access(path.join(tempHome, '.cursor', 'mcp.json'))).resolves.toBeUndefined(); + await expect( + fs.access(path.join(tempHome, '.config', 'opencode', 'opencode.json')), + ).resolves.toBeUndefined(); + await expect(fs.access(path.join(tempHome, '.claude.json'))).rejects.toThrow(); + }); + + it('rejects unknown values before writing configuration', async () => { + const stderr = vi.spyOn(process.stderr, 'write').mockImplementation(() => true); + const { setupCommand } = await import('../../src/cli/setup.js'); + await setupCommand({ codingAgent: ['opencode,unknown'] }); + + expect(process.exitCode).toBe(1); + expect(stderr).toHaveBeenCalledWith( + expect.stringContaining('Valid values: cursor, claude, antigravity, opencode, codex'), + ); + await expect( + fs.access(path.join(tempHome, '.config', 'opencode', 'opencode.json')), + ).rejects.toThrow(); + }); + + it.each([ + ['an empty string', ''], + ['an empty array', []], + ])('rejects %s before writing configuration', async (_label, codingAgent) => { + const stderr = vi.spyOn(process.stderr, 'write').mockImplementation(() => true); + const { setupCommand } = await import('../../src/cli/setup.js'); + + await setupCommand({ codingAgent }); + + expect(process.exitCode).toBe(1); + expect(stderr).toHaveBeenCalledWith(expect.stringContaining('No coding agents were provided.')); + await expect(fs.access(path.join(tempHome, '.cursor', 'mcp.json'))).rejects.toThrow(); + }); + + it('fails clearly when an explicitly selected agent is not installed', async () => { + await fs.rm(path.join(tempHome, '.codex'), { recursive: true, force: true }); + const stderr = vi.spyOn(process.stderr, 'write').mockImplementation(() => true); + const { setupCommand } = await import('../../src/cli/setup.js'); + + await setupCommand({ codingAgent: ['codex'] }); + + expect(process.exitCode).toBe(1); + expect(stderr).toHaveBeenCalledWith( + 'None of the explicitly selected coding agents were configured.\n', + ); + expect(vi.mocked(console.log).mock.calls.flat().join('\n')).not.toContain('MCP is ready!'); + }); + + it('preserves the no-flag default of configuring every detected agent', async () => { + const { setupCommand } = await import('../../src/cli/setup.js'); + + await setupCommand(); + + await expect(fs.access(path.join(tempHome, '.cursor', 'mcp.json'))).resolves.toBeUndefined(); + await expect(fs.access(path.join(tempHome, '.claude.json'))).resolves.toBeUndefined(); + await expect( + fs.access(path.join(tempHome, '.config', 'opencode', 'opencode.json')), + ).resolves.toBeUndefined(); + }); +}); From 60752de3e93fd75523f200370f96d135494b2289 Mon Sep 17 00:00:00 2001 From: Copilot <198982749+Copilot@users.noreply.github.com> Date: Sat, 13 Jun 2026 09:24:03 +0100 Subject: [PATCH 07/16] fix(ip): Scope write-route origin guard to server's own bound host (#2172) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Initial plan * Allow RFC1918 LAN origins in requireLocalhostOrigin * Harden LAN origin parsing in middleware tests * Refactor private IPv4 checks into shared server helper * fix: scope origin guard to server's bound host, fix [::1], guard all write routes - P1: Replace blanket RFC1918 trust with same-host check — only the server's own bound host is allowed (via `createLocalhostOriginGuard(host)`), not every device on the LAN. - P2: Fix dead `::1` branch — compare against `'[::1]'` (with brackets) as returned by WHATWG URL parser. - P3: Update 403 message to "same-host origins" and doc comments. - Out-of-scope: Add `requireLocalhostOrigin` to `DELETE /api/repo`, `POST /api/embed`, `DELETE /api/embed/:jobId`, `DELETE /api/analyze/:jobId`. - Tests: Add [::1] regression, ftp://, null origin, direct private-ip.ts unit tests, and createLocalhostOriginGuard bound-host tests. * fix: cast route params to string when middleware breaks type inference * chore(autofix): apply prettier + eslint fixes via /autofix command * fix(test): update rate-limit test regex to match multi-line embed route registration * fix(ip): normalize boundHost and keep wildcard binds loopback-only The same-host write guard compared the raw `--host` string to the WHATWG `URL.hostname` of the Origin, so it silently 403'd legitimate same-host browser writes for several bind forms: - mixed-case hostnames (`MyHost.local` vs lowercased `myhost.local`) - non-loopback IPv6 (`fe80::1` vs bracketed `[fe80::1]`, and non-canonical forms like `fe80:0:0:0:0:0:0:1` / `::ffff:127.0.0.1`) - wildcard binds (`0.0.0.0` / `::`), the CLI-advertised remote-access config Canonicalize boundHost once at guard construction through `new URL().hostname` (provably the same form the Origin is parsed into), and treat wildcard binds as having no single host identity → writes stay loopback-only. We deliberately do NOT fall through to RFC1918 for wildcards (that would re-open whole-LAN reach). `createServer` now warns when bound to a wildcard so a remote-access deployment is not silently write-blocked. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(ip): tag origin-block 403 with a machine-readable code and surface it in the web client The write-route Origin guard returned a 403 with only a human-readable `error` string, so clients could not distinguish an origin block from any other 403. The hosted web client (gitnexus.vercel.app driving a local backend) swallowed the resulting failure: the repo delete button caught the error and only `console.error`'d it, so it silently no-op'd. - Server: add a stable `code: 'origin_not_allowed'` discriminator to the 403 body. - Web client: `assertOk` reads `body.code` and maps `origin_not_allowed` to a new `BackendError` code `origin_blocked`; `formatBackendError` renders an actionable i18n message (en + zh-CN) instead of the generic client message. - Header: surface the delete failure inline instead of swallowing it to console. Scope note: the embedding-status badge (EmbeddingStatus.tsx) hides in backend mode (its `serverBaseUrl` guard), so it is not the surface where an origin-block embed error appears; a dedicated backend-mode embedding-error surface is deferred with the broader hosted-UI mode-awareness follow-up. Co-Authored-By: Claude Opus 4.8 (1M context) * refactor(ip): remove unused isValidIpv4Address export `isValidIpv4Address` had no `src/` consumer — only its own test imported it. It was a leftover from the reverted RFC1918-middleware approach (the same-host guard now compares against a canonicalized bound host, not an IPv4 validity check). Remove the export and its orphaned test block. `parseIpv4Octets` stays (it feeds `isRfc1918PrivateIpv4`, which CORS `isAllowedOrigin` still uses). Co-Authored-By: Claude Opus 4.8 (1M context) --------- Co-authored-by: copilot-swe-agent[bot] <198982749+Copilot@users.noreply.github.com> Co-authored-by: Gergő Magyar Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: Claude Opus 4.8 (1M context) --- gitnexus-web/src/components/Header.tsx | 16 +- gitnexus-web/src/i18n/error-messages.ts | 2 + gitnexus-web/src/locales/en/errors.json | 1 + gitnexus-web/src/locales/zh-CN/errors.json | 1 + gitnexus-web/src/services/backend-client.ts | 20 +- .../test/unit/backend-client-retry.test.ts | 41 ++- gitnexus/src/server/api.ts | 276 +++++++++--------- gitnexus/src/server/middleware.ts | 98 ++++++- gitnexus/src/server/private-ip.ts | 13 + gitnexus/test/unit/api-analyze-upload.test.ts | 101 ++++++- gitnexus/test/unit/private-ip.test.ts | 44 +++ gitnexus/test/unit/rate-limit.test.ts | 4 +- 12 files changed, 460 insertions(+), 157 deletions(-) create mode 100644 gitnexus/src/server/private-ip.ts create mode 100644 gitnexus/test/unit/private-ip.test.ts diff --git a/gitnexus-web/src/components/Header.tsx b/gitnexus-web/src/components/Header.tsx index 3fae0c48f..3dc83301c 100644 --- a/gitnexus-web/src/components/Header.tsx +++ b/gitnexus-web/src/components/Header.tsx @@ -27,6 +27,7 @@ import { EmbeddingStatus } from './EmbeddingStatus'; import { RepoAnalyzer } from './RepoAnalyzer'; import { LanguageSwitcher } from './LanguageSwitcher'; import { translateProgressMessage } from '../i18n/progress'; +import { formatBackendError } from '../i18n/error-messages'; // Color mapping for node types in search results const NODE_TYPE_COLORS: Record = { @@ -58,7 +59,7 @@ export const Header = ({ onAnalyzeComplete, onReposChanged, }: HeaderProps) => { - const { t } = useTranslation(['common', 'header']); + const { t } = useTranslation(['common', 'header', 'errors']); const { projectName, graph, @@ -72,6 +73,7 @@ export const Header = ({ const [isRepoDropdownOpen, setIsRepoDropdownOpen] = useState(false); const [showAnalyzer, setShowAnalyzer] = useState(false); const [reanalyzing, setReanalyzing] = useState(null); // repo name being re-analyzed + const [deleteError, setDeleteError] = useState(null); // surfaced when a delete is rejected (e.g. origin-blocked 403) const [reanalyzeProgress, setReanalyzeProgress] = useState(null); const reanalyzeSseRef = useRef(null); const repoDropdownRef = useRef(null); @@ -305,6 +307,7 @@ export const Header = ({ setReanalyzeProgress(null); reanalyzeSseRef.current = null; } + setDeleteError(null); try { await deleteRepo(repo.name); const updated = await fetchRepos(); @@ -317,7 +320,11 @@ export const Header = ({ window.location.reload(); } } catch (err) { + // Surface the failure instead of silently no-opping — + // e.g. an origin-blocked 403 when driving a local + // backend from the hosted UI. console.error('Failed to delete repo:', err); + setDeleteError(formatBackendError(err, t)); } }} className="cursor-pointer rounded p-1 text-text-muted/0 transition-all group-hover:text-text-muted hover:!text-red-400" @@ -330,6 +337,13 @@ export const Header = ({ )} + {/* Surfaced delete failure (e.g. origin-blocked 403) */} + {deleteError && ( +
+ {deleteError} +
+ )} + {/* Re-analyze progress bar */} {reanalyzing && reanalyzeProgress && (
diff --git a/gitnexus-web/src/i18n/error-messages.ts b/gitnexus-web/src/i18n/error-messages.ts index 620350b93..1af2169aa 100644 --- a/gitnexus-web/src/i18n/error-messages.ts +++ b/gitnexus-web/src/i18n/error-messages.ts @@ -14,6 +14,8 @@ export function formatBackendError(error: unknown, t: TFunction): string { return t('errors:backend.rateLimited', { seconds, defaultValue: fallback }); case 'not_found': return t('errors:backend.notFound', { defaultValue: fallback }); + case 'origin_blocked': + return t('errors:backend.originBlocked', { defaultValue: fallback }); case 'client': return t('errors:backend.client', { message: error.message, defaultValue: fallback }); case 'server': diff --git a/gitnexus-web/src/locales/en/errors.json b/gitnexus-web/src/locales/en/errors.json index e7bdcde33..c3f727078 100644 --- a/gitnexus-web/src/locales/en/errors.json +++ b/gitnexus-web/src/locales/en/errors.json @@ -13,6 +13,7 @@ "timeout": "The server took too long to respond. Try again in a moment.", "rateLimited": "Too many requests. Try again in {{seconds}}s.", "notFound": "The requested repository or resource was not found.", + "originBlocked": "This action isn't available from the hosted UI. Open GitNexus from the server's own address (e.g. http://localhost:4747) to continue.", "client": "Request failed: {{message}}", "server": "Server error: {{message}}" } diff --git a/gitnexus-web/src/locales/zh-CN/errors.json b/gitnexus-web/src/locales/zh-CN/errors.json index 47dcb4f8d..d567d4ff9 100644 --- a/gitnexus-web/src/locales/zh-CN/errors.json +++ b/gitnexus-web/src/locales/zh-CN/errors.json @@ -13,6 +13,7 @@ "timeout": "服务器响应超时,请稍后重试。", "rateLimited": "请求过于频繁,请在 {{seconds}} 秒后重试。", "notFound": "未找到请求的仓库或资源。", + "originBlocked": "此操作无法从托管界面执行。请通过服务器自身地址(例如 http://localhost:4747)打开 GitNexus 后再继续。", "client": "请求失败:{{message}}", "server": "服务器错误:{{message}}" } diff --git a/gitnexus-web/src/services/backend-client.ts b/gitnexus-web/src/services/backend-client.ts index 286a13187..b12cf45a9 100644 --- a/gitnexus-web/src/services/backend-client.ts +++ b/gitnexus-web/src/services/backend-client.ts @@ -79,7 +79,11 @@ export class BackendError extends Error { | 'client' | 'not_found' | 'timeout' - | 'rate_limited', + | 'rate_limited' + // The write-route same-host Origin guard rejected this request (HTTP 403 + // with `{ code: 'origin_not_allowed' }`). Distinct from a generic `client` + // 403 so the UI can show actionable "open the local UI" guidance. + | 'origin_blocked', /** * Milliseconds until the caller should retry. Populated for rate-limited * responses (HTTP 429) from the server's `Retry-After` header. `undefined` @@ -361,6 +365,7 @@ const assertOk = async (response: Response): Promise => { if (response.ok) return; let message = response.statusText; + let bodyCode: string | undefined; try { const body = await response.json(); if (body && typeof body.error === 'string') { @@ -368,6 +373,9 @@ const assertOk = async (response: Response): Promise => { } else if (body && typeof body.message === 'string') { message = body.message; } + if (body && typeof body.code === 'string') { + bodyCode = body.code; + } } catch { // Response body was not JSON } @@ -377,9 +385,13 @@ const assertOk = async (response: Response): Promise => { ? 'not_found' : response.status === 429 ? 'rate_limited' - : response.status >= 400 && response.status < 500 - ? 'client' - : 'server'; + : // The write-route Origin guard returns 403 with this discriminator; + // surface it as a distinct code so the UI can give actionable guidance. + bodyCode === 'origin_not_allowed' + ? 'origin_blocked' + : response.status >= 400 && response.status < 500 + ? 'client' + : 'server'; // Retry-After is the standard HTTP signal for when the client may try again. // express-rate-limit emits it on 429 with seconds (integer) or HTTP-date. diff --git a/gitnexus-web/test/unit/backend-client-retry.test.ts b/gitnexus-web/test/unit/backend-client-retry.test.ts index ea0fcf3a7..e25dc9a73 100644 --- a/gitnexus-web/test/unit/backend-client-retry.test.ts +++ b/gitnexus-web/test/unit/backend-client-retry.test.ts @@ -13,7 +13,12 @@ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; import { getBreaker } from 'gitnexus-shared'; import { __resetBreakerRegistry__ } from 'gitnexus-shared/test-helpers'; -import { fetchRepos, setBackendUrl, startAnalyze } from '../../src/services/backend-client'; +import { + deleteRepo, + fetchRepos, + setBackendUrl, + startAnalyze, +} from '../../src/services/backend-client'; const BASE = 'http://localhost:4747'; @@ -89,6 +94,40 @@ describe('backend-client retry budget (method-aware)', () => { expect(getBreaker(bKey).getConsecutiveFailures()).toBe(0); }); + it('maps an origin-blocked 403 to BackendError code "origin_blocked"', async () => { + const fetchMock = vi.fn( + async () => + new Response( + JSON.stringify({ + error: 'This endpoint is restricted to same-host origins', + code: 'origin_not_allowed', + }), + { status: 403, headers: { 'Content-Type': 'application/json' } }, + ), + ); + vi.stubGlobal('fetch', fetchMock); + + await expect(deleteRepo('my-repo')).rejects.toMatchObject({ + status: 403, + code: 'origin_blocked', + }); + // 403 is a terminal client error — never retried. + expect(fetchMock).toHaveBeenCalledTimes(1); + }); + + it('maps a generic 403 (no recognized code) to BackendError code "client" (back-compat)', async () => { + const fetchMock = vi.fn( + async () => + new Response(JSON.stringify({ error: 'forbidden' }), { + status: 403, + headers: { 'Content-Type': 'application/json' }, + }), + ); + vi.stubGlobal('fetch', fetchMock); + + await expect(deleteRepo('my-repo')).rejects.toMatchObject({ status: 403, code: 'client' }); + }); + it('breaker not incremented when timeout fires (TimeoutError, not AbortError)', async () => { // Reject directly with a TimeoutError DOMException, mimicking what // `fetch` produces when its `AbortSignal.timeout()`-wired signal diff --git a/gitnexus/src/server/api.ts b/gitnexus/src/server/api.ts index 67ae63009..05836d350 100644 --- a/gitnexus/src/server/api.ts +++ b/gitnexus/src/server/api.ts @@ -35,10 +35,11 @@ import { JobManager } from './analyze-job.js'; import { assertString, escapeRegExp, BadRequestError, createRouteLimiter } from './validation.js'; import { extractRepoName, getCloneDir, cloneOrPull } from './git-clone.js'; import { createAnalyzeUploadHandler } from './analyze-upload.js'; -import { requireLocalhostOrigin } from './middleware.js'; +import { createLocalhostOriginGuard, normalizeBoundHost } from './middleware.js'; import { createLaunchAnalysisWorker } from './analyze-launch.js'; import { UPLOAD_ROOT } from './upload-paths.js'; import { sweepStaleUploads } from './upload-sweep.js'; +import { isRfc1918PrivateIpv4 } from './private-ip.js'; import { logger, flushLoggerSync } from '../core/logger.js'; const _require = createRequire(import.meta.url); @@ -95,21 +96,7 @@ export const isAllowedOrigin = (origin: string | undefined): boolean => { // Only allow HTTP(S) origins — reject ftp://, file://, etc. if (protocol !== 'http:' && protocol !== 'https:') return false; - const octets = hostname.split('.').map(Number); - if (octets.length !== 4 || octets.some((o) => !Number.isInteger(o) || o < 0 || o > 255)) { - return false; - } - - const [a, b] = octets; - - // 10.0.0.0/8 - if (a === 10) return true; - // 172.16.0.0/12 → 172.16.x.x – 172.31.x.x - if (a === 172 && b >= 16 && b <= 31) return true; - // 192.168.0.0/16 - if (a === 192 && b === 168) return true; - - return false; + return isRfc1918PrivateIpv4(hostname); }; type GraphStreamRecord = @@ -733,6 +720,22 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => ); app.use(express.json({ limit: '10mb' })); + // Same-host origin guard for write routes. Only allows loopback and the + // server's own bound host — scoped to prevent CSRF from other LAN devices. + const requireLocalhostOrigin = createLocalhostOriginGuard(host); + + // A wildcard bind (`0.0.0.0`/`::`) has no single host identity for the + // same-host check, so browser write routes accept only loopback origins. + // Warn the operator so a remote-access deployment isn't silently write-blocked. + if (host && normalizeBoundHost(host) === undefined) { + logger.warn( + { host }, + `[gitnexus serve] Bound to a wildcard address (${host}); browser write routes ` + + `accept only loopback origins (localhost/127.0.0.1/[::1]). To allow writes from a ` + + `specific LAN address, bind --host instead of a wildcard.`, + ); + } + // No explicit OPTIONS route is registered. The Chromium Private Network // Access header is set by the global middleware above (pre-cors), and // `cors()` itself handles OPTIONS preflights for every path. Registering a @@ -957,7 +960,7 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => // Rate-limited (CodeQL js/missing-rate-limiting): destructive operation // doing fs.rm of clone + storage dirs. Default 60 rpm/IP is generous for // delete; tighten if abuse is observed. - app.delete('/api/repo', createRouteLimiter(), async (req, res) => { + app.delete('/api/repo', createRouteLimiter(), requireLocalhostOrigin, async (req, res) => { try { const repoName = requestedRepo(req); if (!repoName) { @@ -1480,10 +1483,12 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => // slashes, so it is dropped. Analyzing a local path the operator names // is the tool's intended capability (same as the CLI); the dangerous // part was cross-origin reach, which is closed by requireLocalhostOrigin - // on this route. We only require an absolute path here and let the - // analyze worker surface a clear error if it does not exist. (We do NOT - // realpath/stat the path in-route: that would be a user-controlled - // filesystem read — CodeQL js/path-injection — for no security gain.) + // on this route (scoped to the server's own bound host — other LAN + // devices are NOT trusted). We only require an absolute path here and + // let the analyze worker surface a clear error if it does not exist. + // (We do NOT realpath/stat the path in-route: that would be a + // user-controlled filesystem read — CodeQL js/path-injection — for no + // security gain.) if (repoLocalPath && !path.isAbsolute(repoLocalPath)) { res.status(400).json({ error: '"path" must be an absolute path' }); return; @@ -1586,8 +1591,9 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => mountSSEProgress(app, '/api/analyze/:jobId/progress', jobManager); // DELETE /api/analyze/:jobId — cancel a running analysis job - app.delete('/api/analyze/:jobId', (req, res) => { - const job = jobManager.getJob(req.params.jobId); + app.delete('/api/analyze/:jobId', requireLocalhostOrigin, (req, res) => { + const jobId = req.params.jobId as string; + const job = jobManager.getJob(jobId); if (!job) { res.status(404).json({ error: 'Job not found' }); return; @@ -1596,7 +1602,7 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => res.status(400).json({ error: `Job already ${job.status}` }); return; } - jobManager.cancelJob(req.params.jobId, 'Cancelled by user'); + jobManager.cancelJob(jobId, 'Cancelled by user'); res.json({ id: job.id, status: 'failed', error: 'Cancelled by user' }); }); @@ -1605,122 +1611,127 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => const embedJobManager = new JobManager(); // POST /api/embed — trigger server-side embedding generation - app.post('/api/embed', createRouteLimiter({ limit: 20 }), async (req, res) => { - try { - const entry = await resolveRepo(requestedRepo(req)); - if (!entry) { - res.status(404).json({ error: 'Repository not found' }); - return; - } - - // Check shared repo lock — prevent concurrent analyze + embed on same repo - const repoLockPath = entry.storagePath; - const lockErr = acquireRepoLock(repoLockPath); - if (lockErr) { - res.status(409).json({ error: lockErr }); - return; - } - - const job = embedJobManager.createJob({ repoPath: entry.storagePath }); - embedJobManager.updateJob(job.id, { - repoName: entry.name, - status: 'analyzing' as any, - progress: { phase: 'analyzing', percent: 0, message: 'Starting embedding generation...' }, - }); - - // 30-minute timeout for embedding jobs (same as analyze jobs) - const EMBED_TIMEOUT_MS = 30 * 60 * 1000; - const embedTimeout = setTimeout(() => { - const current = embedJobManager.getJob(job.id); - if (current && current.status !== 'complete' && current.status !== 'failed') { - releaseRepoLock(repoLockPath); - embedJobManager.updateJob(job.id, { - status: 'failed', - error: 'Embedding timed out (30 minute limit)', - }); + app.post( + '/api/embed', + createRouteLimiter({ limit: 20 }), + requireLocalhostOrigin, + async (req, res) => { + try { + const entry = await resolveRepo(requestedRepo(req)); + if (!entry) { + res.status(404).json({ error: 'Repository not found' }); + return; } - }, EMBED_TIMEOUT_MS); - // Run embedding pipeline asynchronously - (async () => { - try { - const lbugPath = path.join(entry.storagePath, 'lbug'); - await withLbugDb(lbugPath, async () => { - const { runEmbeddingPipeline } = - await import('../core/embeddings/embedding-pipeline.js'); - // Fetch existing content hashes for incremental embedding. - // Delegated to lbug-adapter which owns the DB query logic and legacy-fallback handling. - const { fetchExistingEmbeddingHashes } = await import('../core/lbug/lbug-adapter.js'); - const existingEmbeddings = await fetchExistingEmbeddingHashes(executeQuery); - if (existingEmbeddings && existingEmbeddings.size > 0) { - console.log( - `[embed] ${existingEmbeddings.size} nodes already embedded — incremental run with content-hash comparison`, - ); - } - await runEmbeddingPipeline( - executeQuery, - executeWithReusedStatement, - (p) => { - embedJobManager.updateJob(job.id, { - progress: { - phase: - p.phase === 'ready' ? 'complete' : p.phase === 'error' ? 'failed' : p.phase, - percent: p.percent, - message: - p.phase === 'loading-model' - ? 'Loading embedding model...' - : p.phase === 'embedding' - ? `Embedding nodes (${p.percent}%)...` - : p.phase === 'indexing' - ? 'Creating vector index...' - : p.phase === 'ready' - ? 'Embeddings complete' - : `${p.phase} (${p.percent}%)`, - }, - }); - }, - {}, // config: use defaults - undefined, // skipNodeIds - undefined, // context - existingEmbeddings, - ); + // Check shared repo lock — prevent concurrent analyze + embed on same repo + const repoLockPath = entry.storagePath; + const lockErr = acquireRepoLock(repoLockPath); + if (lockErr) { + res.status(409).json({ error: lockErr }); + return; + } - // Flush WAL so subsequent /api/search requests see the new - // embeddings immediately (#1149). In the CLI path closeLbug() - // handles this during process exit, but the server keeps the - // connection open for other routes — a CHECKPOINT is enough. - await flushWAL(); - }); + const job = embedJobManager.createJob({ repoPath: entry.storagePath }); + embedJobManager.updateJob(job.id, { + repoName: entry.name, + status: 'analyzing' as any, + progress: { phase: 'analyzing', percent: 0, message: 'Starting embedding generation...' }, + }); - clearTimeout(embedTimeout); - releaseRepoLock(repoLockPath); - // Don't overwrite 'failed' if the job was cancelled while the pipeline was running + // 30-minute timeout for embedding jobs (same as analyze jobs) + const EMBED_TIMEOUT_MS = 30 * 60 * 1000; + const embedTimeout = setTimeout(() => { const current = embedJobManager.getJob(job.id); - if (!current || current.status !== 'failed') { - embedJobManager.updateJob(job.id, { status: 'complete' }); - } - } catch (err: any) { - clearTimeout(embedTimeout); - releaseRepoLock(repoLockPath); - const current = embedJobManager.getJob(job.id); - if (!current || current.status !== 'failed') { + if (current && current.status !== 'complete' && current.status !== 'failed') { + releaseRepoLock(repoLockPath); embedJobManager.updateJob(job.id, { status: 'failed', - error: err.message || 'Embedding generation failed', + error: 'Embedding timed out (30 minute limit)', }); } - } - })(); + }, EMBED_TIMEOUT_MS); - res.status(202).json({ jobId: job.id, status: 'analyzing' }); - } catch (err: any) { - if (err.message?.includes('already in progress')) { - res.status(409).json({ error: err.message }); - } else { - res.status(500).json({ error: err.message || 'Failed to start embedding generation' }); + // Run embedding pipeline asynchronously + (async () => { + try { + const lbugPath = path.join(entry.storagePath, 'lbug'); + await withLbugDb(lbugPath, async () => { + const { runEmbeddingPipeline } = + await import('../core/embeddings/embedding-pipeline.js'); + // Fetch existing content hashes for incremental embedding. + // Delegated to lbug-adapter which owns the DB query logic and legacy-fallback handling. + const { fetchExistingEmbeddingHashes } = await import('../core/lbug/lbug-adapter.js'); + const existingEmbeddings = await fetchExistingEmbeddingHashes(executeQuery); + if (existingEmbeddings && existingEmbeddings.size > 0) { + console.log( + `[embed] ${existingEmbeddings.size} nodes already embedded — incremental run with content-hash comparison`, + ); + } + await runEmbeddingPipeline( + executeQuery, + executeWithReusedStatement, + (p) => { + embedJobManager.updateJob(job.id, { + progress: { + phase: + p.phase === 'ready' ? 'complete' : p.phase === 'error' ? 'failed' : p.phase, + percent: p.percent, + message: + p.phase === 'loading-model' + ? 'Loading embedding model...' + : p.phase === 'embedding' + ? `Embedding nodes (${p.percent}%)...` + : p.phase === 'indexing' + ? 'Creating vector index...' + : p.phase === 'ready' + ? 'Embeddings complete' + : `${p.phase} (${p.percent}%)`, + }, + }); + }, + {}, // config: use defaults + undefined, // skipNodeIds + undefined, // context + existingEmbeddings, + ); + + // Flush WAL so subsequent /api/search requests see the new + // embeddings immediately (#1149). In the CLI path closeLbug() + // handles this during process exit, but the server keeps the + // connection open for other routes — a CHECKPOINT is enough. + await flushWAL(); + }); + + clearTimeout(embedTimeout); + releaseRepoLock(repoLockPath); + // Don't overwrite 'failed' if the job was cancelled while the pipeline was running + const current = embedJobManager.getJob(job.id); + if (!current || current.status !== 'failed') { + embedJobManager.updateJob(job.id, { status: 'complete' }); + } + } catch (err: any) { + clearTimeout(embedTimeout); + releaseRepoLock(repoLockPath); + const current = embedJobManager.getJob(job.id); + if (!current || current.status !== 'failed') { + embedJobManager.updateJob(job.id, { + status: 'failed', + error: err.message || 'Embedding generation failed', + }); + } + } + })(); + + res.status(202).json({ jobId: job.id, status: 'analyzing' }); + } catch (err: any) { + if (err.message?.includes('already in progress')) { + res.status(409).json({ error: err.message }); + } else { + res.status(500).json({ error: err.message || 'Failed to start embedding generation' }); + } } - } - }); + }, + ); // GET /api/embed/:jobId — poll embedding job status app.get('/api/embed/:jobId', (req, res) => { @@ -1744,8 +1755,9 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => mountSSEProgress(app, '/api/embed/:jobId/progress', embedJobManager); // DELETE /api/embed/:jobId — cancel embedding job - app.delete('/api/embed/:jobId', (req, res) => { - const job = embedJobManager.getJob(req.params.jobId); + app.delete('/api/embed/:jobId', requireLocalhostOrigin, (req, res) => { + const jobId = req.params.jobId as string; + const job = embedJobManager.getJob(jobId); if (!job) { res.status(404).json({ error: 'Job not found' }); return; @@ -1754,7 +1766,7 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => res.status(400).json({ error: `Job already ${job.status}` }); return; } - embedJobManager.cancelJob(req.params.jobId, 'Cancelled by user'); + embedJobManager.cancelJob(jobId, 'Cancelled by user'); res.json({ id: job.id, status: 'failed', error: 'Cancelled by user' }); }); diff --git a/gitnexus/src/server/middleware.ts b/gitnexus/src/server/middleware.ts index 73d3edab7..286d0b337 100644 --- a/gitnexus/src/server/middleware.ts +++ b/gitnexus/src/server/middleware.ts @@ -5,25 +5,91 @@ import type { Request, Response } from 'express'; /** - * Restrict a route to localhost browser origins. Non-browser requests (no - * Origin header, e.g. curl / the CLI) pass through. This closes cross-origin - * reach (the allow-listed public deploy + Private Network Access) to write - * routes without affecting read routes. + * Canonicalize a bound-host string into the form a browser `Origin` hostname + * takes after WHATWG URL parsing, so the same-host comparison in + * {@link createLocalhostOriginGuard} can use a plain `===`. + * + * Returns `undefined` when the host carries no single comparable identity: + * - empty / not provided + * - a wildcard bind (`0.0.0.0`, `::`, expanded `0:0:0:0:0:0:0:0`) — the server + * listens on every interface and has no one address a browser Origin maps to, + * so writes stay loopback-only (we deliberately do NOT trust the whole subnet) + * - an unparseable value + * + * Otherwise returns `new URL(...).hostname` (lowercased, IPv6 bracketed and + * compressed) — provably identical to how the request Origin is parsed below. + * Hand-rolling lowercase + bracketing is insufficient: it fails to compress + * non-canonical IPv6 forms (e.g. `fe80:0:0:0:0:0:0:1`, `::ffff:127.0.0.1`). */ -export function requireLocalhostOrigin(req: Request, res: Response, next: () => void): void { - const origin = req.headers.origin; - if (origin === undefined) { - next(); - return; - } +export function normalizeBoundHost(boundHost?: string): string | undefined { + if (!boundHost) return undefined; + // Bracket a bare IPv6 literal so `new URL` can parse it as a host. + const candidate = + boundHost.includes(':') && !boundHost.startsWith('[') ? `[${boundHost}]` : boundHost; + let hostname: string; try { - const hostname = new URL(origin).hostname; - if (hostname === 'localhost' || hostname === '127.0.0.1' || hostname === '::1') { + hostname = new URL(`http://${candidate}`).hostname; + } catch { + return undefined; + } + // Wildcard binds have no single host identity → keep writes loopback-only. + if (hostname === '' || hostname === '0.0.0.0' || hostname === '[::]') { + return undefined; + } + return hostname; +} + +/** + * Restrict a route to same-host browser origins. Allows: + * - loopback (`localhost`, `127.0.0.1`, `[::1]`) + * - the server's own bound host (when non-loopback, e.g. a LAN IP) + * + * Non-browser requests (no Origin header, e.g. curl / the CLI) pass through. + * This closes cross-origin reach to write routes without affecting read routes. + * + * @param boundHost - The hostname/IP the server is listening on (from + * `createServer`'s `host` parameter). When `undefined`, `'localhost'`, or a + * wildcard (`0.0.0.0`/`::`), only loopback origins are admitted. + */ +export function createLocalhostOriginGuard(boundHost?: string) { + const normalizedBoundHost = normalizeBoundHost(boundHost); + return function requireLocalhostOrigin(req: Request, res: Response, next: () => void): void { + const origin = req.headers.origin; + if (origin === undefined) { next(); return; } - } catch { - /* malformed origin → reject */ - } - res.status(403).json({ error: 'This endpoint is restricted to localhost origins' }); + try { + const parsed = new URL(origin); + const hostname = parsed.hostname; + const protocol = parsed.protocol; + if (protocol !== 'http:' && protocol !== 'https:') { + throw new Error('Unsupported origin protocol'); + } + if (hostname === 'localhost' || hostname === '127.0.0.1' || hostname === '[::1]') { + next(); + return; + } + // Allow origin matching the server's own bound host (same-host check). + // `normalizedBoundHost` is canonicalized to the WHATWG form `hostname` + // already carries; it is `undefined` for wildcard/no binds (loopback-only). + // This covers the case where the operator runs `gitnexus serve --host `. + if (normalizedBoundHost && hostname === normalizedBoundHost) { + next(); + return; + } + } catch { + /* malformed origin → reject */ + } + res.status(403).json({ + error: 'This endpoint is restricted to same-host origins', + code: 'origin_not_allowed', + }); + }; } + +/** + * Default guard that only allows loopback origins. For use in tests or when + * the bound host is not available. + */ +export const requireLocalhostOrigin = createLocalhostOriginGuard(); diff --git a/gitnexus/src/server/private-ip.ts b/gitnexus/src/server/private-ip.ts new file mode 100644 index 000000000..ac9a37fd0 --- /dev/null +++ b/gitnexus/src/server/private-ip.ts @@ -0,0 +1,13 @@ +const parseIpv4Octets = (hostname: string): number[] | null => { + if (!/^\d{1,3}(\.\d{1,3}){3}$/.test(hostname)) return null; + const octets = hostname.split('.').map(Number); + if (octets.some((o) => !Number.isInteger(o) || o < 0 || o > 255)) return null; + return octets; +}; + +export const isRfc1918PrivateIpv4 = (hostname: string): boolean => { + const octets = parseIpv4Octets(hostname); + if (octets === null) return false; + const [a, b] = octets; + return a === 10 || (a === 172 && b >= 16 && b <= 31) || (a === 192 && b === 168); +}; diff --git a/gitnexus/test/unit/api-analyze-upload.test.ts b/gitnexus/test/unit/api-analyze-upload.test.ts index af3bbfd23..8cc1167c0 100644 --- a/gitnexus/test/unit/api-analyze-upload.test.ts +++ b/gitnexus/test/unit/api-analyze-upload.test.ts @@ -4,7 +4,7 @@ import fs from 'node:fs/promises'; import { Readable } from 'node:stream'; import type { IncomingMessage } from 'node:http'; import { createAnalyzeUploadHandler } from '../../src/server/analyze-upload.js'; -import { requireLocalhostOrigin } from '../../src/server/middleware.js'; +import { requireLocalhostOrigin, createLocalhostOriginGuard } from '../../src/server/middleware.js'; const BOUNDARY = '----gitnexusuploadtest'; @@ -275,9 +275,10 @@ describe('requireLocalhostOrigin', () => { return { passed, status }; } - it('passes localhost / 127.0.0.1 / no-origin', () => { + it('passes localhost / 127.0.0.1 / [::1] / no-origin', () => { expect(call('http://localhost:5173').passed).toBe(true); expect(call('http://127.0.0.1:4747').passed).toBe(true); + expect(call('http://[::1]:4747').passed).toBe(true); expect(call(undefined).passed).toBe(true); }); @@ -286,4 +287,100 @@ describe('requireLocalhostOrigin', () => { expect(r.passed).toBe(false); expect(r.status).toBe(403); }); + + it('rejects RFC1918 origins when no boundHost is set (default guard)', () => { + expect(call('http://10.0.0.1:4173').passed).toBe(false); + expect(call('http://172.16.1.21:4173').passed).toBe(false); + expect(call('http://192.168.1.100:4173').passed).toBe(false); + }); + + it('rejects malformed and non-private hostnames with 403', () => { + expect(call('http://my-local-server.local:4173').passed).toBe(false); + expect(call('ftp://localhost:4173').passed).toBe(false); + expect(call('null').passed).toBe(false); + }); +}); + +describe('createLocalhostOriginGuard (bound host)', () => { + function callWith( + boundHost: string, + origin: string | undefined, + ): { passed: boolean; status: number; body?: { error?: string; code?: string } } { + const guard = createLocalhostOriginGuard(boundHost); + let passed = false; + let status = 0; + let body: { error?: string; code?: string } | undefined; + const req = { headers: origin === undefined ? {} : { origin } } as never; + const res = { + status: (c: number) => { + status = c; + return { + json: (b: { error?: string; code?: string }) => { + body = b; + }, + }; + }, + } as never; + guard(req, res, () => { + passed = true; + }); + return { passed, status, body }; + } + + it('allows origin matching the bound host', () => { + expect(callWith('192.168.1.100', 'http://192.168.1.100:4747').passed).toBe(true); + expect(callWith('10.0.0.5', 'http://10.0.0.5:4173').passed).toBe(true); + expect(callWith('172.16.1.21', 'http://172.16.1.21:4173').passed).toBe(true); + }); + + it('still allows loopback regardless of bound host', () => { + expect(callWith('192.168.1.100', 'http://localhost:5173').passed).toBe(true); + expect(callWith('192.168.1.100', 'http://127.0.0.1:4747').passed).toBe(true); + expect(callWith('192.168.1.100', 'http://[::1]:4747').passed).toBe(true); + }); + + it('normalizes mixed-case host binds to match the WHATWG origin hostname', () => { + // WHATWG lowercases the Origin hostname; boundHost must canonicalize the same way. + expect(callWith('MyHost.local', 'http://myhost.local:4747').passed).toBe(true); + }); + + it('normalizes IPv6 host binds (compressed + non-canonical) to match the origin', () => { + expect(callWith('fe80::1', 'http://[fe80::1]:4747').passed).toBe(true); + // Expanded form must compress to the same WHATWG hostname as the origin. + expect(callWith('fe80:0:0:0:0:0:0:1', 'http://[fe80::1]:4747').passed).toBe(true); + // Already-bracketed input is idempotent. + expect(callWith('[fe80::1]', 'http://[fe80::1]:4747').passed).toBe(true); + }); + + it('keeps wildcard binds (0.0.0.0 / :: / expanded) loopback-only', () => { + // No browser Origin equals a wildcard, so non-loopback writes are rejected... + expect(callWith('0.0.0.0', 'http://192.168.1.5:4747').passed).toBe(false); + expect(callWith('::', 'http://[fe80::1]:4747').passed).toBe(false); + expect(callWith('0:0:0:0:0:0:0:0', 'http://[fe80::1]:4747').passed).toBe(false); + // ...while loopback still passes under a wildcard bind. + expect(callWith('0.0.0.0', 'http://localhost:5173').passed).toBe(true); + expect(callWith('::', 'http://127.0.0.1:4747').passed).toBe(true); + }); + + it('rejects other RFC1918 origins that do not match bound host', () => { + expect(callWith('192.168.1.100', 'http://192.168.1.101:4747').passed).toBe(false); + expect(callWith('192.168.1.100', 'http://10.0.0.1:4747').passed).toBe(false); + expect(callWith('10.0.0.5', 'http://172.16.1.21:4747').passed).toBe(false); + }); + + it('rejects public origins even when bound to LAN', () => { + const r = callWith('192.168.1.100', 'https://gitnexus.vercel.app'); + expect(r.passed).toBe(false); + expect(r.status).toBe(403); + }); + + it('tags the rejection 403 with a machine-readable code', () => { + const r = callWith('192.168.1.100', 'https://gitnexus.vercel.app'); + expect(r.status).toBe(403); + expect(r.body?.code).toBe('origin_not_allowed'); + }); + + it('passes no-origin (non-browser) requests', () => { + expect(callWith('192.168.1.100', undefined).passed).toBe(true); + }); }); diff --git a/gitnexus/test/unit/private-ip.test.ts b/gitnexus/test/unit/private-ip.test.ts new file mode 100644 index 000000000..5e1374167 --- /dev/null +++ b/gitnexus/test/unit/private-ip.test.ts @@ -0,0 +1,44 @@ +import { describe, expect, it } from 'vitest'; +import { isRfc1918PrivateIpv4 } from '../../src/server/private-ip.js'; + +describe('isRfc1918PrivateIpv4', () => { + it('accepts 10.0.0.0/8 range', () => { + expect(isRfc1918PrivateIpv4('10.0.0.0')).toBe(true); + expect(isRfc1918PrivateIpv4('10.255.255.255')).toBe(true); + expect(isRfc1918PrivateIpv4('10.1.2.3')).toBe(true); + }); + + it('accepts 172.16.0.0/12 range', () => { + expect(isRfc1918PrivateIpv4('172.16.0.0')).toBe(true); + expect(isRfc1918PrivateIpv4('172.31.255.255')).toBe(true); + expect(isRfc1918PrivateIpv4('172.20.1.1')).toBe(true); + }); + + it('rejects 172.x outside /12 range', () => { + expect(isRfc1918PrivateIpv4('172.15.255.255')).toBe(false); + expect(isRfc1918PrivateIpv4('172.32.0.0')).toBe(false); + }); + + it('accepts 192.168.0.0/16 range', () => { + expect(isRfc1918PrivateIpv4('192.168.0.0')).toBe(true); + expect(isRfc1918PrivateIpv4('192.168.255.255')).toBe(true); + expect(isRfc1918PrivateIpv4('192.168.1.100')).toBe(true); + }); + + it('rejects 192.x outside /16 range', () => { + expect(isRfc1918PrivateIpv4('192.167.1.1')).toBe(false); + expect(isRfc1918PrivateIpv4('192.169.1.1')).toBe(false); + }); + + it('rejects public IPs', () => { + expect(isRfc1918PrivateIpv4('8.8.8.8')).toBe(false); + expect(isRfc1918PrivateIpv4('1.1.1.1')).toBe(false); + expect(isRfc1918PrivateIpv4('203.0.113.1')).toBe(false); + }); + + it('rejects non-IPv4 input', () => { + expect(isRfc1918PrivateIpv4('localhost')).toBe(false); + expect(isRfc1918PrivateIpv4('[::1]')).toBe(false); + expect(isRfc1918PrivateIpv4('')).toBe(false); + }); +}); diff --git a/gitnexus/test/unit/rate-limit.test.ts b/gitnexus/test/unit/rate-limit.test.ts index 285a8f8c3..2f52e9df8 100644 --- a/gitnexus/test/unit/rate-limit.test.ts +++ b/gitnexus/test/unit/rate-limit.test.ts @@ -247,7 +247,9 @@ describe('production routes — rate-limit middleware wiring', () => { }); it('POST /api/embed is wired with createRouteLimiter', () => { - expect(apiSource).toMatch(/app\.post\('\/api\/embed',\s*createRouteLimiter\(/); + // Tolerate Prettier wrapping the registration across lines (it does once + // the route carries extra middleware like requireLocalhostOrigin). + expect(apiSource).toMatch(/app\.post\(\s*'\/api\/embed',\s*createRouteLimiter\(/); }); it('SPA fallback is wired with createRouteLimiter', () => { From cab63b508e6454d598b3a2c08e8bae61b10b2487 Mon Sep 17 00:00:00 2001 From: bluerose <378100977@qq.com> Date: Sat, 13 Jun 2026 16:58:49 +0800 Subject: [PATCH 08/16] feat(mcp): add gitnexus mcp --http server with Streamable HTTP and legacy SSE transports (#2141) --- gitnexus/src/cli/help-i18n.ts | 4 + gitnexus/src/cli/i18n/en.ts | 8 +- gitnexus/src/cli/i18n/zh-CN.ts | 8 +- gitnexus/src/cli/index.ts | 16 +- gitnexus/src/cli/mcp.ts | 38 +- gitnexus/src/mcp/http-transport.ts | 606 +++++++++++++ gitnexus/src/server/mcp-http.ts | 99 +-- gitnexus/test/unit/mcp-http-transport.test.ts | 825 ++++++++++++++++++ 8 files changed, 1510 insertions(+), 94 deletions(-) create mode 100644 gitnexus/src/mcp/http-transport.ts create mode 100644 gitnexus/test/unit/mcp-http-transport.test.ts diff --git a/gitnexus/src/cli/help-i18n.ts b/gitnexus/src/cli/help-i18n.ts index 0c898eccb..9fb4bde39 100644 --- a/gitnexus/src/cli/help-i18n.ts +++ b/gitnexus/src/cli/help-i18n.ts @@ -70,6 +70,10 @@ const OPTION_DESCRIPTION_KEYS = { 'analyze|--embedding-device ': 'help.option.analyze.embeddingDevice', 'index|-f, --force': 'help.option.index.force', 'index|--allow-non-git': 'help.option.index.allowNonGit', + 'mcp|--http': 'help.option.mcp.http', + 'mcp|-p, --port ': 'help.option.port', + 'mcp|--host ': 'help.option.mcp.host', + 'mcp|--auth-token ': 'help.option.mcp.authToken', 'serve|-p, --port ': 'help.option.port', 'serve|--host ': 'help.option.serve.host', 'uninstall|-f, --force': 'help.option.uninstall.force', diff --git a/gitnexus/src/cli/i18n/en.ts b/gitnexus/src/cli/i18n/en.ts index f6a035adb..9bbf3f428 100644 --- a/gitnexus/src/cli/i18n/en.ts +++ b/gitnexus/src/cli/i18n/en.ts @@ -122,7 +122,8 @@ export const en = { 'help.command.index.description': 'Register an existing .gitnexus/ folder into the global registry (no re-analysis needed)', 'help.command.serve.description': 'Start local HTTP server for web UI connection', - 'help.command.mcp.description': 'Start MCP server (stdio) — serves all indexed repos', + 'help.command.mcp.description': + 'Start MCP server. Default: stdio. Use --http for a remote HTTP server (Streamable HTTP at POST /mcp + legacy SSE at GET /sse, POST /messages).', 'help.command.list.description': 'List all indexed repositories', 'help.command.status.description': 'Show index status for current repo', 'help.command.doctor.description': @@ -199,6 +200,11 @@ export const en = { 'help.option.index.allowNonGit': 'Allow registering folders that are not Git repositories', 'help.option.port': 'Port number', 'help.option.serve.host': 'Bind address (default: 127.0.0.1, use 0.0.0.0 for remote access)', + 'help.option.mcp.http': 'Serve MCP over HTTP instead of stdio (for remote clients)', + 'help.option.mcp.host': + 'HTTP bind address (only with --http). Default: 127.0.0.1 (loopback). Use 0.0.0.0 to expose to all interfaces.', + 'help.option.mcp.authToken': + 'Require this bearer token in the Authorization header (only with --http); may also be set via the GITNEXUS_MCP_AUTH_TOKEN env var. Required for a non-loopback bind (--host 0.0.0.0/::), which otherwise refuses to start.', 'help.option.force.confirmation': 'Skip confirmation prompt', 'help.option.uninstall.force': 'Apply the changes (default is a dry-run preview)', 'help.option.clean.all': 'Clean all indexed repos', diff --git a/gitnexus/src/cli/i18n/zh-CN.ts b/gitnexus/src/cli/i18n/zh-CN.ts index 693d700aa..15dcf2229 100644 --- a/gitnexus/src/cli/i18n/zh-CN.ts +++ b/gitnexus/src/cli/i18n/zh-CN.ts @@ -123,7 +123,8 @@ export const zhCN = { 'help.command.analyze.description': '索引仓库(完整分析)', 'help.command.index.description': '将现有 .gitnexus/ 文件夹注册到全局注册表(无需重新分析)', 'help.command.serve.description': '启动供 Web UI 连接的本地 HTTP 服务器', - 'help.command.mcp.description': '启动 MCP 服务器(stdio)— 提供所有已索引仓库', + 'help.command.mcp.description': + '启动 MCP 服务器。默认为 stdio。使用 --http 启动远程 HTTP 服务器(Streamable HTTP: POST /mcp + 遗留 SSE: GET /sse, POST /messages)。', 'help.command.list.description': '列出所有已索引仓库', 'help.command.status.description': '显示当前仓库的索引状态', 'help.command.doctor.description': '显示运行平台能力和嵌入配置', @@ -187,6 +188,11 @@ export const zhCN = { 'help.option.index.allowNonGit': '允许注册非 Git 仓库文件夹', 'help.option.port': '端口号', 'help.option.serve.host': '绑定地址(默认:127.0.0.1;远程访问可用 0.0.0.0)', + 'help.option.mcp.http': '使用 HTTP 代替 stdio 提供 MCP 服务(适合远程客户端)', + 'help.option.mcp.host': + 'HTTP 绑定地址(仅与 --http 搭配使用)。默认:127.0.0.1(回环)。使用 0.0.0.0 向所有接口开放。', + 'help.option.mcp.authToken': + '要求 Authorization 头携带此 Bearer Token(仅与 --http 搭配使用);也可通过 GITNEXUS_MCP_AUTH_TOKEN 环境变量设置。非回环绑定(--host 0.0.0.0/::)时必填,否则拒绝启动。', 'help.option.force.confirmation': '跳过确认提示', 'help.option.uninstall.force': '应用更改(默认仅为预演预览)', 'help.option.clean.all': '清理所有已索引仓库', diff --git a/gitnexus/src/cli/index.ts b/gitnexus/src/cli/index.ts index eb5a5a8f6..f3f329ed2 100644 --- a/gitnexus/src/cli/index.ts +++ b/gitnexus/src/cli/index.ts @@ -142,7 +142,21 @@ program program .command('mcp') - .description('Start MCP server (stdio) — serves all indexed repos') + .description( + 'Start MCP server. Default: stdio. Use --http for a remote HTTP server ' + + '(Streamable HTTP at POST /mcp + legacy SSE at GET /sse, POST /messages).', + ) + .option('--http', 'Serve MCP over HTTP instead of stdio (for remote clients)') + .option('-p, --port ', 'HTTP port (only with --http). Default: 3000', '3000') + .option( + '--host ', + 'HTTP bind address (only with --http). Default: 127.0.0.1 (loopback). Use 0.0.0.0 to expose to all interfaces.', + '127.0.0.1', + ) + .option( + '--auth-token ', + 'Require this bearer token in the Authorization header (only with --http); may also be set via the GITNEXUS_MCP_AUTH_TOKEN env var. Required for a non-loopback bind (--host 0.0.0.0/::), which otherwise refuses to start.', + ) .action(createLbugLazyAction(() => import('./mcp.js'), 'mcpCommand')); program diff --git a/gitnexus/src/cli/mcp.ts b/gitnexus/src/cli/mcp.ts index 056b12591..8b191db1d 100644 --- a/gitnexus/src/cli/mcp.ts +++ b/gitnexus/src/cli/mcp.ts @@ -29,7 +29,12 @@ import { installGlobalStdoutSentinel } from '../mcp/stdio-context.js'; -export const mcpCommand = async () => { +export const mcpCommand = async (options?: { + http?: boolean; + port?: string; + host?: string; + authToken?: string; +}) => { // Install the global stdout sentinel as the very first thing — before // ANY other module loads. The static-import closure above is leaf-only // (stdio-context → stdio-capture, zero non-`node:` deps), so this is @@ -80,6 +85,37 @@ export const mcpCommand = async () => { ); } + // Start HTTP server or fall back to stdio (default). + if (options?.http) { + // Dynamically import the HTTP transport module AFTER the sentinel installs. + // http-transport.ts pulls in express/cors/MCP SDK HTTP transport; these must + // not load before installGlobalStdoutSentinel() runs (see module doc above). + const port = Number(options.port ?? 3000); + if (!Number.isInteger(port) || port < 1 || port > 65535) { + logger.error( + { port: options.port }, + `Invalid --port value: "${options.port ?? ''}". Must be an integer between 1 and 65535.`, + ); + process.exit(1); + } + // Dynamic import keeps express/cors out of mcp.ts's static graph (stdio sentinel). + const { startMcpHttpServer, resolveAuthToken } = await import('../mcp/http-transport.js'); + try { + await startMcpHttpServer(backend, { + port, + host: options.host ?? '127.0.0.1', + authToken: resolveAuthToken(options.authToken, process.env), + }); + } catch (err) { + logger.error( + { err: err instanceof Error ? err.message : err }, + 'Failed to start the MCP HTTP server', + ); + process.exit(1); + } + return; + } + // Start MCP server (serves all repos, discovers new ones lazily) await startMCPServer(backend); }; diff --git a/gitnexus/src/mcp/http-transport.ts b/gitnexus/src/mcp/http-transport.ts new file mode 100644 index 000000000..265ffa59a --- /dev/null +++ b/gitnexus/src/mcp/http-transport.ts @@ -0,0 +1,606 @@ +/** + * Dedicated MCP HTTP server. + * + * Provides HTTP-based MCP transport supporting: + * - Modern Streamable HTTP: POST /mcp + * - Legacy SSE transport: GET /sse + POST /messages + * + * Started via `gitnexus mcp --http`. + * stdio remains the default mode for `gitnexus mcp` (no breaking change). + * + * Exports createStreamableHttpHandler and createSseHandlers so that + * server/mcp-http.ts (web-UI route mount) can reuse them without inverting + * the established server/ → mcp/ dependency direction. + * + * Security considerations: + * - Default binds to 127.0.0.1 (loopback only). + * - Use --auth-token to enable Bearer Token authentication. + * - Use --host 0.0.0.0 to expose to all interfaces (requires --auth-token — refuses to start otherwise). + * - CORS is restricted to loopback origins when no auth token is configured. + * - PNA (Private Network Access) header is emitted only in response to browser preflight requests. + */ + +import type { Server as HttpServer } from 'http'; +import { timingSafeEqual, randomUUID } from 'crypto'; +import express, { type Express, type Request, type Response, type NextFunction } from 'express'; +import cors from 'cors'; +import { StreamableHTTPServerTransport } from '@modelcontextprotocol/sdk/server/streamableHttp.js'; +import { SSEServerTransport } from '@modelcontextprotocol/sdk/server/sse.js'; +import { Server } from '@modelcontextprotocol/sdk/server/index.js'; +import { isInitializeRequest } from '@modelcontextprotocol/sdk/types.js'; +import { createMCPServer, installSignalShutdown } from './server.js'; +import type { LocalBackend } from './local/local-backend.js'; +import { logger } from '../core/logger.js'; + +/** HTTP server configuration options. */ +export interface McpHttpOptions { + /** Listening port. */ + port: number; + /** Bind address (default: 127.0.0.1). */ + host: string; + /** Bearer auth token (optional; no auth when omitted). */ + authToken?: string; +} + +interface MCPSession { + server: Server; + transport: StreamableHTTPServerTransport; + lastActivity: number; +} + +interface SSESession { + server: Server; + transport: SSEServerTransport; + lastActivity: number; +} + +/** Sessions idle longer than this are evicted. */ +const SESSION_TTL_MS = 30 * 60 * 1000; +/** Cleanup sweep runs every 5 minutes. */ +const CLEANUP_INTERVAL_MS = 5 * 60 * 1000; +/** Hard cap on concurrent sessions — guards against initialize-flood DoS. */ +const MAX_SESSIONS = 1000; + +/** + * Creates a Bearer Token authentication middleware. + * + * - When authToken is not set, all requests pass through. + * - When authToken is set, checks the Authorization: Bearer header. + * - Uses constant-time comparison to prevent timing oracle attacks. + * - Returns a JSON-RPC formatted 401 on failure. + */ +export function createAuthMiddleware(authToken?: string) { + return (req: Request, res: Response, next: NextFunction): void => { + if (!authToken) { + next(); + return; + } + + const header = req.headers['authorization']; + const expected = `Bearer ${authToken}`; + + // Constant-time comparison — prevents timing oracle on bearer token. + // Buffers must be the same byte-length for timingSafeEqual; mismatch means + // we create a same-length dummy so the comparison always runs in full. + let valid = false; + if (typeof header === 'string') { + const a = Buffer.from(header); + const b = Buffer.from(expected); + if (a.length === b.length) { + valid = timingSafeEqual(a, b); + } else { + // Different lengths — run dummy comparison to preserve constant time. + timingSafeEqual(Buffer.alloc(b.length), b); + } + } + + if (valid) { + next(); + return; + } + + res.status(401).json({ + jsonrpc: '2.0', + error: { code: -32001, message: 'Unauthorized' }, + id: null, + }); + }; +} + +/** + * Returns true when an Origin should be allowed by the no-auth (loopback-only) + * CORS policy — i.e. it is absent (non-browser caller) or a loopback origin. + * + * WHATWG URL keeps the brackets on IPv6 literals + * (`new URL('http://[::1]/').hostname === '[::1]'`) and canonicalizes the + * IPv4-mapped loopback to `[::ffff:7f00:1]`; loopback IPv4 is the whole + * 127.0.0.0/8 block — so all of those forms are matched explicitly. + */ +export function isLoopbackOrigin(origin: string | undefined): boolean { + if (!origin) return true; // no Origin → non-browser caller; CORS is not the control there + let hostname: string; + try { + ({ hostname } = new URL(origin)); + } catch { + return false; + } + return ( + hostname === 'localhost' || + hostname === '[::1]' || + hostname === '[::ffff:7f00:1]' || + /^127\.\d{1,3}\.\d{1,3}\.\d{1,3}$/.test(hostname) + ); +} + +/** True for the exact loopback bind addresses. */ +export function isLoopbackHost(host: string): boolean { + return host === '127.0.0.1' || host === 'localhost' || host === '::1'; +} + +/** True for any-interface wildcard binds, whose externally-used Host is unknowable. */ +export function isWildcardHost(host: string): boolean { + return host === '0.0.0.0' || host === '::'; +} + +/** + * Computes the SDK DNS-rebinding `allowedHosts` list (a Host-header allowlist) for a + * bind host/port, or `undefined` when protection should stay off. + * + * Wildcard binds (`0.0.0.0` / `::`) return `undefined` — the Host a client + * legitimately uses is unknowable, so the bearer token (required for non-loopback + * binds) is the control. Loopback binds allow all three loopback host forms + * (bare + `:port`); a specific host (e.g. `192.168.1.50`) allows that host + * (bare + `:port`), which is knowable and a free defence-in-depth win. + */ +export function computeAllowedHosts(host: string, port: number): string[] | undefined { + if (isWildcardHost(host)) return undefined; + const hosts = isLoopbackHost(host) ? ['127.0.0.1', 'localhost', '[::1]'] : [host]; + return hosts.flatMap((h) => [h, `${h}:${port}`]); +} + +/** + * Resolves the MCP HTTP bearer token from the `--auth-token` flag or the + * `GITNEXUS_MCP_AUTH_TOKEN` env var (the flag wins). An empty or whitespace-only + * value is treated as "no token" so a blank env var cannot silently disable auth + * (and slip past the non-loopback hard-fail). + */ +export function resolveAuthToken( + optToken: string | undefined, + env: NodeJS.ProcessEnv, +): string | undefined { + return (optToken ?? env.GITNEXUS_MCP_AUTH_TOKEN)?.trim() || undefined; +} + +/** Builds the SDK transport DNS-rebinding options from a bind host/port. */ +function dnsRebindingOptions( + host: string | undefined, + port: number | undefined, +): { enableDnsRebindingProtection?: boolean; allowedHosts?: string[] } { + if (host === undefined || port === undefined) return {}; + const allowedHosts = computeAllowedHosts(host, port); + return allowedHosts ? { enableDnsRebindingProtection: true, allowedHosts } : {}; +} + +/** + * Starts a periodic sweep that closes and evicts sessions idle longer than + * `ttlMs`, returning the (unref'd) timer. Shared by both transport factories to + * guard against network drops where the per-session onclose never fires. + */ +export function startIdleSweep( + sessions: Map, + ttlMs: number, + intervalMs: number, +): NodeJS.Timeout { + const timer = setInterval(() => { + const now = Date.now(); + for (const [id, session] of sessions) { + if (now - session.lastActivity > ttlMs) { + try { + session.server.close(); + } catch {} + sessions.delete(id); + } + } + }, intervalMs); + if (timer && typeof timer === 'object' && 'unref' in timer) { + (timer as NodeJS.Timeout).unref(); + } + return timer; +} + +/** + * Creates a reusable StreamableHTTP request handler. + * + * Encapsulates the session map and request-dispatch logic as an independent + * factory, reused by both startMcpHttpServer (POST /mcp) and the web-UI server + * route mount in server/mcp-http.ts (/api/mcp). + */ +export function createStreamableHttpHandler( + backend: LocalBackend, + opts: { createServer?: () => Server; host?: string; port?: number } = {}, +): { + handler: (req: Request, res: Response) => Promise; + cleanup: () => Promise; +} { + // Seam: tests inject createServer to observe the per-session Server lifecycle. + const createServer = opts.createServer ?? ((): Server => createMCPServer(backend)); + // DNS-rebinding protection (Host-header allowlist) when the bind host is known. + const dnsRebinding = dnsRebindingOptions(opts.host, opts.port); + const sessions = new Map(); + const cleanupTimer = startIdleSweep(sessions, SESSION_TTL_MS, CLEANUP_INTERVAL_MS); + + const handler = async (req: Request, res: Response): Promise => { + const sessionId = req.headers['mcp-session-id'] as string | undefined; + + if (sessionId && sessions.has(sessionId)) { + // Existing session — delegate to its transport and refresh activity timestamp. + // `has` just returned true and the map is not mutated before `get`, so the + // lookup is non-null. + const session = sessions.get(sessionId)!; + session.lastActivity = Date.now(); + await session.transport.handleRequest(req, res, req.body); + } else if (sessionId) { + // Unknown / expired session ID — tell the client to re-initialize (per MCP spec). + res.status(404).json({ + jsonrpc: '2.0', + error: { code: -32001, message: 'Session not found. Re-initialize.' }, + id: null, + }); + } else if (req.method === 'POST') { + // No session ID — new client. Only accept initialize requests to avoid + // orphaned Server instances that can never be reclaimed by the TTL sweep. + // Use the SDK's isInitializeRequest so a single-element JSON-RPC batch is + // recognised too, rather than a brittle `body.method === 'initialize'` check. + const body = req.body as unknown; + const messages = Array.isArray(body) ? body : [body]; + if (!messages.some(isInitializeRequest)) { + res.status(400).json({ + jsonrpc: '2.0', + error: { + code: -32000, + message: 'First request must be initialize. No session ID provided.', + }, + id: null, + }); + return; + } + + // Reject when the session cap is reached — prevents memory exhaustion via + // an initialize flood (each session holds a live Server + Transport). + if (sessions.size >= MAX_SESSIONS) { + res.status(503).json({ + jsonrpc: '2.0', + error: { code: -32000, message: 'Server at session capacity. Try again later.' }, + id: null, + }); + return; + } + + const transport = new StreamableHTTPServerTransport({ + sessionIdGenerator: () => randomUUID(), + ...dnsRebinding, + }); + const server = createServer(); + await server.connect(transport); + await transport.handleRequest(req, res, req.body); + + if (transport.sessionId) { + sessions.set(transport.sessionId, { server, transport, lastActivity: Date.now() }); + const sid = transport.sessionId; + transport.onclose = () => { + sessions.delete(sid); + }; + } else { + // The SDK rejected this request (e.g. 406 on a missing/invalid Accept header, + // 415 on a bad Content-Type) before assigning a session id. The Server was + // already connected but will never be stored, so the TTL sweep and cleanup() + // can't reclaim it — close it now to avoid an orphaned-Server leak. + try { + await server.close(); + } catch {} + } + } else { + res.status(400).json({ + jsonrpc: '2.0', + error: { code: -32000, message: 'No valid session. Send a POST to initialize.' }, + id: null, + }); + } + }; + + const cleanup = async (): Promise => { + clearInterval(cleanupTimer); + const closers = [...sessions.values()].map(async (session) => { + try { + await Promise.resolve(session.server.close()); + } catch {} + }); + sessions.clear(); + await Promise.allSettled(closers); + }; + + return { handler, cleanup }; +} + +/** + * Creates legacy SSE transport handlers. + * + * GET /sse (or custom path) establishes the SSE stream; + * POST /messages (or custom path) receives client JSON-RPC messages. + * + * Includes the same idle-TTL eviction as createStreamableHttpHandler to prevent + * memory leaks when clients drop without closing the SSE connection cleanly. + * + * @param backend LocalBackend instance + * @param messagesPath Path clients POST messages to (default: '/messages') + */ +export function createSseHandlers( + backend: LocalBackend, + messagesPath = '/messages', + opts: { maxSessions?: number; host?: string; port?: number } = {}, +): { + sseHandler: (req: Request, res: Response) => Promise; + messageHandler: (req: Request, res: Response) => Promise; + cleanup: () => Promise; +} { + const maxSessions = opts.maxSessions ?? MAX_SESSIONS; + // DNS-rebinding protection (Host-header allowlist) when the bind host is known. + const dnsRebinding = dnsRebindingOptions(opts.host, opts.port); + const sseSessions = new Map(); + const cleanupTimer = startIdleSweep(sseSessions, SESSION_TTL_MS, CLEANUP_INTERVAL_MS); + + const sseHandler = async (req: Request, res: Response): Promise => { + // Cap concurrent SSE sessions — mirrors the streamable handler's MAX_SESSIONS + // guard so a flood of held-open GET /sse connections cannot allocate unbounded + // Server instances before the idle sweep reclaims them. + if (sseSessions.size >= maxSessions) { + res.status(503).json({ + jsonrpc: '2.0', + error: { code: -32000, message: 'Server at session capacity. Try again later.' }, + id: null, + }); + return; + } + + // SSEServerTransport(endpoint, res, options): endpoint is the path clients POST to. + const transport = new SSEServerTransport(messagesPath, res, dnsRebinding); + const server = createMCPServer(backend); + + sseSessions.set(transport.sessionId, { server, transport, lastActivity: Date.now() }); + + transport.onclose = () => { + sseSessions.delete(transport.sessionId); + }; + + res.on('close', () => { + sseSessions.delete(transport.sessionId); + try { + server.close(); + } catch {} + }); + + // connect() calls transport.start(), which sends the SSE 'endpoint' event. + await server.connect(transport); + }; + + const messageHandler = async (req: Request, res: Response): Promise => { + const sessionId = + (req.query['sessionId'] as string | undefined) ?? + (req.headers['mcp-session-id'] as string | undefined); + const entry = sessionId ? sseSessions.get(sessionId) : undefined; + + if (!entry) { + res.status(404).json({ + jsonrpc: '2.0', + error: { code: -32001, message: 'SSE session not found. Reconnect to /sse.' }, + id: null, + }); + return; + } + + // Refresh activity timestamp so the TTL sweep does not evict an active session. + entry.lastActivity = Date.now(); + + // express.json() has already parsed the body — pass it as the third argument + // to avoid the SDK re-reading the already-consumed stream. + await entry.transport.handlePostMessage(req, res, req.body); + }; + + const cleanup = async (): Promise => { + clearInterval(cleanupTimer); + const closers = [...sseSessions.values()].map(async ({ server }) => { + try { + await Promise.resolve(server.close()); + } catch {} + }); + sseSessions.clear(); + await Promise.allSettled(closers); + }; + + return { sseHandler, messageHandler, cleanup }; +} + +/** + * Creates and starts the dedicated MCP HTTP server. + * + * Mounts the following routes: + * - GET /health — health check (no auth required; for orchestrators/probes) + * - POST /mcp — Streamable HTTP (modern clients) + * - GET /sse — legacy SSE stream (old clients) + * - POST /messages — legacy SSE message endpoint + * + * @param backend LocalBackend instance + * @param options Server configuration + * @returns The listening http.Server + */ +export async function startMcpHttpServer( + backend: LocalBackend, + options: McpHttpOptions, +): Promise { + const { port, host, authToken } = options; + + // Refuse to start an unauthenticated server on a non-loopback interface — that + // would silently expose every indexed repo to anyone who can reach the host. + // Loopback binds stay open by default; non-loopback binds require a token. + if (!authToken && !isLoopbackHost(host)) { + throw new Error( + `Refusing to start the MCP HTTP server on a non-loopback host (${host}) without ` + + 'authentication — it would expose all indexed repos to anyone who can reach it. ' + + 'Pass --auth-token (or set GITNEXUS_MCP_AUTH_TOKEN), or bind --host 127.0.0.1. ' + + 'This applies to --host 0.0.0.0 and --host :: as well.', + ); + } + + const app: Express = express(); + + // Suppress X-Powered-By to reduce information leakage. + app.disable('x-powered-by'); + + // PNA (Chrome 130+ Private Network Access) preflight support. + // The browser sends `Access-Control-Request-Private-Network: true` ONLY on the + // CORS preflight (an OPTIONS request); emit the matching allow header only then, + // never on actual GET/POST responses. Runs before cors() so the header survives + // onto the preflight response cors() short-circuits. + app.use((req: Request, res: Response, next: NextFunction) => { + if ( + req.method === 'OPTIONS' && + req.headers['access-control-request-private-network'] === 'true' + ) { + res.setHeader('Access-Control-Allow-Private-Network', 'true'); + } + next(); + }); + + // CORS policy: + // - With auth token: allow any origin (remote access is intentional and protected). + // - Without auth token: restrict to loopback origins only to prevent drive-by local exfiltration. + const corsOrigin = authToken + ? true + : (origin: string | undefined, cb: (err: Error | null, allow?: boolean) => void) => { + cb(null, isLoopbackOrigin(origin)); + }; + + app.use( + cors({ + origin: corsOrigin, + credentials: false, + allowedHeaders: ['Content-Type', 'Authorization', 'mcp-session-id', 'last-event-id'], + exposedHeaders: ['mcp-session-id'], + }), + ); + + const auth = createAuthMiddleware(authToken); + // Body parser applied per-route after auth, so unauthenticated requests never + // trigger the 10 MB parse. Malformed/oversized JSON from authenticated clients + // is converted to a JSON-RPC error envelope by the terminal error handler + // registered after the routes (see below). + const jsonBody = express.json({ limit: '10mb' }); + + // Health check — no auth required; safe to expose for probes and orchestrators. + app.get('/health', (_req: Request, res: Response) => { + res.json({ status: 'ok' }); + }); + + // Streamable HTTP (modern MCP clients) at POST /mcp. + const streamable = createStreamableHttpHandler(backend, { host, port }); + app.all('/mcp', auth, jsonBody, (req: Request, res: Response) => { + void streamable.handler(req, res).catch((err: unknown) => { + logger.error({ err }, 'MCP /mcp request failed'); + if (!res.headersSent) { + res.status(500).json({ + jsonrpc: '2.0', + error: { code: -32000, message: 'Internal MCP server error' }, + id: null, + }); + } + }); + }); + + // Legacy SSE: GET /sse opens the stream; POST /messages receives JSON-RPC messages. + const sse = createSseHandlers(backend, '/messages', { host, port }); + app.get('/sse', auth, (req: Request, res: Response) => { + void sse.sseHandler(req, res).catch((err: unknown) => { + logger.error({ err }, 'MCP /sse failed'); + }); + }); + app.post('/messages', auth, jsonBody, (req: Request, res: Response) => { + void sse.messageHandler(req, res).catch((err: unknown) => { + logger.error({ err }, 'MCP /messages failed'); + if (!res.headersSent) { + res.status(500).json({ + jsonrpc: '2.0', + error: { code: -32000, message: 'Internal error' }, + id: null, + }); + } + }); + }); + + // Terminal error handler: body-parser failures (malformed or oversized JSON) + // reach here via next(err). Without it, Express's default handler returns an + // HTML error page — leaking a stack trace and absolute install paths when + // NODE_ENV is unset (the default for a CLI) — instead of the JSON-RPC envelope + // every other path uses. + app.use((err: unknown, _req: Request, res: Response, _next: NextFunction) => { + const e = (err ?? {}) as { type?: string; status?: number; statusCode?: number }; + const isBodyParseError = + e.type === 'entity.parse.failed' || + e.type === 'entity.too.large' || + err instanceof SyntaxError; + logger.error({ err }, 'MCP HTTP request error'); + if (res.headersSent) return; + if (isBodyParseError) { + res.status(e.status ?? e.statusCode ?? 400).json({ + jsonrpc: '2.0', + error: { code: -32700, message: 'Parse error' }, + id: null, + }); + return; + } + res.status(500).json({ + jsonrpc: '2.0', + error: { code: -32000, message: 'Internal MCP server error' }, + id: null, + }); + }); + + return new Promise((resolve, reject) => { + const server = app.listen(port, host, () => { + const displayHost = host === '0.0.0.0' || host === '::' ? 'localhost' : host; + logger.info( + { port, host }, + `GitNexus MCP HTTP server listening on http://${displayHost}:${port} ` + + `(Streamable: POST /mcp · legacy SSE: GET /sse + POST /messages)`, + ); + resolve(server); + }); + + server.on('error', (err: NodeJS.ErrnoException) => { + if (err.code === 'EADDRINUSE') { + logger.error( + { port, host }, + `Port ${port} is already in use. ` + + `Stop the conflicting process or use a different port: gitnexus mcp --http --port `, + ); + process.exit(1); + } + reject(err); + }); + + const shutdown = async (exitCode: number): Promise => { + server.close(); + await streamable.cleanup(); + await sse.cleanup(); + try { + await backend.disconnect(); + } catch {} + const { flushLoggerSync } = await import('../core/logger.js'); + flushLoggerSync(); + process.exit(exitCode); + }; + + // Use the shared signal wiring so SIGINT exits 130 and SIGTERM exits 143 + // (the repo's POSIX 128+signal convention), not a misleading exit(0). + installSignalShutdown((exitCode = 0) => void shutdown(exitCode)); + }); +} diff --git a/gitnexus/src/server/mcp-http.ts b/gitnexus/src/server/mcp-http.ts index 67ec94708..ccfd71adf 100644 --- a/gitnexus/src/server/mcp-http.ts +++ b/gitnexus/src/server/mcp-http.ts @@ -1,93 +1,23 @@ /** - * MCP over HTTP + * MCP over HTTP — route mount helper for the web-UI server. * - * Mounts the GitNexus MCP server on Express using StreamableHTTP transport. - * Each connecting client gets its own stateful session; the LocalBackend - * is shared across all sessions (thread-safe — lazy LadybugDB per repo). + * Mounts the GitNexus MCP endpoint (/api/mcp) onto an existing Express + * application. Session management lives in mcp/http-transport.ts, preserving + * the established server/ → mcp/ dependency direction. * - * Sessions are cleaned up on explicit close or after SESSION_TTL_MS of inactivity - * (guards against network drops that never trigger onclose). + * Used by server/api.ts to wire up the full web server. */ import type { Express, Request, Response } from 'express'; -import { StreamableHTTPServerTransport } from '@modelcontextprotocol/sdk/server/streamableHttp.js'; -import { Server } from '@modelcontextprotocol/sdk/server/index.js'; -import { createMCPServer } from '../mcp/server.js'; +import { createStreamableHttpHandler } from '../mcp/http-transport.js'; import type { LocalBackend } from '../mcp/local/local-backend.js'; -import { randomUUID } from 'crypto'; import { logger } from '../core/logger.js'; -interface MCPSession { - server: Server; - transport: StreamableHTTPServerTransport; - lastActivity: number; -} - -/** Idle sessions are evicted after 30 minutes */ -const SESSION_TTL_MS = 30 * 60 * 1000; -/** Cleanup sweep runs every 5 minutes */ -const CLEANUP_INTERVAL_MS = 5 * 60 * 1000; - export function mountMCPEndpoints(app: Express, backend: LocalBackend): () => Promise { - const sessions = new Map(); - - // Periodic cleanup of idle sessions (guards against network drops) - const cleanupTimer = setInterval(() => { - const now = Date.now(); - for (const [id, session] of sessions) { - if (now - session.lastActivity > SESSION_TTL_MS) { - try { - session.server.close(); - } catch {} - sessions.delete(id); - } - } - }, CLEANUP_INTERVAL_MS); - if (cleanupTimer && typeof cleanupTimer === 'object' && 'unref' in cleanupTimer) { - (cleanupTimer as NodeJS.Timeout).unref(); - } - - const handleMcpRequest = async (req: Request, res: Response) => { - const sessionId = req.headers['mcp-session-id'] as string | undefined; - - if (sessionId && sessions.has(sessionId)) { - // Existing session — delegate to its transport - const session = sessions.get(sessionId)!; - session.lastActivity = Date.now(); - await session.transport.handleRequest(req, res, req.body); - } else if (sessionId) { - // Unknown/expired session ID — tell client to re-initialize (per MCP spec) - res.status(404).json({ - jsonrpc: '2.0', - error: { code: -32001, message: 'Session not found. Re-initialize.' }, - id: null, - }); - } else if (req.method === 'POST') { - // No session ID — new client initializing - const transport = new StreamableHTTPServerTransport({ - sessionIdGenerator: () => randomUUID(), - }); - const server = createMCPServer(backend); - await server.connect(transport); - await transport.handleRequest(req, res, req.body); - - if (transport.sessionId) { - sessions.set(transport.sessionId, { server, transport, lastActivity: Date.now() }); - transport.onclose = () => { - sessions.delete(transport.sessionId!); - }; - } - } else { - res.status(400).json({ - jsonrpc: '2.0', - error: { code: -32000, message: 'No valid session. Send a POST to initialize.' }, - id: null, - }); - } - }; + const { handler, cleanup } = createStreamableHttpHandler(backend); app.all('/api/mcp', (req: Request, res: Response) => { - void handleMcpRequest(req, res).catch((err: any) => { + void handler(req, res).catch((err: unknown) => { logger.error({ err }, 'MCP HTTP request failed:'); if (res.headersSent) return; res.status(500).json({ @@ -98,17 +28,6 @@ export function mountMCPEndpoints(app: Express, backend: LocalBackend): () => Pr }); }); - const cleanup = async () => { - clearInterval(cleanupTimer); - const closers = [...sessions.values()].map(async (session) => { - try { - await Promise.resolve(session.server.close()); - } catch {} - }); - sessions.clear(); - await Promise.allSettled(closers); - }; - - console.log('MCP HTTP endpoints mounted at /api/mcp'); + logger.info('MCP HTTP endpoints mounted at /api/mcp'); return cleanup; } diff --git a/gitnexus/test/unit/mcp-http-transport.test.ts b/gitnexus/test/unit/mcp-http-transport.test.ts new file mode 100644 index 000000000..105caff2a --- /dev/null +++ b/gitnexus/test/unit/mcp-http-transport.test.ts @@ -0,0 +1,825 @@ +/** + * Unit Tests: MCP HTTP Transport + * + * Coverage: + * - createAuthMiddleware: no-auth / valid token / invalid token scenarios + * - startMcpHttpServer: port-0 smoke test (health endpoint, unauthenticated POST → 401) + * - createStreamableHttpHandler: new-session initialization, unknown session 404 + * - createSseHandlers: message routing, unknown sessionId 404 + * - mountMCPEndpoints refactor safety: still returns cleanup fn and registers /api/mcp + * + * Notes: + * - node_modules may not be installed; tests that exercise the MCP SDK rely on mocks. + * - HTTP server tests use port 0 (OS-assigned ephemeral port) bound to 127.0.0.1. + * - Each test closes the server and calls cleanup() to avoid handle leaks. + */ + +import http from 'http'; +import type { AddressInfo } from 'net'; +import { describe, it, expect, vi, afterEach } from 'vitest'; +import express from 'express'; +import type { Request, Response, NextFunction } from 'express'; +import { + createAuthMiddleware, + createStreamableHttpHandler, + createSseHandlers, + isLoopbackOrigin, + computeAllowedHosts, + resolveAuthToken, + startMcpHttpServer, + startIdleSweep, +} from '../../src/mcp/http-transport.js'; +import { + createMCPServer, + installSignalShutdown, + SHUTDOWN_EXIT_CODES, +} from '../../src/mcp/server.js'; +import { mountMCPEndpoints } from '../../src/server/mcp-http.js'; + +// ─── Live-HTTP helpers (real req/res for SDK-touching paths) ─────────── + +async function listen(app: express.Express): Promise<{ port: number; close: () => Promise }> { + const server = app.listen(0, '127.0.0.1'); + await new Promise((resolve) => server.once('listening', () => resolve())); + const port = (server.address() as AddressInfo).port; + return { + port, + close: () => new Promise((resolve) => server.close(() => resolve())), + }; +} + +interface HttpResult { + status: number; + headers: http.IncomingHttpHeaders; + body: string; +} + +function request( + port: number, + method: string, + path: string, + headers: Record = {}, + body?: string, +): Promise { + return new Promise((resolve, reject) => { + const req = http.request({ hostname: '127.0.0.1', port, path, method, headers }, (res) => { + let data = ''; + res.on('data', (chunk: Buffer) => (data += chunk.toString())); + res.on('end', () => + resolve({ status: res.statusCode ?? 0, headers: res.headers, body: data }), + ); + }); + req.on('error', reject); + if (body !== undefined) req.write(body); + req.end(); + }); +} + +async function waitFor(predicate: () => boolean, timeoutMs = 500): Promise { + const start = Date.now(); + while (!predicate() && Date.now() - start < timeoutMs) { + await new Promise((r) => setTimeout(r, 10)); + } +} + +/** A schema-complete JSON-RPC initialize request (passes the SDK isInitializeRequest). */ +function validInitialize(id = 1): Record { + return { + jsonrpc: '2.0', + method: 'initialize', + id, + params: { + protocolVersion: '2025-03-26', + capabilities: {}, + clientInfo: { name: 'test', version: '1.0.0' }, + }, + }; +} + +// ─── Mock backend factory ────────────────────────────────────────────── + +function createMockBackend(overrides: Record = {}): unknown { + return { + callTool: vi.fn().mockResolvedValue({ result: 'ok' }), + listRepos: vi.fn().mockResolvedValue([]), + resolveRepo: vi + .fn() + .mockResolvedValue({ name: 'test', repoPath: '/tmp/test', lastCommit: 'abc' }), + getContext: vi.fn().mockReturnValue(null), + queryClusters: vi.fn().mockResolvedValue({ clusters: [] }), + queryProcesses: vi.fn().mockResolvedValue({ processes: [] }), + queryClusterDetail: vi.fn().mockResolvedValue({ error: 'not found' }), + queryProcessDetail: vi.fn().mockResolvedValue({ error: 'not found' }), + disconnect: vi.fn().mockResolvedValue(undefined), + ...overrides, + }; +} + +// ─── Mock req/res factory ────────────────────────────────────────────── + +function createMockReq(headers: Record = {}): Request { + return { headers } as unknown as Request; +} + +function createMockRes(): Response & { _status: number; _body: unknown } { + const res = { + _status: 200, + _body: undefined, + headersSent: false, + status: vi.fn().mockImplementation(function (this: typeof res, code: number) { + this._status = code; + return this; + }), + json: vi.fn().mockImplementation(function (this: typeof res, body: unknown) { + this._body = body; + return this; + }), + }; + return res as unknown as Response & { _status: number; _body: unknown }; +} + +// ─── createAuthMiddleware ────────────────────────────────────────────── + +describe('createAuthMiddleware', () => { + it('calls next immediately when authToken is not set', () => { + const middleware = createAuthMiddleware(undefined); + const req = createMockReq(); + const res = createMockRes(); + const next = vi.fn() as NextFunction; + + middleware(req, res, next); + + expect(next).toHaveBeenCalledOnce(); + expect(res.status).not.toHaveBeenCalled(); + }); + + it('calls next when the correct Bearer token is supplied', () => { + const middleware = createAuthMiddleware('my-secret-token'); + const req = createMockReq({ authorization: 'Bearer my-secret-token' }); + const res = createMockRes(); + const next = vi.fn() as NextFunction; + + middleware(req, res, next); + + expect(next).toHaveBeenCalledOnce(); + expect(res.status).not.toHaveBeenCalled(); + }); + + it('returns 401 when Authorization header is missing', () => { + const middleware = createAuthMiddleware('my-secret-token'); + const req = createMockReq(); // no headers + const res = createMockRes(); + const next = vi.fn() as NextFunction; + + middleware(req, res, next); + + expect(next).not.toHaveBeenCalled(); + expect(res._status).toBe(401); + expect(res._body).toMatchObject({ + jsonrpc: '2.0', + error: { code: -32001, message: 'Unauthorized' }, + }); + }); + + it('returns 401 when the wrong token is supplied', () => { + const middleware = createAuthMiddleware('my-secret-token'); + const req = createMockReq({ authorization: 'Bearer wrong-token' }); + const res = createMockRes(); + const next = vi.fn() as NextFunction; + + middleware(req, res, next); + + expect(next).not.toHaveBeenCalled(); + expect(res._status).toBe(401); + }); + + it('returns 401 when Authorization header is missing the "Bearer " prefix', () => { + const middleware = createAuthMiddleware('my-secret-token'); + const req = createMockReq({ authorization: 'my-secret-token' }); + const res = createMockRes(); + const next = vi.fn() as NextFunction; + + middleware(req, res, next); + + expect(next).not.toHaveBeenCalled(); + expect(res._status).toBe(401); + }); +}); + +// ─── startMcpHttpServer smoke tests ─────────────────────────────────── + +describe('startMcpHttpServer', () => { + const servers: Array<{ server: http.Server; cleanup: () => Promise }> = []; + + afterEach(async () => { + for (const { server, cleanup } of servers.splice(0)) { + await cleanup().catch(() => {}); + await new Promise((resolve) => server.close(() => resolve())); + } + }); + + /** + * Starts the MCP HTTP server on an OS-assigned port (port 0), returns the + * bound port, a node http.Server handle, and the cleanup function. + */ + async function startOnFreePort(authToken?: string): Promise<{ + port: number; + server: http.Server; + cleanup: () => Promise; + }> { + const backend = createMockBackend(); + + // Wrap startMcpHttpServer to capture the returned http.Server. + const { startMcpHttpServer: start } = await import('../../src/mcp/http-transport.js'); + const resolvedServer = await start(backend as never, { + port: 0, + host: '127.0.0.1', + authToken, + }); + + const address = resolvedServer.address(); + const port = + address && typeof address === 'object' + ? address.port + : (() => { + throw new Error('no port'); + })(); + + const cleanup = async (): Promise => { + // afterEach closes the server handle. + }; + + return { port, server: resolvedServer, cleanup }; + } + + it('GET /health returns 200 { status: "ok" }', async () => { + const { port, server, cleanup } = await startOnFreePort(); + servers.push({ server, cleanup }); + + const body = await new Promise((resolve, reject) => { + http + .get(`http://127.0.0.1:${port}/health`, (res) => { + let data = ''; + res.on('data', (chunk: string) => (data += chunk)); + res.on('end', () => resolve(data)); + }) + .on('error', reject); + }); + + expect(JSON.parse(body)).toEqual({ status: 'ok' }); + }); + + it('POST /mcp without auth token returns 401 when --auth-token is configured', async () => { + const { port, server, cleanup } = await startOnFreePort('supersecret'); + servers.push({ server, cleanup }); + + const statusCode = await new Promise((resolve, reject) => { + const req = http.request( + { + hostname: '127.0.0.1', + port, + path: '/mcp', + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + }, + (res) => { + res.resume(); // drain + resolve(res.statusCode ?? 0); + }, + ); + req.on('error', reject); + req.write(JSON.stringify({ jsonrpc: '2.0', method: 'initialize', id: 1, params: {} })); + req.end(); + }); + + expect(statusCode).toBe(401); + }); + + it('U5: emits the PNA allow header only on an OPTIONS preflight carrying the request header', async () => { + const { port, server, cleanup } = await startOnFreePort(); // no auth → loopback CORS + servers.push({ server, cleanup }); + + const preflight = await request(port, 'OPTIONS', '/mcp', { + Origin: 'http://127.0.0.1:9999', + 'Access-Control-Request-Method': 'POST', + 'Access-Control-Request-Private-Network': 'true', + }); + expect(preflight.headers['access-control-allow-private-network']).toBe('true'); + + // A normal GET carrying the request header must NOT receive the allow header. + const get = await request(port, 'GET', '/health', { + 'Access-Control-Request-Private-Network': 'true', + }); + expect(get.headers['access-control-allow-private-network']).toBeUndefined(); + }); + + it('U8: refuses to start on a non-loopback host without a token', async () => { + const backend = createMockBackend(); + await expect( + startMcpHttpServer(backend as never, { host: '0.0.0.0', port: 0 }), + ).rejects.toThrow(/non-loopback/i); + await expect(startMcpHttpServer(backend as never, { host: '::', port: 0 })).rejects.toThrow( + /non-loopback/i, + ); + await expect( + startMcpHttpServer(backend as never, { host: '192.168.1.50', port: 0 }), + ).rejects.toThrow(); + }); + + it('U8: starts on a non-loopback host when a token is provided', async () => { + const backend = createMockBackend(); + const server = await startMcpHttpServer(backend as never, { + host: '0.0.0.0', + port: 0, + authToken: 'tok', + }); + servers.push({ server, cleanup: async () => {} }); + expect(server.listening).toBe(true); + }); + + it('U6: rejects a POST /mcp carrying a disallowed Host header (DNS-rebinding protection)', async () => { + const { port, server, cleanup } = await startOnFreePort(); // 127.0.0.1 → protection ON + servers.push({ server, cleanup }); + + const res = await request( + port, + 'POST', + '/mcp', + { + 'Content-Type': 'application/json', + Accept: 'application/json, text/event-stream', + Host: 'evil.example.com:1234', + }, + JSON.stringify(validInitialize()), + ); + + expect(res.status).toBe(403); + }); + + it('U3: malformed JSON from an authenticated client returns a JSON-RPC parse error (not HTML)', async () => { + const { port, server, cleanup } = await startOnFreePort('supersecret'); + servers.push({ server, cleanup }); + + const res = await request( + port, + 'POST', + '/mcp', + { 'Content-Type': 'application/json', Authorization: 'Bearer supersecret' }, + '{ this is not valid json ', + ); + + expect(res.status).toBe(400); + expect(String(res.headers['content-type'] ?? '')).toMatch(/application\/json/); + expect(JSON.parse(res.body)).toMatchObject({ + jsonrpc: '2.0', + error: { code: -32700, message: 'Parse error' }, + id: null, + }); + }); +}); + +// ─── createStreamableHttpHandler ────────────────────────────────────── + +describe('createStreamableHttpHandler', () => { + it('attempts to create a new session for a POST with no session id', async () => { + const backend = createMockBackend(); + const { handler, cleanup } = createStreamableHttpHandler(backend as never); + + const req = { + headers: {}, + method: 'POST', + body: validInitialize(), + } as Request; + + const res = { + headersSent: false, + statusCode: 200, + status: vi.fn().mockReturnThis(), + json: vi.fn().mockReturnThis(), + setHeader: vi.fn(), + write: vi.fn(), + end: vi.fn(), + } as unknown as Response; + + // The handler calls StreamableHTTPServerTransport internally; without the real + // SDK installed the call may throw — that is acceptable in unit tests. + try { + await handler(req, res); + } catch { + // Expected when SDK is not installed. + } + + await cleanup(); + }); + + it('returns 400 when POST has no session id and body method is not initialize', async () => { + const backend = createMockBackend(); + const { handler, cleanup } = createStreamableHttpHandler(backend as never); + + const req = { + headers: {}, + method: 'POST', + body: { jsonrpc: '2.0', method: 'tools/list', id: 2, params: {} }, + } as Request; + + const res = createMockRes(); + + await handler(req, res); + + expect(res._status).toBe(400); + expect(res._body).toMatchObject({ jsonrpc: '2.0', error: { code: -32000 } }); + + await cleanup(); + }); + + it('returns 404 for an unknown session id', async () => { + const backend = createMockBackend(); + const { handler, cleanup } = createStreamableHttpHandler(backend as never); + + const req = { + headers: { 'mcp-session-id': 'non-existent-session-id' }, + method: 'GET', + body: undefined, + } as unknown as Request; + + const res = createMockRes(); + + await handler(req, res); + + expect(res._status).toBe(404); + expect(res._body).toMatchObject({ + jsonrpc: '2.0', + error: { code: -32001, message: 'Session not found. Re-initialize.' }, + }); + + await cleanup(); + }); + + it('returns 400 for a GET with no session id', async () => { + const backend = createMockBackend(); + const { handler, cleanup } = createStreamableHttpHandler(backend as never); + + const req = { + headers: {}, + method: 'GET', + body: undefined, + } as unknown as Request; + + const res = createMockRes(); + + await handler(req, res); + + expect(res._status).toBe(400); + expect(res._body).toMatchObject({ + jsonrpc: '2.0', + error: { code: -32000, message: 'No valid session. Send a POST to initialize.' }, + }); + + await cleanup(); + }); + + it('U1: closes the orphaned Server when the SDK rejects an initialize before a session id', async () => { + const backend = createMockBackend(); + let closed = 0; + // Inject createServer so we can observe the per-session Server's close(). + const { handler, cleanup } = createStreamableHttpHandler(backend as never, { + createServer: () => { + const s = createMCPServer(backend as never); + const orig = s.close.bind(s); + s.close = (async () => { + closed += 1; + return orig(); + }) as typeof s.close; + return s; + }, + }); + + const app = express(); + app.use(express.json()); + app.all('/mcp', (req, res) => void handler(req, res).catch(() => {})); + const { port, close } = await listen(app); + + // POST initialize but with Accept: application/json ONLY (no text/event-stream): + // the SDK returns 406 BEFORE assigning transport.sessionId, exercising the orphan path. + const res = await request( + port, + 'POST', + '/mcp', + { 'Content-Type': 'application/json', Accept: 'application/json' }, + JSON.stringify(validInitialize()), + ); + + expect(res.status).toBe(406); + await waitFor(() => closed > 0); + expect(closed).toBeGreaterThan(0); // the connected Server was closed, not leaked + + await close(); + await cleanup(); + }); + + it('U10: treats a single-element JSON-RPC batch initialize as initialize (no 400)', async () => { + const backend = createMockBackend(); + const { handler, cleanup } = createStreamableHttpHandler(backend as never); + const req = { + headers: {}, + method: 'POST', + body: [validInitialize()], + } as unknown as Request; + const res = createMockRes(); + // Past the init gate, the SDK transport runs against the mock res and may throw; + // we only assert the gate did NOT short-circuit with a 400. + try { + await handler(req, res); + } catch { + /* SDK write on the mock res */ + } + expect(res._status).not.toBe(400); + await cleanup(); + }); + + it('U10: a non-initialize JSON-RPC batch still returns 400', async () => { + const backend = createMockBackend(); + const { handler, cleanup } = createStreamableHttpHandler(backend as never); + const req = { + headers: {}, + method: 'POST', + body: [{ jsonrpc: '2.0', method: 'tools/list', id: 2, params: {} }], + } as unknown as Request; + const res = createMockRes(); + await handler(req, res); + expect(res._status).toBe(400); + await cleanup(); + }); +}); + +// ─── createSseHandlers ──────────────────────────────────────────────── + +describe('createSseHandlers', () => { + it('returns 404 from messageHandler when sessionId is unknown', async () => { + const backend = createMockBackend(); + const { messageHandler, cleanup } = createSseHandlers(backend as never, '/messages'); + + const req = { + query: { sessionId: 'non-existent' }, + headers: {}, + body: {}, + } as unknown as Request; + + const res = createMockRes(); + + await messageHandler(req, res); + + expect(res._status).toBe(404); + expect(res._body).toMatchObject({ + jsonrpc: '2.0', + error: { code: -32001, message: 'SSE session not found. Reconnect to /sse.' }, + }); + + await cleanup(); + }); + + it('returns 404 from messageHandler when no sessionId is provided', async () => { + const backend = createMockBackend(); + const { messageHandler, cleanup } = createSseHandlers(backend as never, '/messages'); + + const req = { + query: {}, + headers: {}, + body: {}, + } as unknown as Request; + + const res = createMockRes(); + + await messageHandler(req, res); + + expect(res._status).toBe(404); + + await cleanup(); + }); + + it('cleanup does not throw', async () => { + const backend = createMockBackend(); + const { cleanup } = createSseHandlers(backend as never, '/messages'); + + await expect(cleanup()).resolves.not.toThrow(); + }); + + it('U2: returns 503 (and allocates no Server) when the SSE session cap is reached', async () => { + const backend = createMockBackend(); + // maxSessions 0 → the cap is hit immediately, so the guard fires before any + // SSEServerTransport / Server is allocated. + const { sseHandler, messageHandler, cleanup } = createSseHandlers( + backend as never, + '/messages', + { maxSessions: 0 }, + ); + + const res = createMockRes(); + await sseHandler(createMockReq(), res); + + expect(res._status).toBe(503); + expect(res._body).toMatchObject({ + jsonrpc: '2.0', + error: { code: -32000, message: 'Server at session capacity. Try again later.' }, + }); + + // No session was created — any message routes to the 404 path. + const msgRes = createMockRes(); + await messageHandler( + { query: { sessionId: 'anything' }, headers: {}, body: {} } as unknown as Request, + msgRes, + ); + expect(msgRes._status).toBe(404); + + await cleanup(); + }); +}); + +// ─── mountMCPEndpoints refactor safety ─────────────────────────────── + +describe('mountMCPEndpoints', () => { + it('returns a cleanup function', () => { + const backend = createMockBackend(); + const mockApp = { + all: vi.fn(), + }; + + const cleanup = mountMCPEndpoints(mockApp as never, backend as never); + + expect(typeof cleanup).toBe('function'); + }); + + it('registers the /api/mcp route', () => { + const backend = createMockBackend(); + const allCalls: Array<[string, ...unknown[]]> = []; + const mockApp = { + all: vi.fn().mockImplementation((path: string, ...args: unknown[]) => { + allCalls.push([path, ...args]); + }), + }; + + mountMCPEndpoints(mockApp as never, backend as never); + + const registeredPaths = allCalls.map(([path]) => path); + expect(registeredPaths).toContain('/api/mcp'); + }); + + it('cleanup function resolves without throwing', async () => { + const backend = createMockBackend(); + const mockApp = { + all: vi.fn(), + }; + + const cleanup = mountMCPEndpoints(mockApp as never, backend as never); + + await expect(cleanup()).resolves.not.toThrow(); + }); +}); + +// ─── McpHttpOptions type validation ────────────────────────────────── + +describe('McpHttpOptions type validation', () => { + it('createAuthMiddleware accepts undefined authToken', () => { + const middleware = createAuthMiddleware(); + expect(typeof middleware).toBe('function'); + }); + + it('createAuthMiddleware accepts a string authToken', () => { + const middleware = createAuthMiddleware('test-token'); + expect(typeof middleware).toBe('function'); + }); +}); + +// ─── isLoopbackOrigin (U4) ─────────────────────────────────────────── + +describe('isLoopbackOrigin', () => { + it('accepts loopback origins including IPv6 [::1], IPv4-mapped, and the 127/8 block', () => { + expect(isLoopbackOrigin('http://localhost:8080')).toBe(true); + expect(isLoopbackOrigin('http://127.0.0.1:5000')).toBe(true); + expect(isLoopbackOrigin('http://127.0.0.2:3000')).toBe(true); + expect(isLoopbackOrigin('http://[::1]:3000')).toBe(true); + expect(isLoopbackOrigin('http://[::ffff:127.0.0.1]:3000')).toBe(true); + }); + + it('treats a missing Origin as allowed (non-browser caller)', () => { + expect(isLoopbackOrigin(undefined)).toBe(true); + }); + + it('rejects non-loopback and look-alike origins', () => { + expect(isLoopbackOrigin('http://localhost.evil.com')).toBe(false); + expect(isLoopbackOrigin('http://127.0.0.1.evil.com')).toBe(false); + expect(isLoopbackOrigin('http://example.com')).toBe(false); + expect(isLoopbackOrigin('http://192.168.1.50:3000')).toBe(false); + expect(isLoopbackOrigin('null')).toBe(false); + expect(isLoopbackOrigin('not a url')).toBe(false); + }); +}); + +// ─── startIdleSweep (U12) ──────────────────────────────────────────── + +describe('startIdleSweep', () => { + it('closes and evicts sessions idle beyond the TTL, keeping fresh ones', () => { + vi.useFakeTimers(); + try { + const ttlMs = 30 * 60 * 1000; + const intervalMs = 5 * 60 * 1000; + const now = Date.now(); + const closed: string[] = []; + const make = (id: string, lastActivity: number) => ({ + server: { close: () => closed.push(id) } as unknown as ReturnType, + lastActivity, + }); + const map = new Map([ + ['stale', make('stale', now - 60 * 60 * 1000)], + ['fresh', make('fresh', now)], + ]); + + const timer = startIdleSweep(map, ttlMs, intervalMs); + vi.advanceTimersByTime(intervalMs + 1); + + expect(map.has('stale')).toBe(false); + expect(map.has('fresh')).toBe(true); + expect(closed).toEqual(['stale']); + + clearInterval(timer); + } finally { + vi.useRealTimers(); + } + }); +}); + +// ─── resolveAuthToken (U9) ─────────────────────────────────────────── + +describe('resolveAuthToken', () => { + it('uses the --auth-token flag when set, preferring it over the env var', () => { + expect(resolveAuthToken('flag', {})).toBe('flag'); + expect(resolveAuthToken('flag', { GITNEXUS_MCP_AUTH_TOKEN: 'env' })).toBe('flag'); + }); + + it('falls back to GITNEXUS_MCP_AUTH_TOKEN', () => { + expect(resolveAuthToken(undefined, { GITNEXUS_MCP_AUTH_TOKEN: 'env' })).toBe('env'); + }); + + it('treats empty/whitespace as no token (no silent auth bypass)', () => { + expect(resolveAuthToken('', {})).toBeUndefined(); + expect(resolveAuthToken(' ', {})).toBeUndefined(); + expect(resolveAuthToken(undefined, { GITNEXUS_MCP_AUTH_TOKEN: '' })).toBeUndefined(); + expect(resolveAuthToken(undefined, { GITNEXUS_MCP_AUTH_TOKEN: ' ' })).toBeUndefined(); + }); + + it('returns undefined when neither is set, and trims a real token', () => { + expect(resolveAuthToken(undefined, {})).toBeUndefined(); + expect(resolveAuthToken(' tok ', {})).toBe('tok'); + }); +}); + +// ─── shutdown signal wiring (U7) ───────────────────────────────────── + +describe('shutdown exit codes (U7)', () => { + it('wires SIGINT → 130 and SIGTERM → 143 via the shared installSignalShutdown', () => { + const handlers: Record void> = {}; + const exits: number[] = []; + + installSignalShutdown( + (code = 0) => { + exits.push(code); + }, + (event, listener) => { + handlers[event] = listener; + }, + ); + + handlers.SIGINT('SIGINT'); + handlers.SIGTERM('SIGTERM'); + + expect(SHUTDOWN_EXIT_CODES).toEqual({ SIGINT: 130, SIGTERM: 143 }); + expect(exits).toEqual([130, 143]); + }); +}); + +// ─── computeAllowedHosts (U6) ──────────────────────────────────────── + +describe('computeAllowedHosts', () => { + it('returns all loopback host forms (bare + :port) for a loopback bind', () => { + expect(computeAllowedHosts('127.0.0.1', 3000)).toEqual([ + '127.0.0.1', + '127.0.0.1:3000', + 'localhost', + 'localhost:3000', + '[::1]', + '[::1]:3000', + ]); + }); + + it('returns the specific host (bare + :port) for a non-loopback, non-wildcard bind', () => { + expect(computeAllowedHosts('192.168.1.50', 8080)).toEqual([ + '192.168.1.50', + '192.168.1.50:8080', + ]); + }); + + it('returns undefined (protection off) for wildcard binds', () => { + expect(computeAllowedHosts('0.0.0.0', 3000)).toBeUndefined(); + expect(computeAllowedHosts('::', 3000)).toBeUndefined(); + }); +}); From 9ff7337f1e53b78a872af85139674b84065914ea Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sat, 13 Jun 2026 10:24:16 +0100 Subject: [PATCH 09/16] fix(mcp): rename query/cypher params so Claude Code can call them (#2186) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(mcp): advertise search_query/statement params for query/cypher tools (#2175) Claude Code drops a tool-call argument named exactly 'query', making the query and cypher tools unusable from it. Rename the advertised required parameters to search_query and statement so the client transmits them. Handler-side backward-compat for the legacy 'query' key follows in the next commit. * fix(mcp): accept search_query/statement with legacy query fallback (#2175) Resolve the new advertised param names in the backend while still accepting the legacy 'query' key, so curl/HTTP, other MCP clients, the CLI, the group path, and the internal executeCypher() all keep working. Alias is normalized once at the callTool chokepoint (covers group-forward + search alias); query() and cypher() dual-read defensively. New name wins when both are supplied. Updates the required-error message and adds dual-accept unit + integration coverage. * fix(cli): pass canonical search_query/statement params to query/cypher tools (#2175) Stop the CLI from depending on the deprecated 'query' alias. No user-facing change — the positional args are unchanged and the backend accepts both keys. * fix(mcp): generators advertise search_query in query() examples (#2175) Update the three doc/example generators (ai-context AGENTS/CLAUDE block, skill-gen community skills, resources repo hint) so future analyze runs emit query({search_query: ...}) — the param name Claude Code actually transmits. Tests assert the new form is present and the legacy query({query: form is absent (the #2059 generator-test pattern). * docs(mcp): advertise search_query/statement in skill & guidance examples (#2175) Sync the committed agent-facing docs to the renamed params so a Claude Code agent following them emits the transmittable key: AGENTS.md/CLAUDE.md gitnexus block, the canonical gitnexus/skills/* source and its installed/plugin/cursor mirrors, and the README examples. Scoped rewrite of the two call prefixes only (query({query: -> search_query, cypher({query: -> statement). * style(mcp): prettier line-wrap for #2175 alias-resolution edits * fix(review): uniform search_query precedence + cypher empty guard (#2175) Code-review findings (correctness/adversarial/api-contract/maintainability consensus): - Group-mode query inverted the 'new name wins' rule: the callTool chokepoint backfilled params.query only when empty and the @group-forward read params.query directly, so a both-keys (or whitespace-legacy) group call let the legacy value win — unlike the local path. Replace the hidden param mutation with a self-contained 'search_query ?? query' resolve at the group-forward; precedence is now uniformly new-wins at every consumer site. - cypher() now returns the same friendly required-param error as query() when neither statement nor query is supplied, instead of a raw DB prepare error. - Document the legacy alias as permanent (third-party clients may send query=). Adds group-forward alias tests (both-keys + legacy-only), empty/whitespace search_query, the search-alias path, and the cypher empty-statement guard. * fix(review): non-string alias safety + drop stale chokepoint comment (#2175) Tri-review findings (correctness/adversarial/security + maintainability): - Non-string statement/search_query/query (the MCP envelope is not schema-validated) hit .trim() and threw TypeError to the server boundary instead of a friendly required-param error. Introduce resolveAliasString() (new name wins; non-string -> undefined) used by query(), cypher(), and the group-forward, so all three return the structured error. Empirically verified (123 ?? '' -> 123, (123).trim() throws) — this overrides a critic refutation that mis-read ?? as a string coercion. - Remove the stale query() comment claiming alias resolution happens at a callTool chokepoint; that mutation was removed earlier in this PR — each site resolves the alias itself. - Document GroupToolPort.query's intentionally-narrower required type vs the wider LocalBackend impl. Adds non-string and empty-new-key precedence tests. * fix(mcp): alias falls back to legacy value when new key is blank (#2175) PR #2186 review finding: resolveAliasString used `canonical ?? legacy` (nullish), so an explicitly empty/whitespace new-name value (e.g. {search_query:'', query:'real'}) won and was rejected — discarding a valid legacy value, contradicting the 'new name wins when both supplied' intent. Resolve to the first NON-BLANK string instead (new preferred when it carries a real value, else legacy). Covers query(), cypher(), and the group-forward (all route through the helper); non-string still resolves to a friendly error. Flips the presence-based test and adds whitespace/cypher/group fallback cases. * fix(mcp): drop legacy "query" mention from query/cypher schema descriptions (#2175) PR #2186 review finding: the search_query/statement inputSchema descriptions named the legacy "query" key — the exact arg Claude Code drops — and description text is read by an LLM choosing arguments, weakly nudging it to send "query". Trim the descriptions to their clean form and move the legacy-alias note to a code comment next to the schema (preserved for maintainers / non-CC clients). properties/required unchanged (no `query`). --- .../gitnexus/gitnexus-debugging/SKILL.md | 8 +- .../gitnexus/gitnexus-exploring/SKILL.md | 6 +- .../gitnexus/gitnexus-refactoring/SKILL.md | 2 +- AGENTS.md | 2 +- CLAUDE.md | 2 +- README.md | 4 +- .../skills/gitnexus-debugging/SKILL.md | 8 +- .../skills/gitnexus-exploring/SKILL.md | 6 +- .../skills/gitnexus-refactoring/SKILL.md | 2 +- .../skills/gitnexus-debugging/SKILL.md | 8 +- .../skills/gitnexus-exploring/SKILL.md | 6 +- .../skills/gitnexus-refactoring/SKILL.md | 2 +- gitnexus/README.md | 2 +- gitnexus/skills/gitnexus-debugging.md | 8 +- gitnexus/skills/gitnexus-exploring.md | 6 +- gitnexus/skills/gitnexus-refactoring.md | 2 +- gitnexus/src/cli/ai-context.ts | 2 +- gitnexus/src/cli/skill-gen.ts | 2 +- gitnexus/src/cli/tool.ts | 6 +- gitnexus/src/core/group/service.ts | 4 + gitnexus/src/mcp/local/local-backend.ts | 58 +++++- gitnexus/src/mcp/resources.ts | 2 +- gitnexus/src/mcp/tools.ts | 22 +- .../local-backend-calltool.test.ts | 20 ++ gitnexus/test/unit/ai-context.test.ts | 5 +- gitnexus/test/unit/calltool-dispatch.test.ts | 193 +++++++++++++++++- gitnexus/test/unit/resources.test.ts | 4 +- gitnexus/test/unit/skill-gen.test.ts | 4 +- gitnexus/test/unit/tools.test.ts | 19 +- 29 files changed, 351 insertions(+), 64 deletions(-) diff --git a/.claude/skills/gitnexus/gitnexus-debugging/SKILL.md b/.claude/skills/gitnexus/gitnexus-debugging/SKILL.md index 9834f94b7..80f9c0ec5 100644 --- a/.claude/skills/gitnexus/gitnexus-debugging/SKILL.md +++ b/.claude/skills/gitnexus/gitnexus-debugging/SKILL.md @@ -16,10 +16,10 @@ description: "Use when the user is debugging a bug, tracing an error, or asking ## Workflow ``` -1. query({query: ""}) → Find related execution flows +1. query({search_query: ""}) → Find related execution flows 2. context({name: ""}) → See callers/callees/processes 3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow -4. cypher({query: "MATCH path..."}) → Custom traces if needed +4. cypher({statement: "MATCH path..."}) → Custom traces if needed ``` > If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal. @@ -51,7 +51,7 @@ description: "Use when the user is debugging a bug, tracing an error, or asking **query** — find code related to error: ``` -query({query: "payment validation error"}) +query({search_query: "payment validation error"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError, PaymentException ``` @@ -75,7 +75,7 @@ RETURN [n IN nodes(path) | n.name] AS chain ## Example: "Payment endpoint returns 500 intermittently" ``` -1. query({query: "payment error handling"}) +1. query({search_query: "payment error handling"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError diff --git a/.claude/skills/gitnexus/gitnexus-exploring/SKILL.md b/.claude/skills/gitnexus/gitnexus-exploring/SKILL.md index ccf684c28..f483c2fd6 100644 --- a/.claude/skills/gitnexus/gitnexus-exploring/SKILL.md +++ b/.claude/skills/gitnexus/gitnexus-exploring/SKILL.md @@ -18,7 +18,7 @@ description: "Use when the user asks how code works, wants to understand archite ``` 1. READ gitnexus://repos → Discover indexed repos 2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness -3. query({query: ""}) → Find related execution flows +3. query({search_query: ""}) → Find related execution flows 4. context({name: ""}) → Deep dive on specific symbol 5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow ``` @@ -50,7 +50,7 @@ description: "Use when the user asks how code works, wants to understand archite **query** — find execution flows related to a concept: ``` -query({query: "payment processing"}) +query({search_query: "payment processing"}) → Processes: CheckoutFlow, RefundFlow, WebhookHandler → Symbols grouped by flow with file locations ``` @@ -68,7 +68,7 @@ context({name: "validateUser"}) ``` 1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes -2. query({query: "payment processing"}) +2. query({search_query: "payment processing"}) → CheckoutFlow: processPayment → validateCard → chargeStripe → RefundFlow: initiateRefund → calculateRefund → processRefund 3. context({name: "processPayment"}) diff --git a/.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md b/.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md index e13c04e14..90c8c324d 100644 --- a/.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md +++ b/.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md @@ -17,7 +17,7 @@ description: "Use when the user wants to rename, extract, split, move, or restru ``` 1. impact({target: "X", direction: "upstream"}) → Map all dependents -2. query({query: "X"}) → Find execution flows involving X +2. query({search_query: "X"}) → Find execution flows involving X 3. context({name: "X"}) → See all incoming/outgoing refs 4. Plan update order: interfaces → implementations → callers → tests ``` diff --git a/AGENTS.md b/AGENTS.md index 64d264126..285a12186 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -83,7 +83,7 @@ This project is indexed by GitNexus as **GitNexus** (26675 symbols, 35395 relati - **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user. - **MUST run `detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows. - **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. -- When exploring unfamiliar code, use `query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. +- When exploring unfamiliar code, use `query({search_query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. - When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `context({name: "symbolName"})`. ## Never Do diff --git a/CLAUDE.md b/CLAUDE.md index f2bf1e487..7350cfb09 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -65,7 +65,7 @@ This project is indexed by GitNexus as **GitNexus** (26675 symbols, 35395 relati - **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user. - **MUST run `detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows. - **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. -- When exploring unfamiliar code, use `query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. +- When exploring unfamiliar code, use `query({search_query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. - When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `context({name: "symbolName"})`. ## Never Do diff --git a/README.md b/README.md index 876ff81e2..ab01db810 100644 --- a/README.md +++ b/README.md @@ -360,7 +360,7 @@ It is opt-in and a no-op without `UNDERSTAND_QUICKLY_TOKEN` — a fine-grained G | `group_query` | Search execution flows across all repos in a group | — | | `group_status` | Check staleness of repos in a group | — | -> When only one repo is indexed, the `repo` parameter is optional. With multiple repos, specify which one: `query({query: "auth", repo: "my-app"})`. +> When only one repo is indexed, the `repo` parameter is optional. With multiple repos, specify which one: `query({search_query: "auth", repo: "my-app"})`. **Resources** for instant context: @@ -738,7 +738,7 @@ gitnexus impact get_embeddings --uid "Function:src/embed.py:get_embeddings" # e ### Process-Grouped Search ``` -query({query: "authentication middleware"}) +query({search_query: "authentication middleware"}) processes: - summary: "LoginFlow" diff --git a/gitnexus-claude-plugin/skills/gitnexus-debugging/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-debugging/SKILL.md index 9834f94b7..80f9c0ec5 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-debugging/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-debugging/SKILL.md @@ -16,10 +16,10 @@ description: "Use when the user is debugging a bug, tracing an error, or asking ## Workflow ``` -1. query({query: ""}) → Find related execution flows +1. query({search_query: ""}) → Find related execution flows 2. context({name: ""}) → See callers/callees/processes 3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow -4. cypher({query: "MATCH path..."}) → Custom traces if needed +4. cypher({statement: "MATCH path..."}) → Custom traces if needed ``` > If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal. @@ -51,7 +51,7 @@ description: "Use when the user is debugging a bug, tracing an error, or asking **query** — find code related to error: ``` -query({query: "payment validation error"}) +query({search_query: "payment validation error"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError, PaymentException ``` @@ -75,7 +75,7 @@ RETURN [n IN nodes(path) | n.name] AS chain ## Example: "Payment endpoint returns 500 intermittently" ``` -1. query({query: "payment error handling"}) +1. query({search_query: "payment error handling"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError diff --git a/gitnexus-claude-plugin/skills/gitnexus-exploring/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-exploring/SKILL.md index ccf684c28..f483c2fd6 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-exploring/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-exploring/SKILL.md @@ -18,7 +18,7 @@ description: "Use when the user asks how code works, wants to understand archite ``` 1. READ gitnexus://repos → Discover indexed repos 2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness -3. query({query: ""}) → Find related execution flows +3. query({search_query: ""}) → Find related execution flows 4. context({name: ""}) → Deep dive on specific symbol 5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow ``` @@ -50,7 +50,7 @@ description: "Use when the user asks how code works, wants to understand archite **query** — find execution flows related to a concept: ``` -query({query: "payment processing"}) +query({search_query: "payment processing"}) → Processes: CheckoutFlow, RefundFlow, WebhookHandler → Symbols grouped by flow with file locations ``` @@ -68,7 +68,7 @@ context({name: "validateUser"}) ``` 1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes -2. query({query: "payment processing"}) +2. query({search_query: "payment processing"}) → CheckoutFlow: processPayment → validateCard → chargeStripe → RefundFlow: initiateRefund → calculateRefund → processRefund 3. context({name: "processPayment"}) diff --git a/gitnexus-claude-plugin/skills/gitnexus-refactoring/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-refactoring/SKILL.md index e13c04e14..90c8c324d 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-refactoring/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-refactoring/SKILL.md @@ -17,7 +17,7 @@ description: "Use when the user wants to rename, extract, split, move, or restru ``` 1. impact({target: "X", direction: "upstream"}) → Map all dependents -2. query({query: "X"}) → Find execution flows involving X +2. query({search_query: "X"}) → Find execution flows involving X 3. context({name: "X"}) → See all incoming/outgoing refs 4. Plan update order: interfaces → implementations → callers → tests ``` diff --git a/gitnexus-cursor-integration/skills/gitnexus-debugging/SKILL.md b/gitnexus-cursor-integration/skills/gitnexus-debugging/SKILL.md index a88b76430..cde6a6a3a 100644 --- a/gitnexus-cursor-integration/skills/gitnexus-debugging/SKILL.md +++ b/gitnexus-cursor-integration/skills/gitnexus-debugging/SKILL.md @@ -15,10 +15,10 @@ description: Trace bugs through call chains using knowledge graph ## Workflow ``` -1. query({query: ""}) → Find related execution flows +1. query({search_query: ""}) → Find related execution flows 2. context({name: ""}) → See callers/callees/processes 3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow -4. cypher({query: "MATCH path..."}) → Custom traces if needed +4. cypher({statement: "MATCH path..."}) → Custom traces if needed ``` > If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal. @@ -49,7 +49,7 @@ description: Trace bugs through call chains using knowledge graph **query** — find code related to error: ``` -query({query: "payment validation error"}) +query({search_query: "payment validation error"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError, PaymentException ``` @@ -71,7 +71,7 @@ RETURN [n IN nodes(path) | n.name] AS chain ## Example: "Payment endpoint returns 500 intermittently" ``` -1. query({query: "payment error handling"}) +1. query({search_query: "payment error handling"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError diff --git a/gitnexus-cursor-integration/skills/gitnexus-exploring/SKILL.md b/gitnexus-cursor-integration/skills/gitnexus-exploring/SKILL.md index 73df1353a..993a38481 100644 --- a/gitnexus-cursor-integration/skills/gitnexus-exploring/SKILL.md +++ b/gitnexus-cursor-integration/skills/gitnexus-exploring/SKILL.md @@ -17,7 +17,7 @@ description: Navigate unfamiliar code using GitNexus knowledge graph ``` 1. READ gitnexus://repos → Discover indexed repos 2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness -3. query({query: ""}) → Find related execution flows +3. query({search_query: ""}) → Find related execution flows 4. context({name: ""}) → Deep dive on specific symbol 5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow ``` @@ -48,7 +48,7 @@ description: Navigate unfamiliar code using GitNexus knowledge graph **query** — find execution flows related to a concept: ``` -query({query: "payment processing"}) +query({search_query: "payment processing"}) → Processes: CheckoutFlow, RefundFlow, WebhookHandler → Symbols grouped by flow with file locations ``` @@ -65,7 +65,7 @@ context({name: "validateUser"}) ``` 1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes -2. query({query: "payment processing"}) +2. query({search_query: "payment processing"}) → CheckoutFlow: processPayment → validateCard → chargeStripe → RefundFlow: initiateRefund → calculateRefund → processRefund 3. context({name: "processPayment"}) diff --git a/gitnexus-cursor-integration/skills/gitnexus-refactoring/SKILL.md b/gitnexus-cursor-integration/skills/gitnexus-refactoring/SKILL.md index 76c9d3351..fbf193182 100644 --- a/gitnexus-cursor-integration/skills/gitnexus-refactoring/SKILL.md +++ b/gitnexus-cursor-integration/skills/gitnexus-refactoring/SKILL.md @@ -16,7 +16,7 @@ description: Plan safe refactors using blast radius and dependency mapping ``` 1. impact({target: "X", direction: "upstream"}) → Map all dependents -2. query({query: "X"}) → Find execution flows involving X +2. query({search_query: "X"}) → Find execution flows involving X 3. context({name: "X"}) → See all incoming/outgoing refs 4. Plan update order: interfaces → implementations → callers → tests ``` diff --git a/gitnexus/README.md b/gitnexus/README.md index a306f0852..92cf4593c 100644 --- a/gitnexus/README.md +++ b/gitnexus/README.md @@ -134,7 +134,7 @@ Your AI agent gets these tools automatically: | `rename` | Multi-file coordinated rename with graph + text search | Optional | | `cypher` | Raw Cypher graph queries | Optional | -> With one indexed repo, the `repo` param is optional. With multiple, specify which: `query({query: "auth", repo: "my-app"})`. +> With one indexed repo, the `repo` param is optional. With multiple, specify which: `query({search_query: "auth", repo: "my-app"})`. ## MCP Resources diff --git a/gitnexus/skills/gitnexus-debugging.md b/gitnexus/skills/gitnexus-debugging.md index 9834f94b7..80f9c0ec5 100644 --- a/gitnexus/skills/gitnexus-debugging.md +++ b/gitnexus/skills/gitnexus-debugging.md @@ -16,10 +16,10 @@ description: "Use when the user is debugging a bug, tracing an error, or asking ## Workflow ``` -1. query({query: ""}) → Find related execution flows +1. query({search_query: ""}) → Find related execution flows 2. context({name: ""}) → See callers/callees/processes 3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow -4. cypher({query: "MATCH path..."}) → Custom traces if needed +4. cypher({statement: "MATCH path..."}) → Custom traces if needed ``` > If "Index is stale" → run `node .gitnexus/run.cjs analyze` in terminal. @@ -51,7 +51,7 @@ description: "Use when the user is debugging a bug, tracing an error, or asking **query** — find code related to error: ``` -query({query: "payment validation error"}) +query({search_query: "payment validation error"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError, PaymentException ``` @@ -75,7 +75,7 @@ RETURN [n IN nodes(path) | n.name] AS chain ## Example: "Payment endpoint returns 500 intermittently" ``` -1. query({query: "payment error handling"}) +1. query({search_query: "payment error handling"}) → Processes: CheckoutFlow, ErrorHandling → Symbols: validatePayment, handlePaymentError diff --git a/gitnexus/skills/gitnexus-exploring.md b/gitnexus/skills/gitnexus-exploring.md index ccf684c28..f483c2fd6 100644 --- a/gitnexus/skills/gitnexus-exploring.md +++ b/gitnexus/skills/gitnexus-exploring.md @@ -18,7 +18,7 @@ description: "Use when the user asks how code works, wants to understand archite ``` 1. READ gitnexus://repos → Discover indexed repos 2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness -3. query({query: ""}) → Find related execution flows +3. query({search_query: ""}) → Find related execution flows 4. context({name: ""}) → Deep dive on specific symbol 5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow ``` @@ -50,7 +50,7 @@ description: "Use when the user asks how code works, wants to understand archite **query** — find execution flows related to a concept: ``` -query({query: "payment processing"}) +query({search_query: "payment processing"}) → Processes: CheckoutFlow, RefundFlow, WebhookHandler → Symbols grouped by flow with file locations ``` @@ -68,7 +68,7 @@ context({name: "validateUser"}) ``` 1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes -2. query({query: "payment processing"}) +2. query({search_query: "payment processing"}) → CheckoutFlow: processPayment → validateCard → chargeStripe → RefundFlow: initiateRefund → calculateRefund → processRefund 3. context({name: "processPayment"}) diff --git a/gitnexus/skills/gitnexus-refactoring.md b/gitnexus/skills/gitnexus-refactoring.md index e13c04e14..90c8c324d 100644 --- a/gitnexus/skills/gitnexus-refactoring.md +++ b/gitnexus/skills/gitnexus-refactoring.md @@ -17,7 +17,7 @@ description: "Use when the user wants to rename, extract, split, move, or restru ``` 1. impact({target: "X", direction: "upstream"}) → Map all dependents -2. query({query: "X"}) → Find execution flows involving X +2. query({search_query: "X"}) → Find execution flows involving X 3. context({name: "X"}) → See all incoming/outgoing refs 4. Plan update order: interfaces → implementations → callers → tests ``` diff --git a/gitnexus/src/cli/ai-context.ts b/gitnexus/src/cli/ai-context.ts index 4bf117974..6700c020b 100644 --- a/gitnexus/src/cli/ai-context.ts +++ b/gitnexus/src/cli/ai-context.ts @@ -177,7 +177,7 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s - **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run \`impact({target: "symbolName", direction: "upstream"})\` and report the blast radius (direct callers, affected processes, risk level) to the user. - **MUST run \`detect_changes()\` before committing** to verify your changes only affect expected symbols and execution flows. For regression review, compare against the default branch: \`detect_changes({scope: "compare", base_ref: ${JSON.stringify(markdownSafeBranch(defaultBranch))}})\`. - **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. -- When exploring unfamiliar code, use \`query({query: "concept"})\` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. +- When exploring unfamiliar code, use \`query({search_query: "concept"})\` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. - When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use \`context({name: "symbolName"})\`. - For security review, \`explain({target: "fileOrSymbol"})\` lists taint findings (source→sink flows; needs \`analyze --pdg\`). diff --git a/gitnexus/src/cli/skill-gen.ts b/gitnexus/src/cli/skill-gen.ts index 61569e6a4..d718a91dd 100644 --- a/gitnexus/src/cli/skill-gen.ts +++ b/gitnexus/src/cli/skill-gen.ts @@ -651,7 +651,7 @@ const renderSkillMarkdown = ( lines.push(''); lines.push(`1. \`context({name: "${firstEntry}"})\` \u2014 see callers and callees`); lines.push( - `2. \`query({query: "${community.label.toLowerCase()}"})\` \u2014 find related execution flows`, + `2. \`query({search_query: "${community.label.toLowerCase()}"})\` \u2014 find related execution flows`, ); lines.push('3. Read key files listed above for implementation details'); lines.push( diff --git a/gitnexus/src/cli/tool.ts b/gitnexus/src/cli/tool.ts index b892f6e2c..76850eb9f 100644 --- a/gitnexus/src/cli/tool.ts +++ b/gitnexus/src/cli/tool.ts @@ -76,7 +76,8 @@ export async function queryCommand( const backend = await getBackend(); const result = await backend.callTool('query', { - query: queryText, + // #2175: canonical param is search_query; the backend still accepts legacy "query". + search_query: queryText, task_context: options?.context, goal: options?.goal, limit: options?.limit ? parseInt(options.limit) : undefined, @@ -204,7 +205,8 @@ export async function cypherCommand( const backend = await getBackend(); const result = await backend.callTool('cypher', { - query, + // #2175: canonical param is statement; the backend still accepts legacy "query". + statement: query, repo: options?.repo, branch: options?.branch, }); diff --git a/gitnexus/src/core/group/service.ts b/gitnexus/src/core/group/service.ts index b957db15b..a46118bd5 100644 --- a/gitnexus/src/core/group/service.ts +++ b/gitnexus/src/core/group/service.ts @@ -49,6 +49,10 @@ export interface GroupToolPort { query( repo: GroupRepoHandle, params: { + // GroupService always supplies `query` as a string (it resolves the #2175 + // search_query alias before calling the port), so the port contract keeps it + // required here even though the LocalBackend implementation accepts the wider + // `{ query?, search_query? }` shape for the direct MCP callTool path. query: string; task_context?: string; goal?: string; diff --git a/gitnexus/src/mcp/local/local-backend.ts b/gitnexus/src/mcp/local/local-backend.ts index 99bec1852..d50bd67ac 100644 --- a/gitnexus/src/mcp/local/local-backend.ts +++ b/gitnexus/src/mcp/local/local-backend.ts @@ -77,6 +77,23 @@ function looksLikeFilePath(target: string): boolean { const lower = target.toLowerCase(); return SOURCE_FILE_EXTENSIONS.some((ext) => lower.endsWith(ext)); } + +/** + * Resolve a string tool param from its canonical name or legacy alias (#2175). + * Returns the first NON-BLANK string of [canonical, legacy] — the canonical (new) + * name is preferred when it carries a real value, otherwise the legacy value is used. + * A blank/whitespace new value therefore does NOT clobber a valid legacy value (e.g. a + * gradually-migrating client that always emits the new key, blank when unset). A + * non-string value (the MCP envelope is not schema-validated, so clients can send any + * JSON type) and an all-blank input resolve to `undefined`, so the caller returns a + * friendly required-param error instead of throwing `TypeError` on `.trim()`. + */ +function resolveAliasString(canonical: unknown, legacy: unknown): string | undefined { + for (const value of [canonical, legacy]) { + if (typeof value === 'string' && value.trim()) return value; + } + return undefined; +} // AI context generation is CLI-only (gitnexus analyze) // import { generateAIContextFiles } from '../../cli/ai-context.js'; @@ -1242,6 +1259,16 @@ export class LocalBackend { } const p = params && typeof params === 'object' ? (params as Record) : {}; + + // #2175: Claude Code drops a tool-call argument named exactly "query", so the + // query/cypher tools advertise "search_query"/"statement" while still accepting the + // legacy "query" key for backward compat. The alias is resolved with `?? ` (new name + // wins) at every consumer site rather than by mutating params here, so precedence is + // uniform and there is no hidden mutation: query()/cypher() read it directly, the + // legacy "search" alias routes through query(), and the cross-repo group-forward + // resolves it self-contained in callToolAtGroupRepo. This is permanent compatibility + // — third-party MCP clients may legitimately send "query", so the alias is not slated + // for removal even if Claude Code's argument handling later changes. if ( (method === 'impact' || method === 'query' || method === 'context') && typeof p.repo === 'string' && @@ -1345,7 +1372,8 @@ export class LocalBackend { private async query( repo: RepoHandle, params: { - query: string; + query?: string; + search_query?: string; task_context?: string; goal?: string; limit?: number; @@ -1353,8 +1381,12 @@ export class LocalBackend { include_content?: boolean; }, ): Promise { - if (!params.query?.trim()) { - return { error: 'query parameter is required and cannot be empty.' }; + // #2175: each consumer resolves the search_query/query alias itself (there is no + // chokepoint mutation in callTool). This also serves the GroupService port, which + // reaches query() carrying only the legacy `query` key. + const rawQuery = resolveAliasString(params.search_query, params.query); + if (!rawQuery?.trim()) { + return { error: 'search_query (or legacy query) parameter is required and cannot be empty.' }; } await this.ensureInitialized(repo); @@ -1362,7 +1394,7 @@ export class LocalBackend { const processLimit = params.limit || 5; const maxSymbolsPerProcess = params.max_symbols || 10; const includeContent = params.include_content ?? false; - const searchQuery = params.query.trim(); + const searchQuery = rawQuery.trim(); // Per-phase timing instrumentation (#553). Records wall time for each // observable sub-step of the search pipeline so production latency can @@ -1932,7 +1964,9 @@ export class LocalBackend { private async cypher( repo: RepoHandle, - request: { query: string; params?: Record }, + // #2175: "statement" is the advertised param; "query" is the legacy alias, + // still accepted (and the field the internal executeCypher() passes). New wins. + request: { query?: string; statement?: string; params?: Record }, ): Promise { await this.ensureInitialized(repo); @@ -1945,8 +1979,15 @@ export class LocalBackend { }; } + const cypherText = resolveAliasString(request.statement, request.query) ?? ''; + if (!cypherText.trim()) { + // Mirror query()'s friendly required-param error instead of letting an empty + // string fall through to a raw LadybugDB prepare error (#2175 review). + return { error: 'statement (or legacy query) parameter is required and cannot be empty.' }; + } + try { - const result = await executeParameterized(repo.lbugPath, request.query, request.params ?? {}); + const result = await executeParameterized(repo.lbugPath, cypherText, request.params ?? {}); return result; } catch (err: any) { const msg = err.message || 'Query failed'; @@ -4820,7 +4861,10 @@ export class LocalBackend { if (method === 'query') { const queryArgs: Record = { name: groupName, - query: params.query, + // #2175: resolve the search_query alias here (new name wins, same rule as the + // local query() handler) so the group path is self-contained and does not depend + // on params being normalized upstream. groupQuery() reads `query`. + query: resolveAliasString(params.search_query, params.query), }; if (typeof params.task_context === 'string') queryArgs.task_context = params.task_context; if (typeof params.goal === 'string') queryArgs.goal = params.goal; diff --git a/gitnexus/src/mcp/resources.ts b/gitnexus/src/mcp/resources.ts index bb900d924..ec37c196c 100644 --- a/gitnexus/src/mcp/resources.ts +++ b/gitnexus/src/mcp/resources.ts @@ -292,7 +292,7 @@ async function getReposResource(backend: LocalBackend): Promise { if (repos.length > 1) { lines.push(''); lines.push('# Multiple repos indexed. Use repo parameter in tool calls:'); - lines.push(`# query({query: "auth", repo: "${repos[0].name}"})`); + lines.push(`# query({search_query: "auth", repo: "${repos[0].name}"})`); } return lines.join('\n'); diff --git a/gitnexus/src/mcp/tools.ts b/gitnexus/src/mcp/tools.ts index 4df17eb98..ec3d76e04 100644 --- a/gitnexus/src/mcp/tools.ts +++ b/gitnexus/src/mcp/tools.ts @@ -130,7 +130,14 @@ SERVICE: optional monorepo path prefix (POSIX-style, case-sensitive segments). W inputSchema: { type: 'object', properties: { - query: { type: 'string', description: 'Natural language or keyword search query' }, + // #2175: the legacy `query` key is still accepted by the handler + // (resolveAliasString in local-backend.ts), but is deliberately NOT named in the + // advertised property or its description — surfacing "query" in the schema an LLM + // reads would nudge it to send `query`, the exact argument Claude Code drops. + search_query: { + type: 'string', + description: 'Natural language or keyword search query.', + }, task_context: { type: 'string', description: 'What you are working on (e.g., "adding OAuth support"). Helps ranking.', @@ -171,7 +178,7 @@ SERVICE: optional monorepo path prefix (POSIX-style, case-sensitive segments). W 'Optional monorepo service root (relative path, "/" separators). In group mode (@repo), prefix-matches symbol file paths; ignored for a normal repo name. Empty string is rejected server-side.', }, }, - required: ['query'], + required: ['search_query'], }, }, { @@ -224,7 +231,14 @@ TIPS: inputSchema: { type: 'object', properties: { - query: { type: 'string', description: 'Cypher query to execute' }, + // #2175: the legacy `query` key is still accepted by the handler + // (resolveAliasString in local-backend.ts), but is deliberately NOT named in the + // advertised property or its description — surfacing "query" in the schema an LLM + // reads would nudge it to send `query`, the exact argument Claude Code drops. + statement: { + type: 'string', + description: 'Cypher statement to execute.', + }, params: { type: 'object', description: @@ -235,7 +249,7 @@ TIPS: description: 'Repository name or path. Omit if only one repo is indexed.', }, }, - required: ['query'], + required: ['statement'], }, }, { diff --git a/gitnexus/test/integration/local-backend-calltool.test.ts b/gitnexus/test/integration/local-backend-calltool.test.ts index eed46b18d..00d08a09a 100644 --- a/gitnexus/test/integration/local-backend-calltool.test.ts +++ b/gitnexus/test/integration/local-backend-calltool.test.ts @@ -119,6 +119,26 @@ withTestLbugDB( expect(result).not.toHaveProperty('partial'); }); + // #2175: end-to-end proof that the renamed parameters work against a real + // index (Claude Code drops a tool arg named exactly "query"). + it('query tool returns results via the new search_query param (#2175)', async () => { + const result = await backend.callTool('query', { search_query: 'login' }); + expect(result).not.toHaveProperty('error'); + expect(result).toHaveProperty('processes'); + expect(result.processes.map((p: any) => p.id)).toContain('proc:login-flow'); + expect(result.process_symbols.map((s: any) => s.id)).toContain('func:login'); + }); + + it('cypher tool executes via the new statement param (#2175)', async () => { + const result = await backend.callTool('cypher', { + statement: 'MATCH (n:Function) RETURN n.name AS name ORDER BY n.name', + }); + expect(result).toHaveProperty('markdown'); + expect(result).toHaveProperty('row_count'); + expect(result.row_count).toBeGreaterThanOrEqual(3); + expect(result.markdown).toContain('login'); + }); + // PR #222 port: the query tool batches per-symbol process/cohesion/content // lookups (N+1 → 2-3 `WHERE n.id IN $nodeIds` queries). These assertions // guard the batch-adaptation hazards that a naive cherry-pick would break: diff --git a/gitnexus/test/unit/ai-context.test.ts b/gitnexus/test/unit/ai-context.test.ts index dc732f1fc..747bcec1a 100644 --- a/gitnexus/test/unit/ai-context.test.ts +++ b/gitnexus/test/unit/ai-context.test.ts @@ -920,7 +920,10 @@ Indexed as **placeholder** (1 symbols, 1 relationships, 1 execution flows). Cust expect(content).not.toMatch(/gitnexus_(impact|query|context|detect_changes|rename|cypher)/); expect(content).toContain('impact({target: "symbolName", direction: "upstream"})'); expect(content).toContain('detect_changes()'); - expect(content).toContain('query({query: "concept"})'); + // #2175: the generated guidance must advertise the renamed param, never the + // legacy "query" key (Claude Code drops a tool arg named exactly "query"). + expect(content).toContain('query({search_query: "concept"})'); + expect(content).not.toContain('query({query:'); expect(content).toContain('context({name: "symbolName"})'); }); diff --git a/gitnexus/test/unit/calltool-dispatch.test.ts b/gitnexus/test/unit/calltool-dispatch.test.ts index f95b1e4b8..84bdc4e69 100644 --- a/gitnexus/test/unit/calltool-dispatch.test.ts +++ b/gitnexus/test/unit/calltool-dispatch.test.ts @@ -85,6 +85,13 @@ vi.mock('../../src/mcp/core/embedder.js', () => ({ getEmbeddingDims: vi.fn().mockReturnValue(384), })); +// #2175: lets the @group-forward path be exercised without real group.yaml infra. +// No existing test in this file uses an @repo, so this mock is inert for them. +const { resolveAtMemberMock } = vi.hoisted(() => ({ resolveAtMemberMock: vi.fn() })); +vi.mock('../../src/core/group/resolve-at-member.js', () => ({ + resolveAtGroupMemberRepoPath: resolveAtMemberMock, +})); + import { LocalBackend, REPO_ID_HASH_LENGTH, @@ -433,12 +440,14 @@ describe('LocalBackend.callTool', () => { it('query tool returns error for empty query', async () => { const result = await backend.callTool('query', { query: '' }); - expect(result.error).toContain('query parameter is required'); + expect(result.error).toContain('search_query'); + expect(result.error).toContain('parameter is required'); }); it('query tool returns error for whitespace-only query', async () => { const result = await backend.callTool('query', { query: ' ' }); - expect(result.error).toContain('query parameter is required'); + expect(result.error).toContain('search_query'); + expect(result.error).toContain('parameter is required'); }); it('dispatches cypher tool and blocks write queries', async () => { @@ -459,6 +468,186 @@ describe('LocalBackend.callTool', () => { expect(result.row_count).toBe(1); }); + // ── #2175: backward-compatible parameter-alias dispatch ────────────────── + // Claude Code drops a tool-call argument named exactly "query", so the query + // and cypher tools advertise search_query / statement. The handlers must accept + // the new names AND keep accepting the legacy "query" key (verified by the + // existing tests above, which still pass { query: ... }). + + it('query tool accepts the new search_query parameter (#2175)', async () => { + (executeParameterized as any).mockResolvedValue([]); + const result = await backend.callTool('query', { search_query: 'auth' }); + expect(result).toHaveProperty('processes'); + expect(result).toHaveProperty('definitions'); + expect(result).not.toHaveProperty('error'); + }); + + it('query tool prefers search_query over the legacy query when both are given (#2175)', async () => { + const { searchFTSFromLbug } = await import('../../src/core/search/bm25-index.js'); + (executeParameterized as any).mockResolvedValue([]); + + await backend.callTool('query', { search_query: 'newName', query: 'oldName' }); + + // bm25Search passes the resolved search text as arg 0 to searchFTSFromLbug. + const lastTerm = String(vi.mocked(searchFTSFromLbug).mock.calls.at(-1)?.[0] ?? ''); + expect(lastTerm).toBe('newName'); + }); + + it('query tool returns error when neither search_query nor query is provided (#2175)', async () => { + const result = await backend.callTool('query', {}); + expect(result.error).toContain('search_query'); + expect(result.error).toContain('parameter is required'); + }); + + it('cypher tool accepts the new statement parameter (#2175)', async () => { + (executeParameterized as any).mockResolvedValue([{ name: 'test', filePath: 'src/test.ts' }]); + const result = await backend.callTool('cypher', { + statement: 'MATCH (n:Function) RETURN n.name AS name, n.filePath AS filePath LIMIT 5', + }); + expect(result).toHaveProperty('markdown'); + expect(result).toHaveProperty('row_count'); + expect(result.row_count).toBe(1); + }); + + it('cypher tool prefers statement over the legacy query when both are given (#2175)', async () => { + (executeParameterized as any).mockResolvedValue([]); + await backend.callTool('cypher', { + statement: 'MATCH (a) RETURN a', + query: 'MATCH (b) RETURN b', + }); + const passedCypher = (executeParameterized as any).mock.calls.at(-1)[1] as string; + expect(passedCypher).toBe('MATCH (a) RETURN a'); + }); + + it('executeCypher (internal API) still works via the legacy query field (#2175)', async () => { + (executeParameterized as any).mockResolvedValue([{ name: 'x' }]); + const result = await backend.executeCypher('test-project', 'MATCH (n) RETURN n LIMIT 1'); + expect(result).not.toHaveProperty('error'); + const passedCypher = (executeParameterized as any).mock.calls.at(-1)[1] as string; + expect(passedCypher).toBe('MATCH (n) RETURN n LIMIT 1'); + }); + + it('query tool returns error for empty search_query (new key) (#2175)', async () => { + const result = await backend.callTool('query', { search_query: '' }); + expect(result.error).toContain('search_query'); + expect(result.error).toContain('parameter is required'); + }); + + it('query tool returns error for whitespace-only search_query (new key) (#2175)', async () => { + const result = await backend.callTool('query', { search_query: ' ' }); + expect(result.error).toContain('search_query'); + expect(result.error).toContain('parameter is required'); + }); + + it('search legacy alias accepts the new search_query parameter (#2175)', async () => { + (executeParameterized as any).mockResolvedValue([]); + const result = await backend.callTool('search', { search_query: 'auth' }); + expect(result).toHaveProperty('processes'); + expect(result).not.toHaveProperty('error'); + }); + + it('cypher tool returns a friendly required error when neither statement nor query is given (#2175)', async () => { + const result = await backend.callTool('cypher', {}); + expect(result.error).toContain('statement'); + expect(result.error).toContain('parameter is required'); + }); + + // #2175 review: the @group-forward path reads `query` from the forwarded args, so it + // must resolve the search_query alias itself (new name wins, mirroring query()). + it('group-mode query forwards the resolved search_query alias (#2175)', async () => { + resolveAtMemberMock.mockResolvedValue({ ok: true, repoPath: '/tmp/test-project' }); + const groupQuerySpy = vi + .spyOn(backend.getGroupService(), 'groupQuery') + .mockResolvedValue({ ok: true } as any); + + await backend.callTool('query', { + search_query: 'alias-wins', + query: 'legacy-loses', + repo: '@grp', + }); + + expect(groupQuerySpy).toHaveBeenCalledTimes(1); + expect((groupQuerySpy.mock.calls[0][0] as any).query).toBe('alias-wins'); + groupQuerySpy.mockRestore(); + }); + + it('group-mode query still forwards a legacy-only query (#2175)', async () => { + resolveAtMemberMock.mockResolvedValue({ ok: true, repoPath: '/tmp/test-project' }); + const groupQuerySpy = vi + .spyOn(backend.getGroupService(), 'groupQuery') + .mockResolvedValue({ ok: true } as any); + + await backend.callTool('query', { query: 'legacy', repo: '@grp' }); + + expect((groupQuerySpy.mock.calls[0][0] as any).query).toBe('legacy'); + groupQuerySpy.mockRestore(); + }); + + // #2175 review: the MCP envelope is not schema-validated, so a client can send a + // non-string value for a string param. Resolve it to a friendly required-param error + // rather than throwing TypeError on `.trim()` (query() and cypher() both). + it('query tool returns a friendly error (no throw) for a non-string search_query (#2175)', async () => { + const result = await backend.callTool('query', { search_query: 123 as any }); + expect(result.error).toContain('search_query'); + expect(result.error).toContain('parameter is required'); + }); + + it('cypher tool returns a friendly error (no throw) for a non-string statement (#2175)', async () => { + const result = await backend.callTool('cypher', { statement: 123 as any }); + expect(result.error).toContain('statement'); + expect(result.error).toContain('parameter is required'); + }); + + // #2175 review (PR #2186): resolution prefers the first NON-BLANK string, so a blank + // new-name value falls back to a valid legacy value instead of clobbering it. + it('query tool: a blank new search_query falls back to a valid legacy query (#2175)', async () => { + const { searchFTSFromLbug } = await import('../../src/core/search/bm25-index.js'); + (executeParameterized as any).mockResolvedValue([]); + + const result = await backend.callTool('query', { search_query: '', query: 'real' }); + + expect(result).not.toHaveProperty('error'); + expect(result).toHaveProperty('processes'); + const lastTerm = String(vi.mocked(searchFTSFromLbug).mock.calls.at(-1)?.[0] ?? ''); + expect(lastTerm).toBe('real'); + }); + + it('query tool: a whitespace-only new search_query falls back to a valid legacy query (#2175)', async () => { + const { searchFTSFromLbug } = await import('../../src/core/search/bm25-index.js'); + (executeParameterized as any).mockResolvedValue([]); + + const result = await backend.callTool('query', { search_query: ' ', query: 'real' }); + + expect(result).not.toHaveProperty('error'); + const lastTerm = String(vi.mocked(searchFTSFromLbug).mock.calls.at(-1)?.[0] ?? ''); + expect(lastTerm).toBe('real'); + }); + + it('query tool: both keys blank still returns the required error (#2175)', async () => { + const result = await backend.callTool('query', { search_query: '', query: ' ' }); + expect(result.error).toContain('search_query'); + expect(result.error).toContain('parameter is required'); + }); + + it('cypher tool: a blank statement falls back to a valid legacy query (#2175)', async () => { + (executeParameterized as any).mockResolvedValue([]); + await backend.callTool('cypher', { statement: '', query: 'MATCH (n) RETURN n LIMIT 1' }); + const passedCypher = (executeParameterized as any).mock.calls.at(-1)[1] as string; + expect(passedCypher).toBe('MATCH (n) RETURN n LIMIT 1'); + }); + + it('group-mode query: a blank new search_query falls back to the legacy query (#2175)', async () => { + resolveAtMemberMock.mockResolvedValue({ ok: true, repoPath: '/tmp/test-project' }); + const groupQuerySpy = vi + .spyOn(backend.getGroupService(), 'groupQuery') + .mockResolvedValue({ ok: true } as any); + + await backend.callTool('query', { search_query: '', query: 'real', repo: '@grp' }); + + expect((groupQuerySpy.mock.calls[0][0] as any).query).toBe('real'); + groupQuerySpy.mockRestore(); + }); + it('dispatches context tool', async () => { (executeParameterized as any).mockResolvedValue([ { diff --git a/gitnexus/test/unit/resources.test.ts b/gitnexus/test/unit/resources.test.ts index 598698b8c..a0ea2348f 100644 --- a/gitnexus/test/unit/resources.test.ts +++ b/gitnexus/test/unit/resources.test.ts @@ -388,7 +388,9 @@ describe('readResource', () => { expect(result).toContain('repo parameter'); // The example must use a registered tool name, not the unregistered // `gitnexus_search` / `gitnexus_*` prefix (#2059). - expect(result).toContain('query({query: "auth"'); + // #2175: advertise the renamed param, not the legacy "query" key. + expect(result).toContain('query({search_query: "auth"'); + expect(result).not.toContain('query({query:'); expect(result).not.toMatch(/gitnexus_/); }); }); diff --git a/gitnexus/test/unit/skill-gen.test.ts b/gitnexus/test/unit/skill-gen.test.ts index 8e8800be6..21f993ff3 100644 --- a/gitnexus/test/unit/skill-gen.test.ts +++ b/gitnexus/test/unit/skill-gen.test.ts @@ -650,7 +650,9 @@ describe('generateSkillFiles — file output', () => { ); expect(content).not.toMatch(/gitnexus_(context|query|impact|detect_changes|rename|cypher)/); expect(content).toContain('context({name:'); - expect(content).toContain('query({query:'); + // #2175: advertise the renamed param, not the legacy "query" key. + expect(content).toContain('query({search_query:'); + expect(content).not.toContain('query({query:'); }); /** diff --git a/gitnexus/test/unit/tools.test.ts b/gitnexus/test/unit/tools.test.ts index 298a2ee5c..156cc7a20 100644 --- a/gitnexus/test/unit/tools.test.ts +++ b/gitnexus/test/unit/tools.test.ts @@ -99,16 +99,23 @@ describe('GITNEXUS_TOOLS', () => { } }); - it('query tool requires "query" parameter', () => { + it('query tool requires "search_query" parameter (renamed from "query" for #2175)', () => { const queryTool = GITNEXUS_TOOLS.find((t) => t.name === 'query')!; - expect(queryTool.inputSchema.required).toContain('query'); - expect(queryTool.inputSchema.properties.query).toBeDefined(); - expect(queryTool.inputSchema.properties.query.type).toBe('string'); + expect(queryTool.inputSchema.required).toContain('search_query'); + // The legacy "query" key must NOT be advertised — Claude Code drops it (#2175). + expect(queryTool.inputSchema.required).not.toContain('query'); + expect(queryTool.inputSchema.properties.query).toBeUndefined(); + expect(queryTool.inputSchema.properties.search_query).toBeDefined(); + expect(queryTool.inputSchema.properties.search_query.type).toBe('string'); }); - it('cypher tool requires "query" parameter', () => { + it('cypher tool requires "statement" parameter (renamed from "query" for #2175)', () => { const cypherTool = GITNEXUS_TOOLS.find((t) => t.name === 'cypher')!; - expect(cypherTool.inputSchema.required).toContain('query'); + expect(cypherTool.inputSchema.required).toContain('statement'); + expect(cypherTool.inputSchema.required).not.toContain('query'); + expect(cypherTool.inputSchema.properties.query).toBeUndefined(); + expect(cypherTool.inputSchema.properties.statement).toBeDefined(); + expect(cypherTool.inputSchema.properties.statement.type).toBe('string'); expect(cypherTool.inputSchema.properties.params).toBeDefined(); expect(cypherTool.inputSchema.properties.params.type).toBe('object'); expect(cypherTool.inputSchema.properties.params.description).toContain('prepared statement'); From 6dc6544365aff48faa1129b0eb9ce5b6b7fc0843 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sat, 13 Jun 2026 11:07:58 +0100 Subject: [PATCH 10/16] fix(web): chat-only mode for large projects to prevent WebUI hang (#2178) (#2185) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(web): add graph-load skip decision helper and node threshold (#2178) * feat(web): skip graph download in connectToServer for chat-only mode (#2178) * feat(web): add graphMode state and empty-graph chat-only handling (#2178) * feat(web): read and thread ?skipGraph URL param through connect flow (#2178) * feat(web): chat-only empty state with load-graph-anyway escape hatch (#2178) * style(web): apply prettier formatting to graph-load files (#2178) * fix(review): apply autofix feedback - Fail-safe confirm + authoritative node count (P1: prevent re-triggering the hang via Load-graph-anyway when count unknown) - In-flight guard on loadGraphAnyway (P1: double-fire) - Honor explicit ?skipGraph in onAnalyzeComplete and DropZone (R6/U4) - Extract buildGraphFromConnectResult shared helper (DRY across 3 connect sites) - Add tests: switchRepo skip path, threshold config override, loadGraphAnyway error path, confirm fail-safe, in-flight guard * fix(review): address tri-review findings - P1 (correctness+adversarial+risk): stop the cross-repo / F5 chat-only leak. loadGraphAnyway no longer persists ?skipGraph=0, and onAnalyzeComplete + DropZone no longer inherit a stale ?skipGraph for a different repo — both could bypass auto-detect and re-trigger the #2178 hang. ?skipGraph is now a bookmark hint honored only by the initial auto-connect; in-session repo changes auto-detect. - P2 (performance): auto-detect now also skips on edge count (edge-driven force-layout cliff), not just nodes; LARGE_GRAPH_EDGE_THRESHOLD default 50K. - P2 (julik): reset graphMode/chatOnlyNodeCount at the top of switchRepo so a failed switch can't leave a stale chat-only overlay. - P2 (julik): set serverBaseUrl before awaiting handleServerConnect in auto-connect so the Load-graph-anyway button isn't briefly a no-op. - P2 (risk): hide the misleading '0 nodes / 0 edges' stats in chat-only mode (Header + StatusBar). - P2 (performance): guard the GraphCanvas layout effect against the empty chat-only graph. - Tests: edge-threshold decision + connectToServer edge-trigger; load-anyway no longer asserts URL persistence. * fix(web): make Load-graph-anyway cancellable, unmount-safe, fail-safe confirm (#2178) - AbortController + mountedRef: cancel the in-flight download on unmount and guard every post-await setState by the mounted ref (an abort surfaces as a BackendError, not a DOMException AbortError, so name-checks would miss it) - Stale-result guard: a load-anyway that resolves after a concurrent switchRepo no longer clobbers the new repo's graph/mode/count - GraphCanvas confirm fails SAFE (treat as declined) when window.confirm is unavailable or throws, instead of silently proceeding into a large download * fix(web): make the AI agent and chat surface aware of chat-only mode (#2178) - buildDynamicSystemPrompt + createGraphRAGAgent take a chatOnly flag and append a note (both prompt branches) that supersedes VISUAL GROUNDING: the graph isn't loaded, [[Type:Name]] node citations won't highlight, prefer [[path:START-END]] - initializeAgent resolves chatOnly = opts ?? graphModeRef.current==='chatOnly': connect-flow callers (handleServerConnect, switchRepo, loadGraphAnyway re-init) pass it explicitly; lazy/settings re-inits fall back to live mode via the ref - loadGraphAnyway re-inits the agent (chatOnly:false) after a full load so the prompt drops the note - RightPanel shows a chat-only banner so the degradation is visible where AI output renders (en + zh-CN) * fix(web): streaming circuit breaker for graphs with missing size stats (#2178) - GraphTooLargeError + a mid-stream breaker in parseNdjsonGraphResponse: count nodes/relationships as they arrive and abort (cancel reader in try/finally, then throw) the moment either crosses its limit — reusing the existing node/ edge thresholds, no new magic constant. Throwing right after the offending push means a later error record in the same chunk can't pre-empt it. - fetchGraph gains optional maxNodes/maxEdges (off by default → existing callers unchanged). connectToServer arms them only for auto-detect downloads (skipGraph !== false) and catches GraphTooLargeError → chat-only, re-throwing every other error. This backstops the no-stats fail-open path that could otherwise re-trigger the original hang. * chore(autofix): apply prettier + eslint fixes via /autofix command --------- Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gitnexus-web/src/App.tsx | 43 ++- gitnexus-web/src/components/DropZone.tsx | 4 + gitnexus-web/src/components/GraphCanvas.tsx | 69 +++- gitnexus-web/src/components/Header.tsx | 6 +- gitnexus-web/src/components/RightPanel.tsx | 9 + gitnexus-web/src/components/StatusBar.tsx | 6 +- gitnexus-web/src/config/ui-constants.ts | 34 ++ gitnexus-web/src/core/llm/agent.ts | 18 +- gitnexus-web/src/core/llm/context-builder.ts | 24 +- .../src/hooks/app-state/graph.test.tsx | 17 + gitnexus-web/src/hooks/app-state/graph.tsx | 27 ++ gitnexus-web/src/hooks/useAppState.tsx | 163 ++++++++- gitnexus-web/src/lib/apply-connect-result.ts | 46 +++ gitnexus-web/src/lib/graph-load-decision.ts | 84 +++++ gitnexus-web/src/locales/en/chat.json | 3 + gitnexus-web/src/locales/en/graph.json | 11 +- gitnexus-web/src/locales/zh-CN/chat.json | 3 + gitnexus-web/src/locales/zh-CN/graph.json | 11 +- gitnexus-web/src/services/backend-client.ts | 111 ++++++- gitnexus-web/src/vite-env.d.ts | 14 + gitnexus-web/test/unit/agent-prompt.test.ts | 30 ++ .../test/unit/graph-load-decision.test.ts | 145 ++++++++ .../test/unit/load-graph-anyway.test.tsx | 242 ++++++++++++++ .../test/unit/server-connection.test.ts | 314 ++++++++++++++++++ 24 files changed, 1387 insertions(+), 47 deletions(-) create mode 100644 gitnexus-web/src/lib/apply-connect-result.ts create mode 100644 gitnexus-web/src/lib/graph-load-decision.ts create mode 100644 gitnexus-web/test/unit/graph-load-decision.test.ts create mode 100644 gitnexus-web/test/unit/load-graph-anyway.test.tsx diff --git a/gitnexus-web/src/App.tsx b/gitnexus-web/src/App.tsx index 68f7d4968..6141104b5 100644 --- a/gitnexus-web/src/App.tsx +++ b/gitnexus-web/src/App.tsx @@ -10,7 +10,7 @@ import { StatusBar } from './components/StatusBar'; import { FileTreePanel } from './components/FileTreePanel'; import { CodeReferencesPanel } from './components/CodeReferencesPanel'; import { getActiveProviderConfig } from './core/llm/settings-service'; -import { createKnowledgeGraph } from './core/graph/graph'; +import { buildGraphFromConnectResult } from './lib/apply-connect-result'; import { connectToServer, fetchRepos, @@ -21,6 +21,7 @@ import { type BackendRepo, } from './services/backend-client'; import { ERROR_RESET_DELAY_MS } from './config/ui-constants'; +import { parseSkipGraphParam } from './lib/graph-load-decision'; import { formatBackendError } from './i18n/error-messages'; import { useTranslation } from 'react-i18next'; @@ -30,6 +31,8 @@ const AppContent = () => { viewMode, setViewMode, setGraph, + setGraphMode, + setChatOnlyNodeCount, setProgress, setProjectName, progress, @@ -66,15 +69,14 @@ const AppContent = () => { setProjectName(projectName); setCurrentRepo(projectName); - // Build KnowledgeGraph from server data for visualization - const graph = createKnowledgeGraph(); - for (const node of result.nodes) { - graph.addNode(node); - } - for (const rel of result.relationships) { - graph.addRelationship(rel); - } - setGraph(graph); + // Build KnowledgeGraph from server data for visualization. In chat-only + // mode the graph download was skipped, so the shared builder keeps an + // empty (but non-null) graph and flags the mode so the UI shows the + // chat-only empty state, with the node count captured for its notice. + const built = buildGraphFromConnectResult(result); + setGraph(built.graph); + setGraphMode(built.graphMode); + setChatOnlyNodeCount(built.graphMode === 'chatOnly' ? built.nodeCount : null); // Persist the active project in the URL for bookmarkability and F5 refresh resilience const urlObj = new URL(window.location.href); @@ -84,10 +86,11 @@ const AppContent = () => { // Transition directly to exploring view setViewMode('exploring'); - // Initialize agent with backend queries, then start embeddings + // Initialize agent with backend queries, then start embeddings. Pass the + // chat-only flag so the agent's prompt matches the loaded/skipped graph (#2178). try { if (getActiveProviderConfig()) { - await initializeAgent(projectName); + await initializeAgent(projectName, { chatOnly: result.graphSkipped }); } startEmbeddingsWithFallback(); } catch (err) { @@ -97,6 +100,8 @@ const AppContent = () => { [ setViewMode, setGraph, + setGraphMode, + setChatOnlyNodeCount, setProjectName, setCurrentRepo, initializeAgent, @@ -116,6 +121,9 @@ const AppContent = () => { const params = new URLSearchParams(window.location.search); const serverUrlParam = params.get('server'); const projectParam = params.get('project'); + // `?skipGraph=1` forces chat-only, `?skipGraph=0` forces a full graph; + // absent → auto-detect by node count. Bookmarkable / survives F5 (#2178). + const skipGraphParam = parseSkipGraphParam(params.get('skipGraph')); if (!serverUrlParam && !projectParam) return; autoConnectRan.current = true; @@ -162,15 +170,19 @@ const AppContent = () => { }, undefined, projectParam || undefined, - { awaitAnalysis: true }, // enable backend hold-queue for repos still being analyzed + { awaitAnalysis: true, skipGraph: skipGraphParam }, // hold-queue + chat-only control (#2178) ); }; tryConnect() .then(async (result) => { + // Set serverBaseUrl BEFORE handleServerConnect: the latter transitions + // to 'exploring' (rendering the chat-only overlay + its "Load graph + // anyway" button) and then awaits agent init, leaving a window where + // loadGraphAnyway would silently no-op on a still-null serverBaseUrl. + setServerBaseUrl(baseUrl); await handleServerConnect(result); setProgress(null); - setServerBaseUrl(baseUrl); fetchRepos() .then((repos) => setAvailableRepos(repos)) .catch((e) => console.warn('Failed to fetch repo list:', e)); @@ -261,6 +273,9 @@ const AppContent = () => { try { const repos = await fetchRepos(); setAvailableRepos(repos); + // Auto-detect by size for a freshly-analyzed repo (#2178). A stale + // ?skipGraph from a previously-viewed repo must NOT leak in here — + // that would bypass the size guard and could re-trigger the hang. const result = await connectToServer(url, undefined, undefined, repoName); await handleServerConnect(result); setServerBaseUrl(normalizeServerUrl(url)); diff --git a/gitnexus-web/src/components/DropZone.tsx b/gitnexus-web/src/components/DropZone.tsx index 521b64115..389f99869 100644 --- a/gitnexus-web/src/components/DropZone.tsx +++ b/gitnexus-web/src/components/DropZone.tsx @@ -206,6 +206,10 @@ export const DropZone = ({ onServerConnect }: DropZoneProps) => { const abortController = new AbortController(); abortControllerRef.current = abortController; try { + // Landing-screen repo selection auto-detects by size (#2178). The + // ?skipGraph URL param is a bookmark hint for the initial auto-connect + // only; honoring a stale value for a different repo here would risk the + // hang it is meant to prevent. const result = await connectToServer( detectedBackendUrl, (p, downloaded, total) => { diff --git a/gitnexus-web/src/components/GraphCanvas.tsx b/gitnexus-web/src/components/GraphCanvas.tsx index d0880dbe1..70030eb88 100644 --- a/gitnexus-web/src/components/GraphCanvas.tsx +++ b/gitnexus-web/src/components/GraphCanvas.tsx @@ -27,6 +27,8 @@ import type { GraphNode } from 'gitnexus-shared'; import { QueryFAB } from './QueryFAB'; import Graph from 'graphology'; import { useTranslation } from 'react-i18next'; +import { LARGE_GRAPH_NODE_THRESHOLD } from '../config/ui-constants'; +import { shouldConfirmGraphLoad } from '../lib/graph-load-decision'; export interface GraphCanvasHandle { focusNode: (nodeId: string) => void; @@ -55,6 +57,9 @@ export const GraphCanvas = forwardRef((_, ref) => { animatedNodes, graphViewMode, setGraphViewMode, + graphMode, + chatOnlyNodeCount, + loadGraphAnyway, } = useAppState(); const [hoveredNodeName, setHoveredNodeName] = useState(null); @@ -193,7 +198,10 @@ export const GraphCanvas = forwardRef((_, ref) => { // Update Sigma graph when KnowledgeGraph changes useEffect(() => { - if (!graph) return; + // Skip layout work in chat-only mode: `graph` is non-null but empty, the + // overlay covers the canvas, and this guard also future-proofs against a + // transient where a populated graph is set while mode is still chat-only. + if (!graph || graphMode === 'chatOnly') return; let sigmaGraph: Graph; @@ -218,7 +226,7 @@ export const GraphCanvas = forwardRef((_, ref) => { } setSigmaGraph(sigmaGraph); - }, [graph, nodeById, setSigmaGraph, graphViewMode]); + }, [graph, graphMode, nodeById, setSigmaGraph, graphViewMode]); // Update node visibility when filters change useEffect(() => { @@ -256,6 +264,37 @@ export const GraphCanvas = forwardRef((_, ref) => { resetZoom(); }, [setSelectedNode, setSigmaSelectedNode, resetZoom]); + // Chat-only mode (#2178): the graph download was skipped. `chatOnlyNodeCount` + // comes from app state (captured at connect time), so it is authoritative and + // available immediately — not derived from the async `availableRepos` list. + const handleLoadGraphAnyway = useCallback(() => { + // Warn before re-triggering a potentially browser-hanging download. Confirm + // whenever the count is large OR unknown — never silently re-load a graph we + // can't size, which would risk re-introducing the original #2178 hang. Skip + // the prompt only when the count is known to be below the threshold (a small + // repo force-skipped via ?skipGraph=1). + const needsConfirm = shouldConfirmGraphLoad(chatOnlyNodeCount, LARGE_GRAPH_NODE_THRESHOLD); + if (needsConfirm) { + // Fail SAFE, not open: if there's no usable confirm dialog (some embedded + // webviews) or it throws, treat it as declined rather than loading a + // graph we couldn't warn about (#2178). + const canPrompt = typeof window !== 'undefined' && typeof window.confirm === 'function'; + if (!canPrompt) return; + let confirmed = false; + try { + confirmed = window.confirm( + chatOnlyNodeCount != null + ? t('canvas.chatOnly.loadAnywayWarning', { count: chatOnlyNodeCount.toLocaleString() }) + : t('canvas.chatOnly.loadAnywayWarningUnknown'), + ); + } catch { + return; + } + if (!confirmed) return; + } + void loadGraphAnyway(); + }, [chatOnlyNodeCount, loadGraphAnyway, t]); + return (
{/* Background gradient */} @@ -324,6 +363,32 @@ export const GraphCanvas = forwardRef((_, ref) => { className="sigma-container h-full w-full cursor-grab active:cursor-grabbing" /> + {/* Chat-only empty state (#2178): graph download was skipped for a large + project. Chat works normally; offer an explicit "load anyway" escape. */} + {graphMode === 'chatOnly' && ( +
+
+

+ {t('canvas.chatOnly.title')} +

+

+ {chatOnlyNodeCount != null + ? t('canvas.chatOnly.descriptionWithCount', { + count: chatOnlyNodeCount.toLocaleString(), + }) + : t('canvas.chatOnly.description')} +

+

{t('canvas.chatOnly.citationNote')}

+ +
+
+ )} + {/* Hovered node tooltip - only show when NOT selected */} {hoveredNodeName && !sigmaSelectedNode && (
diff --git a/gitnexus-web/src/components/Header.tsx b/gitnexus-web/src/components/Header.tsx index 3dc83301c..03bdd6041 100644 --- a/gitnexus-web/src/components/Header.tsx +++ b/gitnexus-web/src/components/Header.tsx @@ -63,6 +63,7 @@ export const Header = ({ const { projectName, graph, + graphMode, openChatPanel, isRightPanelOpen, rightPanelTab, @@ -467,8 +468,9 @@ export const Header = ({ ✨ - {/* Stats */} - {graph && ( + {/* Stats — hidden in chat-only mode, where the empty-but-non-null graph + would otherwise show a misleading "0 nodes / 0 edges" (#2178). */} + {graph && graphMode !== 'chatOnly' && (
{t('common:counts.nodes', { count: nodeCount })} {t('common:counts.edges', { count: edgeCount })} diff --git a/gitnexus-web/src/components/RightPanel.tsx b/gitnexus-web/src/components/RightPanel.tsx index 193d2c1e6..7038895d8 100644 --- a/gitnexus-web/src/components/RightPanel.tsx +++ b/gitnexus-web/src/components/RightPanel.tsx @@ -23,6 +23,7 @@ export const RightPanel = () => { isRightPanelOpen, setRightPanelOpen, graph, + graphMode, addCodeReference, // LLM / chat state chatMessages, @@ -283,6 +284,14 @@ export const RightPanel = () => {
+ {/* Chat-only notice: the graph wasn't loaded for this large project, so + inline node citations won't pin in the (absent) graph view (#2178). */} + {graphMode === 'chatOnly' && ( +
+ {t('chat:chatOnly.banner')} +
+ )} + {/* Status / errors */} {agentError && (
diff --git a/gitnexus-web/src/components/StatusBar.tsx b/gitnexus-web/src/components/StatusBar.tsx index 4193d4f21..1097aba6a 100644 --- a/gitnexus-web/src/components/StatusBar.tsx +++ b/gitnexus-web/src/components/StatusBar.tsx @@ -5,7 +5,7 @@ import { useTranslation } from 'react-i18next'; import { translateProgressMessage } from '../i18n/progress'; export const StatusBar = () => { - const { graph, progress } = useAppState(); + const { graph, graphMode, progress } = useAppState(); const { t } = useTranslation(['common', 'graph']); const nodeCount = graph?.nodes.length ?? 0; @@ -68,7 +68,9 @@ export const StatusBar = () => { {/* Right - Stats */}
- {graph && ( + {/* Suppress counts in chat-only mode: the empty-but-non-null graph would + otherwise show a misleading "0 nodes / 0 edges" for a large repo (#2178). */} + {graph && graphMode !== 'chatOnly' && ( <> {t('common:counts.nodes', { count: nodeCount })} • diff --git a/gitnexus-web/src/config/ui-constants.ts b/gitnexus-web/src/config/ui-constants.ts index 1bdb9bae1..6dfba2b2b 100644 --- a/gitnexus-web/src/config/ui-constants.ts +++ b/gitnexus-web/src/config/ui-constants.ts @@ -8,6 +8,40 @@ export const DEFAULT_BACKEND_URL = export const DEFAULT_OLLAMA_BASE_URL = 'http://localhost:11434'; export const DEFAULT_OPENROUTER_BASE_URL = 'https://openrouter.ai/api/v1'; +/** + * Default node-count above which the WebUI connects in chat-only mode (skips + * the full graph download). Grounded in sigma.js/graphology prior art: ~10K + * nodes render smoothly, complex-styled rendering struggles past ~5K, and the + * force-layout degrades beyond ~50K edges. GitNexus renders labeled nodes with + * force layout and has ~1.7x more edges than nodes, so the edge cliff is crossed + * around ~25-30K nodes. Override at deploy time via + * window.__GITNEXUS_CONFIG__.largeGraphNodeThreshold. See issue #2178. + */ +const DEFAULT_LARGE_GRAPH_NODE_THRESHOLD = 25_000; + +/** + * Default edge-count above which the WebUI connects in chat-only mode. The + * browser force-layout cliff is edge-driven (degrades beyond ~50K edges), and + * GitNexus graphs carry more edges than nodes, so an edge-heavy but node-light + * repo can still hang even when under the node threshold. Override via + * window.__GITNEXUS_CONFIG__.largeGraphEdgeThreshold. See issue #2178. + */ +const DEFAULT_LARGE_GRAPH_EDGE_THRESHOLD = 50_000; + +const resolveThreshold = (override: number | undefined, fallback: number): number => + // Ignore non-finite, NaN, or non-positive overrides — fall back to the default. + typeof override === 'number' && Number.isFinite(override) && override > 0 ? override : fallback; + +export const LARGE_GRAPH_NODE_THRESHOLD = resolveThreshold( + typeof window !== 'undefined' ? window.__GITNEXUS_CONFIG__?.largeGraphNodeThreshold : undefined, + DEFAULT_LARGE_GRAPH_NODE_THRESHOLD, +); + +export const LARGE_GRAPH_EDGE_THRESHOLD = resolveThreshold( + typeof window !== 'undefined' ? window.__GITNEXUS_CONFIG__?.largeGraphEdgeThreshold : undefined, + DEFAULT_LARGE_GRAPH_EDGE_THRESHOLD, +); + /** Minimum Node.js version required by the gitnexus CLI (injected by Vite from package.json engines). */ declare const __REQUIRED_NODE_VERSION__: string; export const REQUIRED_NODE_VERSION = __REQUIRED_NODE_VERSION__; diff --git a/gitnexus-web/src/core/llm/agent.ts b/gitnexus-web/src/core/llm/agent.ts index 9263cfe2c..c10748fd0 100644 --- a/gitnexus-web/src/core/llm/agent.ts +++ b/gitnexus-web/src/core/llm/agent.ts @@ -33,7 +33,11 @@ import type { AgentStreamChunk, AgentHistoryMessage, } from './types'; -import { type CodebaseContext, buildDynamicSystemPrompt } from './context-builder'; +import { + type CodebaseContext, + buildDynamicSystemPrompt, + CHAT_ONLY_PROMPT_NOTE, +} from './context-builder'; import { DEFAULT_OLLAMA_BASE_URL, DEFAULT_OPENROUTER_BASE_URL } from '../../config/ui-constants'; import { DeepSeekChatOpenAI, @@ -357,14 +361,20 @@ export const createGraphRAGAgent = ( config: ProviderConfig, backend: GraphRAGBackend, codebaseContext?: CodebaseContext, + chatOnly = false, ) => { const model = createChatModel(config); const tools = createGraphRAGTools(backend); - // Use dynamic prompt if context is provided, otherwise use base prompt + // Use dynamic prompt if context is provided, otherwise use base prompt. The + // chat-only note (graph not loaded, #2178) must apply in BOTH branches — when + // codebaseContext is absent, buildDynamicSystemPrompt is never called, so + // append the note here too. const systemPrompt = codebaseContext - ? buildDynamicSystemPrompt(BASE_SYSTEM_PROMPT, codebaseContext) - : BASE_SYSTEM_PROMPT; + ? buildDynamicSystemPrompt(BASE_SYSTEM_PROMPT, codebaseContext, chatOnly) + : chatOnly + ? `${BASE_SYSTEM_PROMPT}${CHAT_ONLY_PROMPT_NOTE}` + : BASE_SYSTEM_PROMPT; // Log the full prompt for debugging if (import.meta.env.DEV) { diff --git a/gitnexus-web/src/core/llm/context-builder.ts b/gitnexus-web/src/core/llm/context-builder.ts index 1deb595cc..3c0cdd77a 100644 --- a/gitnexus-web/src/core/llm/context-builder.ts +++ b/gitnexus-web/src/core/llm/context-builder.ts @@ -413,7 +413,27 @@ export function formatContextForPrompt(context: CodebaseContext): string { * Build the complete dynamic system prompt * Context is appended at the END so core instructions remain at the top */ -export function buildDynamicSystemPrompt(basePrompt: string, context: CodebaseContext): string { +/** + * Note appended in chat-only mode (graph download skipped for a large project, + * #2178). It supersedes the static VISUAL GROUNDING section in BASE_SYSTEM_PROMPT + * so the agent stops claiming the user sees a graph or that node citations + * highlight — neither is true when the in-memory graph is empty. + */ +export const CHAT_ONLY_PROMPT_NOTE = ` + +--- + +## ⚠️ CHAT-ONLY MODE (graph not loaded) +The knowledge graph is NOT loaded in the UI for this project (it was too large to render). This OVERRIDES the VISUAL GROUNDING section above: +- \`[[Type:Name]]\` node citations will NOT highlight anything — avoid relying on them. +- Prefer \`[[path:START-END]]\` file citations, which still resolve and open the file. +- All your tools (search, cypher, grep, read) work normally against the backend; only the visual graph is absent.`; + +export function buildDynamicSystemPrompt( + basePrompt: string, + context: CodebaseContext, + chatOnly = false, +): string { const contextSection = formatContextForPrompt(context); // Append context at the END - keeps core instructions at top for better adherence @@ -422,5 +442,5 @@ export function buildDynamicSystemPrompt(basePrompt: string, context: CodebaseCo --- ## 📦 CURRENT CODEBASE -${contextSection}`; +${contextSection}${chatOnly ? CHAT_ONLY_PROMPT_NOTE : ''}`; } diff --git a/gitnexus-web/src/hooks/app-state/graph.test.tsx b/gitnexus-web/src/hooks/app-state/graph.test.tsx index 12d946ee2..9bbf76d1f 100644 --- a/gitnexus-web/src/hooks/app-state/graph.test.tsx +++ b/gitnexus-web/src/hooks/app-state/graph.test.tsx @@ -19,4 +19,21 @@ describe('GraphState', () => { }); expect(result.current.graphViewMode).toBe('tree'); }); + + it('should default graphMode to "full"', () => { + const { result } = renderHook(() => useGraphState(), { wrapper }); + expect(result.current.graphMode).toBe('full'); + }); + + it('should switch graphMode to "chatOnly" and back', () => { + const { result } = renderHook(() => useGraphState(), { wrapper }); + act(() => { + result.current.setGraphMode('chatOnly'); + }); + expect(result.current.graphMode).toBe('chatOnly'); + act(() => { + result.current.setGraphMode('full'); + }); + expect(result.current.graphMode).toBe('full'); + }); }); diff --git a/gitnexus-web/src/hooks/app-state/graph.tsx b/gitnexus-web/src/hooks/app-state/graph.tsx index aa6524a5f..d02b6d16d 100644 --- a/gitnexus-web/src/hooks/app-state/graph.tsx +++ b/gitnexus-web/src/hooks/app-state/graph.tsx @@ -2,6 +2,9 @@ import { createContext, useContext, useCallback, useMemo, useState, ReactNode } import type { GraphNode, NodeLabel } from 'gitnexus-shared'; import type { KnowledgeGraph } from '../../core/graph/types'; import { DEFAULT_VISIBLE_LABELS, DEFAULT_VISIBLE_EDGES, type EdgeType } from '../../lib/constants'; +import type { GraphMode } from '../../lib/apply-connect-result'; + +export type { GraphMode }; interface GraphStateContextValue { graph: KnowledgeGraph | null; @@ -18,6 +21,22 @@ interface GraphStateContextValue { setHighlightedNodeIds: (ids: Set) => void; graphViewMode: 'force' | 'tree' | 'circles'; setGraphViewMode: (mode: 'force' | 'tree' | 'circles') => void; + /** + * Whether the in-memory graph was downloaded ('full') or skipped for a large + * project ('chatOnly'). In chat-only mode `graph` is an empty-but-non-null + * KnowledgeGraph so existing `graph?.` consumers keep working; this flag is + * the explicit signal that drives the chat-only empty-state UI. See #2178. + */ + graphMode: GraphMode; + setGraphMode: (mode: GraphMode) => void; + /** + * Node count of the connected repo when in chat-only mode (from the connect + * result's repo stats), or null when unknown. Used to size and gate the + * chat-only empty-state notice and its "load anyway" warning without waiting + * on the async `availableRepos` list. See #2178. + */ + chatOnlyNodeCount: number | null; + setChatOnlyNodeCount: (count: number | null) => void; } const GraphStateContext = createContext(null); @@ -30,6 +49,8 @@ export const GraphStateProvider = ({ children }: { children: ReactNode }) => { const [depthFilter, setDepthFilter] = useState(null); const [highlightedNodeIds, setHighlightedNodeIds] = useState>(new Set()); const [graphViewMode, setGraphViewMode] = useState<'force' | 'tree' | 'circles'>('force'); + const [graphMode, setGraphMode] = useState('full'); + const [chatOnlyNodeCount, setChatOnlyNodeCount] = useState(null); const toggleLabelVisibility = useCallback((label: NodeLabel) => { setVisibleLabels((prev) => @@ -59,6 +80,10 @@ export const GraphStateProvider = ({ children }: { children: ReactNode }) => { setHighlightedNodeIds, graphViewMode, setGraphViewMode, + graphMode, + setGraphMode, + chatOnlyNodeCount, + setChatOnlyNodeCount, }), [ graph, @@ -68,6 +93,8 @@ export const GraphStateProvider = ({ children }: { children: ReactNode }) => { depthFilter, highlightedNodeIds, graphViewMode, + graphMode, + chatOnlyNodeCount, ], ); diff --git a/gitnexus-web/src/hooks/useAppState.tsx b/gitnexus-web/src/hooks/useAppState.tsx index 5c118450b..e3a3daaec 100644 --- a/gitnexus-web/src/hooks/useAppState.tsx +++ b/gitnexus-web/src/hooks/useAppState.tsx @@ -10,7 +10,7 @@ import { } from 'react'; import type { GraphNode, NodeLabel, PipelineProgress } from 'gitnexus-shared'; import type { KnowledgeGraph } from '../core/graph/types'; -import { createKnowledgeGraph } from '../core/graph/graph'; +import { buildGraphFromConnectResult } from '../lib/apply-connect-result'; import type { LLMSettings, AgentStreamChunk, @@ -43,7 +43,7 @@ import { ERROR_RESET_DELAY_MS } from '../config/ui-constants'; import i18n from '../i18n'; import { normalizePath } from '../lib/path-resolution'; import { FILE_REF_REGEX, NODE_REF_REGEX } from '../lib/grounding-patterns'; -import { GraphStateProvider, useGraphState } from './app-state/graph'; +import { GraphStateProvider, useGraphState, type GraphMode } from './app-state/graph'; export const AUTO_START_EMBEDDINGS_STORAGE_KEY = 'gitnexus.autoStartEmbeddings'; @@ -127,6 +127,13 @@ interface AppState { graphViewMode: 'force' | 'tree' | 'circles'; setGraphViewMode: (mode: 'force' | 'tree' | 'circles') => void; + // Graph load mode (full download vs chat-only / skipped graph) + graphMode: GraphMode; + setGraphMode: (mode: GraphMode) => void; + // Connected repo's node count while in chat-only mode (null when unknown) + chatOnlyNodeCount: number | null; + setChatOnlyNodeCount: (count: number | null) => void; + // Query state highlightedNodeIds: Set; setHighlightedNodeIds: (ids: Set) => void; @@ -163,6 +170,8 @@ interface AppState { setAvailableRepos: (repos: BackendRepo[]) => void; switchRepo: (repoName: string) => Promise; setCurrentRepo: (repoName: string) => void; + /** Download the full graph for the current repo after a chat-only connect (#2178). */ + loadGraphAnyway: () => Promise; // Worker API (shared across app) runQuery: (cypher: string) => Promise; @@ -195,7 +204,7 @@ interface AppState { // LLM methods refreshLLMSettings: () => void; - initializeAgent: (overrideProjectName?: string) => Promise; + initializeAgent: (overrideProjectName?: string, opts?: { chatOnly?: boolean }) => Promise; sendChatMessage: (message: string) => Promise; stopChatResponse: () => void; clearChat: () => void; @@ -238,6 +247,10 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => { setHighlightedNodeIds, graphViewMode, setGraphViewMode, + graphMode, + setGraphMode, + chatOnlyNodeCount, + setChatOnlyNodeCount, } = useGraphState(); // Right Panel @@ -591,13 +604,24 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => { const chatAbortRef = useRef(null); const chatStateRef = useRef<'idle' | 'streaming' | 'aborting'>('idle'); + // Mirror graphMode into a ref so initializeAgent's deferred callers (lazy chat + // init, settings-driven re-init) can read the current mode without re-creating + // the callback; connect-flow callers pass an explicit chatOnly flag. (#2178) + const graphModeRef = useRef(graphMode); + useEffect(() => { + graphModeRef.current = graphMode; + }, [graphMode]); + const initializeAgent = useCallback( - async (overrideProjectName?: string): Promise => { + async (overrideProjectName?: string, opts?: { chatOnly?: boolean }): Promise => { const config = getActiveProviderConfig(); if (!config) { setAgentError('Please configure an LLM provider in settings'); return; } + // Explicit flag from connect-flow callers (race-safe); otherwise fall back + // to live mode via the ref so deferred callers stay correct too. (#2178) + const chatOnly = opts?.chatOnly ?? graphModeRef.current === 'chatOnly'; setIsAgentInitializing(true); setAgentError(null); @@ -628,7 +652,7 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => { backendReadFile(filePath, { repo }).then((r) => r.content), }; - agentRef.current = createGraphRAGAgent(config, backend, codebaseContext); + agentRef.current = createGraphRAGAgent(config, backend, codebaseContext, chatOnly); setIsAgentReady(true); setAgentError(null); if (import.meta.env.DEV) { @@ -1153,9 +1177,15 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => { setCodeReferences([]); setCodePanelOpen(false); setCodeReferenceFocus(null); + // Reset graph-load mode up front so a FAILED switch can't leave the + // previous repo's stale chat-only overlay showing (#2178). The success + // path re-derives the mode from the connect result below. + setGraphMode('full'); + setChatOnlyNodeCount(null); let connectedRepo: BackendRepo | undefined; let pNameStr = repoName || 'server-project'; + let connectedChatOnly = false; try { const result: ConnectResult = await connectToServer( @@ -1205,10 +1235,14 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => { connectedRepo = result.repoInfo; pNameStr = pName; - const newGraph = createKnowledgeGraph(); - for (const node of result.nodes) newGraph.addNode(node); - for (const rel of result.relationships) newGraph.addRelationship(rel); - setGraph(newGraph); + // In chat-only mode the graph download was skipped; the shared builder + // keeps an empty (but non-null) graph so existing `graph?.` consumers + // stay happy, and reports the mode + node count in lockstep. + const built = buildGraphFromConnectResult(result); + setGraph(built.graph); + setGraphMode(built.graphMode); + setChatOnlyNodeCount(built.graphMode === 'chatOnly' ? built.nodeCount : null); + connectedChatOnly = built.graphMode === 'chatOnly'; } catch (err: unknown) { console.error('Repo switch failed:', err); setProgress({ @@ -1227,9 +1261,13 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => { } if (pNameStr) { - // Persist the selected project in the URL so a refresh re-opens it + // Persist the selected project in the URL so a refresh re-opens it. + // Drop any `?skipGraph` override: a deliberate repo switch should make a + // fresh per-repo decision (auto-detect) on the next refresh rather than + // carry the previous repo's forced mode (#2178). const urlObj = new URL(window.location.href); urlObj.searchParams.set('project', pNameStr); + urlObj.searchParams.delete('skipGraph'); window.history.replaceState(null, '', urlObj.toString()); } @@ -1241,7 +1279,7 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => { // Re-initialize agent with the new repo's graph context try { if (getActiveProviderConfig()) { - await initializeAgent(pNameStr); + await initializeAgent(pNameStr, { chatOnly: connectedChatOnly }); } setViewMode('exploring'); startEmbeddingsWithFallback(); @@ -1261,6 +1299,8 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => { setViewMode, setProjectName, setGraph, + setGraphMode, + setChatOnlyNodeCount, initializeAgent, startEmbeddingsWithFallback, setHighlightedNodeIds, @@ -1276,6 +1316,102 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => { ], ); + // Load the full graph for the current repo after a chat-only connection. + // This is the escape hatch behind the chat-only empty state (#2178). It + // forces `skipGraph: false` so the size-based auto-detect cannot re-skip it. + // The override is session-scoped (deliberately NOT persisted to the URL): a + // persisted `?skipGraph=0` would leak onto a different repo via the other + // connect entry points and could silently re-trigger the hang on refresh. + const loadGraphInFlightRef = useRef(false); + // Cancels the in-flight load-anyway download; mountedRef gates post-await + // state writes so an unmount mid-download can't setState on a dead instance. + const loadGraphAbortRef = useRef(null); + const loadGraphMountedRef = useRef(true); + useEffect(() => { + return () => { + loadGraphMountedRef.current = false; + loadGraphAbortRef.current?.abort(); + }; + }, []); + const loadGraphAnyway = useCallback(async (): Promise => { + if (!serverBaseUrl) return; + // Guard against a double-trigger (rapid double-click or a racing + // programmatic call) starting two concurrent full-graph downloads. + if (loadGraphInFlightRef.current) return; + loadGraphInFlightRef.current = true; + const repo = repoRef.current; + const controller = new AbortController(); + loadGraphAbortRef.current = controller; + + setProgress({ + phase: 'extracting', + percent: 0, + message: i18n.t('common:progress.downloadingGraph'), + detail: i18n.t('common:progress.validating'), + }); + setViewMode('loading'); + + try { + const result = await connectToServer( + serverBaseUrl, + (phase, downloaded, total) => { + if (phase === 'downloading') { + const pct = total ? Math.round((downloaded / total) * 90) + 5 : 50; + const mb = (downloaded / (1024 * 1024)).toFixed(1); + setProgress({ + phase: 'extracting', + percent: pct, + message: i18n.t('common:progress.downloadingGraph'), + detail: i18n.t('common:progress.downloadedMb', { mb }), + }); + } + }, + controller.signal, + repo, + { awaitAnalysis: true, skipGraph: false }, + ); + + // Bail if we unmounted, or if a concurrent switchRepo changed the active + // repo while this load was in flight (the late result must not clobber the + // new repo's state). Guard keyed on the ref — an abort surfaces as a + // BackendError, not a DOMException AbortError. + if (!loadGraphMountedRef.current || repoRef.current !== repo) return; + + const built = buildGraphFromConnectResult(result); + setGraph(built.graph); + setGraphMode(built.graphMode); + // Full download succeeded → leave chat-only mode; clear the cached count. + setChatOnlyNodeCount(built.graphMode === 'chatOnly' ? built.nodeCount : null); + + setProgress(null); + setViewMode('exploring'); + + // The graph is now loaded — re-init the agent so its system prompt drops + // the chat-only note (#2178, KTD2). Guarded on a configured provider, like + // switchRepo; runs inside the mounted/stale guard above. + if (getActiveProviderConfig()) { + await initializeAgent(repo, { chatOnly: false }); + } + } catch (err) { + if (!loadGraphMountedRef.current || repoRef.current !== repo) return; + console.error('Load graph anyway failed:', err); + // Stay in chat-only mode (the overlay reappears) and return to the view. + setProgress(null); + setViewMode('exploring'); + } finally { + if (loadGraphAbortRef.current === controller) loadGraphAbortRef.current = null; + loadGraphInFlightRef.current = false; + } + }, [ + serverBaseUrl, + setProgress, + setViewMode, + setGraph, + setGraphMode, + setChatOnlyNodeCount, + initializeAgent, + ]); + const removeCodeReference = useCallback( (id: string) => { setCodeReferences((prev) => { @@ -1334,6 +1470,10 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => { setDepthFilter, graphViewMode, setGraphViewMode, + graphMode, + setGraphMode, + chatOnlyNodeCount, + setChatOnlyNodeCount, highlightedNodeIds, setHighlightedNodeIds, aiCitationHighlightedNodeIds, @@ -1362,6 +1502,7 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => { setAvailableRepos, switchRepo, setCurrentRepo, + loadGraphAnyway, runQuery, isDatabaseReady, // Embedding state and methods diff --git a/gitnexus-web/src/lib/apply-connect-result.ts b/gitnexus-web/src/lib/apply-connect-result.ts new file mode 100644 index 000000000..1b7028869 --- /dev/null +++ b/gitnexus-web/src/lib/apply-connect-result.ts @@ -0,0 +1,46 @@ +import type { ConnectResult } from '../services/backend-client'; +import type { KnowledgeGraph } from '../core/graph/types'; +import { createKnowledgeGraph } from '../core/graph/graph'; + +/** + * Whether the in-memory graph was downloaded ('full') or skipped for a large + * project ('chatOnly'). Defined here (not in the graph state slice) so the + * shared connect-result builder and the state slice agree on one source. In + * chat-only mode `graph` is an empty-but-non-null KnowledgeGraph so existing + * `graph?.` consumers keep working; the flag drives the chat-only UI. See #2178. + */ +export type GraphMode = 'full' | 'chatOnly'; + +export interface BuiltGraph { + graph: KnowledgeGraph; + graphMode: GraphMode; + /** + * Node count for the connected repo (from `repoInfo.stats.nodes`), or null + * when the backend did not report it. Captured here at connect time so the + * chat-only notice and its size warning have an authoritative value that does + * not depend on the async `availableRepos` list having loaded yet. + */ + nodeCount: number | null; +} + +/** + * Build the in-memory KnowledgeGraph from a connect result and derive the + * graph mode + node count. In chat-only mode (`graphSkipped`) the node/relation + * loops are skipped, leaving an empty-but-non-null graph. + * + * Shared by every connect entry point — App.handleServerConnect, switchRepo, + * and loadGraphAnyway — so the build, the mode flag, and the node count stay in + * lockstep instead of drifting across three near-identical copies. See #2178. + */ +export function buildGraphFromConnectResult(result: ConnectResult): BuiltGraph { + const graph = createKnowledgeGraph(); + if (!result.graphSkipped) { + for (const node of result.nodes) graph.addNode(node); + for (const rel of result.relationships) graph.addRelationship(rel); + } + return { + graph, + graphMode: result.graphSkipped ? 'chatOnly' : 'full', + nodeCount: result.repoInfo.stats?.nodes ?? null, + }; +} diff --git a/gitnexus-web/src/lib/graph-load-decision.ts b/gitnexus-web/src/lib/graph-load-decision.ts new file mode 100644 index 000000000..955e3f297 --- /dev/null +++ b/gitnexus-web/src/lib/graph-load-decision.ts @@ -0,0 +1,84 @@ +/** + * Pure decision logic for the WebUI's chat-only / skip-graph connection mode + * (issue #2178). Kept free of React and network concerns so it can be unit + * tested directly and reused at every connect entry point. + * + * The WebUI hangs on very large projects because the connect flow downloads the + * entire knowledge graph into memory. The AI chat does not need that graph (it + * calls the backend HTTP API directly), so we skip the download when the user + * asked for chat-only mode or when the project is large enough to auto-detect. + */ + +export interface SkipGraphDecisionInput { + /** + * Explicit user/URL choice, if any. `true` forces chat-only, `false` forces a + * full graph download, `undefined` defers to auto-detection by size. + */ + explicit: boolean | undefined; + /** Node count reported by the backend (`repoInfo.stats.nodes`), if known. */ + nodeCount: number | null | undefined; + /** Node auto-detect threshold (LARGE_GRAPH_NODE_THRESHOLD). */ + threshold: number; + /** Edge count reported by the backend (`repoInfo.stats.edges`), if known. */ + edgeCount?: number | null | undefined; + /** Edge auto-detect threshold (LARGE_GRAPH_EDGE_THRESHOLD). */ + edgeThreshold?: number; +} + +const isOver = (count: number | null | undefined, threshold: number | undefined): boolean => + typeof threshold === 'number' && + typeof count === 'number' && + Number.isFinite(count) && + count > threshold; + +/** + * Decide whether to skip the graph download. + * + * - An explicit boolean choice always wins (override in both directions). + * - Otherwise auto-detect: skip when EITHER the node count OR the edge count is + * known and strictly greater than its threshold. Edges matter because the + * browser force-layout cliff is edge-driven and GitNexus graphs carry more + * edges than nodes — an edge-heavy but node-light repo can still hang. + * - Missing/unknown counts fail open to a full download (we never skip purely + * because we couldn't read the size). + */ +export function decideSkipGraph({ + explicit, + nodeCount, + threshold, + edgeCount, + edgeThreshold, +}: SkipGraphDecisionInput): boolean { + if (typeof explicit === 'boolean') return explicit; + return isOver(nodeCount, threshold) || isOver(edgeCount, edgeThreshold); +} + +/** + * Whether to prompt for confirmation before loading the full graph from the + * chat-only escape hatch ("Load graph anyway"). Confirm whenever the node count + * is large OR unknown — never silently re-load a graph we cannot size, which + * would risk re-introducing the original browser hang (#2178). Skip the prompt + * only when the count is known to be at or below the threshold (a small repo + * that was force-skipped via `?skipGraph=1`). + */ +export function shouldConfirmGraphLoad( + nodeCount: number | null | undefined, + threshold: number, +): boolean { + if (typeof nodeCount !== 'number' || !Number.isFinite(nodeCount)) return true; + return nodeCount > threshold; +} + +/** + * Parse the `?skipGraph` URL parameter into the tri-state used by + * {@link decideSkipGraph}. Accepts `1`/`true` (chat-only) and `0`/`false` + * (full graph), case-insensitively. Anything else — including a missing + * parameter — yields `undefined` (auto-detect). + */ +export function parseSkipGraphParam(value: string | null | undefined): boolean | undefined { + if (value == null) return undefined; + const normalized = value.trim().toLowerCase(); + if (normalized === '1' || normalized === 'true') return true; + if (normalized === '0' || normalized === 'false') return false; + return undefined; +} diff --git a/gitnexus-web/src/locales/en/chat.json b/gitnexus-web/src/locales/en/chat.json index 4d976c335..a768877b6 100644 --- a/gitnexus-web/src/locales/en/chat.json +++ b/gitnexus-web/src/locales/en/chat.json @@ -29,6 +29,9 @@ "configureAI": "Configure AI", "connecting": "Connecting" }, + "chatOnly": { + "banner": "Graph not loaded (large project). Chat works normally; inline node citations won't highlight in the graph view." + }, "roles": { "you": "You", "assistant": "Nexus AI" diff --git a/gitnexus-web/src/locales/en/graph.json b/gitnexus-web/src/locales/en/graph.json index 072e47c2d..72c883456 100644 --- a/gitnexus-web/src/locales/en/graph.json +++ b/gitnexus-web/src/locales/en/graph.json @@ -129,7 +129,16 @@ "runLayout": "Run Layout Again", "layoutOptimizing": "Layout optimizing...", "turnOffHighlights": "Turn off all highlights", - "turnOnHighlights": "Turn on AI highlights" + "turnOnHighlights": "Turn on AI highlights", + "chatOnly": { + "title": "Graph not loaded", + "description": "This is a large project, so the graph was skipped to keep the browser responsive. AI chat works normally.", + "descriptionWithCount": "This project has {{count}} nodes, so the graph was skipped to keep the browser responsive. AI chat works normally.", + "citationNote": "While the graph is unloaded, inline file citations from chat won't auto-open in the Code panel.", + "loadAnyway": "Load graph anyway", + "loadAnywayWarning": "This project has {{count}} nodes. Loading the full graph may make the browser slow or unresponsive. Continue?", + "loadAnywayWarningUnknown": "This may be a large project. Loading the full graph may make the browser slow or unresponsive. Continue?" + } }, "processes": { "unknownStep": "Unknown", diff --git a/gitnexus-web/src/locales/zh-CN/chat.json b/gitnexus-web/src/locales/zh-CN/chat.json index f83c5ad75..6804b6b53 100644 --- a/gitnexus-web/src/locales/zh-CN/chat.json +++ b/gitnexus-web/src/locales/zh-CN/chat.json @@ -29,6 +29,9 @@ "configureAI": "配置 AI", "connecting": "连接中" }, + "chatOnly": { + "banner": "图谱未加载(大型项目)。对话功能正常;内联节点引用不会在图谱视图中高亮。" + }, "roles": { "you": "你", "assistant": "Nexus AI" diff --git a/gitnexus-web/src/locales/zh-CN/graph.json b/gitnexus-web/src/locales/zh-CN/graph.json index 7fc7c71e7..6dba980b7 100644 --- a/gitnexus-web/src/locales/zh-CN/graph.json +++ b/gitnexus-web/src/locales/zh-CN/graph.json @@ -129,7 +129,16 @@ "runLayout": "重新运行布局", "layoutOptimizing": "正在优化布局...", "turnOffHighlights": "关闭全部高亮", - "turnOnHighlights": "开启 AI 高亮" + "turnOnHighlights": "开启 AI 高亮", + "chatOnly": { + "title": "图谱未加载", + "description": "这是一个大型项目,已跳过图谱加载以保持浏览器响应。AI 对话功能正常可用。", + "descriptionWithCount": "该项目包含 {{count}} 个节点,已跳过图谱加载以保持浏览器响应。AI 对话功能正常可用。", + "citationNote": "图谱未加载时,对话中的内联文件引用不会自动在代码面板中打开。", + "loadAnyway": "仍然加载图谱", + "loadAnywayWarning": "该项目包含 {{count}} 个节点。加载完整图谱可能导致浏览器变慢或无响应。是否继续?", + "loadAnywayWarningUnknown": "这可能是一个大型项目。加载完整图谱可能导致浏览器变慢或无响应。是否继续?" + } }, "processes": { "unknownStep": "未知", diff --git a/gitnexus-web/src/services/backend-client.ts b/gitnexus-web/src/services/backend-client.ts index b12cf45a9..b6f2e34a5 100644 --- a/gitnexus-web/src/services/backend-client.ts +++ b/gitnexus-web/src/services/backend-client.ts @@ -8,6 +8,8 @@ import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; import { CircuitOpenError, ResilientFetchExhaustedError, resilientFetch } from 'gitnexus-shared'; +import { LARGE_GRAPH_NODE_THRESHOLD, LARGE_GRAPH_EDGE_THRESHOLD } from '../config/ui-constants'; +import { decideSkipGraph } from '../lib/graph-load-decision'; // ── Types ────────────────────────────────────────────────────────────────── @@ -96,6 +98,24 @@ export class BackendError extends Error { } } +/** + * Thrown by the graph stream parser when the streamed node/relationship count + * crosses the size limit mid-download (#2178). It is the backstop for the case + * pre-fetch stats can't cover (absent/stale `stats.nodes`/`stats.edges` on a + * genuinely large repo). `connectToServer` catches it and falls into chat-only + * mode instead of letting the full graph hang the browser. + */ +export class GraphTooLargeError extends Error { + constructor( + message: string, + public readonly nodeCount: number, + public readonly relationshipCount: number, + ) { + super(message); + this.name = 'GraphTooLargeError'; + } +} + // ── SSE Utility ──────────────────────────────────────────────────────────── export interface SSEHandlers { @@ -539,13 +559,18 @@ export const fetchRepoInfo = async ( return { ...data, repoPath: data.repoPath ?? data.path }; }; -/** Fetch the graph (nodes + relationships). Content stripped by default. */ +/** Fetch the graph (nodes + relationships). Content stripped by default. + * `maxNodes`/`maxEdges` arm a streaming circuit breaker (#2178): if the streamed + * count crosses either limit, the download aborts with a GraphTooLargeError + * instead of materializing a graph that would hang the browser. Off by default. */ export const fetchGraph = async ( repo?: string, opts?: { includeContent?: boolean; signal?: AbortSignal; onProgress?: (downloaded: number, total: number | null) => void; + maxNodes?: number; + maxEdges?: number; }, ): Promise<{ nodes: GraphNode[]; relationships: GraphRelationship[] }> => { const params = [repoParam(repo), opts?.includeContent ? 'includeContent=true' : '', 'stream=true'] @@ -558,7 +583,7 @@ export const fetchGraph = async ( const contentType = response.headers.get('Content-Type') || ''; if (contentType.includes('application/x-ndjson')) { - return parseNdjsonGraphResponse(response, opts?.onProgress); + return parseNdjsonGraphResponse(response, opts?.onProgress, opts?.maxNodes, opts?.maxEdges); } if (!opts?.onProgress || !response.body) { @@ -592,6 +617,8 @@ export const fetchGraph = async ( const parseNdjsonGraphResponse = async ( response: Response, onProgress?: (downloaded: number, total: number | null) => void, + maxNodes?: number, + maxEdges?: number, ): Promise<{ nodes: GraphNode[]; relationships: GraphRelationship[] }> => { if (!response.body) { throw new BackendError('No response body', response.status, 'server'); @@ -606,6 +633,14 @@ const parseNdjsonGraphResponse = async ( let buffer = ''; let downloaded = 0; + // Streaming circuit breaker (#2178): enforce the size limits mid-download as a + // backstop when pre-fetch stats were missing. Same `> threshold` comparison as + // decideSkipGraph. Throwing immediately after the offending push means a later + // error record in the same chunk is never reached — the breaker wins. + const overLimit = (): boolean => + (typeof maxNodes === 'number' && nodes.length > maxNodes) || + (typeof maxEdges === 'number' && relationships.length > maxEdges); + const parseLine = (line: string) => { const trimmed = line.trim(); if (!trimmed) return; @@ -628,6 +663,20 @@ const parseNdjsonGraphResponse = async ( } }; + const tripBreaker = async () => { + // Free the socket promptly; never let a cancel rejection mask the breaker. + try { + await reader.cancel(); + } catch { + // ignore — we're aborting anyway + } + throw new GraphTooLargeError( + `Graph exceeds the size limit (nodes=${nodes.length}, relationships=${relationships.length})`, + nodes.length, + relationships.length, + ); + }; + while (true) { const { done, value } = await reader.read(); if (done) break; @@ -640,11 +689,13 @@ const parseNdjsonGraphResponse = async ( buffer = lines.pop() || ''; for (const line of lines) { parseLine(line); + if (overLimit()) await tripBreaker(); } } buffer += decoder.decode(); parseLine(buffer); + if (overLimit()) await tripBreaker(); return { nodes, relationships }; }; @@ -904,6 +955,14 @@ export interface ConnectResult { nodes: GraphNode[]; relationships: GraphRelationship[]; repoInfo: BackendRepo; + /** + * True when the graph download was skipped (chat-only mode) — either because + * the caller asked for it or because the project exceeded the auto-detect + * node threshold. When true, `nodes`/`relationships` are empty and graph + * visualization is unavailable, but AI chat and all backend-API features + * work normally. See issue #2178. + */ + graphSkipped: boolean; } /** @@ -911,13 +970,15 @@ export interface ConnectResult { * Content is NOT included (use readFile/grep for file access). * Pass `awaitAnalysis: true` when the repo may still be cloning/analyzing — * this enables the backend hold-queue and a 5-minute fetch timeout. + * Pass `skipGraph: true`/`false` to force chat-only / full-graph mode; omit it + * to auto-detect from the project's node count (LARGE_GRAPH_NODE_THRESHOLD). */ export async function connectToServer( url: string, onProgress?: (phase: string, downloaded: number, total: number | null) => void, signal?: AbortSignal, repoName?: string, - opts?: { awaitAnalysis?: boolean }, + opts?: { awaitAnalysis?: boolean; skipGraph?: boolean }, ): Promise { const baseUrl = normalizeServerUrl(url); setBackendUrl(baseUrl); @@ -925,11 +986,45 @@ export async function connectToServer( onProgress?.('validating', 0, null); const repoInfo = await fetchRepoInfo(repoName, { awaitAnalysis: opts?.awaitAnalysis }); - onProgress?.('downloading', 0, null); - const { nodes, relationships } = await fetchGraph(repoName, { - signal, - onProgress: (downloaded, total) => onProgress?.('downloading', downloaded, total), + // Decide whether to skip the (potentially huge) graph download. The AI chat + // talks to the backend HTTP API directly and does not need the in-memory + // graph, so for large projects — or when the caller explicitly asked for + // chat-only mode — we connect instantly without materializing the graph. + // repoInfo is already fetched above, so the node-count check costs no extra + // round-trip. See issue #2178. + const skipGraph = decideSkipGraph({ + explicit: opts?.skipGraph, + nodeCount: repoInfo.stats?.nodes, + threshold: LARGE_GRAPH_NODE_THRESHOLD, + edgeCount: repoInfo.stats?.edges, + edgeThreshold: LARGE_GRAPH_EDGE_THRESHOLD, }); - return { nodes, relationships, repoInfo }; + if (skipGraph) { + return { nodes: [], relationships: [], repoInfo, graphSkipped: true }; + } + + // Arm the streaming circuit breaker for auto-detect downloads as a backstop + // for the no-stats fail-open case (#2178). An explicit "load anyway" + // (skipGraph === false) opts out — the user has accepted the cost. + const enforceLimits = opts?.skipGraph !== false; + + onProgress?.('downloading', 0, null); + try { + const { nodes, relationships } = await fetchGraph(repoName, { + signal, + onProgress: (downloaded, total) => onProgress?.('downloading', downloaded, total), + maxNodes: enforceLimits ? LARGE_GRAPH_NODE_THRESHOLD : undefined, + maxEdges: enforceLimits ? LARGE_GRAPH_EDGE_THRESHOLD : undefined, + }); + return { nodes, relationships, repoInfo, graphSkipped: false }; + } catch (err) { + // The breaker tripped mid-stream → fall into chat-only, the same result the + // pre-fetch skip path produces. Re-throw every other error (genuine + // BackendErrors must still surface to the caller's catch). + if (err instanceof GraphTooLargeError) { + return { nodes: [], relationships: [], repoInfo, graphSkipped: true }; + } + throw err; + } } diff --git a/gitnexus-web/src/vite-env.d.ts b/gitnexus-web/src/vite-env.d.ts index 4a8d41b00..cbbe32bf1 100644 --- a/gitnexus-web/src/vite-env.d.ts +++ b/gitnexus-web/src/vite-env.d.ts @@ -3,5 +3,19 @@ interface Window { __GITNEXUS_CONFIG__?: { backendUrl?: string; + /** + * Node-count above which the WebUI connects in chat-only mode by default + * (skips the full graph download to avoid hanging the browser on very + * large projects). Override at deploy time; falls back to + * LARGE_GRAPH_NODE_THRESHOLD in config/ui-constants.ts. See issue #2178. + */ + largeGraphNodeThreshold?: number; + /** + * Edge-count above which the WebUI connects in chat-only mode by default. + * The browser force-layout cliff is edge-driven, so this guards edge-heavy + * repos that fall under the node threshold. Falls back to + * LARGE_GRAPH_EDGE_THRESHOLD in config/ui-constants.ts. See issue #2178. + */ + largeGraphEdgeThreshold?: number; }; } diff --git a/gitnexus-web/test/unit/agent-prompt.test.ts b/gitnexus-web/test/unit/agent-prompt.test.ts index 6c47c6fd2..c2cb1392f 100644 --- a/gitnexus-web/test/unit/agent-prompt.test.ts +++ b/gitnexus-web/test/unit/agent-prompt.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it } from 'vitest'; import { BASE_SYSTEM_PROMPT } from '../../src/core/llm/agent'; +import { buildDynamicSystemPrompt, type CodebaseContext } from '../../src/core/llm/context-builder'; import { createGraphRAGTools, GRAPH_RAG_TOOL_NAMES, @@ -7,6 +8,19 @@ import { } from '../../src/core/llm/tools'; import { NODE_REF_REGEX } from '../../src/lib/grounding-patterns'; +const MINIMAL_CONTEXT: CodebaseContext = { + stats: { + projectName: 'proj', + fileCount: 0, + functionCount: 0, + classCount: 0, + interfaceCount: 0, + methodCount: 0, + }, + hotspots: [], + folderTree: '', +}; + /** Legacy or phantom tool names that must not appear in the system prompt. */ const FORBIDDEN_TOOL_NAMES = [ 'hybrid_search', @@ -80,3 +94,19 @@ describe('BASE_SYSTEM_PROMPT tool parity', () => { expect(BASE_SYSTEM_PROMPT).not.toMatch(/\b(?:use|call|invoke)\s+`?highlight_in_graph/i); }); }); + +describe('buildDynamicSystemPrompt chat-only mode (#2178)', () => { + it('appends a chat-only note that overrides VISUAL GROUNDING when chatOnly', () => { + const prompt = buildDynamicSystemPrompt(BASE_SYSTEM_PROMPT, MINIMAL_CONTEXT, true); + expect(prompt).toContain('CHAT-ONLY MODE'); + expect(prompt).toMatch(/node citations will NOT highlight/i); + expect(prompt).toContain('[[path:START-END]]'); + }); + + it('leaves the prompt unchanged when chatOnly is false/omitted', () => { + const full = buildDynamicSystemPrompt(BASE_SYSTEM_PROMPT, MINIMAL_CONTEXT); + const explicitFalse = buildDynamicSystemPrompt(BASE_SYSTEM_PROMPT, MINIMAL_CONTEXT, false); + expect(full).toBe(explicitFalse); + expect(full).not.toContain('CHAT-ONLY MODE'); + }); +}); diff --git a/gitnexus-web/test/unit/graph-load-decision.test.ts b/gitnexus-web/test/unit/graph-load-decision.test.ts new file mode 100644 index 000000000..d4713a4da --- /dev/null +++ b/gitnexus-web/test/unit/graph-load-decision.test.ts @@ -0,0 +1,145 @@ +import { describe, expect, it } from 'vitest'; +import { + decideSkipGraph, + parseSkipGraphParam, + shouldConfirmGraphLoad, +} from '../../src/lib/graph-load-decision'; + +const THRESHOLD = 25_000; +const EDGE_THRESHOLD = 50_000; + +describe('decideSkipGraph', () => { + it('auto-detects: skips when node count exceeds the threshold', () => { + expect(decideSkipGraph({ explicit: undefined, nodeCount: 300_000, threshold: THRESHOLD })).toBe( + true, + ); + }); + + it('auto-detects: keeps the full graph for small projects', () => { + expect(decideSkipGraph({ explicit: undefined, nodeCount: 500, threshold: THRESHOLD })).toBe( + false, + ); + }); + + it('explicit choice overrides auto-detection in both directions', () => { + // Force chat-only even for a tiny repo. + expect(decideSkipGraph({ explicit: true, nodeCount: 10, threshold: THRESHOLD })).toBe(true); + // Force a full graph even for a huge repo. + expect(decideSkipGraph({ explicit: false, nodeCount: 999_999, threshold: THRESHOLD })).toBe( + false, + ); + }); + + it('uses strictly-greater comparison at the threshold boundary', () => { + expect( + decideSkipGraph({ explicit: undefined, nodeCount: THRESHOLD, threshold: THRESHOLD }), + ).toBe(false); + expect( + decideSkipGraph({ explicit: undefined, nodeCount: THRESHOLD + 1, threshold: THRESHOLD }), + ).toBe(true); + }); + + it('fails open to a full download when the node count is unknown', () => { + expect( + decideSkipGraph({ explicit: undefined, nodeCount: undefined, threshold: THRESHOLD }), + ).toBe(false); + expect(decideSkipGraph({ explicit: undefined, nodeCount: null, threshold: THRESHOLD })).toBe( + false, + ); + expect(decideSkipGraph({ explicit: undefined, nodeCount: NaN, threshold: THRESHOLD })).toBe( + false, + ); + }); + + it('skips on the edge count even when nodes are under the node threshold', () => { + // Edge-heavy, node-light repo: 20K nodes (< 25K) but 80K edges (> 50K). + expect( + decideSkipGraph({ + explicit: undefined, + nodeCount: 20_000, + threshold: THRESHOLD, + edgeCount: 80_000, + edgeThreshold: EDGE_THRESHOLD, + }), + ).toBe(true); + }); + + it('does not skip when both node and edge counts are under their thresholds', () => { + expect( + decideSkipGraph({ + explicit: undefined, + nodeCount: 5_000, + threshold: THRESHOLD, + edgeCount: 10_000, + edgeThreshold: EDGE_THRESHOLD, + }), + ).toBe(false); + }); + + it('explicit choice overrides the edge auto-detect too', () => { + expect( + decideSkipGraph({ + explicit: false, + nodeCount: 1, + threshold: THRESHOLD, + edgeCount: 999_999, + edgeThreshold: EDGE_THRESHOLD, + }), + ).toBe(false); + }); + + it('fails open when edge count is unknown and nodes are under threshold', () => { + expect( + decideSkipGraph({ + explicit: undefined, + nodeCount: 5_000, + threshold: THRESHOLD, + edgeCount: undefined, + edgeThreshold: EDGE_THRESHOLD, + }), + ).toBe(false); + }); +}); + +describe('parseSkipGraphParam', () => { + it('parses affirmative values to true', () => { + expect(parseSkipGraphParam('1')).toBe(true); + expect(parseSkipGraphParam('true')).toBe(true); + expect(parseSkipGraphParam('TRUE')).toBe(true); + expect(parseSkipGraphParam(' true ')).toBe(true); + }); + + it('parses negative values to false', () => { + expect(parseSkipGraphParam('0')).toBe(false); + expect(parseSkipGraphParam('false')).toBe(false); + expect(parseSkipGraphParam('False')).toBe(false); + }); + + it('returns undefined for missing or unrecognized values', () => { + expect(parseSkipGraphParam(null)).toBeUndefined(); + expect(parseSkipGraphParam(undefined)).toBeUndefined(); + expect(parseSkipGraphParam('')).toBeUndefined(); + expect(parseSkipGraphParam('yes')).toBeUndefined(); + expect(parseSkipGraphParam('2')).toBeUndefined(); + }); +}); + +describe('shouldConfirmGraphLoad', () => { + it('confirms for a large repo', () => { + expect(shouldConfirmGraphLoad(300_000, THRESHOLD)).toBe(true); + expect(shouldConfirmGraphLoad(THRESHOLD + 1, THRESHOLD)).toBe(true); + }); + + it('does NOT confirm for a small repo at or below the threshold', () => { + expect(shouldConfirmGraphLoad(500, THRESHOLD)).toBe(false); + expect(shouldConfirmGraphLoad(THRESHOLD, THRESHOLD)).toBe(false); + }); + + it('confirms (fail-safe) when the node count is unknown', () => { + // The key regression guard: an unknown count must NOT silently re-load, + // which would risk re-introducing the #2178 hang. + expect(shouldConfirmGraphLoad(null, THRESHOLD)).toBe(true); + expect(shouldConfirmGraphLoad(undefined, THRESHOLD)).toBe(true); + expect(shouldConfirmGraphLoad(NaN, THRESHOLD)).toBe(true); + }); +}); diff --git a/gitnexus-web/test/unit/load-graph-anyway.test.tsx b/gitnexus-web/test/unit/load-graph-anyway.test.tsx new file mode 100644 index 000000000..10c15c9cc --- /dev/null +++ b/gitnexus-web/test/unit/load-graph-anyway.test.tsx @@ -0,0 +1,242 @@ +import { afterEach, describe, expect, it, vi } from 'vitest'; +import { renderHook, act } from '@testing-library/react'; +import { AppStateProvider, useAppState } from '../../src/hooks/useAppState'; + +afterEach(() => { + vi.restoreAllMocks(); + // Reset the URL mutated by loadGraphAnyway's persistence. + window.history.replaceState(null, '', '/'); +}); + +const repoInfoResponse = () => + new Response( + JSON.stringify({ + name: 'big-repo', + path: '/r/big-repo', + repoPath: '/r/big-repo', + indexedAt: '2026-06-13T00:00:00Z', + stats: { nodes: 300_000, edges: 600_000 }, + }), + { status: 200, headers: { 'Content-Type': 'application/json' } }, + ); + +const graphNdjsonResponse = () => { + const body = + '{"type":"node","data":{"id":"File:a.ts","label":"File","properties":{"name":"a.ts","filePath":"a.ts"}}}\n' + + '{"type":"relationship","data":{"id":"r1","type":"CONTAINS","sourceId":"File:a.ts","targetId":"File:a.ts"}}\n'; + return new Response(body, { + status: 200, + headers: { 'Content-Type': 'application/x-ndjson' }, + }); +}; + +describe('loadGraphAnyway (chat-only escape hatch, #2178)', () => { + it('forces a full graph download and flips graphMode back to full', async () => { + const fetchMock = vi.fn((url: string) => { + if (url.includes('/api/repo')) return Promise.resolve(repoInfoResponse()); + if (url.includes('/api/graph')) return Promise.resolve(graphNdjsonResponse()); + return Promise.resolve( + new Response('{}', { status: 200, headers: { 'Content-Type': 'application/json' } }), + ); + }); + vi.stubGlobal('fetch', fetchMock); + + const { result } = renderHook(() => useAppState(), { wrapper: AppStateProvider }); + + act(() => { + result.current.setServerBaseUrl('http://localhost:4747'); + result.current.setCurrentRepo('big-repo'); + result.current.setGraphMode('chatOnly'); + }); + + await act(async () => { + await result.current.loadGraphAnyway(); + }); + + // Despite the 300K node count, skipGraph:false forces the download. + expect(result.current.graphMode).toBe('full'); + expect(result.current.graph?.nodeCount).toBe(1); + const graphCalls = fetchMock.mock.calls.filter(([u]) => String(u).includes('/api/graph')); + expect(graphCalls.length).toBeGreaterThan(0); + // The override is session-scoped — deliberately NOT persisted to the URL, so + // it cannot leak onto a different repo or re-trigger the hang on F5 (#2178). + expect(window.location.search).not.toContain('skipGraph'); + }); + + it('no-ops when there is no server connection', async () => { + const fetchMock = vi.fn(); + vi.stubGlobal('fetch', fetchMock); + + const { result } = renderHook(() => useAppState(), { wrapper: AppStateProvider }); + + await act(async () => { + await result.current.loadGraphAnyway(); + }); + + expect(fetchMock).not.toHaveBeenCalled(); + }); + + it('guards against a concurrent double-invocation (only one download)', async () => { + const fetchMock = vi.fn((url: string) => { + if (url.includes('/api/repo')) return Promise.resolve(repoInfoResponse()); + if (url.includes('/api/graph')) return Promise.resolve(graphNdjsonResponse()); + return Promise.resolve( + new Response('{}', { status: 200, headers: { 'Content-Type': 'application/json' } }), + ); + }); + vi.stubGlobal('fetch', fetchMock); + + const { result } = renderHook(() => useAppState(), { wrapper: AppStateProvider }); + act(() => { + result.current.setServerBaseUrl('http://localhost:4747'); + result.current.setCurrentRepo('big-repo'); + result.current.setGraphMode('chatOnly'); + }); + + await act(async () => { + // Fire twice synchronously — the second call must be dropped by the guard. + const a = result.current.loadGraphAnyway(); + const b = result.current.loadGraphAnyway(); + await Promise.all([a, b]); + }); + + const graphCalls = fetchMock.mock.calls.filter(([u]) => String(u).includes('/api/graph')); + expect(graphCalls).toHaveLength(1); + }); + + it('stays in chat-only mode when the full-graph download fails', async () => { + const fetchMock = vi.fn((url: string) => { + if (url.includes('/api/repo')) return Promise.resolve(repoInfoResponse()); + if (url.includes('/api/graph')) + return Promise.resolve(new Response('{"error":"boom"}', { status: 500 })); + return Promise.resolve( + new Response('{}', { status: 200, headers: { 'Content-Type': 'application/json' } }), + ); + }); + vi.stubGlobal('fetch', fetchMock); + + const { result } = renderHook(() => useAppState(), { wrapper: AppStateProvider }); + act(() => { + result.current.setServerBaseUrl('http://localhost:4747'); + result.current.setCurrentRepo('big-repo'); + result.current.setGraphMode('chatOnly'); + }); + + await act(async () => { + await result.current.loadGraphAnyway(); + }); + + // Failure leaves the user in chat-only mode (overlay reappears), view restored. + expect(result.current.graphMode).toBe('chatOnly'); + expect(result.current.viewMode).toBe('exploring'); + expect(window.location.search).not.toContain('skipGraph=0'); + }); + + it('discards a stale result when the active repo changed mid-load', async () => { + let resolveGraph: (r: Response) => void = () => {}; + const graphPromise = new Promise((res) => { + resolveGraph = res; + }); + const fetchMock = vi.fn((url: string) => { + if (url.includes('/api/repo')) return Promise.resolve(repoInfoResponse()); + if (url.includes('/api/graph')) return graphPromise; + return Promise.resolve( + new Response('{}', { status: 200, headers: { 'Content-Type': 'application/json' } }), + ); + }); + vi.stubGlobal('fetch', fetchMock); + + const { result } = renderHook(() => useAppState(), { wrapper: AppStateProvider }); + act(() => { + result.current.setServerBaseUrl('http://localhost:4747'); + result.current.setCurrentRepo('repo-A'); + result.current.setGraphMode('chatOnly'); + }); + + let loadPromise: Promise = Promise.resolve(); + act(() => { + loadPromise = result.current.loadGraphAnyway(); // captures repo-A + }); + // A concurrent switch changes the active repo while the load is in flight. + act(() => { + result.current.setCurrentRepo('repo-B'); + }); + await act(async () => { + resolveGraph(graphNdjsonResponse()); + await loadPromise; + }); + + // The stale repo-A result must NOT flip the (now repo-B) view to full. + expect(result.current.graphMode).toBe('chatOnly'); + }); + + it('does not throw or apply state when unmounted mid-load', async () => { + let resolveGraph: (r: Response) => void = () => {}; + const graphPromise = new Promise((res) => { + resolveGraph = res; + }); + const fetchMock = vi.fn((url: string) => { + if (url.includes('/api/repo')) return Promise.resolve(repoInfoResponse()); + if (url.includes('/api/graph')) return graphPromise; + return Promise.resolve( + new Response('{}', { status: 200, headers: { 'Content-Type': 'application/json' } }), + ); + }); + vi.stubGlobal('fetch', fetchMock); + + const { result, unmount } = renderHook(() => useAppState(), { wrapper: AppStateProvider }); + act(() => { + result.current.setServerBaseUrl('http://localhost:4747'); + result.current.setCurrentRepo('big-repo'); + result.current.setGraphMode('chatOnly'); + }); + + let loadPromise: Promise = Promise.resolve(); + act(() => { + loadPromise = result.current.loadGraphAnyway(); + }); + unmount(); // fires cleanup: mountedRef=false + abort + await act(async () => { + resolveGraph(graphNdjsonResponse()); + await loadPromise; // resolves without setState-after-unmount throwing + }); + }); +}); + +describe('switchRepo auto-detect (chat-only, #2178)', () => { + afterEach(() => { + vi.restoreAllMocks(); + window.history.replaceState(null, '', '/'); + }); + + it('enters chat-only mode and captures the node count for a large repo', async () => { + const fetchMock = vi.fn((url: string) => { + if (url.includes('/api/repo')) return Promise.resolve(repoInfoResponse()); + if (url.includes('/api/repos')) + return Promise.resolve( + new Response('[]', { status: 200, headers: { 'Content-Type': 'application/json' } }), + ); + if (url.includes('/api/graph')) return Promise.resolve(graphNdjsonResponse()); + return Promise.resolve( + new Response('{}', { status: 200, headers: { 'Content-Type': 'application/json' } }), + ); + }); + vi.stubGlobal('fetch', fetchMock); + + const { result } = renderHook(() => useAppState(), { wrapper: AppStateProvider }); + act(() => { + result.current.setServerBaseUrl('http://localhost:4747'); + }); + + await act(async () => { + await result.current.switchRepo('big-repo'); + }); + + // 300K nodes > threshold → auto-skip, empty graph, count captured, no graph download. + expect(result.current.graphMode).toBe('chatOnly'); + expect(result.current.graph?.nodeCount).toBe(0); + expect(result.current.chatOnlyNodeCount).toBe(300_000); + const graphCalls = fetchMock.mock.calls.filter(([u]) => String(u).includes('/api/graph')); + expect(graphCalls).toHaveLength(0); + }); +}); diff --git a/gitnexus-web/test/unit/server-connection.test.ts b/gitnexus-web/test/unit/server-connection.test.ts index dc1e79e7c..b5767ec33 100644 --- a/gitnexus-web/test/unit/server-connection.test.ts +++ b/gitnexus-web/test/unit/server-connection.test.ts @@ -1,12 +1,34 @@ import { afterEach, describe, expect, it, vi } from 'vitest'; import { + connectToServer, fetchGraph, getBackendUrl, + GraphTooLargeError, normalizeServerUrl, setBackendUrl, validateBackendUrl, } from '../../src/services/backend-client'; +// ── NDJSON stream helpers for the U3 circuit-breaker tests ── +const ndjsonStream = (lines: string[]): ReadableStream => { + const encoder = new TextEncoder(); + return new ReadableStream({ + start(controller) { + for (const l of lines) controller.enqueue(encoder.encode(l)); + controller.close(); + }, + }); +}; +const ndjsonResponse = (lines: string[]): Response => + new Response(ndjsonStream(lines), { + status: 200, + headers: { 'Content-Type': 'application/x-ndjson' }, + }); +const nodeLine = (i: number): string => + `{"type":"node","data":{"id":"n${i}","label":"Function","properties":{"name":"f${i}"}}}\n`; +const relLine = (i: number): string => + `{"type":"relationship","data":{"id":"r${i}","type":"CALLS","sourceId":"n0","targetId":"n${i}"}}\n`; + describe('normalizeServerUrl', () => { it('adds http:// to localhost', () => { expect(normalizeServerUrl('localhost:4747')).toBe('http://localhost:4747'); @@ -172,6 +194,151 @@ describe('fetchGraph', () => { }); }); +describe('connectToServer skipGraph (chat-only mode)', () => { + const repoInfo = (nodes: number | undefined) => ({ + name: 'big-repo', + path: '/repos/big-repo', + repoPath: '/repos/big-repo', + indexedAt: '2026-06-13T00:00:00Z', + ...(nodes !== undefined ? { stats: { nodes, edges: nodes * 2 } } : {}), + }); + + // Routes /api/repo to the repo info and /api/graph to the supplied handler; + // any other path returns an empty 200 so the breaker stays closed. + const makeFetchMock = (nodes: number | undefined) => { + const graphHandler = vi.fn( + () => + new Response('{"nodes":[],"relationships":[]}', { + status: 200, + headers: { 'Content-Type': 'application/json' }, + }), + ); + const fetchMock = vi.fn((url: string) => { + if (url.includes('/api/repo')) { + return Promise.resolve( + new Response(JSON.stringify(repoInfo(nodes)), { + status: 200, + headers: { 'Content-Type': 'application/json' }, + }), + ); + } + if (url.includes('/api/graph')) { + return Promise.resolve(graphHandler()); + } + return Promise.resolve( + new Response('{}', { status: 200, headers: { 'Content-Type': 'application/json' } }), + ); + }); + return { fetchMock, graphHandler }; + }; + + const graphRequests = (fetchMock: ReturnType) => + fetchMock.mock.calls.filter(([u]: unknown[]) => String(u).includes('/api/graph')); + + it('skips the graph download when skipGraph is true (even for a tiny repo)', async () => { + const { fetchMock } = makeFetchMock(5); + vi.stubGlobal('fetch', fetchMock); + + const result = await connectToServer( + 'http://localhost:4747', + undefined, + undefined, + 'big-repo', + { + skipGraph: true, + }, + ); + + expect(result.graphSkipped).toBe(true); + expect(result.nodes).toEqual([]); + expect(result.relationships).toEqual([]); + expect(result.repoInfo.name).toBe('big-repo'); + expect(graphRequests(fetchMock)).toHaveLength(0); + }); + + it('downloads the graph when skipGraph is false (even for a huge repo)', async () => { + const { fetchMock } = makeFetchMock(300_000); + vi.stubGlobal('fetch', fetchMock); + + const result = await connectToServer( + 'http://localhost:4747', + undefined, + undefined, + 'big-repo', + { + skipGraph: false, + }, + ); + + expect(result.graphSkipped).toBe(false); + expect(graphRequests(fetchMock).length).toBeGreaterThan(0); + }); + + it('auto-detects a large project and skips the graph (no explicit flag)', async () => { + const { fetchMock } = makeFetchMock(300_000); + vi.stubGlobal('fetch', fetchMock); + + const result = await connectToServer('http://localhost:4747', undefined, undefined, 'big-repo'); + + expect(result.graphSkipped).toBe(true); + expect(graphRequests(fetchMock)).toHaveLength(0); + }); + + it('downloads the graph for a small project (no explicit flag)', async () => { + const { fetchMock } = makeFetchMock(500); + vi.stubGlobal('fetch', fetchMock); + + const result = await connectToServer('http://localhost:4747', undefined, undefined, 'big-repo'); + + expect(result.graphSkipped).toBe(false); + expect(graphRequests(fetchMock).length).toBeGreaterThan(0); + }); + + it('fails open to a full download when node stats are missing', async () => { + const { fetchMock } = makeFetchMock(undefined); + vi.stubGlobal('fetch', fetchMock); + + const result = await connectToServer('http://localhost:4747', undefined, undefined, 'big-repo'); + + expect(result.graphSkipped).toBe(false); + expect(graphRequests(fetchMock).length).toBeGreaterThan(0); + }); + + it('auto-detects an edge-heavy repo (nodes under, edges over the threshold)', async () => { + // 10K nodes (< 25K node threshold) but 80K edges (> 50K edge threshold). + const fetchMock = vi.fn((url: string) => { + if (url.includes('/api/repo')) { + return Promise.resolve( + new Response( + JSON.stringify({ + name: 'edgy-repo', + path: '/repos/edgy-repo', + repoPath: '/repos/edgy-repo', + indexedAt: '2026-06-13T00:00:00Z', + stats: { nodes: 10_000, edges: 80_000 }, + }), + { status: 200, headers: { 'Content-Type': 'application/json' } }, + ), + ); + } + return Promise.resolve( + new Response('{}', { status: 200, headers: { 'Content-Type': 'application/json' } }), + ); + }); + vi.stubGlobal('fetch', fetchMock); + + const result = await connectToServer( + 'http://localhost:4747', + undefined, + undefined, + 'edgy-repo', + ); + + expect(result.graphSkipped).toBe(true); + expect(fetchMock.mock.calls.filter(([u]) => String(u).includes('/api/graph'))).toHaveLength(0); + }); +}); + describe('DEFAULT_BACKEND_URL resolution', () => { afterEach(() => { delete window.__GITNEXUS_CONFIG__; @@ -203,6 +370,34 @@ describe('DEFAULT_BACKEND_URL resolution', () => { }); }); +describe('LARGE_GRAPH_NODE_THRESHOLD resolution', () => { + afterEach(() => { + delete window.__GITNEXUS_CONFIG__; + vi.resetModules(); + }); + + it('defaults to 25000 when no config is injected', async () => { + delete window.__GITNEXUS_CONFIG__; + const { LARGE_GRAPH_NODE_THRESHOLD } = await import('../../src/config/ui-constants'); + expect(LARGE_GRAPH_NODE_THRESHOLD).toBe(25_000); + }); + + it('uses a valid positive override', async () => { + window.__GITNEXUS_CONFIG__ = { largeGraphNodeThreshold: 100_000 }; + const { LARGE_GRAPH_NODE_THRESHOLD } = await import('../../src/config/ui-constants'); + expect(LARGE_GRAPH_NODE_THRESHOLD).toBe(100_000); + }); + + it('ignores NaN, zero, and negative overrides (falls back to default)', async () => { + for (const bad of [NaN, 0, -10]) { + window.__GITNEXUS_CONFIG__ = { largeGraphNodeThreshold: bad }; + vi.resetModules(); + const { LARGE_GRAPH_NODE_THRESHOLD } = await import('../../src/config/ui-constants'); + expect(LARGE_GRAPH_NODE_THRESHOLD, `override=${bad}`).toBe(25_000); + } + }); +}); + describe('validateBackendUrl', () => { it('allows http:// URLs', () => { expect(() => validateBackendUrl('http://localhost:4747')).not.toThrow(); @@ -259,3 +454,122 @@ describe('setBackendUrl', () => { expect(getBackendUrl()).toBe('http://localhost:4747'); }); }); + +describe('fetchGraph streaming size breaker (#2178)', () => { + it('throws GraphTooLargeError when node count exceeds maxNodes', async () => { + setBackendUrl('http://localhost:4747'); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue(ndjsonResponse([nodeLine(0), nodeLine(1), nodeLine(2)])), + ); + await expect(fetchGraph('repo', { maxNodes: 2 })).rejects.toBeInstanceOf(GraphTooLargeError); + }); + + it('completes when node count is at or below maxNodes (== not >)', async () => { + setBackendUrl('http://localhost:4747'); + vi.stubGlobal('fetch', vi.fn().mockResolvedValue(ndjsonResponse([nodeLine(0), nodeLine(1)]))); + const result = await fetchGraph('repo', { maxNodes: 2 }); + expect(result.nodes).toHaveLength(2); + }); + + it('trips on the edge counter for a node-light stream', async () => { + setBackendUrl('http://localhost:4747'); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue(ndjsonResponse([nodeLine(0), relLine(1), relLine(2), relLine(3)])), + ); + await expect(fetchGraph('repo', { maxNodes: 1000, maxEdges: 2 })).rejects.toBeInstanceOf( + GraphTooLargeError, + ); + }); + + it('never trips when no limits are passed (default behavior unchanged)', async () => { + setBackendUrl('http://localhost:4747'); + vi.stubGlobal( + 'fetch', + vi + .fn() + .mockResolvedValue(ndjsonResponse([nodeLine(0), nodeLine(1), nodeLine(2), relLine(3)])), + ); + const result = await fetchGraph('repo'); + expect(result.nodes).toHaveLength(3); + expect(result.relationships).toHaveLength(1); + }); + + it('breaker wins over a later error record in the same stream', async () => { + setBackendUrl('http://localhost:4747'); + vi.stubGlobal( + 'fetch', + vi + .fn() + .mockResolvedValue( + ndjsonResponse([ + nodeLine(0), + nodeLine(1), + nodeLine(2), + '{"type":"error","error":"late boom"}\n', + ]), + ), + ); + await expect(fetchGraph('repo', { maxNodes: 2 })).rejects.toBeInstanceOf(GraphTooLargeError); + }); +}); + +describe('connectToServer streaming breaker (no-stats fail-open backstop, #2178)', () => { + afterEach(() => { + delete window.__GITNEXUS_CONFIG__; + vi.resetModules(); + }); + + // Re-import with a tiny threshold so a 3-record stream exercises the breaker. + const setupTinyThreshold = async () => { + window.__GITNEXUS_CONFIG__ = { largeGraphNodeThreshold: 2, largeGraphEdgeThreshold: 2 }; + vi.resetModules(); + const mod = await import('../../src/services/backend-client'); + mod.setBackendUrl('http://localhost:4747'); + return mod; + }; + + const repoNoStats = () => + new Response( + JSON.stringify({ name: 'r', path: '/r', repoPath: '/r', indexedAt: '2026-06-13T00:00:00Z' }), + { status: 200, headers: { 'Content-Type': 'application/json' } }, + ); + + it('falls into chat-only when an auto-detect stream exceeds the threshold (absent stats)', async () => { + const { connectToServer: connect } = await setupTinyThreshold(); + const fetchMock = vi.fn((url: string) => { + if (url.includes('/api/repo')) return Promise.resolve(repoNoStats()); + if (url.includes('/api/graph')) + return Promise.resolve(ndjsonResponse([nodeLine(0), nodeLine(1), nodeLine(2)])); + return Promise.resolve( + new Response('{}', { status: 200, headers: { 'Content-Type': 'application/json' } }), + ); + }); + vi.stubGlobal('fetch', fetchMock); + + const result = await connect('http://localhost:4747', undefined, undefined, 'r'); + expect(result.graphSkipped).toBe(true); + expect(result.nodes).toEqual([]); + expect(result.relationships).toEqual([]); + }); + + it('does NOT enforce the breaker for an explicit load-anyway (skipGraph:false)', async () => { + const { connectToServer: connect } = await setupTinyThreshold(); + const fetchMock = vi.fn((url: string) => { + if (url.includes('/api/repo')) return Promise.resolve(repoNoStats()); + if (url.includes('/api/graph')) + return Promise.resolve(ndjsonResponse([nodeLine(0), nodeLine(1), nodeLine(2)])); + return Promise.resolve( + new Response('{}', { status: 200, headers: { 'Content-Type': 'application/json' } }), + ); + }); + vi.stubGlobal('fetch', fetchMock); + + const result = await connect('http://localhost:4747', undefined, undefined, 'r', { + skipGraph: false, + }); + expect(result.graphSkipped).toBe(false); + expect(result.nodes).toHaveLength(3); + }); +}); From 912285064a52c7947b1b7c445d87d30163029677 Mon Sep 17 00:00:00 2001 From: Minidoracat Date: Sat, 13 Jun 2026 18:52:14 +0800 Subject: [PATCH 11/16] perf(hooks): cmdline-first Linux db-lock scan, drop the lsof fallback (#2180) (#2183) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf(hooks): cmdline-first Linux db-lock scan, drop the lsof fallback (#2180) The probe's Linux scan was O(processes × fds) — stat every fd of every process — so on a busy host it blew its budget and fell through to lsof, which then timed out (~2 s) and fail-closed. Every Grep/Glob/Bash hook spent ~2 s of CPU to conclude 'couldn't tell'. Rewrite linuxProcScanFindGitNexusServer (name kept; return type now tri-state 'owned' | 'not-owned' | 'timeout') as three phases: 0. /proc//comm prefilter — kernel task->comm, never touches the target's memory maps; truncation-safe whitelist match (comm is capped at 15 visible chars). Calibrated to what a real server reports: @ladybugdb/core's worker_threads rename the main thread to 'MainThread', so that is whitelisted alongside the launcher basenames — omitting it would blind the probe to every server. 1. bounded /proc//cmdline read (openSync+readSync, default 16 KiB with a floor of 4 KiB and a bounded escalation up to a hard ceiling) so a D-state holder cannot stall the hook and the mcp/serve mode token is never clipped off a long interpreter path. 2. dev+ino fd match for the 0–2 survivors only. Dispatch: 'owned' and 'timeout' both map to true. Timeout is now fail-closed (overload self-throttle) instead of falling through to lsof; the Linux lsof fallback is removed entirely. End-to-end semantics on busy hosts are unchanged (the old lsof arm also fail-closed there) — the ~2 s of wasted work and the orphan-spawning lsof are what's gone. macOS lsof+ps and Windows Restart Manager paths are untouched. Also: fix the budget parse bug (Number(raw && trim()) treated '0' as 1200; now parseInt-then-validate, with <= 0 an explicit immediate timeout) and add GITNEXUS_HOOK_PROC_ROOT so the Linux scan can be unit tested against a fixture procfs instead of the host's real /proc. Measured on a 583-process host with 6 background gitnexus mcp servers: owner detection 6–12 ms (was ~1216 ms + lsof timeout), ~100x. Tests: new hook-db-lock-probe.test.ts drives all three phases against a fake procfs (comm-truncation safety, Phase 0 trap, 4 KiB-boundary owner-miss guard, budget=0 immediate timeout, EACCES fail-closed) plus a live-/proc e2e that pins the fd-visible lbug-handle property against a real subprocess holder. The lsof/ps owner-detection suites are relaned to macOS (Linux no longer takes that path); the lsof orphan-reaping suite is removed (no lsof is spawned on Linux now) with a rationale note. Note: pre-commit typecheck skipped; remaining tsc errors are pre-existing on main (none in files touched here). * fix(hooks): honest EACCES verdict + real escalation coverage (#2183 review) Addresses the tri-review (maintainer + Codex): - [P2] Phase-2 fd-dir EACCES no longer claims 'owned'. /proc//fd is owner-only (0500), so a cross-user/root gitnexus server serving ANY repo cleared Phase 0+1 and hit EACCES here, and the old catch returned 'owned' — falsely claiming it locks THIS repo's lbug (dev+ino never compared) and permanently suppressing augment. Split the failure shapes: ENOENT -> continue (raced away); EACCES/EPERM and transient EIO/ESTALE -> 'timeout' (unverifiable -> fail-closed, but honest, not a false ownership claim); ENOTDIR/other structural errors -> continue (not a real fd dir). Same fail-closed dispatcher outcome, no false 'owned', plus a GITNEXUS_DEBUG diagnostic so an operator can tell this skip path from a real owner. - [P2] The escalation test now actually iterates the escalation loop: the gitnexus token sits under 4 KB while the mode token is padded past GITNEXUS_HOOK_PROC_CMDLINE_MAX=4096, and a readSync spy asserts >1 read (the old 9 KB-under-16 KB-cap shape read once and never escalated). - escalation loop now re-checks the budget each iteration and returns a distinct timeout sentinel (never '' — an empty string would read as 'not a candidate' and could drop a real owner -> fail-open); the caller maps it to 'timeout'. - GITNEXUS_HOOK_PROC_ROOT is gated to test context so a stray production env export can't disable Linux owner detection (fail-open). - New uid-agnostic spy tests pin every fd-readdir errno branch (EACCES/EPERM/EIO/ESTALE -> timeout, ENOTDIR -> not-owned) regardless of the runner's uid (the disk chmod-000 tests no-op under root). Note: pre-commit typecheck skipped; remaining tsc errors are pre-existing on main (none in files touched here). * fix(hooks): drop the always-true outOfBudget presence guard (CodeQL #2183) CodeQL flagged `typeof outOfBudget === 'function' && outOfBudget()` as unneeded defensive code: readLinuxCmdline has a single caller (linuxProcScanFindGitNexusServer) that always passes the callback, so the typeof guard is dead. Drop it, leaving `if (outOfBudget())`, and note the invariant in the comment. Mirrored in the byte-identical plugin copy. * fix(hooks): parse numeric hook env with Number() so scientific notation works (#2183 review) getCmdlineMaxBytes and resolveLinuxProcBudgetMs parsed their env via Number.parseInt(raw, 10), so a value like "16e3" silently became 16 (parseInt stops at 'e') instead of 16000. Switch both to Number(String(raw).trim()), which honors scientific notation and is stricter on trailing garbage ("123abc" -> NaN -> default) — matching the repo-majority Number()+isFinite env idiom (src/cli/analyze.ts, src/core/embeddings/hf-env.ts). The two functions had DIFFERENT guard skeletons, so a verbatim swap would regress the budget: resolveLinuxProcBudgetMs used `raw != null ?` with no empty-string short-circuit, and Number("")===0 (vs parseInt("")===NaN) would make a set-but-empty GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS="" resolve to budget 0 => immediate fail-CLOSED timeout => augment permanently skipped. Added the `&& String(raw).trim()` guard so ''/whitespace fall to the 1200 default while "0" still parses to the deliberate #2180 immediate-timeout vector. Exported both helpers for white-box tests (the values are otherwise only observable indirectly through scan timing) and added platform-independent coverage: "16e3"->16000, ""/whitespace->1200 (the regression guard), "0"->0, "123abc"/unset->1200, cmdline "8e3"->8000, "2e3"/""/unset->16384. Both byte-identical hook-db-lock-probe.cjs copies updated together. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(hooks): allocUnsafe the per-chunk cmdline read buffer (#2183 review) readLinuxCmdline allocated each per-chunk read buffer with Buffer.alloc(chunkCap), zero-filling memory that readSync immediately and fully overwrites. Switch the hot read buffer to Buffer.allocUnsafe — safe because readSync initializes exactly [0, bytes), only buf.subarray(0, bytes) is consumed, and Buffer.concat deep-copies that slice into `collected`, so the uninitialized tail can never reach the decoded cmdline. The zero-length `collected = Buffer.alloc(0)` is left unchanged (allocUnsafe gains nothing on a 0-length buffer). The existing D3 multi-chunk decode tests cover the read path and stay green. Both byte-identical hook-db-lock-probe.cjs copies updated together. Co-Authored-By: Claude Opus 4.8 (1M context) * test(hooks): harden the live /proc owner-detection e2e against CI flake (#2183 review) Two flake mechanisms, fixed without weakening what the e2e proves: - Holder readiness (the genuine false-FAIL): the pid-file poll was 200x25ms=5s; a loaded runner can be slow to spawn the child, tripping expect(holderPid).toBeGreaterThan(0). Widened to ~10s and raised the per-test timeout 20s -> 40s. - Scan budget (kept the assertion honest): the live scan ran at the default 1200ms. Because the dispatcher maps a budget 'timeout' to owned=TRUE, a busy host exhausting 1200ms before reaching the holder would make the assertion pass for the WRONG reason (a hollow timeout, not real fd-visible detection). Set a generous explicit 10000ms budget via the existing setEnv() helper so the module afterEach restores it (replacing the raw `delete process.env...` that bypassed env tracking). Raised the coarse timing regression guard to sit ABOVE the budget (5000 -> 15000) so a legitimately-slow-but-correct scan can't trip it. The load-bearing asserts (dev+ino fd-visibility precheck, owned===true for our own lbug) are unchanged. Verified the e2e executes (not skipped) on Linux. Co-Authored-By: Claude Opus 4.8 (1M context) * chore(changelog): empty the root CHANGELOG [Unreleased] section Per maintainer request, nothing should sit under [Unreleased] in the root CHANGELOG.md (the release-owned changelog is gitnexus/CHANGELOG.md, whose [Unreleased] is already empty). Removes all three accumulated blocks — Fixed (#2163), Performance (#2180), Changed (KuzuDB->LadybugDB) — leaving only the [Unreleased] header above [1.5.3]. Pure removal; no release sections touched. Co-Authored-By: Claude Opus 4.8 (1M context) --------- Co-authored-by: Gergő Magyar Co-authored-by: Claude Opus 4.8 (1M context) --- CHANGELOG.md | 11 - .../hooks/hook-db-lock-probe.cjs | 365 ++++++++- gitnexus/hooks/claude/hook-db-lock-probe.cjs | 365 ++++++++- gitnexus/scripts/cross-platform-tests.ts | 1 + .../integration/antigravity-hook-e2e.test.ts | 8 +- gitnexus/test/unit/hook-db-lock-probe.test.ts | 723 ++++++++++++++++++ gitnexus/test/unit/hooks.test.ts | 340 ++------ gitnexus/test/utils/hook-test-helpers.ts | 51 ++ 8 files changed, 1537 insertions(+), 327 deletions(-) create mode 100644 gitnexus/test/unit/hook-db-lock-probe.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index efb1b531f..1bf60be80 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,17 +4,6 @@ All notable changes to GitNexus will be documented in this file. ## [Unreleased] -### Fixed - -- **Hook db-lock probe no longer strands unkillable `lsof`/`ps` orphans** — the probe's `lsof`/`ps` subprocesses are now wrapped in a self-tested coreutils `timeout`/`gtimeout` (`timeout -k 1 …`), so a hook SIGKILLed by the runner's 10s timeout can no longer leave `lsof` running forever (orphan lifetime bounded at ~3s); `acquireHookSlot` now also gates the probe itself, capping concurrent probes at 3 per repo. Opt out with `GITNEXUS_HOOK_TIMEOUT_PATH=disabled`. (#2163) -- **Hook augment CLI no longer strands orphans either** — `runGitNexusCli` in the Claude, plugin, and Antigravity hook adapters now wraps the `gitnexus augment` subprocess (the longest-lived hook child: 7s local / 12s npx inner budgets) in the same self-tested coreutils `timeout` guard as the probe's `lsof`/`ps`, with a budget of ceil(inner/1000)+1 seconds — strictly above the inner `spawnSync` timeout, so on the supervised path Node's SIGTERM still fires first and observable behavior is unchanged. Once the hook itself has been SIGKILLed the guard takes over, with per-branch semantics: on the direct-exec branches (the CLI is the guard's child) it SIGTERMs at budget and `-k 1` SIGKILLs 1s later; on the npx branches (the CLI is a *grandchild* behind npx) it uses `-s KILL`, SIGKILLing the whole process group at budget — a TERM-first guard there would only kill the obedient npx parent and exit before its `-k` escalation fires, stranding a SIGTERM-immune CLI. Two npx-branch caveats remain (both no worse than the pre-fix behavior, where the grandchild received no signal at all): the group-wide KILL is coreutils semantics, so a busybox `timeout` — which passes the self-test — still signals only its direct child and cannot reach the grandchild; and on the supervised path (hook alive, inner `spawnSync` timeout SIGTERMs the guard) coreutils forwards TERM rather than KILL, so a SIGTERM-immune CLI grandchild is still not reaped there. The guard self-test now also requires exit-status propagation (`sh -c 'exit 42'` must yield 42), so an always-exit-0 stub at `GITNEXUS_HOOK_TIMEOUT_PATH` can no longer be adopted and silently swallow the probe and augment. Windows, `GITNEXUS_HOOK_TIMEOUT_PATH=disabled`, and Unix hosts with no usable coreutils `timeout`/`gtimeout` at all (e.g. macOS without Homebrew coreutils) keep the exact pre-wrap unguarded invocation — every guard-less Unix run, whatever the reason (disabled, nothing usable, probe version skew), is now diagnosed once per hook run under `GITNEXUS_DEBUG`. The Cursor hook is not wrapped yet (it does not install the probe helper) but now reports its slot-saturated skip under `GITNEXUS_DEBUG`. (#2163 follow-up) - -### Changed -- Migrated from KuzuDB to LadybugDB v0.15 (`@ladybugdb/core`, `@ladybugdb/wasm-core`) -- Renamed all internal paths from `kuzu` to `lbug` (storage: `.gitnexus/kuzu` → `.gitnexus/lbug`) -- Added automatic cleanup of stale KuzuDB index files -- LadybugDB v0.15 requires explicit VECTOR extension loading for semantic search - ## [1.5.3] - 2026-04-01 ### Added diff --git a/gitnexus-claude-plugin/hooks/hook-db-lock-probe.cjs b/gitnexus-claude-plugin/hooks/hook-db-lock-probe.cjs index de0fa5e85..948581110 100644 --- a/gitnexus-claude-plugin/hooks/hook-db-lock-probe.cjs +++ b/gitnexus-claude-plugin/hooks/hook-db-lock-probe.cjs @@ -3,14 +3,36 @@ * with a command line that looks like a GitNexus MCP/serve server? * * Backends (no user-installed Sysinternals): - * - Linux: scan procfs under /proc (per-PID fd entries) via stat(2) (dev+inode); works without lsof; - * optional lsof fallback when proc scan finds nothing. + * - Linux: cmdline-first procfs scan under /proc, no lsof at all (#2180). Three + * phases, cheapest first: (0) read /proc//comm — a tiny task->comm read + * that never touches the target's mm — and keep only PIDs whose comm is a + * plausible node/gitnexus server; (1) read up to GITNEXUS_HOOK_PROC_CMDLINE_MAX + * bytes of /proc//cmdline via openSync+readSync (bounded, so a D-state + * holder stuck on mmap_lock or a giant argv can't wedge the hook) and prefilter + * with isGitNexusServerCommand; (2) only for the 0..N survivors, stat their + * /proc//fd/* and compare dev+inode against the target lbug. The lbug + * handle is fd-visible (a @ladybugdb/core property), so this finds every real + * owner without scanning every fd of every process. * - macOS / *BSD / etc.: trusted lsof + ps (absolute paths first). * - Windows: Restart Manager (rstrtmgr) via bundled PowerShell script + * Win32_Process for command lines; trusted powershell.exe under %SystemRoot%. * - * Fail-open on most errors; fail-closed only on lsof ETIMEDOUT (Unix) or - * PowerShell ETIMEDOUT (Windows), matching the hook contract. + * Fail matrix: + * - Linux proc scan: owner found -> fail-closed (skip augment); budget exhausted + * (GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS) -> fail-CLOSED (#2180). This is a + * deliberate change from the old "timeout -> fail-open then try lsof" path. + * End-to-end the busy-host outcome is unchanged: the old code's lsof fallback + * ETIMEDOUT'd on the very hosts where the scan ran out of budget and ALSO + * failed closed there — the lsof leg only ever added 1-2s of dead work plus + * the orphan-storm risk it caused (#2163). What changes is that an overloaded + * host now self-throttles immediately (the throttle the incident needed) + * instead of paying for a doomed lsof. Mid-load hosts that used to fall + * through to a successful lsof now answer from the scan directly (faster) or, + * if even the scan can't finish in budget, fail closed (self-throttle) — a + * bounded, documented tradeoff, never an orphan. + * - macOS / other Unix: fail-open on most errors; fail-closed only on lsof + * ETIMEDOUT, matching the hook contract. + * - Windows: fail-closed only on PowerShell ETIMEDOUT. * * Unix subprocess containment contract (#2163): * - lsof/ps are wrapped in coreutils `timeout`/`gtimeout` when a working @@ -46,6 +68,16 @@ function isGitNexusServerCommand(command) { return hasServerMode && hasGitNexus; } +// GITNEXUS_DEBUG-gated stderr diagnostics. Reuses the exact gating predicate the +// Windows ps1-load warning already uses (===' 1' / ==='true') so there is one +// debug convention in this file, and writes via process.stderr.write (NOT a +// spawn) so it never perturbs the windowsHide spawn-count invariant. +function debugLog(msg) { + if (process.env.GITNEXUS_DEBUG === '1' || process.env.GITNEXUS_DEBUG === 'true') { + process.stderr.write(`[GitNexus hook] ${msg}\n`); + } +} + function resolveHookBinary(tool) { const envKey = tool === 'lsof' ? 'GITNEXUS_HOOK_LSOF_PATH' : 'GITNEXUS_HOOK_PS_PATH'; const fromEnv = process.env[envKey]; @@ -242,59 +274,325 @@ function hasGitNexusServerOwnerWindows(dbPathAbs, myPid) { return false; } -function readLinuxCmdline(pidStr) { +// The procfs root every Linux scan path reads from. Production is always /proc; +// GITNEXUS_HOOK_PROC_ROOT only exists so unit tests can inject a fixture tree +// (comm + cmdline + fd symlinks) and assert the three-phase logic without +// scanning the real, ~hundreds-of-process /proc of the test host. +// +// Test-only gate (F4): the override is honored ONLY under a test runner — +// vitest injects VITEST="true" and NODE_ENV="test" into every worker (verified; +// a production hook is `node .cjs` with neither set). Without the gate, a +// production env that accidentally leaked GITNEXUS_HOOK_PROC_ROOT (pointing at an +// empty/bad tree) would make readdirSync find no pids -> 'not-owned' -> Linux +// owner detection silently OFF (fail-OPEN: augment races the real server for the +// lbug, the #1492 class). Gating to the test signal makes that leak inert in +// production (always /proc) while the fake-procfs unit tests, which run under +// vitest, still inject freely. Unset env (or non-test context) => /proc, so the +// production path is byte-for-byte the historical behavior. +function isTestContext() { + return ( + process.env.VITEST === 'true' || process.env.VITEST === '1' || process.env.NODE_ENV === 'test' + ); +} +function getProcRoot() { + if (!isTestContext()) return '/proc'; + const raw = process.env.GITNEXUS_HOOK_PROC_ROOT; + return raw && String(raw).trim() ? String(raw) : '/proc'; +} + +// Max bytes read from /proc//cmdline in Phase 1. Bounded by default so a +// D-state holder wedged on mmap_lock, or a process with a pathological multi-MB +// argv, can't stall the hook. 16 KiB comfortably clears a realistic +// `node mcp` line +// (the `mcp`/`serve` mode token lives at the very tail, so the cap must be large +// enough to reach it — see PROC_CMDLINE_FLOOR escalation below). Overridable for +// tests; never goes below PROC_CMDLINE_FLOOR. +const PROC_CMDLINE_FLOOR = 4096; +function getCmdlineMaxBytes() { + const raw = process.env.GITNEXUS_HOOK_PROC_CMDLINE_MAX; + // Number() (not parseInt) so "8e3" reads as 8000, not 8 (parseInt stops at + // 'e'). The `raw && String(raw).trim()` guard keeps empty/whitespace on the + // default; trailing garbage ("8abc") now -> NaN -> default (stricter). + const n = raw && String(raw).trim() ? Number(String(raw).trim()) : NaN; + if (Number.isFinite(n) && n >= PROC_CMDLINE_FLOOR) return n; + return 16384; +} + +// Phase 0 comm prefilter. /proc//comm is the kernel task->comm string, +// capped at 16 bytes INCLUDING the trailing NUL — i.e. at most 15 visible +// chars, truncated by the kernel with no marker. So a process whose real name +// is longer than 15 chars shows a 15-char prefix here. The match below is +// therefore truncation-safe in BOTH directions (a whitelist name that is a +// prefix of comm, or comm that is a prefix of a whitelist name, both count) to +// guarantee we never drop a real owner at this cheap stage — Phase 2's dev+ino +// fd check is the real authority; Phase 0/1 only exist to skip the overwhelming +// majority (kernel threads, shells, editors) cheaply. +// +// The whitelist is calibrated against what a real `gitnexus mcp`/`serve` server +// actually reports for comm. Observed on production hosts: the server renames +// its main thread, so comm reads `MainThread` (via @ladybugdb/core's +// worker_threads setup), NOT `node` — omitting it would blind the probe to +// every real server (#1492-class owner miss). We also keep the plausible +// launcher/runtime basenames in case a future build does not rename the thread. +// Conservative by design: over-collecting a few extra candidates only costs a +// bounded number of Phase 1 cmdline reads. +const COMM_CANDIDATES = ['node', 'gitnexus', 'bun', 'deno', 'npm', 'npx', 'MainThread']; +function commLooksLikeServer(comm) { + const c = comm.trim(); + if (!c) return false; + for (const name of COMM_CANDIDATES) { + if (name === c || name.startsWith(c) || c.startsWith(name)) return true; + } + return false; +} + +function readProcComm(procRoot, pidStr) { try { - return fs.readFileSync(`/proc/${pidStr}/cmdline`, 'utf8').replace(/\0+/g, ' ').trim(); + return fs + .readFileSync(path.join(procRoot, pidStr, 'comm'), 'utf8') + .replace(/\0+/g, '') + .trim(); } catch { return ''; } } -function linuxProcScanFindGitNexusServer(dbPathAbs, myPid) { +// Timeout sentinel for readLinuxCmdline (F3). MUST be distinct from the +// "unreadable/empty" return value (''): '' flows through isGitNexusServerCommand +// as a NON-candidate (both regexes are false on ''), so the Phase 1 caller +// `continue`s past it — correct for a raced/openSync-failed pid, but a FAIL-OPEN +// bug if it ever meant "I ran out of budget mid-read" (a real owner whose +// escalation timed out would be silently dropped, racing the lbug -> #1492). A +// unique Symbol can never collide with any cmdline string, so the caller can +// branch on it explicitly and map a mid-read timeout to the tri-state 'timeout' +// (fail-CLOSED) instead of swallowing it as a non-candidate. +const CMDLINE_TIMEOUT = Symbol('gitnexus.cmdline.timeout'); + +// Bounded /proc//cmdline read for Phase 1. openSync+readSync (not +// readFileSync) so a D-state holder cannot stall the hook on a huge or +// never-EOF argv: we read at most `cap` bytes and stop. cmdline separates argv +// with NULs; convert to spaces for isGitNexusServerCommand. +// +// Owner-miss guard for the 4 KB cap: the `gitnexus` token usually sits in the +// first path component while the `mcp`/`serve` mode token is the LAST argv, so +// a naive 4 KB read could clip the mode token off a server launched with a very +// long interpreter path and silently miss a real owner. We mitigate two ways: +// (a) the default cap (16 KiB) already clears realistic lines; (b) if the first +// read fills the cap AND already contains the `gitnexus` token but no mode +// token yet, we keep reading in bounded chunks (up to a hard ceiling) until the +// mode token appears or the file ends — so a genuine server is never missed for +// want of a few more bytes, while non-candidates still pay only the initial +// bounded read. +// +// Budget (F3): the escalation loop above is the one place a SINGLE pathological +// candidate could read up to HARD_CEIL (256 KiB) before the next scan-level +// budget check, weakening the timeout contract. `outOfBudget` (the scan's shared +// deadline callback) is checked once per escalation iteration; on expiry we +// return CMDLINE_TIMEOUT (NOT '') so the caller can fail-closed honestly rather +// than mistake the partial read for a non-candidate. Reads that simply can't +// open / error out still return '' (genuinely "not a readable candidate"). +function readLinuxCmdline(procRoot, pidStr, cap, outOfBudget) { + const file = path.join(procRoot, pidStr, 'cmdline'); + let fd; + try { + fd = fs.openSync(file, 'r'); + } catch { + return ''; + } + try { + const HARD_CEIL = 262144; // 256 KiB absolute ceiling for the escalation path + let collected = Buffer.alloc(0); + let offset = 0; + let chunkCap = cap; + for (;;) { + // allocUnsafe is safe here: readSync fills exactly [0, bytes), only + // buf.subarray(0, bytes) is consumed, and Buffer.concat deep-copies that + // slice into `collected`, so the uninitialized tail never reaches decode. + const buf = Buffer.allocUnsafe(chunkCap); + const bytes = fs.readSync(fd, buf, 0, chunkCap, offset); + if (bytes <= 0) break; + collected = Buffer.concat([collected, buf.subarray(0, bytes)]); + offset += bytes; + const text = collected.toString('utf8').replace(/\0+/g, ' '); + // Stop early when we can already decide "owner": has both the gitnexus + // token and a mode token. Keep going only when gitnexus is present but + // the mode token might be just past the boundary. + const hasGitNexus = + /(?:^|[/\\\s])gitnexus(?:\.cmd)?(?:\s|$)/.test(text) || + /node_modules[/\\]gitnexus[/\\]/.test(text); + const hasMode = /(?:^|\s)(mcp|serve)(?:\s|$)/.test(text); + if (hasMode) break; // decided (positive); isGitNexusServerCommand re-checks below + if (bytes < chunkCap) break; // EOF: full cmdline read, definitive + if (!hasGitNexus) break; // not a candidate; do not escalate the read + if (offset >= HARD_CEIL) break; // bounded escalation only + // Budget gate the escalation: a single huge-argv candidate must not burn + // the whole scan deadline before we re-check. Return the timeout sentinel + // (never '') so the caller fails closed instead of treating us as a + // non-candidate. The sole caller (linuxProcScanFindGitNexusServer) always + // passes outOfBudget, so no presence guard is needed. + if (outOfBudget()) return CMDLINE_TIMEOUT; + chunkCap = cap; // keep reading more in cap-sized chunks + } + return collected.toString('utf8').replace(/\0+/g, ' ').trim(); + } catch { + return ''; + } finally { + try { + fs.closeSync(fd); + } catch { + /* ignore */ + } + } +} + +function resolveLinuxProcBudgetMs() { const raw = process.env.GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS; - const budget = Number(raw && String(raw).trim()) ? Number.parseInt(String(raw), 10) : 1200; + // Gate on the STRING's emptiness, NOT the parsed number's truthiness — the + // old `Number(raw && trim()) ? ... : 1200` form treated "0" as falsy and + // silently fell back to 1200 (#2180). Use Number() (not parseInt) so "16e3" + // reads as 16000, not 16 (parseInt stops at 'e'). The `&& String(raw).trim()` + // guard is load-bearing: without it a set-but-empty/whitespace value would be + // `Number("")===0` => budget 0 => immediate fail-CLOSED timeout (augment + // permanently skipped). With it, ''/whitespace => NaN => 1200 default, while a + // finite "0" still parses to an explicit, deterministic "no budget" => + // immediate timeout. Non-numeric / unset => default 1200. + const n = raw != null && String(raw).trim() ? Number(String(raw).trim()) : NaN; + if (!Number.isFinite(n)) return 1200; + return n; // may be <= 0, meaning "out of budget on the first check" +} + +// Returns one of: 'owned' (a non-self process with a GitNexus-server cmdline +// holds the target lbug fd), 'not-owned' (scan completed, no such owner), or +// 'timeout' (the per-scan budget was exhausted before a verdict). The name is +// pinned by a source-contract test; only the return TYPE changed (#2180: +// boolean -> tri-state, so the dispatcher can fail-closed on 'timeout'). +function linuxProcScanFindGitNexusServer(dbPathAbs, myPid) { + const budget = resolveLinuxProcBudgetMs(); + // A non-positive budget is an explicit, deterministic "no time to scan" => + // immediate timeout (the #2180 test vector, and the only correct reading of + // the fixed parse: "0" must NOT mean 1200). Returning before any procfs read + // keeps it instantaneous regardless of host load. + if (budget <= 0) return 'timeout'; + const procRoot = getProcRoot(); + const cmdlineCap = getCmdlineMaxBytes(); const start = Date.now(); + const outOfBudget = () => Date.now() - start > budget; + let targetStat; try { targetStat = fs.statSync(dbPathAbs); } catch { - return false; + // Caller already existsSync'd the path; a stat failure here is a transient + // race, treat as no owner (historical semantics). + return 'not-owned'; } + let procEntries; try { - procEntries = fs.readdirSync('/proc', { withFileTypes: true }); + procEntries = fs.readdirSync(procRoot, { withFileTypes: true }); } catch { - return false; + return 'not-owned'; } + + // Phase 0 + Phase 1: collect the few PIDs whose comm AND cmdline look like a + // GitNexus server, without touching any fd yet. + const candidates = []; for (const ent of procEntries) { - if (Date.now() - start > budget) return false; + if (outOfBudget()) return 'timeout'; if (!ent.isDirectory() || !/^\d+$/.test(ent.name)) continue; const pid = Number.parseInt(ent.name, 10); if (!Number.isFinite(pid) || pid === myPid) continue; - const fdDir = path.join('/proc', ent.name, 'fd'); + + // Phase 0: cheap comm prefilter. + const comm = readProcComm(procRoot, ent.name); + if (!comm) continue; // unreadable comm (kernel thread, raced exit) -> skip + if (!commLooksLikeServer(comm)) continue; + + // Phase 1: bounded cmdline read + isGitNexusServerCommand prefilter. + if (outOfBudget()) return 'timeout'; + const cmdline = readLinuxCmdline(procRoot, ent.name, cmdlineCap, outOfBudget); + // F3: a mid-read budget timeout returns the CMDLINE_TIMEOUT sentinel (a + // Symbol, never a string). Fail CLOSED on it rather than letting it fall + // through isGitNexusServerCommand as a non-candidate — a real owner whose + // escalation timed out must not be silently dropped (would fail-OPEN). + if (cmdline === CMDLINE_TIMEOUT) return 'timeout'; + if (!isGitNexusServerCommand(cmdline)) continue; + candidates.push(ent.name); + } + + // Phase 2: only now stat the fds of the (typically 0-2) survivors. + for (const pidStr of candidates) { + if (outOfBudget()) return 'timeout'; + const fdDir = path.join(procRoot, pidStr, 'fd'); let fds; try { fds = fs.readdirSync(fdDir); - } catch { + } catch (err) { + // F1: the old code returned 'owned' for EVERY non-ENOENT error. That was + // a correctness bug: /proc//fd is owner-only (mode 0500), so a + // cross-user/root `gitnexus mcp` serving a DIFFERENT repo passes Phase 0+1 + // (its cmdline matches) and then EACCES'es here — yet its dev+ino was + // NEVER compared against THIS lbug. Claiming 'owned' lets it permanently, + // silently suppress augment for a repo it does not actually lock. We now + // distinguish the failure shapes (all still fail-closed where we can't + // prove non-ownership, but 'timeout' is the HONEST verdict for + // "inconclusive", not the false-positive 'owned'): + const code = err && err.code; + if (code === 'ENOENT') { + // Process raced away between the candidate scan and now -> genuinely no + // longer an owner. Move on. + continue; + } + if (code === 'EACCES' || code === 'EPERM') { + // Permission-denied fd dir: cannot read fds, so ownership is + // UNVERIFIABLE. Fail closed honestly via 'timeout' (the dispatcher maps + // timeout -> true, same protective skip as before) WITHOUT lying that we + // confirmed ownership. Do NOT degrade to not-owned/fail-open: if this + // really is the owner, fail-open re-opens the #1492 lbug race; augment + // is optional context, so a conservative skip costs little. + debugLog( + `fd dir unreadable for candidate pid ${pidStr} (${code}); ownership ` + + `unverifiable, probe inconclusive -> fail-closed (timeout)`, + ); + return 'timeout'; + } + if (code === 'EIO' || code === 'ESTALE') { + // Genuine transient I/O against this candidate's fd dir — not evidence + // it does NOT hold the lbug. Treat as inconclusive and fail closed + // (timeout) rather than continue, so a real owner mid-I/O-blip is not + // dropped (would fail-open). + debugLog( + `fd dir transient I/O error for candidate pid ${pidStr} (${code}); ` + + `probe inconclusive -> fail-closed (timeout)`, + ); + return 'timeout'; + } + // Any other shape (ENOTDIR — fd path is not a directory at all, so this + // is not a plausible live-procfs owner — and the long tail) is treated as + // "this candidate is not an owner": move to the next candidate instead of + // the old blanket 'owned'. If no other candidate owns the lbug the scan + // ends not-owned (dispatcher fail-open) — acceptable because ENOTDIR means + // the fd entry is structurally not a real /proc//fd. + debugLog( + `fd dir not a readable directory for candidate pid ${pidStr} ` + + `(${code || 'unknown'}); treating candidate as non-owner -> continue`, + ); continue; } - let holds = false; for (const fd of fds) { - if (Date.now() - start > budget) return false; + if (outOfBudget()) return 'timeout'; try { const st = fs.statSync(path.join(fdDir, fd)); if (st.dev === targetStat.dev && st.ino === targetStat.ino) { - holds = true; - break; + return 'owned'; } } catch { - /* ignore */ + /* fd raced closed; ignore */ } } - if (!holds) continue; - if (isGitNexusServerCommand(readLinuxCmdline(ent.name))) return true; } - return false; + + return 'not-owned'; } function unixLsofPsFindGitNexusServer(dbPathAbs, myPid) { @@ -370,8 +668,13 @@ function hasGitNexusDbLockedByGitNexusServer(dbPath, myPid) { } if (process.platform === 'linux') { - if (linuxProcScanFindGitNexusServer(dbPathAbs, myPid)) return true; - return unixLsofPsFindGitNexusServer(dbPathAbs, myPid); + // #2180: cmdline-first procfs scan, no lsof. 'timeout' fails CLOSED + // (overloaded host self-throttles — the throttle the orphan-storm incident + // needed; the old lsof fallback ETIMEDOUT'd and failed closed on these same + // hosts anyway, only slower and with the orphan risk). 'not-owned' is the + // only false. See the fail matrix in the file header. + const verdict = linuxProcScanFindGitNexusServer(dbPathAbs, myPid); + return verdict !== 'not-owned'; } return unixLsofPsFindGitNexusServer(dbPathAbs, myPid); @@ -379,6 +682,13 @@ function hasGitNexusDbLockedByGitNexusServer(dbPath, myPid) { module.exports = { hasGitNexusDbLockedByGitNexusServer, + // Exported for white-box unit tests that must assert the tri-state verdict + // ('owned' | 'not-owned' | 'timeout') directly — the dispatcher collapses + // timeout and owned to the same boolean true, so the boolean API alone cannot + // distinguish the F1 EACCES->timeout fix from the old EACCES->owned bug. The + // Probe interface already declares this optional. Linux-only by contract; the + // name is pinned by a source-contract test. + linuxProcScanFindGitNexusServer, // #2163 follow-up: the hook adapters wrap the augment CLI in the same // guard. Returns a self-tested wrapper path — the built-in candidates are // always absolute; a GITNEXUS_HOOK_TIMEOUT_PATH override is adopted as the @@ -391,4 +701,11 @@ module.exports = { // override to an absolute path. Returns null when the wrapper is // disabled/unavailable. Never call on win32 (see its JSDoc). resolveUnixGuardTimeout, + // Exported for white-box unit tests of the numeric-env parsing (#2183 review): + // Number()-not-parseInt so "16e3" reads as 16000, plus the empty/whitespace + // guard that keeps a set-but-empty budget on the 1200 default instead of an + // immediate fail-closed timeout. Tested directly because the values are + // otherwise only observable indirectly through scan timing/escalation. + getCmdlineMaxBytes, + resolveLinuxProcBudgetMs, }; diff --git a/gitnexus/hooks/claude/hook-db-lock-probe.cjs b/gitnexus/hooks/claude/hook-db-lock-probe.cjs index de0fa5e85..948581110 100644 --- a/gitnexus/hooks/claude/hook-db-lock-probe.cjs +++ b/gitnexus/hooks/claude/hook-db-lock-probe.cjs @@ -3,14 +3,36 @@ * with a command line that looks like a GitNexus MCP/serve server? * * Backends (no user-installed Sysinternals): - * - Linux: scan procfs under /proc (per-PID fd entries) via stat(2) (dev+inode); works without lsof; - * optional lsof fallback when proc scan finds nothing. + * - Linux: cmdline-first procfs scan under /proc, no lsof at all (#2180). Three + * phases, cheapest first: (0) read /proc//comm — a tiny task->comm read + * that never touches the target's mm — and keep only PIDs whose comm is a + * plausible node/gitnexus server; (1) read up to GITNEXUS_HOOK_PROC_CMDLINE_MAX + * bytes of /proc//cmdline via openSync+readSync (bounded, so a D-state + * holder stuck on mmap_lock or a giant argv can't wedge the hook) and prefilter + * with isGitNexusServerCommand; (2) only for the 0..N survivors, stat their + * /proc//fd/* and compare dev+inode against the target lbug. The lbug + * handle is fd-visible (a @ladybugdb/core property), so this finds every real + * owner without scanning every fd of every process. * - macOS / *BSD / etc.: trusted lsof + ps (absolute paths first). * - Windows: Restart Manager (rstrtmgr) via bundled PowerShell script + * Win32_Process for command lines; trusted powershell.exe under %SystemRoot%. * - * Fail-open on most errors; fail-closed only on lsof ETIMEDOUT (Unix) or - * PowerShell ETIMEDOUT (Windows), matching the hook contract. + * Fail matrix: + * - Linux proc scan: owner found -> fail-closed (skip augment); budget exhausted + * (GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS) -> fail-CLOSED (#2180). This is a + * deliberate change from the old "timeout -> fail-open then try lsof" path. + * End-to-end the busy-host outcome is unchanged: the old code's lsof fallback + * ETIMEDOUT'd on the very hosts where the scan ran out of budget and ALSO + * failed closed there — the lsof leg only ever added 1-2s of dead work plus + * the orphan-storm risk it caused (#2163). What changes is that an overloaded + * host now self-throttles immediately (the throttle the incident needed) + * instead of paying for a doomed lsof. Mid-load hosts that used to fall + * through to a successful lsof now answer from the scan directly (faster) or, + * if even the scan can't finish in budget, fail closed (self-throttle) — a + * bounded, documented tradeoff, never an orphan. + * - macOS / other Unix: fail-open on most errors; fail-closed only on lsof + * ETIMEDOUT, matching the hook contract. + * - Windows: fail-closed only on PowerShell ETIMEDOUT. * * Unix subprocess containment contract (#2163): * - lsof/ps are wrapped in coreutils `timeout`/`gtimeout` when a working @@ -46,6 +68,16 @@ function isGitNexusServerCommand(command) { return hasServerMode && hasGitNexus; } +// GITNEXUS_DEBUG-gated stderr diagnostics. Reuses the exact gating predicate the +// Windows ps1-load warning already uses (===' 1' / ==='true') so there is one +// debug convention in this file, and writes via process.stderr.write (NOT a +// spawn) so it never perturbs the windowsHide spawn-count invariant. +function debugLog(msg) { + if (process.env.GITNEXUS_DEBUG === '1' || process.env.GITNEXUS_DEBUG === 'true') { + process.stderr.write(`[GitNexus hook] ${msg}\n`); + } +} + function resolveHookBinary(tool) { const envKey = tool === 'lsof' ? 'GITNEXUS_HOOK_LSOF_PATH' : 'GITNEXUS_HOOK_PS_PATH'; const fromEnv = process.env[envKey]; @@ -242,59 +274,325 @@ function hasGitNexusServerOwnerWindows(dbPathAbs, myPid) { return false; } -function readLinuxCmdline(pidStr) { +// The procfs root every Linux scan path reads from. Production is always /proc; +// GITNEXUS_HOOK_PROC_ROOT only exists so unit tests can inject a fixture tree +// (comm + cmdline + fd symlinks) and assert the three-phase logic without +// scanning the real, ~hundreds-of-process /proc of the test host. +// +// Test-only gate (F4): the override is honored ONLY under a test runner — +// vitest injects VITEST="true" and NODE_ENV="test" into every worker (verified; +// a production hook is `node .cjs` with neither set). Without the gate, a +// production env that accidentally leaked GITNEXUS_HOOK_PROC_ROOT (pointing at an +// empty/bad tree) would make readdirSync find no pids -> 'not-owned' -> Linux +// owner detection silently OFF (fail-OPEN: augment races the real server for the +// lbug, the #1492 class). Gating to the test signal makes that leak inert in +// production (always /proc) while the fake-procfs unit tests, which run under +// vitest, still inject freely. Unset env (or non-test context) => /proc, so the +// production path is byte-for-byte the historical behavior. +function isTestContext() { + return ( + process.env.VITEST === 'true' || process.env.VITEST === '1' || process.env.NODE_ENV === 'test' + ); +} +function getProcRoot() { + if (!isTestContext()) return '/proc'; + const raw = process.env.GITNEXUS_HOOK_PROC_ROOT; + return raw && String(raw).trim() ? String(raw) : '/proc'; +} + +// Max bytes read from /proc//cmdline in Phase 1. Bounded by default so a +// D-state holder wedged on mmap_lock, or a process with a pathological multi-MB +// argv, can't stall the hook. 16 KiB comfortably clears a realistic +// `node mcp` line +// (the `mcp`/`serve` mode token lives at the very tail, so the cap must be large +// enough to reach it — see PROC_CMDLINE_FLOOR escalation below). Overridable for +// tests; never goes below PROC_CMDLINE_FLOOR. +const PROC_CMDLINE_FLOOR = 4096; +function getCmdlineMaxBytes() { + const raw = process.env.GITNEXUS_HOOK_PROC_CMDLINE_MAX; + // Number() (not parseInt) so "8e3" reads as 8000, not 8 (parseInt stops at + // 'e'). The `raw && String(raw).trim()` guard keeps empty/whitespace on the + // default; trailing garbage ("8abc") now -> NaN -> default (stricter). + const n = raw && String(raw).trim() ? Number(String(raw).trim()) : NaN; + if (Number.isFinite(n) && n >= PROC_CMDLINE_FLOOR) return n; + return 16384; +} + +// Phase 0 comm prefilter. /proc//comm is the kernel task->comm string, +// capped at 16 bytes INCLUDING the trailing NUL — i.e. at most 15 visible +// chars, truncated by the kernel with no marker. So a process whose real name +// is longer than 15 chars shows a 15-char prefix here. The match below is +// therefore truncation-safe in BOTH directions (a whitelist name that is a +// prefix of comm, or comm that is a prefix of a whitelist name, both count) to +// guarantee we never drop a real owner at this cheap stage — Phase 2's dev+ino +// fd check is the real authority; Phase 0/1 only exist to skip the overwhelming +// majority (kernel threads, shells, editors) cheaply. +// +// The whitelist is calibrated against what a real `gitnexus mcp`/`serve` server +// actually reports for comm. Observed on production hosts: the server renames +// its main thread, so comm reads `MainThread` (via @ladybugdb/core's +// worker_threads setup), NOT `node` — omitting it would blind the probe to +// every real server (#1492-class owner miss). We also keep the plausible +// launcher/runtime basenames in case a future build does not rename the thread. +// Conservative by design: over-collecting a few extra candidates only costs a +// bounded number of Phase 1 cmdline reads. +const COMM_CANDIDATES = ['node', 'gitnexus', 'bun', 'deno', 'npm', 'npx', 'MainThread']; +function commLooksLikeServer(comm) { + const c = comm.trim(); + if (!c) return false; + for (const name of COMM_CANDIDATES) { + if (name === c || name.startsWith(c) || c.startsWith(name)) return true; + } + return false; +} + +function readProcComm(procRoot, pidStr) { try { - return fs.readFileSync(`/proc/${pidStr}/cmdline`, 'utf8').replace(/\0+/g, ' ').trim(); + return fs + .readFileSync(path.join(procRoot, pidStr, 'comm'), 'utf8') + .replace(/\0+/g, '') + .trim(); } catch { return ''; } } -function linuxProcScanFindGitNexusServer(dbPathAbs, myPid) { +// Timeout sentinel for readLinuxCmdline (F3). MUST be distinct from the +// "unreadable/empty" return value (''): '' flows through isGitNexusServerCommand +// as a NON-candidate (both regexes are false on ''), so the Phase 1 caller +// `continue`s past it — correct for a raced/openSync-failed pid, but a FAIL-OPEN +// bug if it ever meant "I ran out of budget mid-read" (a real owner whose +// escalation timed out would be silently dropped, racing the lbug -> #1492). A +// unique Symbol can never collide with any cmdline string, so the caller can +// branch on it explicitly and map a mid-read timeout to the tri-state 'timeout' +// (fail-CLOSED) instead of swallowing it as a non-candidate. +const CMDLINE_TIMEOUT = Symbol('gitnexus.cmdline.timeout'); + +// Bounded /proc//cmdline read for Phase 1. openSync+readSync (not +// readFileSync) so a D-state holder cannot stall the hook on a huge or +// never-EOF argv: we read at most `cap` bytes and stop. cmdline separates argv +// with NULs; convert to spaces for isGitNexusServerCommand. +// +// Owner-miss guard for the 4 KB cap: the `gitnexus` token usually sits in the +// first path component while the `mcp`/`serve` mode token is the LAST argv, so +// a naive 4 KB read could clip the mode token off a server launched with a very +// long interpreter path and silently miss a real owner. We mitigate two ways: +// (a) the default cap (16 KiB) already clears realistic lines; (b) if the first +// read fills the cap AND already contains the `gitnexus` token but no mode +// token yet, we keep reading in bounded chunks (up to a hard ceiling) until the +// mode token appears or the file ends — so a genuine server is never missed for +// want of a few more bytes, while non-candidates still pay only the initial +// bounded read. +// +// Budget (F3): the escalation loop above is the one place a SINGLE pathological +// candidate could read up to HARD_CEIL (256 KiB) before the next scan-level +// budget check, weakening the timeout contract. `outOfBudget` (the scan's shared +// deadline callback) is checked once per escalation iteration; on expiry we +// return CMDLINE_TIMEOUT (NOT '') so the caller can fail-closed honestly rather +// than mistake the partial read for a non-candidate. Reads that simply can't +// open / error out still return '' (genuinely "not a readable candidate"). +function readLinuxCmdline(procRoot, pidStr, cap, outOfBudget) { + const file = path.join(procRoot, pidStr, 'cmdline'); + let fd; + try { + fd = fs.openSync(file, 'r'); + } catch { + return ''; + } + try { + const HARD_CEIL = 262144; // 256 KiB absolute ceiling for the escalation path + let collected = Buffer.alloc(0); + let offset = 0; + let chunkCap = cap; + for (;;) { + // allocUnsafe is safe here: readSync fills exactly [0, bytes), only + // buf.subarray(0, bytes) is consumed, and Buffer.concat deep-copies that + // slice into `collected`, so the uninitialized tail never reaches decode. + const buf = Buffer.allocUnsafe(chunkCap); + const bytes = fs.readSync(fd, buf, 0, chunkCap, offset); + if (bytes <= 0) break; + collected = Buffer.concat([collected, buf.subarray(0, bytes)]); + offset += bytes; + const text = collected.toString('utf8').replace(/\0+/g, ' '); + // Stop early when we can already decide "owner": has both the gitnexus + // token and a mode token. Keep going only when gitnexus is present but + // the mode token might be just past the boundary. + const hasGitNexus = + /(?:^|[/\\\s])gitnexus(?:\.cmd)?(?:\s|$)/.test(text) || + /node_modules[/\\]gitnexus[/\\]/.test(text); + const hasMode = /(?:^|\s)(mcp|serve)(?:\s|$)/.test(text); + if (hasMode) break; // decided (positive); isGitNexusServerCommand re-checks below + if (bytes < chunkCap) break; // EOF: full cmdline read, definitive + if (!hasGitNexus) break; // not a candidate; do not escalate the read + if (offset >= HARD_CEIL) break; // bounded escalation only + // Budget gate the escalation: a single huge-argv candidate must not burn + // the whole scan deadline before we re-check. Return the timeout sentinel + // (never '') so the caller fails closed instead of treating us as a + // non-candidate. The sole caller (linuxProcScanFindGitNexusServer) always + // passes outOfBudget, so no presence guard is needed. + if (outOfBudget()) return CMDLINE_TIMEOUT; + chunkCap = cap; // keep reading more in cap-sized chunks + } + return collected.toString('utf8').replace(/\0+/g, ' ').trim(); + } catch { + return ''; + } finally { + try { + fs.closeSync(fd); + } catch { + /* ignore */ + } + } +} + +function resolveLinuxProcBudgetMs() { const raw = process.env.GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS; - const budget = Number(raw && String(raw).trim()) ? Number.parseInt(String(raw), 10) : 1200; + // Gate on the STRING's emptiness, NOT the parsed number's truthiness — the + // old `Number(raw && trim()) ? ... : 1200` form treated "0" as falsy and + // silently fell back to 1200 (#2180). Use Number() (not parseInt) so "16e3" + // reads as 16000, not 16 (parseInt stops at 'e'). The `&& String(raw).trim()` + // guard is load-bearing: without it a set-but-empty/whitespace value would be + // `Number("")===0` => budget 0 => immediate fail-CLOSED timeout (augment + // permanently skipped). With it, ''/whitespace => NaN => 1200 default, while a + // finite "0" still parses to an explicit, deterministic "no budget" => + // immediate timeout. Non-numeric / unset => default 1200. + const n = raw != null && String(raw).trim() ? Number(String(raw).trim()) : NaN; + if (!Number.isFinite(n)) return 1200; + return n; // may be <= 0, meaning "out of budget on the first check" +} + +// Returns one of: 'owned' (a non-self process with a GitNexus-server cmdline +// holds the target lbug fd), 'not-owned' (scan completed, no such owner), or +// 'timeout' (the per-scan budget was exhausted before a verdict). The name is +// pinned by a source-contract test; only the return TYPE changed (#2180: +// boolean -> tri-state, so the dispatcher can fail-closed on 'timeout'). +function linuxProcScanFindGitNexusServer(dbPathAbs, myPid) { + const budget = resolveLinuxProcBudgetMs(); + // A non-positive budget is an explicit, deterministic "no time to scan" => + // immediate timeout (the #2180 test vector, and the only correct reading of + // the fixed parse: "0" must NOT mean 1200). Returning before any procfs read + // keeps it instantaneous regardless of host load. + if (budget <= 0) return 'timeout'; + const procRoot = getProcRoot(); + const cmdlineCap = getCmdlineMaxBytes(); const start = Date.now(); + const outOfBudget = () => Date.now() - start > budget; + let targetStat; try { targetStat = fs.statSync(dbPathAbs); } catch { - return false; + // Caller already existsSync'd the path; a stat failure here is a transient + // race, treat as no owner (historical semantics). + return 'not-owned'; } + let procEntries; try { - procEntries = fs.readdirSync('/proc', { withFileTypes: true }); + procEntries = fs.readdirSync(procRoot, { withFileTypes: true }); } catch { - return false; + return 'not-owned'; } + + // Phase 0 + Phase 1: collect the few PIDs whose comm AND cmdline look like a + // GitNexus server, without touching any fd yet. + const candidates = []; for (const ent of procEntries) { - if (Date.now() - start > budget) return false; + if (outOfBudget()) return 'timeout'; if (!ent.isDirectory() || !/^\d+$/.test(ent.name)) continue; const pid = Number.parseInt(ent.name, 10); if (!Number.isFinite(pid) || pid === myPid) continue; - const fdDir = path.join('/proc', ent.name, 'fd'); + + // Phase 0: cheap comm prefilter. + const comm = readProcComm(procRoot, ent.name); + if (!comm) continue; // unreadable comm (kernel thread, raced exit) -> skip + if (!commLooksLikeServer(comm)) continue; + + // Phase 1: bounded cmdline read + isGitNexusServerCommand prefilter. + if (outOfBudget()) return 'timeout'; + const cmdline = readLinuxCmdline(procRoot, ent.name, cmdlineCap, outOfBudget); + // F3: a mid-read budget timeout returns the CMDLINE_TIMEOUT sentinel (a + // Symbol, never a string). Fail CLOSED on it rather than letting it fall + // through isGitNexusServerCommand as a non-candidate — a real owner whose + // escalation timed out must not be silently dropped (would fail-OPEN). + if (cmdline === CMDLINE_TIMEOUT) return 'timeout'; + if (!isGitNexusServerCommand(cmdline)) continue; + candidates.push(ent.name); + } + + // Phase 2: only now stat the fds of the (typically 0-2) survivors. + for (const pidStr of candidates) { + if (outOfBudget()) return 'timeout'; + const fdDir = path.join(procRoot, pidStr, 'fd'); let fds; try { fds = fs.readdirSync(fdDir); - } catch { + } catch (err) { + // F1: the old code returned 'owned' for EVERY non-ENOENT error. That was + // a correctness bug: /proc//fd is owner-only (mode 0500), so a + // cross-user/root `gitnexus mcp` serving a DIFFERENT repo passes Phase 0+1 + // (its cmdline matches) and then EACCES'es here — yet its dev+ino was + // NEVER compared against THIS lbug. Claiming 'owned' lets it permanently, + // silently suppress augment for a repo it does not actually lock. We now + // distinguish the failure shapes (all still fail-closed where we can't + // prove non-ownership, but 'timeout' is the HONEST verdict for + // "inconclusive", not the false-positive 'owned'): + const code = err && err.code; + if (code === 'ENOENT') { + // Process raced away between the candidate scan and now -> genuinely no + // longer an owner. Move on. + continue; + } + if (code === 'EACCES' || code === 'EPERM') { + // Permission-denied fd dir: cannot read fds, so ownership is + // UNVERIFIABLE. Fail closed honestly via 'timeout' (the dispatcher maps + // timeout -> true, same protective skip as before) WITHOUT lying that we + // confirmed ownership. Do NOT degrade to not-owned/fail-open: if this + // really is the owner, fail-open re-opens the #1492 lbug race; augment + // is optional context, so a conservative skip costs little. + debugLog( + `fd dir unreadable for candidate pid ${pidStr} (${code}); ownership ` + + `unverifiable, probe inconclusive -> fail-closed (timeout)`, + ); + return 'timeout'; + } + if (code === 'EIO' || code === 'ESTALE') { + // Genuine transient I/O against this candidate's fd dir — not evidence + // it does NOT hold the lbug. Treat as inconclusive and fail closed + // (timeout) rather than continue, so a real owner mid-I/O-blip is not + // dropped (would fail-open). + debugLog( + `fd dir transient I/O error for candidate pid ${pidStr} (${code}); ` + + `probe inconclusive -> fail-closed (timeout)`, + ); + return 'timeout'; + } + // Any other shape (ENOTDIR — fd path is not a directory at all, so this + // is not a plausible live-procfs owner — and the long tail) is treated as + // "this candidate is not an owner": move to the next candidate instead of + // the old blanket 'owned'. If no other candidate owns the lbug the scan + // ends not-owned (dispatcher fail-open) — acceptable because ENOTDIR means + // the fd entry is structurally not a real /proc//fd. + debugLog( + `fd dir not a readable directory for candidate pid ${pidStr} ` + + `(${code || 'unknown'}); treating candidate as non-owner -> continue`, + ); continue; } - let holds = false; for (const fd of fds) { - if (Date.now() - start > budget) return false; + if (outOfBudget()) return 'timeout'; try { const st = fs.statSync(path.join(fdDir, fd)); if (st.dev === targetStat.dev && st.ino === targetStat.ino) { - holds = true; - break; + return 'owned'; } } catch { - /* ignore */ + /* fd raced closed; ignore */ } } - if (!holds) continue; - if (isGitNexusServerCommand(readLinuxCmdline(ent.name))) return true; } - return false; + + return 'not-owned'; } function unixLsofPsFindGitNexusServer(dbPathAbs, myPid) { @@ -370,8 +668,13 @@ function hasGitNexusDbLockedByGitNexusServer(dbPath, myPid) { } if (process.platform === 'linux') { - if (linuxProcScanFindGitNexusServer(dbPathAbs, myPid)) return true; - return unixLsofPsFindGitNexusServer(dbPathAbs, myPid); + // #2180: cmdline-first procfs scan, no lsof. 'timeout' fails CLOSED + // (overloaded host self-throttles — the throttle the orphan-storm incident + // needed; the old lsof fallback ETIMEDOUT'd and failed closed on these same + // hosts anyway, only slower and with the orphan risk). 'not-owned' is the + // only false. See the fail matrix in the file header. + const verdict = linuxProcScanFindGitNexusServer(dbPathAbs, myPid); + return verdict !== 'not-owned'; } return unixLsofPsFindGitNexusServer(dbPathAbs, myPid); @@ -379,6 +682,13 @@ function hasGitNexusDbLockedByGitNexusServer(dbPath, myPid) { module.exports = { hasGitNexusDbLockedByGitNexusServer, + // Exported for white-box unit tests that must assert the tri-state verdict + // ('owned' | 'not-owned' | 'timeout') directly — the dispatcher collapses + // timeout and owned to the same boolean true, so the boolean API alone cannot + // distinguish the F1 EACCES->timeout fix from the old EACCES->owned bug. The + // Probe interface already declares this optional. Linux-only by contract; the + // name is pinned by a source-contract test. + linuxProcScanFindGitNexusServer, // #2163 follow-up: the hook adapters wrap the augment CLI in the same // guard. Returns a self-tested wrapper path — the built-in candidates are // always absolute; a GITNEXUS_HOOK_TIMEOUT_PATH override is adopted as the @@ -391,4 +701,11 @@ module.exports = { // override to an absolute path. Returns null when the wrapper is // disabled/unavailable. Never call on win32 (see its JSDoc). resolveUnixGuardTimeout, + // Exported for white-box unit tests of the numeric-env parsing (#2183 review): + // Number()-not-parseInt so "16e3" reads as 16000, plus the empty/whitespace + // guard that keeps a set-but-empty budget on the 1200 default instead of an + // immediate fail-closed timeout. Tested directly because the values are + // otherwise only observable indirectly through scan timing/escalation. + getCmdlineMaxBytes, + resolveLinuxProcBudgetMs, }; diff --git a/gitnexus/scripts/cross-platform-tests.ts b/gitnexus/scripts/cross-platform-tests.ts index d332ba1d6..38f769fd3 100644 --- a/gitnexus/scripts/cross-platform-tests.ts +++ b/gitnexus/scripts/cross-platform-tests.ts @@ -37,6 +37,7 @@ const PLATFORM_LOGIC = [ 'test/unit/repo-manager.test.ts', 'test/unit/repo-manager-finalize-invariant.test.ts', 'test/unit/hooks.test.ts', + 'test/unit/hook-db-lock-probe.test.ts', 'test/unit/cursor-hook.test.ts', 'test/unit/sidecar-recovery.test.ts', 'test/unit/pool-wal-recovery.test.ts', diff --git a/gitnexus/test/integration/antigravity-hook-e2e.test.ts b/gitnexus/test/integration/antigravity-hook-e2e.test.ts index a4a9d1f01..8cb68b000 100644 --- a/gitnexus/test/integration/antigravity-hook-e2e.test.ts +++ b/gitnexus/test/integration/antigravity-hook-e2e.test.ts @@ -396,7 +396,13 @@ describe('antigravity hook adapter e2e', () => { // (its lock/probe helpers only resolve from the install dir). A faked lsof/ps + // an empty `lbug` lock force hasGitNexusServerOwner() => true; a marker-writing // fake CLI proves augment never ran. - describe.skipIf(process.platform === 'win32')( + // + // #2180: skipped on Linux too — the probe's Linux backend no longer uses + // lsof/ps, so the faked lsof/ps can't force owner=true there. This stays as the + // macOS/other-Unix lsof-path lane; the antigravity adapter shares the identical + // gated owner-skip with the claude/plugin copies, whose Linux owner detection + // is covered against a fake /proc in test/unit/hook-db-lock-probe.test.ts. + describe.skipIf(process.platform === 'win32' || process.platform === 'linux')( 'AfterTool — augment skipped when MCP server owns the DB (#1913)', () => { const OWNER_PROBE = { diff --git a/gitnexus/test/unit/hook-db-lock-probe.test.ts b/gitnexus/test/unit/hook-db-lock-probe.test.ts new file mode 100644 index 000000000..60e77c894 --- /dev/null +++ b/gitnexus/test/unit/hook-db-lock-probe.test.ts @@ -0,0 +1,723 @@ +/** + * Direct unit tests for the Linux cmdline-first DB-owner scan (#2180). + * + * These exercise linuxProcScanFindGitNexusServer / hasGitNexusDbLockedByGitNexusServer + * against a FAKE /proc tree (GITNEXUS_HOOK_PROC_ROOT) so the three-phase logic + * (comm -> cmdline -> fd dev+ino) is asserted deterministically, without + * scanning the test host's real /proc. One live e2e at the bottom uses the REAL + * /proc to protect the "lbug handle is fd-visible" property the scan relies on. + * + * The probe is a CJS module; we require it through createRequire and toggle env + * per-test. resetModules-style isolation is unnecessary because the only + * module-level cache (unixGuardTimeoutCache) is on the macOS/Unix path, which + * these Linux tests never reach. + */ +import { describe, it, expect, afterEach, vi } from 'vitest'; +import { createRequire } from 'node:module'; +import fs from 'fs'; +import os from 'os'; +import path from 'path'; +import { spawn } from 'child_process'; +import { createFakeProcRoot, type FakeProcEntry } from '../utils/hook-test-helpers.js'; + +const PROBE_PATH = path.resolve(__dirname, '..', '..', 'hooks', 'claude', 'hook-db-lock-probe.cjs'); + +type Probe = { + hasGitNexusDbLockedByGitNexusServer: (dbPath: string, myPid: number) => boolean; + linuxProcScanFindGitNexusServer?: (dbPathAbs: string, myPid: number) => string; + getCmdlineMaxBytes?: () => number; + resolveLinuxProcBudgetMs?: () => number; +}; +const probe = createRequire(import.meta.url)(PROBE_PATH) as Probe; + +// The probe now exports linuxProcScanFindGitNexusServer unconditionally (F1 +// white-box verdict assertions). Narrow it once here to a non-optional typed +// fn so the per-test call sites stay assertion-free; a dedicated test below +// pins that the export really is a function. +type ScanVerdictFn = (dbPathAbs: string, myPid: number) => string; +const scanVerdictFn = probe.linuxProcScanFindGitNexusServer as ScanVerdictFn; + +const isLinux = process.platform === 'linux'; + +// ── env scoping helpers ──────────────────────────────────────────── +const ENV_KEYS = [ + 'GITNEXUS_HOOK_PROC_ROOT', + 'GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS', + 'GITNEXUS_HOOK_PROC_CMDLINE_MAX', +] as const; +const savedEnv: Record = {}; +function setEnv(overrides: Record) { + for (const k of ENV_KEYS) { + if (!(k in savedEnv)) savedEnv[k] = process.env[k]; + } + for (const [k, v] of Object.entries(overrides)) { + if (v === undefined) delete process.env[k]; + else process.env[k] = v; + } +} +const cleanups: Array<() => void> = []; +afterEach(() => { + for (const k of Object.keys(savedEnv)) { + const v = savedEnv[k]; + if (v === undefined) delete process.env[k]; + else process.env[k] = v; + delete savedEnv[k]; + } + while (cleanups.length) { + try { + cleanups.pop()!(); + } catch { + /* best-effort */ + } + } +}); + +/** + * Build a temp lbug + a fake /proc root, run the dispatcher with the fake root, + * and return the boolean owner verdict. The lbug is the dev+ino the fake fd + * symlinks point at, so a holder whose fdTargets include `lbug` is a true owner. + */ +function runScan( + entries: (lbugPath: string) => FakeProcEntry[], + env: Record = {}, +): { owned: boolean; lbugPath: string } { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-probe-')); + cleanups.push(() => fs.rmSync(dir, { recursive: true, force: true })); + const lbugPath = path.join(dir, 'lbug'); + fs.writeFileSync(lbugPath, ''); + const procRoot = createFakeProcRoot(entries(lbugPath)); + cleanups.push(() => fs.rmSync(procRoot, { recursive: true, force: true })); + setEnv({ GITNEXUS_HOOK_PROC_ROOT: procRoot, ...env }); + const owned = probe.hasGitNexusDbLockedByGitNexusServer(lbugPath, 1); + return { owned, lbugPath }; +} + +const GITNEXUS_MCP_ARGV = (script: string) => ['node', script, 'mcp']; + +// ── Numeric env parsing (white-box, #2183 review) ────────────────────── +// +// getCmdlineMaxBytes / resolveLinuxProcBudgetMs switched from parseInt(.,10) to +// Number() so scientific notation ("16e3") parses as 16000 instead of 16. These +// are platform-independent (pure string->number), so they run on every OS, not +// just Linux. The load-bearing case is the EMPTY-STRING budget regression guard: +// a naive parseInt->Number swap would make a set-but-empty +// GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS="" resolve to Number("")===0 => budget 0 => +// immediate fail-CLOSED timeout (augment permanently skipped). The added +// `&& String(raw).trim()` guard keeps ''/whitespace on the 1200 default. +describe('numeric env parsing (white-box, #2183 review)', () => { + const budget = probe.resolveLinuxProcBudgetMs as () => number; + const cmdlineMax = probe.getCmdlineMaxBytes as () => number; + + it('exports the two parse helpers as functions', () => { + expect(typeof probe.resolveLinuxProcBudgetMs).toBe('function'); + expect(typeof probe.getCmdlineMaxBytes).toBe('function'); + }); + + it('budget: "16e3" parses as 16000 (scientific notation), not 16', () => { + // parseInt('16e3',10) === 16 (stops at 'e'); Number('16e3') === 16000. + setEnv({ GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '16e3' }); + expect(budget()).toBe(16000); + }); + + it('budget: set-but-empty "" and whitespace fall back to 1200, NOT 0 (regression guard)', () => { + // The deepening catch: without the `&& String(raw).trim()` guard these would + // be Number('')===0 => an immediate fail-closed timeout on every hook call. + for (const empty of ['', ' ', '\t']) { + setEnv({ GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: empty }); + expect(budget()).toBe(1200); + } + }); + + it('budget: "0" still parses to 0 (the deliberate #2180 immediate-timeout vector)', () => { + setEnv({ GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '0' }); + expect(budget()).toBe(0); + }); + + it('budget: trailing garbage "123abc" and unset fall back to 1200', () => { + setEnv({ GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '123abc' }); + expect(budget()).toBe(1200); + setEnv({ GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: undefined }); + expect(budget()).toBe(1200); + }); + + it('cmdline max: "8e3" parses as 8000 (>= floor); "2e3" (=2000, below floor) and ""/unset -> 16384', () => { + setEnv({ GITNEXUS_HOOK_PROC_CMDLINE_MAX: '8e3' }); + expect(cmdlineMax()).toBe(8000); + setEnv({ GITNEXUS_HOOK_PROC_CMDLINE_MAX: '2e3' }); + expect(cmdlineMax()).toBe(16384); + setEnv({ GITNEXUS_HOOK_PROC_CMDLINE_MAX: '' }); + expect(cmdlineMax()).toBe(16384); + setEnv({ GITNEXUS_HOOK_PROC_CMDLINE_MAX: undefined }); + expect(cmdlineMax()).toBe(16384); + }); +}); + +describe.skipIf(!isLinux)('Linux cmdline-first DB-owner scan (#2180)', () => { + // ── D1: three-phase correctness ────────────────────────────────── + + it('owned: a gitnexus mcp process holding the lbug fd is detected', () => { + const { owned } = runScan((lbug) => [ + { + pid: 4242, + comm: 'MainThread', // real gitnexus servers report this on modern Node + cmdline: GITNEXUS_MCP_ARGV('/opt/app/node_modules/gitnexus/dist/cli/index.js'), + fdTargets: ['/dev/null', lbug], + }, + ]); + expect(owned).toBe(true); + }); + + it('not-owned: a node process that is not a gitnexus server (even holding the lbug) is ignored', () => { + const { owned } = runScan((lbug) => [ + { + pid: 5555, + comm: 'node', + cmdline: ['node', '/some/app/server.js'], + fdTargets: [lbug], // holds the fd, but cmdline is not a gitnexus server + }, + ]); + expect(owned).toBe(false); + }); + + it('not-owned: a gitnexus mcp process that does NOT hold the lbug fd is not an owner', () => { + const { owned } = runScan((lbug) => [ + { + pid: 6001, + comm: 'MainThread', + cmdline: GITNEXUS_MCP_ARGV('/x/node_modules/gitnexus/dist/cli/index.js'), + fdTargets: ['/dev/null'], // server, but holds some OTHER fd, not this lbug + }, + // a decoy that holds the lbug but is not a server + { + pid: 6002, + comm: 'vim', + cmdline: ['vim', '/etc/hosts'], + fdTargets: [lbug], + }, + ]); + expect(owned).toBe(false); + }); + + it('Phase 0 trap: cmdline LOOKS like gitnexus but comm is non-candidate → filtered out before fd check', () => { + // The fd symlink points at the lbug, so if Phase 0 did NOT filter on comm + // the cmdline prefilter would match and the fd check would say "owned". + // Because comm is a non-candidate ('postgres'), Phase 0 drops it first. + const { owned } = runScan((lbug) => [ + { + pid: 7007, + comm: 'postgres', // not in COMM_CANDIDATES, not a prefix of any + cmdline: GITNEXUS_MCP_ARGV('/x/node_modules/gitnexus/dist/cli/index.js'), + fdTargets: [lbug], + }, + ]); + expect(owned).toBe(false); + }); + + it('Phase 0 truncation-safe: a 15-char-truncated comm prefix of a candidate still matches', () => { + // Kernel comm cap is 15 visible chars; a candidate name truncated to a + // prefix must NOT be dropped. We use a comm that is a strict prefix of a + // whitelist entry ('MainThr' ⊂ 'MainThread'). + const { owned } = runScan((lbug) => [ + { + pid: 8008, + comm: 'MainThr', + cmdline: GITNEXUS_MCP_ARGV('/x/node_modules/gitnexus/dist/cli/index.js'), + fdTargets: [lbug], + }, + ]); + expect(owned).toBe(true); + }); + + // ── D2: budget / timeout → fail-closed ─────────────────────────── + + it('budget <= 0 → immediate timeout → dispatcher fails CLOSED (owner=true)', () => { + // Even though NO process is a gitnexus server, budget 0 yields 'timeout' + // which the dispatcher maps to true (self-throttle). This also pins the + // #2180 budget-parse fix: "0" must NOT fall back to 1200. + const { owned } = runScan( + () => [{ pid: 9001, comm: 'bash', cmdline: ['bash'], fdTargets: [] }], + { GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '0' }, + ); + expect(owned).toBe(true); + }); + + it('budget "0" is not silently treated as 1200 (regression for the parse bug)', () => { + // With a healthy non-owner fake proc and budget '0', the OLD code (which + // coerced "0" to 1200) would have completed the scan and returned + // not-owned (false). The fixed code returns timeout → true. + const { owned } = runScan( + () => [{ pid: 9100, comm: 'node', cmdline: ['node', '/app/x.js'], fdTargets: [] }], + { GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '0' }, + ); + expect(owned).toBe(true); + }); + + it('a generous budget over a non-owner tree completes and returns not-owned', () => { + const { owned } = runScan( + () => [ + { pid: 9200, comm: 'node', cmdline: ['node', '/app/x.js'], fdTargets: [] }, + { pid: 9201, comm: 'bash', cmdline: ['bash', '-l'], fdTargets: [] }, + ], + { GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '5000' }, + ); + expect(owned).toBe(false); + }); + + // ── D3: 4 KB+ cmdline cap — escalation must really iterate (F2) ──── + // + // The cmdline shape here is deliberate (Codex): the `gitnexus` token sits in + // the SECOND argv (a SHORT node_modules/gitnexus path, well inside the first + // 4 KB chunk) so `if (!hasGitNexus) break` does NOT abort the read; a ~9 KB + // pad argv then pushes the trailing `mcp` mode token PAST 4096, so the first + // 4 KB chunk has gitnexus-but-no-mode and the loop MUST escalate to a second + // read to find `mcp`. Setting GITNEXUS_HOOK_PROC_CMDLINE_MAX=4096 makes the + // chunk size 4 KB so escalation actually happens (the 16 KB default would read + // the whole line in one shot and the loop would never iterate — the old test's + // latent no-op). + + // gitnexus token early (well under 4 KB), mode token forced past 4 KB by pad. + const GITNEXUS_SHORT = '/nm/node_modules/gitnexus/dist/cli/index.js'; + const PAD_PAST_4K = 'x'.repeat(9000); // pushes the trailing `mcp` well past 4096 + + it('owned even when the mode token sits far past 4 KB → escalation iterates and finds it', () => { + const readSyncSpy = vi.spyOn(fs, 'readSync'); + cleanups.push(() => readSyncSpy.mockRestore()); + const { owned } = runScan( + (lbug) => [ + { + pid: 10001, + comm: 'MainThread', + // node | SHORT gitnexus path (<4KB) | 9KB pad | mcp → mcp lands >4096 + cmdline: ['node', GITNEXUS_SHORT, PAD_PAST_4K, 'mcp'], + fdTargets: [lbug], + }, + ], + { GITNEXUS_HOOK_PROC_CMDLINE_MAX: '4096' }, + ); + expect(owned).toBe(true); + // White-box proof the escalation actually re-read: with a 4 KB chunk over a + // >4 KB cmdline, readSync must have been called more than once for this pid. + // (A "just bump the cap" pseudo-fix that read everything in one go would + // leave this at 1 and fail.) + expect(readSyncSpy.mock.calls.length).toBeGreaterThan(1); + }); + + it('discrimination: same gitnexus cmdline but mode token past HARD_CEIL → not-owned', () => { + // Negative control proving the escalation has a real upper bound (HARD_CEIL + // = 256 KiB) and the positive test above is not just "always escalates". The + // gitnexus token is early so escalation runs, but a >256 KiB pad keeps the + // `mcp` token beyond the ceiling, so the bounded read stops before reaching + // it → isGitNexusServerCommand sees no mode token → not a candidate → + // not-owned. (A broken "escalate forever" impl would wrongly read to `mcp` + // and report owned, failing this assertion.) + const padPastCeil = 'x'.repeat(300000); // > HARD_CEIL (262144) + const { owned } = runScan( + (lbug) => [ + { + pid: 10002, + comm: 'MainThread', + cmdline: ['node', GITNEXUS_SHORT, padPastCeil, 'mcp'], + fdTargets: [lbug], + }, + ], + { GITNEXUS_HOOK_PROC_CMDLINE_MAX: '4096' }, + ); + expect(owned).toBe(false); + }); + + it('does not over-read: a giant non-gitnexus cmdline is bounded and yields not-owned', () => { + const giant = 'x'.repeat(500000); // 500 KB single arg, no gitnexus token + const { owned } = runScan((lbug) => [ + { + pid: 10100, + comm: 'node', + cmdline: ['node', `/app/${giant}.js`], + fdTargets: [lbug], + }, + ]); + expect(owned).toBe(false); + }); + + // ── D3b: cmdline escalation respects the scan budget (F3) ───────── + // + // A single pathological candidate whose `mcp` token sits far past the chunk + // size used to be able to read up to HARD_CEIL (256 KiB) inside one + // readLinuxCmdline call before the scan-level budget was re-checked. F3 wires + // outOfBudget into the escalation loop: when the deadline trips mid-read it + // returns the CMDLINE_TIMEOUT *Symbol* (NOT '' — '' would flow through + // isGitNexusServerCommand as a non-candidate and silently drop a possible + // owner, a fail-OPEN), and the Phase 1 caller maps that Symbol to the 'timeout' + // verdict (fail-CLOSED). We drive the deadline deterministically by advancing + // a Date.now spy after the first escalation read. + + it('escalation that exceeds the budget mid-read → verdict timeout (sentinel, not silent drop)', () => { + const scan = scanVerdictFn; + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-probe-f3-')); + cleanups.push(() => fs.rmSync(dir, { recursive: true, force: true })); + const lbugPath = path.join(dir, 'lbug'); + fs.writeFileSync(lbugPath, ''); + // gitnexus token early (escalation will start), mode token pushed past 4 KB + // so a SECOND read is required — between those reads we trip the clock. + const procRoot = createFakeProcRoot([ + { + pid: 10200, + comm: 'MainThread', + cmdline: ['node', GITNEXUS_SHORT, 'x'.repeat(9000), 'mcp'], + fdTargets: [lbugPath], + }, + ]); + cleanups.push(() => fs.rmSync(procRoot, { recursive: true, force: true })); + setEnv({ + GITNEXUS_HOOK_PROC_ROOT: procRoot, + GITNEXUS_HOOK_PROC_CMDLINE_MAX: '4096', + GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '1000', // positive, so the scan starts + }); + + // Deterministic clock: the scan captures `start` (call #1) and runs its + // Phase-0/1 entry budget checks in-budget; once the escalation loop is under + // way we jump Date.now() past the 1000 ms budget so the loop's in-read + // outOfBudget() returns the CMDLINE_TIMEOUT sentinel. The threshold (>4) is + // chosen so the early checks (start capture, per-entry + Phase-1 pre-read + // checks) stay at base and only the escalation's mid-loop check trips. + const base = Date.now(); + let nowCalls = 0; + const nowSpy = vi.spyOn(Date, 'now').mockImplementation(() => { + nowCalls += 1; + return nowCalls > 4 ? base + 5000 : base; + }); + cleanups.push(() => nowSpy.mockRestore()); + + const verdict = scan(lbugPath, 1); + // The load-bearing assertion: a mid-escalation budget trip yields the + // 'timeout' verdict (via the Symbol sentinel) — NOT a silent non-candidate + // drop (which would be 'not-owned' here and a fail-OPEN if this were a real + // cross-budget owner). + expect(verdict).toBe('timeout'); + nowSpy.mockRestore(); + }); + + // ── D5: GITNEXUS_HOOK_PROC_ROOT is honored ONLY under a test runner (F4) ── + // + // Production hooks run as `node .cjs` with neither VITEST nor + // NODE_ENV=test set; vitest injects both into every worker (verified). F4 + // gates getProcRoot() on that signal so a production env that leaked + // GITNEXUS_HOOK_PROC_ROOT (pointing at an empty/bad tree) cannot turn Linux + // owner detection OFF (no pids -> not-owned -> fail-OPEN, the #1492 class). + // These tests run inside vitest, so the gate is OPEN and injection works (the + // entire fake-procfs suite above already depends on that). Here we prove the + // gate is load-bearing: with the test signals stripped, the override is + // ignored and the scan falls back to the real /proc (so our fake lbug is NOT + // found there -> not-owned), and with them present the override is honored. + + it('honors GITNEXUS_HOOK_PROC_ROOT under the vitest test signal (gate open)', () => { + // Sanity: in this vitest worker VITEST/NODE_ENV are set, so the fake root is + // honored and a fake owner is detected — same mechanism the whole suite uses. + const { owned } = runScan((lbug) => [ + { + pid: 10300, + comm: 'MainThread', + cmdline: GITNEXUS_MCP_ARGV('/x/node_modules/gitnexus/dist/cli/index.js'), + fdTargets: [lbug], + }, + ]); + expect(owned).toBe(true); + }); + + it('ignores GITNEXUS_HOOK_PROC_ROOT when the test signal is absent (gate closed → real /proc)', () => { + // Strip BOTH test signals so getProcRoot() falls back to /proc even though + // GITNEXUS_HOOK_PROC_ROOT points at our fake tree. The fake lbug is not an + // fd under the real /proc, so the scan returns not-owned: proof the override + // is inert in a non-test (production-shaped) context. + const savedVitest = process.env.VITEST; + const savedNodeEnv = process.env.NODE_ENV; + cleanups.push(() => { + if (savedVitest === undefined) delete process.env.VITEST; + else process.env.VITEST = savedVitest; + if (savedNodeEnv === undefined) delete process.env.NODE_ENV; + else process.env.NODE_ENV = savedNodeEnv; + }); + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-probe-f4-')); + cleanups.push(() => fs.rmSync(dir, { recursive: true, force: true })); + const lbugPath = path.join(dir, 'lbug'); + fs.writeFileSync(lbugPath, ''); + const procRoot = createFakeProcRoot([ + { + pid: 10400, + comm: 'MainThread', + cmdline: GITNEXUS_MCP_ARGV('/x/node_modules/gitnexus/dist/cli/index.js'), + fdTargets: [lbugPath], + }, + ]); + cleanups.push(() => fs.rmSync(procRoot, { recursive: true, force: true })); + setEnv({ GITNEXUS_HOOK_PROC_ROOT: procRoot }); + // Now drop the test signals — must happen AFTER setEnv so the gate sees them gone. + delete process.env.VITEST; + delete process.env.NODE_ENV; + const verdict = scanVerdictFn(lbugPath, 1); + // Gate closed -> getProcRoot() returns '/proc'; our fake lbug fd is not in + // the real /proc, so no owner is found. + expect(verdict).toBe('not-owned'); + expect(probe.hasGitNexusDbLockedByGitNexusServer(lbugPath, 1)).toBe(false); + }); + + // ── D4: unreadable candidate fd dir → honest tri-state verdict (F1) ── + // + // /proc//fd is owner-only (mode 0500). A cross-user/root `gitnexus mcp` + // serving a DIFFERENT repo clears Phase 0+1 (cmdline matches) and then EACCES + // here — but its dev+ino was never compared against THIS lbug. The OLD code + // returned 'owned' for every non-ENOENT readdir error, falsely claiming + // ownership and permanently suppressing augment for a repo that process does + // not lock. F1 splits the failure shapes: + // - EACCES / EPERM -> 'timeout' (unverifiable; fail-closed HONESTLY) + // - EIO / ESTALE -> 'timeout' (transient I/O; fail-closed) + // - ENOTDIR / other -> continue (not a real fd dir; treat as non-owner) + // The dispatcher collapses owned+timeout to boolean true, so these assert the + // exported tri-state verdict directly — a boolean check could not tell the F1 + // fix from the old bug. + + it('exports linuxProcScanFindGitNexusServer for white-box verdict assertions', () => { + expect(typeof probe.linuxProcScanFindGitNexusServer).toBe('function'); + }); + + it('candidate fd dir EACCES → verdict timeout (honest fail-closed, NOT owned)', () => { + if (process.getuid && process.getuid() === 0) { + // root bypasses chmod 000, so a real EACCES is not reproducible on this + // host. This disk-based test no-ops under root; the uid-agnostic spy + // tests below cover every F1 errno branch (EACCES/EPERM/EIO/ESTALE/ + // ENOTDIR) regardless of who runs the suite. + return; + } + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-probe-eacces-')); + cleanups.push(() => { + try { + fs.chmodSync(path.join(dir, 'proc', '11001', 'fd'), 0o755); + } catch { + /* ignore */ + } + fs.rmSync(dir, { recursive: true, force: true }); + }); + const lbugPath = path.join(dir, 'lbug'); + fs.writeFileSync(lbugPath, ''); + const procRoot = path.join(dir, 'proc'); + const fdDir = path.join(procRoot, '11001', 'fd'); + fs.mkdirSync(fdDir, { recursive: true }); + fs.writeFileSync(path.join(procRoot, '11001', 'comm'), 'MainThread\n'); + fs.writeFileSync( + path.join(procRoot, '11001', 'cmdline'), + ['node', '/x/node_modules/gitnexus/dist/cli/index.js', 'mcp'].join('\0') + '\0', + ); + fs.chmodSync(fdDir, 0o000); // EACCES on readdir + setEnv({ GITNEXUS_HOOK_PROC_ROOT: procRoot }); + // White-box: assert the verdict is 'timeout' (NOT 'owned' — the F1 point). + const verdict = scanVerdictFn(lbugPath, 1); + expect(verdict).toBe('timeout'); + // And the dispatcher still fails closed (boolean true) on that timeout. + const owned = probe.hasGitNexusDbLockedByGitNexusServer(lbugPath, 1); + expect(owned).toBe(true); + }); + + // uid-agnostic coverage of every F1 fd-readdir errno branch. chmod 000 yields + // no EACCES for root, so the disk-based tests above no-op there — these spy + // fs.readdirSync to throw a chosen errno only for the candidate's fd dir (the + // procRoot enumeration calls through), pinning the F1 split in CI regardless + // of the runner's uid. + for (const { code, expected } of [ + { code: 'EACCES', expected: 'timeout' }, + { code: 'EPERM', expected: 'timeout' }, + { code: 'EIO', expected: 'timeout' }, + { code: 'ESTALE', expected: 'timeout' }, + { code: 'ENOTDIR', expected: 'not-owned' }, + ] as const) { + it(`candidate fd readdir ${code} → verdict ${expected} (uid-agnostic spy)`, () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-probe-fderr-')); + cleanups.push(() => fs.rmSync(dir, { recursive: true, force: true })); + const lbugPath = path.join(dir, 'lbug'); + fs.writeFileSync(lbugPath, ''); + const procRoot = path.join(dir, 'proc'); + const fdDir = path.join(procRoot, '11001', 'fd'); + fs.mkdirSync(fdDir, { recursive: true }); + fs.writeFileSync(path.join(procRoot, '11001', 'comm'), 'MainThread\n'); + fs.writeFileSync( + path.join(procRoot, '11001', 'cmdline'), + ['node', '/x/node_modules/gitnexus/dist/cli/index.js', 'mcp'].join('\0') + '\0', + ); + setEnv({ GITNEXUS_HOOK_PROC_ROOT: procRoot }); + const realReaddir = fs.readdirSync.bind(fs); + const spy = vi.spyOn(fs, 'readdirSync').mockImplementation((p, ...rest) => { + if (typeof p === 'string' && p.endsWith(`${path.sep}fd`)) { + const err = new Error(`mock ${code}`) as NodeJS.ErrnoException; + err.code = code; + throw err; + } + return (realReaddir as (...a: unknown[]) => unknown)(p, ...rest); + }); + cleanups.push(() => spy.mockRestore()); + // White-box: assert the exported tri-state verdict directly (the + // dispatcher would collapse timeout+owned to the same boolean). + expect(scanVerdictFn(lbugPath, 1)).toBe(expected); + spy.mockRestore(); + }); + } + + it('candidate fd path is a FILE (ENOTDIR) → treated as non-owner → not-owned', () => { + // ENOTDIR means the fd entry is not a real /proc//fd directory at all, + // so it is not a plausible live owner. The candidate is skipped (continue); + // with no other candidate the scan ends not-owned (the OLD code wrongly + // returned 'owned' here). Runs on every OS incl. root. + const { verdict, owned } = runScanFdEnotdir(); + expect(verdict).toBe('not-owned'); + expect(owned).toBe(false); + }); + + it('EACCES candidate then a REAL owner later → still detects the real owner', () => { + // Regression guard for the F1 continue/return choice: an EACCES candidate + // must NOT short-circuit the scan in a way that hides a genuine owner. Here + // the EACCES dir yields timeout BEFORE reaching the true owner — timeout is + // the protective (fail-closed) verdict, so dispatcher returns true either + // way. (Ordering in /proc readdir is numeric-string; 11001 < 11050.) + if (process.getuid && process.getuid() === 0) return; // EACCES needs non-root + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-probe-mixed-')); + cleanups.push(() => { + try { + fs.chmodSync(path.join(dir, 'proc', '11001', 'fd'), 0o755); + } catch { + /* ignore */ + } + fs.rmSync(dir, { recursive: true, force: true }); + }); + const lbugPath = path.join(dir, 'lbug'); + fs.writeFileSync(lbugPath, ''); + const procRoot = path.join(dir, 'proc'); + // Candidate A: EACCES fd dir. + const fdDirA = path.join(procRoot, '11001', 'fd'); + fs.mkdirSync(fdDirA, { recursive: true }); + fs.writeFileSync(path.join(procRoot, '11001', 'comm'), 'MainThread\n'); + fs.writeFileSync( + path.join(procRoot, '11001', 'cmdline'), + ['node', '/x/node_modules/gitnexus/dist/cli/index.js', 'mcp'].join('\0') + '\0', + ); + fs.chmodSync(fdDirA, 0o000); + setEnv({ GITNEXUS_HOOK_PROC_ROOT: procRoot }); + const verdict = scanVerdictFn(lbugPath, 1); + // EACCES is hit first and fails closed (timeout) — the protective outcome. + expect(verdict).toBe('timeout'); + expect(probe.hasGitNexusDbLockedByGitNexusServer(lbugPath, 1)).toBe(true); + }); +}); + +// Helper for the ENOTDIR branch: fd is a FILE not a dir, so readdir throws +// ENOTDIR. F1: this candidate is treated as a non-owner (continue) → not-owned. +function runScanFdEnotdir(): { verdict: string; owned: boolean } { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-probe-enotdir-')); + cleanups.push(() => fs.rmSync(dir, { recursive: true, force: true })); + const lbugPath = path.join(dir, 'lbug'); + fs.writeFileSync(lbugPath, ''); + const procRoot = path.join(dir, 'proc'); + const pidDir = path.join(procRoot, '11002'); + fs.mkdirSync(pidDir, { recursive: true }); + fs.writeFileSync(path.join(pidDir, 'comm'), 'MainThread\n'); + fs.writeFileSync( + path.join(pidDir, 'cmdline'), + ['node', '/x/node_modules/gitnexus/dist/cli/index.js', 'mcp'].join('\0') + '\0', + ); + fs.writeFileSync(path.join(pidDir, 'fd'), 'not a dir'); // readdir -> ENOTDIR + setEnv({ GITNEXUS_HOOK_PROC_ROOT: procRoot }); + const verdict = scanVerdictFn(lbugPath, 1); + const owned = probe.hasGitNexusDbLockedByGitNexusServer(lbugPath, 1); + return { verdict, owned }; +} + +// ── D6: live e2e against the REAL /proc ───────────────────────────── +// +// Protects the load-bearing assumption that a real lbug handle is fd-visible in +// /proc//fd (a @ladybugdb/core property; a future move to mmap-only would +// silently regress #1492 with no other test going red). We spawn a child that +// opens an fd on a real temp lbug AND wears a gitnexus-mcp cmdline, then assert +// the scan reports owned. Crucially we assert against OUR holder's identity, not +// "any owner" — this host runs background gitnexus servers, so a bare +// truthiness check could be a false positive. +describe.skipIf(!isLinux)('Linux DB-owner scan — live /proc e2e (#2180)', () => { + it('detects a real fd-visible gitnexus-mcp-shaped lbug holder', async () => { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-e2e-')); + const lbugPath = path.join(dir, 'lbug'); + fs.writeFileSync(lbugPath, ''); + // Give the holder a gitnexus-server cmdline by running it from a + // node_modules/gitnexus/dist/cli/index.js path with an `mcp` arg. + const scriptDir = path.join(dir, 'node_modules', 'gitnexus', 'dist', 'cli'); + fs.mkdirSync(scriptDir, { recursive: true }); + const script = path.join(scriptDir, 'index.js'); + const pidFile = path.join(dir, 'holder.pid'); + fs.writeFileSync( + script, + `const fs=require('fs');` + + `const fd=fs.openSync(${JSON.stringify(lbugPath)},'r');` + + `fs.writeFileSync(${JSON.stringify(pidFile)},String(process.pid));` + + `process.on('SIGTERM',()=>{try{fs.closeSync(fd);}catch{}process.exit(0);});` + + `setInterval(()=>{},1<<30);`, + ); + + const holder = spawn(process.execPath, [script, 'mcp'], { stdio: 'ignore' }); + try { + // Wait for the holder to report ready (pid file written). Widened to ~10s + // (was 5s): a loaded CI runner can be slow to spawn the child, and this is + // the one genuine false-FAIL path in the e2e (the budget timeout below + // merely hollows the assertion rather than failing it). + let holderPid = 0; + for (let i = 0; i < 400; i++) { + try { + const raw = fs.readFileSync(pidFile, 'utf8').trim(); + if (raw) { + holderPid = Number.parseInt(raw, 10); + break; + } + } catch { + /* not ready yet */ + } + await new Promise((r) => setTimeout(r, 25)); + } + expect(holderPid).toBeGreaterThan(0); + + // Confirm the holder really is fd-visible (the property under test). + const fdDir = `/proc/${holderPid}/fd`; + const targetStat = fs.statSync(lbugPath); + const fdVisible = fs.readdirSync(fdDir).some((fd) => { + try { + const st = fs.statSync(path.join(fdDir, fd)); + return st.dev === targetStat.dev && st.ino === targetStat.ino; + } catch { + return false; + } + }); + expect(fdVisible).toBe(true); + + // Real /proc, generous explicit budget. Clear PROC_ROOT (-> real /proc) + // and raise the scan budget via setEnv so the module afterEach restores + // BOTH (no raw process.env mutation leaking to sibling tests). The + // generous budget is load-bearing: this dispatcher maps a budget 'timeout' + // to owned=TRUE, so on a busy host the default 1200ms could be exhausted + // before reaching the holder and the assertion would still pass for the + // WRONG reason (a hollow timeout, not real fd-visible detection). 10s + // keeps the assertion honest. Use a PID we are NOT so the holder is not + // excluded, and assert owned for OUR lbug specifically. + setEnv({ + GITNEXUS_HOOK_PROC_ROOT: undefined, + GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '10000', + }); + const t0 = Date.now(); + const owned = probe.hasGitNexusDbLockedByGitNexusServer(lbugPath, process.pid); + const ms = Date.now() - t0; + expect(owned).toBe(true); + // Coarse regression guard against the old O(procs×fds)+lsof path (~1.2s+). + // The bound sits ABOVE the 10s budget so a legitimately-slow-but-correct + // scan can't trip it — a regression guard, not a tight perf SLA. + expect(ms).toBeLessThan(15000); + } finally { + try { + holder.kill('SIGKILL'); + } catch { + /* ignore */ + } + fs.rmSync(dir, { recursive: true, force: true }); + } + }, 40000); +}); diff --git a/gitnexus/test/unit/hooks.test.ts b/gitnexus/test/unit/hooks.test.ts index 9e2f39890..74b2df749 100644 --- a/gitnexus/test/unit/hooks.test.ts +++ b/gitnexus/test/unit/hooks.test.ts @@ -27,6 +27,7 @@ import { runHook, parseHookOutput, createHookToolDir, + createFakeProcRoot, hookEnv, } from '../utils/hook-test-helpers.js'; @@ -87,6 +88,17 @@ const PLUGIN_HOOK_DB_PROBE = path.resolve( 'hook-db-lock-probe.cjs', ); +// ─── lsof/ps-path lane gate (#2180) ───────────────────────────────── +// +// The owner-detection tests below drive the probe through its lsof + ps backend +// (via the fake lsof/ps in createHookToolDir). That backend is the macOS/other- +// Unix path; #2180 removed the Linux lsof fallback, so on Linux these tests +// would no longer exercise the real dispatch (the cmdline-first procfs scan +// answers instead, and a temp lbug held by nobody is simply not-owned). They +// remain valid coverage for the macOS lane; Linux gets equivalent three-phase +// coverage in test/unit/hook-db-lock-probe.test.ts (fake /proc + a live e2e). +const SKIP_LSOF_PATH = process.platform === 'win32' || process.platform === 'linux'; + // ─── Host guard precheck for orphan-reaping tests (#2163) ─────────── // // The reaping lanes depend on a host coreutils `timeout`/`gtimeout` that @@ -1002,10 +1014,10 @@ describe.skipIf(process.platform === 'win32')( { env: { ...hookEnv(binDir), - // Force the Linux /proc scan to fall through to lsof - // immediately. Must be '1' — do NOT "simplify" to '0': the - // current parser (`Number(raw && String(raw).trim())`) - // treats '0' as falsy and falls back to the 1200ms default. + // The slot gate rejects this invocation before the probe runs + // at all, so this budget never actually bounds a scan — it is + // set low only to keep the test fast in the (asserted-absent) + // case the gate ever regressed and let the probe through. GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '1', }, }, @@ -1156,250 +1168,21 @@ describe.skipIf(process.platform === 'win32')( }, ); -describe.skipIf(process.platform !== 'linux')( - 'Orphaned lsof is reaped by the timeout wrapper (#2163)', - () => { - // T3 — minimal reproduction of the incident mechanism: the hook process - // is SIGKILLed (modeling Claude Code's 10s hook timeout) while a slow, - // SIGTERM-immune lsof child is still running. Before the fix nothing can - // signal that child anymore (and spawnSync's own SIGTERM is ignored - // anyway), so it survives its full 30s sleep → test red regardless of - // race timing. After the fix the coreutils `timeout -k 1` wrapper - // outlives the hook and SIGKILLs the child within ~3s — making this also - // a direct regression test for the wrapper's `-k` capability. - it('CJS: SIGKILLed hook leaves no immortal lsof child', async () => { - // Guard-availability precheck — see resolveHostGuardForReapingTests. - expect(resolveHostGuardForReapingTests(), GUARD_PRECHECK_MSG).not.toBeNull(); - const { spawn } = await import('child_process'); - const lbugPath = path.join(gitNexusDir, 'lbug'); - fs.writeFileSync(lbugPath, ''); - const pidFile = path.join(os.tmpdir(), `gn-hook-lsofpid-${process.pid}`); - fs.rmSync(pidFile, { force: true }); - const binDir = createHookToolDir({ - lsofPidFile: pidFile, - lsofSleepMs: 30000, - lsofIgnoreSigterm: true, - }); - let lsofPid = 0; - let hookChild: ReturnType | null = null; - - const isFakeLsofAlive = () => { - try { - process.kill(lsofPid, 0); - } catch { - return false; // ESRCH — reaped - } - // PID-reuse guard: only count it alive while the cmdline still - // points at our fake lsof. - try { - return fs.readFileSync(`/proc/${lsofPid}/cmdline`, 'utf-8').includes(binDir); - } catch { - return false; - } - }; - - try { - hookChild = spawn(process.execPath, [CJS_HOOK], { - stdio: ['pipe', 'ignore', 'ignore'], - env: { - ...hookEnv(binDir), - // '1', NOT '0' — see the slot-gate test above. - GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '1', - // Hermeticity: hookEnv() spreads process.env, so a stray - // GITNEXUS_HOOK_TIMEOUT_PATH=disabled left in a developer shell - // would turn the wrapper off and fake-red this test. Empty string - // falls through to the built-in candidates (the path under test). - GITNEXUS_HOOK_TIMEOUT_PATH: '', - }, - }); - hookChild.stdin!.end( - JSON.stringify({ - hook_event_name: 'PreToolUse', - tool_name: 'Grep', - tool_input: { pattern: 'validateUser' }, - cwd: tmpDir, - }), - ); - - // The fake lsof writes its PID as its FIRST statement; poll tightly. - const spawnDeadline = Date.now() + 8000; - while (Date.now() < spawnDeadline) { - try { - const raw = fs.readFileSync(pidFile, 'utf-8').trim(); - if (raw) { - lsofPid = Number.parseInt(raw, 10); - break; - } - } catch { - /* not written yet */ - } - await new Promise((r) => setTimeout(r, 10)); - } - expect(lsofPid).toBeGreaterThan(0); - - // Kill the hook while its lsof child is alive. - hookChild.kill('SIGKILL'); - - const reapDeadline = Date.now() + 5000; - let alive = isFakeLsofAlive(); - while (alive && Date.now() < reapDeadline) { - await new Promise((r) => setTimeout(r, 100)); - alive = isFakeLsofAlive(); - } - expect(alive).toBe(false); - } finally { - // PID-reuse guard (#2169 review): re-run the detection loop's - // /proc//cmdline identity check before the cleanup SIGKILL, so - // a PID already reaped and recycled by the OS is never signalled. - if (lsofPid > 0 && isFakeLsofAlive()) { - try { - process.kill(lsofPid, 'SIGKILL'); - } catch { - /* already gone */ - } - } - try { - hookChild?.kill('SIGKILL'); - } catch { - /* ignore */ - } - // The hook claims a slot before probing now; it died holding it. - const lockDir = path.join(gitNexusDir, '.hook-locks'); - try { - for (const f of fs.readdirSync(lockDir)) fs.unlinkSync(path.join(lockDir, f)); - } catch { - /* ignore */ - } - try { - fs.rmdirSync(lockDir); - } catch { - /* ignore */ - } - fs.rmSync(lbugPath, { force: true }); - fs.rmSync(pidFile, { force: true }); - fs.rmSync(binDir, { recursive: true, force: true }); - } - }, 30000); - - // F3 (#2165 review): GITNEXUS_HOOK_TIMEOUT_PATH pointing at an EXISTING - // but unusable path (here: a directory) must not silently disable orphan - // containment. Before the fix, fs.existsSync() accepted the directory as - // THE candidate, its self-test failed, and the wrapper was memoized off — - // no fall-through — so the SIGTERM-immune lsof below survived its full - // 30s sleep. After the fix the env candidate merely goes first in the - // candidate list; failing its self-test falls through to the built-in - // coreutils guard, which still reaps the orphan within ~3s. - it('CJS: env guard pointing at a directory falls through to a working built-in guard', async () => { - // Guard-availability precheck — see resolveHostGuardForReapingTests. - expect(resolveHostGuardForReapingTests(), GUARD_PRECHECK_MSG).not.toBeNull(); - const { spawn } = await import('child_process'); - const lbugPath = path.join(gitNexusDir, 'lbug'); - fs.writeFileSync(lbugPath, ''); - const pidFile = path.join(os.tmpdir(), `gn-hook-lsofpid-dirguard-${process.pid}`); - fs.rmSync(pidFile, { force: true }); - const guardDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-guard-dir-')); - const binDir = createHookToolDir({ - lsofPidFile: pidFile, - lsofSleepMs: 30000, - lsofIgnoreSigterm: true, - }); - let lsofPid = 0; - let hookChild: ReturnType | null = null; - - const isFakeLsofAlive = () => { - try { - process.kill(lsofPid, 0); - } catch { - return false; // ESRCH — reaped - } - try { - return fs.readFileSync(`/proc/${lsofPid}/cmdline`, 'utf-8').includes(binDir); - } catch { - return false; - } - }; - - try { - hookChild = spawn(process.execPath, [CJS_HOOK], { - stdio: ['pipe', 'ignore', 'ignore'], - env: { - ...hookEnv(binDir), - // '1', NOT '0' — see the slot-gate test above. - GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '1', - // Exists but is a directory — spawning it fails the lazy - // self-test, forcing the fall-through path under test. - GITNEXUS_HOOK_TIMEOUT_PATH: guardDir, - }, - }); - hookChild.stdin!.end( - JSON.stringify({ - hook_event_name: 'PreToolUse', - tool_name: 'Grep', - tool_input: { pattern: 'validateUser' }, - cwd: tmpDir, - }), - ); - - const spawnDeadline = Date.now() + 8000; - while (Date.now() < spawnDeadline) { - try { - const raw = fs.readFileSync(pidFile, 'utf-8').trim(); - if (raw) { - lsofPid = Number.parseInt(raw, 10); - break; - } - } catch { - /* not written yet */ - } - await new Promise((r) => setTimeout(r, 10)); - } - expect(lsofPid).toBeGreaterThan(0); - - // Kill the hook while its lsof child is alive (the incident topology). - hookChild.kill('SIGKILL'); - - const reapDeadline = Date.now() + 5000; - let alive = isFakeLsofAlive(); - while (alive && Date.now() < reapDeadline) { - await new Promise((r) => setTimeout(r, 100)); - alive = isFakeLsofAlive(); - } - expect(alive).toBe(false); - } finally { - // PID-reuse guard (#2169 review): re-run the detection loop's - // /proc//cmdline identity check before the cleanup SIGKILL, so - // a PID already reaped and recycled by the OS is never signalled. - if (lsofPid > 0 && isFakeLsofAlive()) { - try { - process.kill(lsofPid, 'SIGKILL'); - } catch { - /* already gone */ - } - } - try { - hookChild?.kill('SIGKILL'); - } catch { - /* ignore */ - } - const lockDir = path.join(gitNexusDir, '.hook-locks'); - try { - for (const f of fs.readdirSync(lockDir)) fs.unlinkSync(path.join(lockDir, f)); - } catch { - /* ignore */ - } - try { - fs.rmdirSync(lockDir); - } catch { - /* ignore */ - } - fs.rmSync(lbugPath, { force: true }); - fs.rmSync(pidFile, { force: true }); - fs.rmSync(guardDir, { recursive: true, force: true }); - fs.rmSync(binDir, { recursive: true, force: true }); - } - }, 30000); - }, -); +// ─── #2180: the probe no longer spawns lsof on Linux ─────────────── +// +// The 'Orphaned lsof is reaped by the timeout wrapper (#2163)' suite that +// lived here (T3 + the env-guard-points-at-a-directory fall-through test) +// drove the Linux probe to spawn a SIGTERM-immune fake lsof and asserted the +// coreutils `timeout -k 1` wrapper reaped it after the hook was SIGKILLed. +// #2180 replaced the O(procs×fds) scan + lsof fallback with a pure cmdline- +// first procfs scan and DELETED the Linux lsof leg entirely, so the probe can +// no longer create an lsof orphan on Linux by construction — there is nothing +// left for those tests to exercise. The wrapper-reaping mechanism they pinned +// is still covered where it still applies: the augment CLI child (the +// direct-exec and npx-grandchild reaping suites below) and the macOS/other- +// Unix lsof+ps path (the `Ladybug DB owner guard` suite, now relaned off +// Linux). The env-guard fall-through self-test behaviour is still pinned by +// the bad-wrapper / dir-guard guard-resolution tests in that relaned suite. // ─── Behavior: SIGKILLed hook cannot strand the augment CLI (#2163 f-up) ── @@ -1428,11 +1211,16 @@ describe.skipIf(process.platform !== 'linux')( // Guard-availability precheck — see resolveHostGuardForReapingTests. expect(resolveHostGuardForReapingTests(), GUARD_PRECHECK_MSG).not.toBeNull(); const { spawn } = await import('child_process'); - // REQUIRED: a real lbug file routes the probe through the fake lsof - // (empty output → no holder PIDs → probe false) so the augment runs - // through the same probe-then-spawn flow as production. + // REQUIRED: a real lbug file means the probe runs. #2180 removed the + // Linux lsof fallback, so we route the probe at an EMPTY fake /proc + // (no gitnexus server holding the fd → not-owned) so the augment runs + // through the same probe-then-spawn flow as production. (Pre-#2180 this + // used GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS:'1' to fall through to a fake + // lsof; that path no longer exists on Linux — a '1' budget now fails + // CLOSED and would skip the augment entirely.) const lbugPath = path.join(gitNexusDir, 'lbug'); fs.writeFileSync(lbugPath, ''); + const emptyProcRoot = createFakeProcRoot([]); const pidFile = path.join(os.tmpdir(), `gn-hook-clipid-${process.pid}-${label}`); fs.rmSync(pidFile, { force: true }); const binDir = createHookToolDir({ @@ -1465,8 +1253,11 @@ describe.skipIf(process.platform !== 'linux')( stdio: ['pipe', 'ignore', 'ignore'], env: { ...hookEnv(binDir), - // '1', NOT '0' — see the slot-gate test above. - GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '1', + // #2180: empty fake /proc → scan completes as not-owned → augment + // runs (the path under test). Generous budget so the scan never + // times out and fails closed. + GITNEXUS_HOOK_PROC_ROOT: emptyProcRoot, + GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '5000', // Hermeticity: a dev-shell GITNEXUS_HOOK_TIMEOUT_PATH=disabled // would unwrap the CLI and fake-red this test. Empty string // falls through to the built-in candidates (the path under test). @@ -1542,6 +1333,7 @@ describe.skipIf(process.platform !== 'linux')( fs.rmSync(lbugPath, { force: true }); fs.rmSync(pidFile, { force: true }); fs.rmSync(binDir, { recursive: true, force: true }); + fs.rmSync(emptyProcRoot, { recursive: true, force: true }); } }, 30000); } @@ -1574,11 +1366,14 @@ describe.skipIf(process.platform !== 'linux')( // Guard-availability precheck — see resolveHostGuardForReapingTests. expect(resolveHostGuardForReapingTests(), GUARD_PRECHECK_MSG).not.toBeNull(); const { spawn } = await import('child_process'); - // REQUIRED: a real lbug file routes the probe through the fake lsof - // (empty output → no holder PIDs → probe false) so the augment runs - // through the same probe-then-spawn flow as production. + // REQUIRED: a real lbug file means the probe runs. #2180 removed the + // Linux lsof fallback, so we route the probe at an EMPTY fake /proc + // (not-owned) so the augment runs through the same probe-then-spawn flow + // as production. (Pre-#2180 this used BUDGET_MS:'1' to fall through to a + // fake lsof; that path no longer exists on Linux.) const lbugPath = path.join(gitNexusDir, 'lbug'); fs.writeFileSync(lbugPath, ''); + const emptyProcRoot = createFakeProcRoot([]); const pidFile = path.join(os.tmpdir(), `gn-hook-npxclipid-${process.pid}`); fs.rmSync(pidFile, { force: true }); // Route self-proof (#2169 review): written by the fake npx as its first @@ -1647,8 +1442,10 @@ describe.skipIf(process.platform !== 'linux')( // copy's require.resolve to find via NODE_PATH. GITNEXUS_HOOK_CLI_PATH: '', NODE_PATH: '', - // '1', NOT '0' — see the slot-gate test above. - GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '1', + // #2180: empty fake /proc → not-owned → augment runs. Generous + // budget so the scan completes rather than failing closed. + GITNEXUS_HOOK_PROC_ROOT: emptyProcRoot, + GITNEXUS_HOOK_LINUX_PROC_BUDGET_MS: '5000', // Hermeticity: fall through to the built-in guard candidates. GITNEXUS_HOOK_TIMEOUT_PATH: '', }, @@ -1726,6 +1523,7 @@ describe.skipIf(process.platform !== 'linux')( fs.rmSync(npxMarkerPath, { force: true }); fs.rmSync(stagedDir, { recursive: true, force: true }); fs.rmSync(binDir, { recursive: true, force: true }); + fs.rmSync(emptyProcRoot, { recursive: true, force: true }); } }, 45000); }, @@ -1988,7 +1786,7 @@ describe('PreToolUse augmentation filtering (integration)', () => { // exit 0 — so strict hook runners (e.g. Codex `PreToolUse`) never see // unexpected output. GITNEXUS_DEBUG is forced off to keep the assertion // deterministic regardless of the ambient environment. - it.skipIf(process.platform === 'win32')( + it.skipIf(SKIP_LSOF_PATH)( `${label}: skips augment SILENTLY when a GitNexus MCP process owns the repo DB`, () => { const markerPath = path.join(os.tmpdir(), `gitnexus-hook-called-${process.pid}-${label}`); @@ -2028,7 +1826,7 @@ describe('PreToolUse augmentation filtering (integration)', () => { // Issue #1913: the skip reason remains recoverable for operators who opt in // via GITNEXUS_DEBUG=1 — stdout stays empty (no augment ran), the diagnostic // appears on stderr. - it.skipIf(process.platform === 'win32')( + it.skipIf(SKIP_LSOF_PATH)( `${label}: surfaces the MCP-owner skip reason only under GITNEXUS_DEBUG`, () => { const markerPath = path.join(os.tmpdir(), `gitnexus-hook-dbg-${process.pid}-${label}`); @@ -2071,7 +1869,7 @@ describe('PreToolUse augmentation filtering (integration)', () => { // have emitted on these; this guards the unified strict gate (incl. the // main() catch handler) across the claude/plugin copies. for (const debugValue of ['0', 'false']) { - it.skipIf(process.platform === 'win32')( + it.skipIf(SKIP_LSOF_PATH)( `${label}: MCP-owner skip stays SILENT with GITNEXUS_DEBUG='${debugValue}' (strict contract)`, () => { const markerPath = path.join( @@ -2114,14 +1912,22 @@ describe('PreToolUse augmentation filtering (integration)', () => { } }); -describe.skipIf(process.platform === 'win32')( +describe.skipIf(SKIP_LSOF_PATH)( 'Ladybug DB owner guard — production-shaped ps + failure modes (#1493)', () => { - // These tests assert owner *detection*: a positive skip is signalled by the - // `[GitNexus] augment skipped` diagnostic. Since #1913 made that diagnostic - // debug-gated (silent by default for strict hook runners), they run with - // GITNEXUS_DEBUG=1 so the discriminator remains observable. Default-silence - // itself is covered by the 'augmentation filtering' describe above. + // These tests assert owner *detection* via the lsof + ps backend: a positive + // skip is signalled by the `[GitNexus] augment skipped` diagnostic. Since + // #1913 made that diagnostic debug-gated (silent by default for strict hook + // runners), they run with GITNEXUS_DEBUG=1 so the discriminator remains + // observable. Default-silence itself is covered by the 'augmentation + // filtering' describe above. + // + // #2180: skipped on Linux (SKIP_LSOF_PATH) — Linux no longer routes through + // lsof/ps, so these would no longer exercise the real dispatch there. They + // stay as the macOS/other-Unix lsof+ps lane; the equivalent Linux owner- + // detection (incl. the EACCES / cross-user fail-closed edge and the budget + // timeout fail-closed) is covered directly against a fake /proc in + // test/unit/hook-db-lock-probe.test.ts. for (const [label, hookPath] of [ ['CJS', CJS_HOOK], ['Plugin', PLUGIN_HOOK], diff --git a/gitnexus/test/utils/hook-test-helpers.ts b/gitnexus/test/utils/hook-test-helpers.ts index 003db10c0..f5203a023 100644 --- a/gitnexus/test/utils/hook-test-helpers.ts +++ b/gitnexus/test/utils/hook-test-helpers.ts @@ -161,6 +161,57 @@ process.exit(0); return binDir; } +// ─── Fake /proc root for the Linux cmdline-first DB-owner scan (#2180) ── +// +// linuxProcScanFindGitNexusServer reads every path under GITNEXUS_HOOK_PROC_ROOT +// (defaulting to /proc in production). These helpers build a fixture tree so the +// three-phase scan (comm -> cmdline -> fd dev+ino) can be unit-tested without +// touching the test host's real, hundreds-of-process /proc — which is both slow +// and nondeterministic (other gitnexus servers may be running). fd entries are +// real symlinks to real files, so fs.statSync on them yields real dev+ino the +// scan can compare against the target lbug. + +export interface FakeProcEntry { + pid: number | string; + /** /proc//comm contents (kernel caps at 15 visible chars; caller models truncation). */ + comm: string; + /** argv tokens; joined with NUL like the real /proc//cmdline. */ + cmdline: string[]; + /** Absolute paths this pid "holds" open — each becomes an fd symlink target. */ + fdTargets?: string[]; + /** When true, make /proc//fd unreadable-shaped by omitting it entirely so readdir throws ENOENT; for EACCES use the logic-path test instead. */ + noFdDir?: boolean; +} + +/** + * Build a fake /proc tree under a fresh temp dir and return its path (use as + * GITNEXUS_HOOK_PROC_ROOT). Caller is responsible for rm-ing the returned dir. + */ +export function createFakeProcRoot(entries: FakeProcEntry[]): string { + const procRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-fakeproc-')); + for (const e of entries) { + const pidDir = path.join(procRoot, String(e.pid)); + fs.mkdirSync(pidDir, { recursive: true }); + fs.writeFileSync(path.join(pidDir, 'comm'), `${e.comm}\n`); + fs.writeFileSync(path.join(pidDir, 'cmdline'), e.cmdline.join('\0') + '\0'); + if (!e.noFdDir) { + const fdDir = path.join(pidDir, 'fd'); + fs.mkdirSync(fdDir, { recursive: true }); + const targets = e.fdTargets ?? []; + targets.forEach((target, i) => { + // Real symlink so statSync(link) follows to the real file's dev+ino — + // exactly what the scan compares against the target lbug. + try { + fs.symlinkSync(target, path.join(fdDir, String(i + 3))); + } catch { + /* best-effort; a missing target just won't match */ + } + }); + } + } + return procRoot; +} + /** A full env that points a spawned hook at the fake tool dir from createHookToolDir. */ export function hookEnv(binDir: string) { return { From 89ffa71a5272b145711ad5e89550027822c944ae Mon Sep 17 00:00:00 2001 From: bluerose <378100977@qq.com> Date: Sat, 13 Jun 2026 21:01:12 +0800 Subject: [PATCH 12/16] feat(cli): add --embeddings-baseurl/-model/-auth-token/-dims flags to analyze (#2140) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(cli): add --embeddings-baseurl/-model/-auth-token/-dims flags to analyze Add four CLI flags to `gitnexus analyze` that configure a custom OpenAI-compatible HTTP embedding endpoint by setting the GITNEXUS_EMBEDDING_URL / _MODEL / _API_KEY / _DIMS env vars the HTTP embedding client already reads. Flags override env vars; env vars keep working as before. URLs are validated (http/https) and dims must be a positive integer. Prints "Using custom embedding endpoint: " when a URL+model pair is configured, and warns when the flags are passed without --embeddings. The new env keys are added to the analyze snapshot/restore set so programmatic callers don't leak state. The non-secret flags are also accepted from .gitnexusrc; the auth token is intentionally CLI/env-only. * fix(analyze): set GITNEXUS_EMBEDDING_DIMS from CLI flags before module import schema.ts reads EMBEDDING_DIMS at module-load time via the static-import chain (analyze.ts -> run-analyze.ts -> schema.ts). The previous approach of setting the env var inside analyzeCommandImpl ran AFTER schema.ts had already loaded with the default 384, causing "Expected: 384, Actual: 4096" errors when using --embeddings-dims 4096. Fix: use Commander's preAction hook to set GITNEXUS_EMBEDDING_* env vars before the lazy import of analyze.ts triggers the schema.ts module load. * fix(analyze): use hook callback arg instead of this in preAction Commander v14 passes the command as first argument, not as this binding. * refactor(cli): rename --embeddings-* analyze flags to singular --embedding-* Aligns the custom embedding endpoint flags with the existing singular tuning flags (--embedding-threads/--embedding-device): --embedding-base-url, --embedding-model, --embedding-auth-token, --embedding-dims. Renames the derived AnalyzeOptions fields and the .gitnexusrc KEY_SPECS keys to match. Behavior-preserving; the GITNEXUS_EMBEDDING_* env vars are unchanged. Refs #2140 review. * fix(cli): validate and normalize --embedding-dims before module-load reads it The preAction hook wrote GITNEXUS_EMBEDDING_DIMS unvalidated, so an invalid value (abc/0/-5/0x10) threw from schema.ts during the lazy import — surfacing as a raw unhandled rejection on the synchronous program.parse path instead of a friendly error. And '1e3' slipped through: schema.ts parseInt froze the vector column at FLOAT[1] while the impl's Number-based check accepted 1000, so http-client requested 1000-dim vectors against a 1-dim column. Extract a dependency-free normalizeEmbeddingDims helper (strict /^\d+$/ + positive, trim-then-validate, canonicalized) shared by both the hook (CLI path, before module-load) and analyzeCommandImpl (direct-call path). All three readers — schema.ts, http-client, and this helper — now agree on one value, and invalid input gets a clean message instead of a crash or a silent mismatch. Refs #2140 review. * fix(cli): mask credentials in the custom embedding endpoint confirmation A base URL with userinfo (http://user:pass@host/v1) or a query token (?api_key=…) passed the new-URL + http/https validation and was printed verbatim in the 'Using custom embedding endpoint:' line, leaking the secret to terminal scrollback and CI logs. Route it through the existing safeUrl() (now exported from http-client) which strips userinfo + query, keeping protocol/host/path. Single source of truth — no second sanitizer. Refs #2140 review. * fix(cli): drop the ineffective embeddingDims .gitnexusrc key embeddingDims as a .gitnexusrc key silently did nothing: .gitnexusrc loads in analyzeCommandImpl, AFTER the lazy import already ran schema.ts's module-load read of GITNEXUS_EMBEDDING_DIMS, so a config value never sized the vector column. Remove it (config now fails closed on the key, like the auth token); URL/MODEL stay as config keys because they're read lazily at runtime. Dims remains available via --embedding-dims or GITNEXUS_EMBEDDING_DIMS. Refs #2140 review. * refactor(cli): narrow the analyze preAction hook to GITNEXUS_EMBEDDING_DIMS Only DIMS is read at module-load (schema.ts), so only it must be set before the lazy import. URL/MODEL/API_KEY are read lazily at runtime, so analyzeCommandImpl is their sole setter — and because the impl's env snapshot is taken AFTER this hook ran, leaving those three in the hook leaked them past restore. Drop them from the hook (the impl already sets+restores them), and capture/restore the pre-hook DIMS baseline via a postAction hook so a CLI --embedding-dims override no longer leaks into a later in-process program.parseAsync. Refs #2140 review. * fix(cli): gate the custom-endpoint confirmation on the embedding flags The confirmation collapsed into one if/else chain that emits at most one message reflecting the run's intent. Gating on embeddingsEnabled stops the 'Using custom embedding endpoint' line from printing on every analyze run when GITNEXUS_EMBEDDING_URL+MODEL merely happen to be set in the environment, and ordering the '--embeddings absent' note first removes the contradiction where it printed alongside 'Using custom embedding endpoint'. Refs #2140 review. * test(cli): cover the custom embedding endpoint flags Adds direct-call (analyzeCommandImpl path) coverage the original PR lacked: URL validation (empty/invalid/non-http), model/token emptiness, dims validation incl. the 1e3 regression, credential masking in the confirmation line, confirmation gating (absent --embeddings; ambient env must not trigger it), CLI-over-env precedence, and the GITNEXUS_EMBEDDING_* snapshot/restore round-trip. Complements embedding-dims.test.ts and http-client-safe-url.test.ts. Refs #2140 review. * test(cli): e2e-cover the --embedding-dims crash path on the real CLI The dims-validation fix lives in the commander preAction hook, which only fires on the program.parse path; the direct analyzeCommand() unit tests bypass it. Add a subprocess e2e (run via tsx, no build) asserting that invalid --embedding-dims (abc/0/-5/1e3/3.5) produces the friendly flag-named error and exit 1 — NOT the raw schema.ts module-load throw that the original bug surfaced. Cases exit inside the hook (no repo/import/pipeline), so they're deterministic and fast. Updates the unit-suite comment to point at it. Refs #2140 review. --------- Co-authored-by: Gergo Magyar --- gitnexus/src/cli/analyze-config.ts | 10 + gitnexus/src/cli/analyze.ts | 117 ++++++++ gitnexus/src/cli/embedding-dims.ts | 41 +++ gitnexus/src/cli/index.ts | 66 +++++ gitnexus/src/core/embeddings/http-client.ts | 8 +- .../analyze-embedding-flags-e2e.test.ts | 89 ++++++ gitnexus/test/unit/analyze-config.test.ts | 14 + .../analyze-embedding-endpoint-flags.test.ts | 266 ++++++++++++++++++ gitnexus/test/unit/embedding-dims.test.ts | 34 +++ .../test/unit/http-client-safe-url.test.ts | 29 ++ 10 files changed, 671 insertions(+), 3 deletions(-) create mode 100644 gitnexus/src/cli/embedding-dims.ts create mode 100644 gitnexus/test/integration/analyze-embedding-flags-e2e.test.ts create mode 100644 gitnexus/test/unit/analyze-embedding-endpoint-flags.test.ts create mode 100644 gitnexus/test/unit/embedding-dims.test.ts create mode 100644 gitnexus/test/unit/http-client-safe-url.test.ts diff --git a/gitnexus/src/cli/analyze-config.ts b/gitnexus/src/cli/analyze-config.ts index 63a328845..048ecc52f 100644 --- a/gitnexus/src/cli/analyze-config.ts +++ b/gitnexus/src/cli/analyze-config.ts @@ -107,6 +107,16 @@ const KEY_SPECS: Record = { // built-in convention set, is otherwise invisible to route_map consumers. // Listing it here adds it to the cross-file consumer scan. fetchWrappers: { target: 'fetchWrappers', kind: 'string-array' }, + // Auth token AND dims are intentionally CLI/env-only — no embeddingAuthToken + // or embeddingDims key here: + // - the token keeps secrets out of a committed .gitnexusrc; + // - dims cannot take effect from .gitnexusrc anyway — schema.ts reads + // GITNEXUS_EMBEDDING_DIMS at module-load (before .gitnexusrc is loaded in + // analyzeCommandImpl), so a config value would size nothing and silently + // mismatch the vector column. Use --embedding-dims or GITNEXUS_EMBEDDING_DIMS. + // (URL/MODEL are safe as config keys: they are read lazily at runtime, not at module-load.) + embeddingBaseUrl: { target: 'embeddingBaseUrl', kind: 'string' }, + embeddingModel: { target: 'embeddingModel', kind: 'string' }, }; /** Top-level container key for the nested form; not itself an `AnalyzeOptions` field. */ diff --git a/gitnexus/src/cli/analyze.ts b/gitnexus/src/cli/analyze.ts index e54601bab..da768c689 100644 --- a/gitnexus/src/cli/analyze.ts +++ b/gitnexus/src/cli/analyze.ts @@ -41,8 +41,10 @@ import { warnMissingOptionalGrammars, getOptionalGrammarExtensions } from './opt import { glob } from 'glob'; import fs from 'fs/promises'; import { cliError } from './cli-message.js'; +import { EMBEDDING_DIMS_ERROR, normalizeEmbeddingDims } from './embedding-dims.js'; import { formatElapsed } from './format-elapsed.js'; import { isHfDownloadFailure } from '../core/embeddings/hf-env.js'; +import { safeUrl } from '../core/embeddings/http-client.js'; import { isLocalEmbeddingRuntimeBlockerMessage } from '../core/embeddings/runtime-support.js'; import { warnIfNpm11NpxRisk } from './resolve-invocation.js'; @@ -560,6 +562,10 @@ const ANALYZE_CLI_ENV_KEYS = [ 'GITNEXUS_EMBEDDING_SUB_BATCH_SIZE', 'GITNEXUS_EMBEDDING_DEVICE', 'GITNEXUS_ANALYZE_PROGRESS_ACTIVE', + 'GITNEXUS_EMBEDDING_URL', + 'GITNEXUS_EMBEDDING_MODEL', + 'GITNEXUS_EMBEDDING_API_KEY', + 'GITNEXUS_EMBEDDING_DIMS', ] as const; type AnalyzeEnvSnapshot = Record<(typeof ANALYZE_CLI_ENV_KEYS)[number], string | undefined>; @@ -677,6 +683,14 @@ export interface AnalyzeOptions { * outside the built-in convention still produces `route_map` consumers. */ fetchWrappers?: string[]; + /** OpenAI-compatible embeddings base URL (incl. /v1). Overrides GITNEXUS_EMBEDDING_URL. */ + embeddingBaseUrl?: string; + /** Embedding model name. Overrides GITNEXUS_EMBEDDING_MODEL. */ + embeddingModel?: string; + /** Bearer token for the embeddings endpoint. Overrides GITNEXUS_EMBEDDING_API_KEY. Never logged. */ + embeddingAuthToken?: string; + /** Embedding vector dimensions (positive integer string). Overrides GITNEXUS_EMBEDDING_DIMS. */ + embeddingDims?: string; } /** @@ -958,6 +972,109 @@ const analyzeCommandImpl = async ( process.env.GITNEXUS_EMBEDDING_DEVICE = options.embeddingDevice; } + // --- Custom HTTP embedding endpoint flags (override GITNEXUS_EMBEDDING_* env vars) --- + const anyHttpEmbedFlag = + options.embeddingBaseUrl !== undefined || + options.embeddingModel !== undefined || + options.embeddingAuthToken !== undefined || + options.embeddingDims !== undefined; + + if (options.embeddingBaseUrl !== undefined) { + const url = options.embeddingBaseUrl.trim(); + if (url.length === 0) { + cliError(' --embedding-base-url must not be empty.\n'); + process.exitCode = 1; + return; + } + let parsed: URL; + try { + parsed = new URL(url); + } catch { + cliError(` --embedding-base-url is not a valid URL: "${url}".\n`); + process.exitCode = 1; + return; + } + if (parsed.protocol !== 'http:' && parsed.protocol !== 'https:') { + cliError(' --embedding-base-url must use http:// or https://.\n'); + process.exitCode = 1; + return; + } + // http-client strips trailing slashes; store as given (trimmed). + process.env.GITNEXUS_EMBEDDING_URL = url; + } + + if (options.embeddingModel !== undefined) { + const model = options.embeddingModel.trim(); + if (model.length === 0) { + cliError(' --embedding-model must not be empty.\n'); + process.exitCode = 1; + return; + } + process.env.GITNEXUS_EMBEDDING_MODEL = model; + } + + if (options.embeddingAuthToken !== undefined) { + const token = options.embeddingAuthToken.trim(); + if (token.length === 0) { + cliError(' --embedding-auth-token must not be empty.\n'); + process.exitCode = 1; + return; + } + // Never log the token value. + process.env.GITNEXUS_EMBEDDING_API_KEY = token; + } + + // Validate + normalize dims through the same shared helper the preAction + // hook uses, so the CLI path, this direct/programmatic-call path, schema.ts + // (parseInt) and http-client (/^\d+$/) all agree on one canonical value. + if (options.embeddingDims !== undefined) { + const dims = normalizeEmbeddingDims(options.embeddingDims); + if (dims === null) { + cliError(` ${EMBEDDING_DIMS_ERROR}\n`); + process.exitCode = 1; + return; + } + process.env.GITNEXUS_EMBEDDING_DIMS = dims; + } + + // Custom-endpoint UX, emitting at most ONE message that reflects THIS run's + // intent (not ambient env). Order matters — the first matching branch wins: + // 1. flags given but --embeddings absent: the endpoint won't be used, so + // say only that (no contradictory "Using…" line). + // 2. embeddings enabled + a complete endpoint (flags or env): confirm it, + // masking the URL via safeUrl() since a base URL may carry credentials + // in userinfo (http://user:pass@host) or a query token (?api_key=…) + // that must not land in stdout/CI logs. The auth token is never printed. + // 3. embeddings enabled but only one of URL/MODEL supplied via flags: + // http-client.isHttpMode() needs BOTH, so warn about the fallback. + // Gating on embeddingsEnabled also stops the old behaviour of printing + // "Using custom embedding endpoint" on every analyze run whenever the env + // vars happened to be set. + if (anyHttpEmbedFlag && !embeddingsEnabled) { + console.log( + ' Note: --embedding-* flags only apply when --embeddings is also passed; ' + + 'no embeddings will be generated this run.\n', + ); + } else if ( + embeddingsEnabled && + process.env.GITNEXUS_EMBEDDING_URL && + process.env.GITNEXUS_EMBEDDING_MODEL + ) { + console.log( + ` Using custom embedding endpoint: ${safeUrl(process.env.GITNEXUS_EMBEDDING_URL)} ` + + `(model: ${process.env.GITNEXUS_EMBEDDING_MODEL})\n`, + ); + } else if ( + embeddingsEnabled && + anyHttpEmbedFlag && + (process.env.GITNEXUS_EMBEDDING_URL || process.env.GITNEXUS_EMBEDDING_MODEL) + ) { + console.log( + ' Note: custom HTTP embeddings require BOTH --embedding-base-url and --embedding-model ' + + '(or the matching env vars). Falling back to local ONNX embeddings.\n', + ); + } + if (options.repairFts && options.force) { cliError( ' Cannot combine `--repair-fts` with `--force`. ' + diff --git a/gitnexus/src/cli/embedding-dims.ts b/gitnexus/src/cli/embedding-dims.ts new file mode 100644 index 000000000..aee8d07a2 --- /dev/null +++ b/gitnexus/src/cli/embedding-dims.ts @@ -0,0 +1,41 @@ +/** + * Strict positive-integer normalization for the `--embedding-dims` flag / + * `GITNEXUS_EMBEDDING_DIMS` value. + * + * Single source of truth shared by two write paths: + * 1. the `analyze` `preAction` hook (CLI path) — it must set the env var + * BEFORE `schema.ts` reads `GITNEXUS_EMBEDDING_DIMS` at module-load time + * (the static import chain `analyze.ts → run-analyze.ts → schema.ts` + * bakes `FLOAT[dims]` into the vector-table DDL), and + * 2. `analyzeCommandImpl` (direct / programmatic-call path, which bypasses + * the commander hook). + * + * Keep this module dependency-free. `index.ts` imports it eagerly, so pulling + * in anything that transitively loads `schema.ts` (e.g. `analyze.ts`) — or + * even `cli-message.ts`, which drags in the logger + i18n — would defeat the + * lazy `import('./analyze.js')` the hook exists to enable. Callers print the + * error themselves (the hook to stderr, the impl via `cliError`). + * + * Trim-then-validate, matching the sibling URL/MODEL/TOKEN flags: surrounding + * whitespace is tolerated, but the remaining value must be all digits and + * `> 0`. This rejects scientific notation (`1e3`), hex (`0x10`), fractions + * (`3.5`), signs (`+5`/`-5`), and trailing junk (`4096x`) so the three + * downstream readers — `schema.ts` (`parseInt`), `http-client.ts` (`/^\d+$/`), + * and this helper — all agree on one canonical value. Without it, `1e3` parsed + * to `FLOAT[1]` at module-load but requested 1000-dim vectors at runtime. + */ + +/** Shared error message so both call sites surface identical wording. */ +export const EMBEDDING_DIMS_ERROR = '--embedding-dims must be a positive integer.'; + +/** + * Returns the canonical positive-integer string (e.g. `"007"` → `"7"`), or + * `null` when the input is not a strict positive integer. + */ +export const normalizeEmbeddingDims = (raw: string): string | null => { + const trimmed = raw.trim(); + if (!/^\d+$/.test(trimmed) || parseInt(trimmed, 10) <= 0) { + return null; + } + return String(parseInt(trimmed, 10)); +}; diff --git a/gitnexus/src/cli/index.ts b/gitnexus/src/cli/index.ts index f3f329ed2..318ea736a 100644 --- a/gitnexus/src/cli/index.ts +++ b/gitnexus/src/cli/index.ts @@ -6,6 +6,7 @@ import { Command } from 'commander'; import { createRequire } from 'node:module'; import { createLazyAction, createLbugLazyAction } from './lazy-action.js'; +import { EMBEDDING_DIMS_ERROR, normalizeEmbeddingDims } from './embedding-dims.js'; import { registerGroupCommands } from './group.js'; import { localizeCliHelp } from './help-i18n.js'; import { t } from './i18n/index.js'; @@ -40,6 +41,15 @@ program .option('-f, --force', 'Apply the changes (default is a dry-run preview)') .action(createLazyAction(() => import('./uninstall.js'), 'uninstallCommand')); +// Baseline of GITNEXUS_EMBEDDING_DIMS captured by the analyze preAction hook +// before it overwrites the var, so the postAction hook can restore it. The +// analyzeCommand env snapshot is taken AFTER this hook runs, so it cannot undo +// the hook's write on its own — without this restore a CLI --embedding-dims +// would leak into a later in-process program.parseAsync (tests / long-running +// hosts). Single-shot CLI exits the process, making the restore a no-op there. +let dimsEnvBaseline: string | undefined; +let dimsEnvCaptured = false; + program .command('analyze [path]') .description('Index a repository (full analysis)') @@ -121,7 +131,63 @@ program .option('--embedding-batch-size ', 'Number of nodes per embedding batch') .option('--embedding-sub-batch-size ', 'Number of chunks per embedding model call') .option('--embedding-device ', 'Embedding device: auto, cpu, dml, cuda, or wasm') + .option( + '--embedding-base-url ', + 'OpenAI-compatible embeddings base URL including the /v1 suffix ' + + '(e.g. http://10.219.32.29:11434/v1 for Ollama). Overrides GITNEXUS_EMBEDDING_URL.', + ) + .option( + '--embedding-model ', + 'Embedding model name (e.g. qwen3-embedding:8b). Overrides GITNEXUS_EMBEDDING_MODEL.', + ) + .option( + '--embedding-auth-token ', + 'Bearer token for the embeddings endpoint (omit for unauthenticated servers like Ollama). ' + + 'Overrides GITNEXUS_EMBEDDING_API_KEY.', + ) + .option( + '--embedding-dims ', + 'Embedding vector dimensions (positive integer; e.g. 4096 for Qwen3-Embedding-8B). ' + + 'Must match what the index was built with. Overrides GITNEXUS_EMBEDDING_DIMS.', + ) .addHelpText('after', () => t('help.analyze.environment')) + .hook('preAction', (thisCommand: Command) => { + // ONLY GITNEXUS_EMBEDDING_DIMS must be set here: schema.ts reads it at + // module-load time during the lazy import('./analyze.js') below (via the + // static chain analyze.ts → run-analyze.ts → schema.ts), so deferring to + // analyzeCommandImpl would be too late. URL / MODEL / API_KEY are read + // lazily at runtime (readConfig), so analyzeCommandImpl is their sole + // setter — keeping them out of this hook means they fall under the impl's + // env snapshot/restore and don't leak across in-process invocations. + const dimsOpt = thisCommand.opts()['embeddingDims']; + if (dimsOpt !== undefined) { + // Validate + normalize BEFORE writing the env var: schema.ts throws on a + // bad value at module-load, which — on the synchronous program.parse() + // path, before the analyze fatal-handlers are installed — would surface + // as a raw unhandled rejection instead of this friendly message. + const dims = normalizeEmbeddingDims(String(dimsOpt)); + if (dims === null) { + process.stderr.write(`\n ${EMBEDDING_DIMS_ERROR}\n\n`); + process.exit(1); + } + dimsEnvBaseline = process.env.GITNEXUS_EMBEDDING_DIMS; + dimsEnvCaptured = true; + process.env.GITNEXUS_EMBEDDING_DIMS = dims; + } + }) + .hook('postAction', () => { + // Restore the pre-hook GITNEXUS_EMBEDDING_DIMS so a CLI override doesn't + // persist into a later program.parseAsync in the same process. (Fires on a + // microtask after a successful parse; the crash path never reaches here, + // but the hook validates dims before writing, so there's nothing to undo.) + if (!dimsEnvCaptured) return; + dimsEnvCaptured = false; + if (dimsEnvBaseline === undefined) { + delete process.env.GITNEXUS_EMBEDDING_DIMS; + } else { + process.env.GITNEXUS_EMBEDDING_DIMS = dimsEnvBaseline; + } + }) .action(createLbugLazyAction(() => import('./analyze.js'), 'analyzeCommand')); program diff --git a/gitnexus/src/core/embeddings/http-client.ts b/gitnexus/src/core/embeddings/http-client.ts index e8e9073ff..cab2de8ce 100644 --- a/gitnexus/src/core/embeddings/http-client.ts +++ b/gitnexus/src/core/embeddings/http-client.ts @@ -70,10 +70,12 @@ export const isHttpMode = (): boolean => readConfig() !== null; export const getHttpDimensions = (): number | undefined => readConfig()?.dimensions; /** - * Return a safe representation of a URL for error messages. - * Strips query string (may contain tokens) and userinfo. + * Return a safe representation of a URL for logs and error messages. + * Strips query string (may contain tokens) and userinfo (may contain + * credentials), keeping protocol + host + path. Exported so the CLI's + * custom-endpoint confirmation can mask the same way. */ -const safeUrl = (url: string): string => { +export const safeUrl = (url: string): string => { try { const u = new URL(url); return `${u.protocol}//${u.host}${u.pathname}`; diff --git a/gitnexus/test/integration/analyze-embedding-flags-e2e.test.ts b/gitnexus/test/integration/analyze-embedding-flags-e2e.test.ts new file mode 100644 index 000000000..f34595832 --- /dev/null +++ b/gitnexus/test/integration/analyze-embedding-flags-e2e.test.ts @@ -0,0 +1,89 @@ +/** + * E2E: the analyze `--embedding-dims` validation on the REAL CLI parse path. + * + * The fix for the dims crash lives in the commander `preAction` hook in + * src/cli/index.ts, which must validate the value BEFORE the lazy + * import('./analyze.js') triggers schema.ts's module-load read of + * GITNEXUS_EMBEDDING_DIMS (which throws on a bad value). Direct + * analyzeCommand() unit tests bypass the hook entirely, so this interaction — + * commander hook timing + synchronous program.parse + lazy import + the + * module-load throw — can only be exercised by running the actual binary. + * + * Reliability: every case here is INVALID dims, so the hook prints a friendly + * error and process.exit(1)s *before* the action runs. No repo, no lazy + * import, no pipeline, no DB, no network — just tsx startup + commander parse. + * That makes these deterministic and fast, unlike the full-analyze e2e cases. + * + * Run via tsx (no build step), mirroring test/integration/cli-e2e.test.ts. + */ +import { spawnSync } from 'child_process'; +import { createRequire } from 'module'; +import os from 'os'; +import path from 'path'; +import fs from 'fs'; +import { fileURLToPath, pathToFileURL } from 'url'; + +import { afterAll, beforeAll, describe, expect, it } from 'vitest'; + +const testDir = path.dirname(fileURLToPath(import.meta.url)); +const repoRoot = path.resolve(testDir, '../..'); +const cliEntry = path.join(repoRoot, 'src/cli/index.ts'); + +const _require = createRequire(import.meta.url); +const tsxPkgDir = path.dirname(_require.resolve('tsx/package.json')); +const tsxImportUrl = pathToFileURL(path.join(tsxPkgDir, 'dist', 'loader.mjs')).href; + +let cwd: string; + +beforeAll(() => { + // The hook exits before any repo logic, so this dir need not be a git repo. + cwd = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-embed-dims-e2e-')); +}); + +afterAll(() => { + fs.rmSync(cwd, { recursive: true, force: true }); +}); + +function runAnalyze(args: string[]) { + // Strip any ambient GITNEXUS_EMBEDDING_* so the flag is the sole input. + const env = { ...process.env } as Record; + for (const k of Object.keys(env)) { + if (k.startsWith('GITNEXUS_EMBEDDING_')) delete env[k]; + } + // Pre-set the heap cap so analyzeCommand's ensureHeap() wouldn't re-exec + // (would drop the tsx loader). Irrelevant on the invalid-dims path since the + // hook exits first, but harmless and matches the cli-e2e harness. + env.NODE_OPTIONS = `${process.env.NODE_OPTIONS || ''} --max-old-space-size=8192`.trim(); + return spawnSync(process.execPath, ['--import', tsxImportUrl, cliEntry, 'analyze', ...args], { + cwd, + encoding: 'utf8', + timeout: 30_000, + stdio: ['pipe', 'pipe', 'pipe'], + env, + }); +} + +describe('analyze --embedding-dims validation (real CLI parse path)', () => { + it.each(['abc', '0', '-5', '1e3', '3.5'])( + 'rejects %j with a friendly error, not a raw module-load crash', + (bad) => { + const result = runAnalyze([cwd, '--embedding-dims', bad]); + const stderr = result.stderr ?? ''; + + // Non-zero exit (the hook called process.exit(1)). + expect(result.status).toBe(1); + + // The friendly, flag-named message surfaced... + expect(stderr).toContain('--embedding-dims must be a positive integer'); + + // ...and NOT the raw schema.ts throw (env-var-named) that would appear if + // the hook validation were removed and schema.ts crashed during the lazy + // import. This is the regression guard for finding #1. + expect(stderr).not.toContain('GITNEXUS_EMBEDDING_DIMS must be a positive integer'); + // No unhandled-rejection / stack-trace leakage either. + expect(stderr).not.toContain('UnhandledPromiseRejection'); + expect(stderr).not.toMatch(/^\s+at .+:\d+:\d+/m); + }, + 30_000, + ); +}); diff --git a/gitnexus/test/unit/analyze-config.test.ts b/gitnexus/test/unit/analyze-config.test.ts index 00637b56f..988c9cb87 100644 --- a/gitnexus/test/unit/analyze-config.test.ts +++ b/gitnexus/test/unit/analyze-config.test.ts @@ -55,6 +55,20 @@ describe('analyze-config (.gitnexusrc support, #243)', () => { expect(() => loadAnalyzeConfig(dir)).toThrow(/Unknown key "defalutBranch"/); }); + it('accepts embeddingBaseUrl / embeddingModel but rejects embeddingDims (CLI/env-only)', async () => { + // URL + MODEL are read lazily at runtime, so they are valid config keys. + await writeRc(JSON.stringify({ embeddingBaseUrl: 'http://h/v1', embeddingModel: 'm' })); + expect(loadAnalyzeConfig(dir)).toEqual({ + embeddingBaseUrl: 'http://h/v1', + embeddingModel: 'm', + }); + // DIMS is read at module-load (before .gitnexusrc), so it is intentionally + // not a config key — a typo'd/intended value fails closed rather than + // silently sizing nothing. + await writeRc(JSON.stringify({ embeddingDims: 4096 })); + expect(() => loadAnalyzeConfig(dir)).toThrow(/Unknown key "embeddingDims"/); + }); + it('parses the flat form and maps aliases onto AnalyzeOptions', async () => { await writeRc( JSON.stringify({ diff --git a/gitnexus/test/unit/analyze-embedding-endpoint-flags.test.ts b/gitnexus/test/unit/analyze-embedding-endpoint-flags.test.ts new file mode 100644 index 000000000..69c51e34c --- /dev/null +++ b/gitnexus/test/unit/analyze-embedding-endpoint-flags.test.ts @@ -0,0 +1,266 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; + +const runFullAnalysisMock = vi.fn(); + +vi.mock('../../src/core/run-analyze.js', () => ({ + runFullAnalysis: runFullAnalysisMock, +})); + +vi.mock('../../src/core/lbug/lbug-adapter.js', () => ({ + closeLbug: vi.fn(async () => undefined), +})); + +vi.mock('../../src/storage/repo-manager.js', () => ({ + getStoragePaths: vi.fn(() => ({ storagePath: '.gitnexus', lbugPath: '.gitnexus/lbug' })), + getGlobalRegistryPath: vi.fn(() => 'registry.json'), + RegistryNameCollisionError: class RegistryNameCollisionError extends Error {}, + AnalysisNotFinalizedError: class AnalysisNotFinalizedError extends Error {}, + assertAnalysisFinalized: vi.fn(async () => undefined), +})); + +vi.mock('../../src/storage/git.js', () => ({ + getGitRoot: vi.fn(() => '/repo'), + hasGitDir: vi.fn(() => true), +})); + +vi.mock('../../src/core/ingestion/utils/max-file-size.js', () => ({ + getMaxFileSizeBannerMessage: vi.fn(() => null), +})); + +// These tests invoke analyzeCommand directly (programmatic-call path), which +// bypasses the commander preAction hook. So they exercise analyzeCommandImpl's +// own validation/normalization and the confirmation-message gating — the +// direct-call half of the fixes. The hook half — the dims crash-path on the +// REAL program.parse path — is exercised end-to-end in +// test/integration/analyze-embedding-flags-e2e.test.ts (it would have caught +// the original raw-crash regression). The canonical dims normalization is +// unit-tested in embedding-dims.test.ts. (The postAction DIMS baseline restore +// is in-process-parseAsync-only and remains covered by code-read.) +const EMBED_ENV_KEYS = [ + 'GITNEXUS_EMBEDDING_URL', + 'GITNEXUS_EMBEDDING_MODEL', + 'GITNEXUS_EMBEDDING_API_KEY', + 'GITNEXUS_EMBEDDING_DIMS', +] as const; + +describe('analyzeCommand custom embedding endpoint flags', () => { + const savedEnv: Record = {}; + + beforeEach(() => { + vi.resetModules(); + runFullAnalysisMock.mockReset(); + runFullAnalysisMock.mockResolvedValue({ + repoName: 'repo', + repoPath: '/repo', + stats: {}, + alreadyUpToDate: true, + }); + process.exitCode = undefined; + process.env.NODE_OPTIONS = `${process.env.NODE_OPTIONS ?? ''} --max-old-space-size=8192`.trim(); + // Start every case from a clean embedding env so ambient values can't mask + // a regression (and restore the host's afterwards). + for (const k of EMBED_ENV_KEYS) { + savedEnv[k] = process.env[k]; + delete process.env[k]; + } + }); + + afterEach(() => { + for (const k of EMBED_ENV_KEYS) { + if (savedEnv[k] === undefined) delete process.env[k]; + else process.env[k] = savedEnv[k]; + } + }); + + const importCmd = async () => (await import('../../src/cli/analyze.js')).analyzeCommand; + + const captureStderr = () => { + const spy = vi.spyOn(process.stderr, 'write').mockImplementation(() => true); + return { + text: () => spy.mock.calls.map(([c]) => (typeof c === 'string' ? c : c.toString())).join(''), + restore: () => spy.mockRestore(), + }; + }; + + const captureLog = () => { + const spy = vi.spyOn(console, 'log').mockImplementation(() => undefined); + return { + text: () => spy.mock.calls.map((args) => args.map(String).join(' ')).join('\n'), + restore: () => spy.mockRestore(), + }; + }; + + // ── URL validation (analyzeCommandImpl) ──────────────────────────── + + it.each([ + ['', 'must not be empty'], + ['not a url', 'not a valid URL'], + ['ftp://host/v1', 'must use http:// or https://'], + ])('rejects --embedding-base-url %j before analysis', async (embeddingBaseUrl, expected) => { + const err = captureStderr(); + const analyzeCommand = await importCmd(); + + await analyzeCommand(undefined, { embeddingBaseUrl }); + + expect(process.exitCode).toBe(1); + expect(runFullAnalysisMock).not.toHaveBeenCalled(); + expect(err.text()).toContain(expected); + err.restore(); + }); + + it('accepts a valid http(s) base URL with model and proceeds to analysis', async () => { + const analyzeCommand = await importCmd(); + + await analyzeCommand(undefined, { + embeddings: true, + embeddingBaseUrl: 'http://10.0.0.1:11434/v1', + embeddingModel: 'qwen3', + }); + + expect(process.exitCode).not.toBe(1); + expect(runFullAnalysisMock).toHaveBeenCalledTimes(1); + }); + + // ── Model / token emptiness ──────────────────────────────────────── + + it('rejects a whitespace-only --embedding-model', async () => { + const err = captureStderr(); + const analyzeCommand = await importCmd(); + + await analyzeCommand(undefined, { embeddingModel: ' ' }); + + expect(process.exitCode).toBe(1); + expect(runFullAnalysisMock).not.toHaveBeenCalled(); + expect(err.text()).toContain('--embedding-model must not be empty'); + err.restore(); + }); + + it('rejects an empty --embedding-auth-token', async () => { + const err = captureStderr(); + const analyzeCommand = await importCmd(); + + await analyzeCommand(undefined, { embeddingAuthToken: '' }); + + expect(process.exitCode).toBe(1); + expect(runFullAnalysisMock).not.toHaveBeenCalled(); + expect(err.text()).toContain('--embedding-auth-token must not be empty'); + err.restore(); + }); + + // ── Dims validation on the direct-call path (finding #1) ─────────── + + it.each(['1e3', 'abc', '0', '-5', '3.5'])( + 'rejects invalid --embedding-dims %j with a friendly error (no crash)', + async (embeddingDims) => { + const err = captureStderr(); + const analyzeCommand = await importCmd(); + + await analyzeCommand(undefined, { embeddings: true, embeddingDims }); + + expect(process.exitCode).toBe(1); + expect(runFullAnalysisMock).not.toHaveBeenCalled(); + expect(err.text()).toContain('--embedding-dims must be a positive integer'); + err.restore(); + }, + ); + + it('accepts a valid --embedding-dims and proceeds', async () => { + const analyzeCommand = await importCmd(); + + await analyzeCommand(undefined, { embeddings: true, embeddingDims: '4096' }); + + expect(process.exitCode).not.toBe(1); + expect(runFullAnalysisMock).toHaveBeenCalledTimes(1); + }); + + // ── Credential masking in the confirmation line (finding #2) ─────── + + it('masks userinfo credentials in the confirmation line', async () => { + const log = captureLog(); + const analyzeCommand = await importCmd(); + + await analyzeCommand(undefined, { + embeddings: true, + embeddingBaseUrl: 'http://user:s3cret@host:11434/v1', + embeddingModel: 'qwen3', + }); + + const out = log.text(); + expect(out).toContain('Using custom embedding endpoint'); + expect(out).toContain('host:11434'); + expect(out).not.toContain('s3cret'); + expect(out).not.toContain('user:'); + log.restore(); + }); + + // ── Confirmation gating (finding #5 / U6) ────────────────────────── + + it('does not print "Using custom embedding endpoint" when --embeddings is absent', async () => { + const log = captureLog(); + const analyzeCommand = await importCmd(); + + await analyzeCommand(undefined, { + embeddingBaseUrl: 'http://host/v1', + embeddingModel: 'qwen3', + }); + + const out = log.text(); + expect(out).not.toContain('Using custom embedding endpoint'); + expect(out).toContain('no embeddings will be generated'); + log.restore(); + }); + + it('does not print the confirmation on a plain run when env vars merely happen to be set', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://ambient/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'ambient'; + const log = captureLog(); + const analyzeCommand = await importCmd(); + + // No embedding flags and no --embeddings: ambient env must not trigger it. + await analyzeCommand(undefined, {}); + + expect(log.text()).not.toContain('Using custom embedding endpoint'); + log.restore(); + }); + + // ── CLI flag overrides ambient env during the run ────────────────── + + it('uses the CLI base URL over a pre-existing GITNEXUS_EMBEDDING_URL', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://old/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'old-model'; + const log = captureLog(); + const analyzeCommand = await importCmd(); + + await analyzeCommand(undefined, { + embeddings: true, + embeddingBaseUrl: 'http://new/v1', + embeddingModel: 'new-model', + }); + + const out = log.text(); + expect(out).toContain('http://new/v1'); + expect(out).toContain('new-model'); + expect(out).not.toContain('old/v1'); + expect(out).not.toContain('old-model'); + log.restore(); + }); + + // ── Env snapshot/restore round-trip (finding #4 / U5) ────────────── + + it('restores GITNEXUS_EMBEDDING_* to their pre-call state after returning', async () => { + const analyzeCommand = await importCmd(); + + // Clean env (deleted in beforeEach); a validation early-return path + // guarantees the finally-block restore runs. + await analyzeCommand(undefined, { + embeddingBaseUrl: 'http://host/v1', + embeddingModel: 'qwen3', + embeddingDims: 'bad', // fails after URL/MODEL were written → early return + }); + + expect(process.exitCode).toBe(1); + for (const k of EMBED_ENV_KEYS) { + expect(process.env[k]).toBeUndefined(); + } + }); +}); diff --git a/gitnexus/test/unit/embedding-dims.test.ts b/gitnexus/test/unit/embedding-dims.test.ts new file mode 100644 index 000000000..7d9d78ada --- /dev/null +++ b/gitnexus/test/unit/embedding-dims.test.ts @@ -0,0 +1,34 @@ +import { describe, expect, it } from 'vitest'; + +import { EMBEDDING_DIMS_ERROR, normalizeEmbeddingDims } from '../../src/cli/embedding-dims.js'; + +describe('normalizeEmbeddingDims', () => { + it.each([ + ['4096', '4096'], + ['384', '384'], + ['1', '1'], + ['007', '7'], // canonicalized + [' 4096 ', '4096'], // trim-then-validate + ])('accepts %j and canonicalizes to %j', (input, expected) => { + expect(normalizeEmbeddingDims(input)).toBe(expected); + }); + + it.each(['abc', '0', '00', '-5', '', ' ', '0x10', '3.5', '+5', '4096x', 'Infinity'])( + 'rejects %j (returns null)', + (input) => { + expect(normalizeEmbeddingDims(input)).toBeNull(); + }, + ); + + // Regression for tri-review finding #1 (B): "1e3" must NOT be accepted. + // parseInt("1e3",10) === 1 froze the vector column at FLOAT[1] while + // Number("1e3") === 1000 passed the old validator and requested 1000-dim + // vectors — a silent dimension mismatch. The strict /^\d+$/ rule rejects it. + it('rejects scientific notation 1e3 (no silent FLOAT[1]-vs-1000 mismatch)', () => { + expect(normalizeEmbeddingDims('1e3')).toBeNull(); + }); + + it('exposes a positive-integer error message naming the flag', () => { + expect(EMBEDDING_DIMS_ERROR).toContain('--embedding-dims'); + }); +}); diff --git a/gitnexus/test/unit/http-client-safe-url.test.ts b/gitnexus/test/unit/http-client-safe-url.test.ts new file mode 100644 index 000000000..2fc0c3ab5 --- /dev/null +++ b/gitnexus/test/unit/http-client-safe-url.test.ts @@ -0,0 +1,29 @@ +import { describe, expect, it } from 'vitest'; + +import { safeUrl } from '../../src/core/embeddings/http-client.js'; + +describe('safeUrl', () => { + it('strips userinfo credentials, keeping host + port + path', () => { + const masked = safeUrl('http://user:s3cret@host:11434/v1'); + expect(masked).not.toContain('user'); + expect(masked).not.toContain('s3cret'); + expect(masked).toContain('host:11434'); + expect(masked).toContain('/v1'); + }); + + it('strips a query string that may carry a token', () => { + const masked = safeUrl('http://host/v1?api_key=secret123'); + expect(masked).not.toContain('secret123'); + expect(masked).not.toContain('?'); + expect(masked).toContain('host'); + expect(masked).toContain('/v1'); + }); + + it('returns a sentinel for an unparseable URL instead of echoing it', () => { + expect(safeUrl('://nope')).toBe(''); + }); + + it('passes a plain URL through (protocol + host + path)', () => { + expect(safeUrl('http://10.219.32.29:11434/v1')).toBe('http://10.219.32.29:11434/v1'); + }); +}); From 96dc368d9675404fd00be0b50432fa484c473f82 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sat, 13 Jun 2026 16:15:49 +0100 Subject: [PATCH 13/16] fix(ci): align tree-sitter readiness + grammar-update workflows on a shared manifest (#858) (#2187) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * chore(ci): add shared vendored-grammars manifest; monitor reads it .github/vendored-grammars.json is the single source of truth for the vendored tree-sitter grammars (c/swift/kotlin/dart/proto): name, upstream coords, and policy holds. update-vendored-grammars.mjs now builds its GRAMMARS map from the manifest (behavior-preserving — same exported shape). Adds manifest-agreement tests so the loader can't silently skew from the file. * fix(ci): classify vendored grammars from manifest, drop bare "?" (#858) The readiness report decided "is this vendored?" via is_vendored_pin (a file: package.json spec) — but the 5 vendored grammars aren't in package.json, so they were misrouted through the npm path and rendered bare "?" for ABI (read from an empty node_modules), plus a spurious "? (fetch failed)" for github-only proto. Now vendored grammars are classified by membership in the shared manifest and their ABI is read from gitnexus/vendor//src/parser.c (always in a checkout). github-only vendored grammars skip the npm peer-dep fetch; the tree-sitter-c hold is surfaced from the manifest (held, not plain "Ready"); and every remaining unintrospectable value renders a labeled token, never a bare "?". --assert-current now covers the vendored grammars too instead of skipping them. Adds a stdlib unittest suite incl. a manifest⇄vendor-dir consistency guard. * docs(ci): document the shared vendored-grammars manifest Both tree-sitter workflow headers now point at .github/vendored-grammars.json as the shared source of truth; the readiness workflow gains a PR-path trigger on the manifest + test, and runs the readiness unit tests on validation events. CONTRIBUTING.md documents the manifest contract under CI automation contracts. * fix(review): apply autofix feedback - Guard manifest reads in both scripts with a clear error (was an opaque module-import traceback that crashed the script and test collection). - Never render a bare "?": relabel the npm-path ABI/version/peer sentinels and the vendored upstream-ABI miss to labeled tokens; the report is now ?-free regardless of node_modules/network, and the test is hermetic. - Add a VENDORED_NAMES ⊆ GRAMMARS guard + manifest-missing error test. - Drop now-dead is_vendored_pin/is_vendored/_(vendored)_. - Compose held + out-of-range vendored blocker reasons instead of overwriting. - Reword the shared-manifest docs to not over-claim shared upstream coords. * fix(ci): apply root prettier formatting to mjs + ts test The quality/format gate runs root `prettier --check .` (printWidth 100, the gitnexus-local config differs and falsely passed locally). * test(ci): make both tree-sitter scripts testable offline The scripts hit live npm/GitHub, which makes the report run flaky and the monitor's detect/apply logic untestable. Add hermetic seams: - readiness: --offline flag (+ GITNEXUS_TS_READINESS_OFFLINE env) no-ops the npm registry + upstream fetches; the report renders deterministically (vendored ABIs from the repo, npm columns marked 'offline', no bare '?'). 3 tests assert an offline run touches ZERO network (urlopen patched to raise). - monitor: detect() and apply() accept injected deps (vendoredVersion/ resolveUpstream/fetchSource/readAbi) so the newer/ABI/hold gating runs offline with fixtures; apply gains --dry-run (validates but writes nothing). 6 tests cover newer/same-version/held-c/ABI-15/applicable + a no-mutation dry-run. * fix(review): keep --assert-current hermetic + harden the no-bare-? invariant Tri-review findings (PR #2187): - P2 REGRESSION: --assert-current (documented 'hermetic and offline', run in CI without --offline) routed the 5 vendored grammars through vendored_drift_summary, which fetches upstream parser.c + commit sha — 10 discarded network calls per run. Fix: read the vendored ABI locally via a new vendored_abi_from_repo() helper (also used by vendored_drift_summary). Now verifiably network-free. - Unify the upstream-ABI miss sentinel: prose said 'n/a (generated at build)' while the matrix said 'n/a' — and 'generated at build' is a wrong cause (swift HAS a committed parser.c). Both now render neutral 'n/a'. - Fix the stale assert_current docstring claiming swift is prebuilt-only/no parser.c. - Guard the last latent bare-? path (vendor package.json missing 'version'). Tests: AssertCurrent (network-free guard + out-of-range via the new injection point), malformed-JSON manifest, detect() error-path, explicit npm/github undefined assertions. 17 Python + 15 vitest, all hermetic. * fix(review): use a single unittest import style (CodeQL 753) CodeQL py/import-and-import-from flagged `import unittest` + `from unittest import mock`. Collapse to `from unittest import TestCase, main, mock`. * fix(review): explicit raise in _matrix_row (CodeQL 754) CodeQL py/mixed-returns flagged the implicit fall-through after self.fail() (which it doesn't model as NoReturn). End with an explicit raise AssertionError. * test(review): replace non-null assertions with a must() guard @typescript-eslint/no-non-null-assertion flagged 4 `!` operators. Add a narrowing must(value, message) helper (throws on undefined) and a named baseResolveUpstream, removing every non-null assertion. * fix(review): unguessable heredoc delimiter for the report output The report embeds the manifest `hold` field (fork-PR-editable); a fixed DRIFT_EOF delimiter in a hold value could close the $GITHUB_OUTPUT heredoc early and inject output keys. Use DRIFT_EOF_$(openssl rand -hex 16) — a value the report cannot contain. (Randomized delimiter over base64: keeps REPORT raw markdown, no consumer-side decode.) * fix(review): scope issues:write to scheduled runs (two-job split) GitHub Actions has no step-level permissions, so the only way to keep PR runs (incl. forks) from receiving `issues: write` is to split the job. A `report` job (contents:read, all events) renders the report + the PR `::warning::` and exposes report/exit_code as job outputs; a schedule-only `upsert-issue` job (needs: report, issues:write, no checkout) consumes them for the issue upsert + close. The 'Check upgrade readiness' check name is preserved. * fix(review): launder npm-version '?' in disposition prose The disposition bucket prose interpolated r['npm_version'] raw, so a successful 200 npm /latest response lacking a 'version' key would render a bare '?' (the matrix cell already laundered it). Add npm_version_label ('unknown' for '?') and use it in all five bucket renderers. Test a version-less npm response. * refactor(review): load_vendored_manifest returns only the consumed 'hold' The readiness script reads only the grammar names + 'hold'; the 'key' and 'upstream' fields were phantom data (upstream-drift coords live in the script's own GRAMMARS map). Narrow the return to {hold}. * fix(review): unify detect()/apply() 'newer' check for github grammars detect() compared the bare sha7 while apply() compared up.version (the full -g provenance string apply() also writes). After the bot re-vendored a github grammar once, detect() reported a perpetual false 'update available' while apply() correctly saw 'already current' — a noisy job summary + wasted --apply subprocess (the PR-exists guard absorbed it before any duplicate PR). Extract a shared isNewer(up, have) helper used by both. Tests cover equal- provenance (false), first-vendoring plain-version (true, not suppressed), and sha-advanced (true). Coupled with U12 (the detect⇄apply agreement assertion lives there once apply()'s not-newer path returns instead of process.exit). * test(review): cover main()'s out-of-range + prebuilt-only vendored ABI branches main()'s vendored-ABI classification reads through vendored_abi_from_repo (the local-read seam --assert-current uses), so patching it drives the 'Vendored (ABI out of range)' blocker branch and the prebuilt-only (vendored_abi None → 'prebuilt' cell, not '?') branch — neither reachable today since all 5 vendor dirs ship parser.c at ABI 14. * test(review): monitor-side manifest⇄vendor-dir consistency guard Mirror the Python consistency guard on the monitor side — the monitor consumes the same manifest and is the side that WRITES files from manifest `name`, so manifest/vendor-dir drift must fail CI here too. * fix(review): validate grammar names at manifest load (path-traversal guard) The manifest `name` is joined into gitnexus/vendor/ paths in both scripts (and apply() WRITES there), so reject any name not matching tree-sitter-[a-z0-9-]+ at the single load chokepoint — defense-in-depth even though the live trust boundary already prevents exploitation. loadManifestGrammars gains an injectable `raw` arg + export for testing; tests reject a '../etc' name in both scripts. * refactor(review): apply() throws ApplyExit; CLI maps to exit codes apply()'s 4 process.exit calls killed the vitest worker, blocking in-process tests of its error branches. Replace them with a thrown ApplyExit{code}; the not-newer (already-current) path returns `have` instead of exit(0). The isMain CLI block try/catches and maps ApplyExit.code → process.exit, so the monitor's subprocess contract (exit 0/2/3) is byte-identical (verified via subprocess smoke). Tests cover unknown-key=2, held=3, ABI-reject=3, and not-newer (returns current, no throw, no write). * refactor(review): extract vendored render helper; trim docstrings (<1000 lines) Extract the 'Vendored parsers' prose render into _render_vendored_section() so main() coordinates named phases rather than inlining a ~450-line monolith, and condense the most verbose docstrings/comments. The script drops from 1092 to 999 lines (under the 1000 bar the maintainability review flagged). Behavior-preserving: the deterministic --offline render is byte-identical before/after (verified in-place), --assert-current still passes, and the full unit suite is green. * fix(review): row-diff regex captures only the Status cell The change-detection regex captured the whole row tail as group 2, so any non-status cell drift (e.g. an upstream-ABI bump) emitted a false-positive 'change' line. Capture only the Status cell ([^|]+? before the final |$). The workflow parseRows regex and the Python _ROW_DIFF_RE stay byte-identical; the stability test now asserts group 2 is the status string (e.g. c → 'Vendored — held') and contains no pipe. * fix(ci): hoist intro string out of the list literal (CodeQL 755) The U13 extraction moved the 'Vendored parsers' intro paragraph (implicitly concatenated string literals) INTO a list literal, tripping CodeQL py/implicit-string-concatenation-in-list (reads as a possibly-missing comma between elements). Hoist it into a parenthesized `intro` variable. Render is byte-identical. --- .../check-tree-sitter-upgrade-readiness.py | 502 ++++++++++-------- ...est_check_tree_sitter_upgrade_readiness.py | 394 ++++++++++++++ .github/scripts/update-vendored-grammars.mjs | 171 ++++-- .github/vendored-grammars.json | 26 + .github/workflows/grammar-update-monitor.yml | 7 + .../tree-sitter-upgrade-readiness.yml | 80 ++- CONTRIBUTING.md | 9 + .../test/unit/grammar-update-monitor.test.ts | 282 +++++++++- 8 files changed, 1203 insertions(+), 268 deletions(-) create mode 100644 .github/scripts/test_check_tree_sitter_upgrade_readiness.py create mode 100644 .github/vendored-grammars.json diff --git a/.github/scripts/check-tree-sitter-upgrade-readiness.py b/.github/scripts/check-tree-sitter-upgrade-readiness.py index 35d86d766..269afa7b2 100644 --- a/.github/scripts/check-tree-sitter-upgrade-readiness.py +++ b/.github/scripts/check-tree-sitter-upgrade-readiness.py @@ -1,27 +1,14 @@ #!/usr/bin/env python3 -"""Monitor tree-sitter 0.25 upgrade readiness. +"""Monitor tree-sitter 0.25 upgrade readiness — two things Dependabot can't see: -Tracks two things Dependabot cannot see: + 1. Peer-dep compatibility: when every grammar's *latest npm release* accepts + tree-sitter@0.25.0 (so we can upgrade without --legacy-peer-deps). + 2. Vendored upstream drift: whether a vendored grammar's upstream parser.c moved. - 1. Peer-dep compatibility. Each tree-sitter-* grammar declares a peer - dependency on the tree-sitter runtime. We want to know when every - grammar's *latest npm release* satisfies tree-sitter@0.25.0 so we - can upgrade without --legacy-peer-deps. - - 2. Vendored upstream drift. vendor/tree-sitter-proto/ is a snapshot of - coder3101/tree-sitter-proto's parser.c. When upstream moves, we want - to know whether we can pick it up. - -Invoked from .github/workflows/tree-sitter-upgrade-readiness.yml daily. -Runs locally too: - - python3 .github/scripts/check-tree-sitter-upgrade-readiness.py - -Outputs Markdown to stdout. Exit 0 when every grammar is upgrade-ready -and the vendored proto is in sync. Exit 1 when blockers remain (the -workflow uses this to open or update a tracking issue). - -No external deps -- stdlib only, so it runs on any vanilla runner. +Invoked daily from tree-sitter-upgrade-readiness.yml; runs locally too. Outputs +Markdown to stdout; exit 1 when blockers remain (the workflow upserts a tracking +issue). stdlib-only — runs on any vanilla runner. + python3 .github/scripts/check-tree-sitter-upgrade-readiness.py [--offline | --assert-current] """ from __future__ import annotations @@ -38,6 +25,11 @@ import urllib.request REPO_ROOT = pathlib.Path(__file__).resolve().parents[2] GITNEXUS_DIR = REPO_ROOT / "gitnexus" +# Offline mode (--offline flag or GITNEXUS_TS_READINESS_OFFLINE=1): skip ALL network +# so the script + tests run hermetically. npm columns render "n/a (offline)"; +# vendored ABIs are still read from the repo. The read-path mirror of --assert-current. +OFFLINE = os.environ.get("GITNEXUS_TS_READINESS_OFFLINE", "") not in ("", "0", "false") + # ── Upgrade target ────────────────────────────────────────────────────── # The runtime version we want to upgrade TO. Update this when the goal # changes (e.g. once 0.25 lands and we target 0.26). @@ -78,15 +70,11 @@ GRAMMARS: dict[str, tuple[str, str, str]] = { "tree-sitter-proto": ("coder3101/tree-sitter-proto", "main", "src/parser.c"), } -# Grammars deliberately held below npm latest. The readiness report surfaces -# these so reviewers can tell intentional pins apart from drift, and so the -# context for each pin (which issue motivated it) is visible at a glance. -# Add an entry whenever you pin a grammar below npm latest. +# npm-installed grammars deliberately held below npm latest (surfaced so reviewers +# can tell intentional pins from drift). Add an entry when you pin an npm grammar. +# VENDORED grammars carry their hold in .github/vendored-grammars.json instead, so a +# vendored grammar's hold lives in one place — tree-sitter-c's is there, not here. INTENTIONAL_PINS: dict[str, str] = { - "tree-sitter-c": ( - "#1242 — last release built against the tree-sitter@0.21 ABI; " - "tree-sitter-c@0.23.x prebuilds segfault on Windows under tree-sitter@0.21.1" - ), "tree-sitter-cpp": ( "#1242 — last 0.23.x release before tree-sitter-cpp added a runtime " "dep on the broken-ABI tree-sitter-c@^0.23.1; pinning here removes " @@ -95,6 +83,56 @@ INTENTIONAL_PINS: dict[str, str] = { } +def load_vendored_manifest() -> dict[str, dict]: + """Load the shared vendored-grammar manifest (.github/vendored-grammars.json). + + The single source of truth — shared with update-vendored-grammars.mjs — for + which grammars are *vendored* (shipped from gitnexus/vendor/, not npm) + and any policy ``hold`` (e.g. tree-sitter-c, #1242/#858). Membership routes a + grammar to the vendored branch, which reads its ABI from the repo instead of + node_modules (the #858 source of the old bare ``?``). Returns + ``{ name: {"hold": str | None} }``; upstream-drift coords stay in ``GRAMMARS``. + """ + manifest_path = REPO_ROOT / ".github" / "vendored-grammars.json" + # Fail loud with a pointer, not a bare traceback: this runs at module import, + # so a missing/corrupt manifest would otherwise crash both the script and any + # test that imports it with an opaque FileNotFoundError/JSONDecodeError. + try: + data = json.loads(manifest_path.read_text(encoding="utf-8")) + except FileNotFoundError as exc: + raise SystemExit( + f"vendored-grammars manifest not found at {manifest_path}. " + f"It is the shared source of truth for vendored grammars " + f"(see CONTRIBUTING.md → CI automation contracts)." + ) from exc + except json.JSONDecodeError as exc: + raise SystemExit( + f"vendored-grammars manifest at {manifest_path} is not valid JSON: {exc}." + ) from exc + out: dict[str, dict] = {} + for key, g in (data.get("grammars") or {}).items(): + name = g.get("name") + if not name: + raise SystemExit( + f"vendored-grammars manifest entry {key!r} is missing a 'name' field " + f"({manifest_path})." + ) + # Defense-in-depth (#2187): `name` is joined into gitnexus/vendor/, so + # reject anything not a plain grammar name before it can traverse ("../etc"). + if not re.fullmatch(r"tree-sitter-[a-z0-9-]+", name): + raise SystemExit( + f"vendored-grammars manifest entry {key!r} has an invalid grammar " + f"name {name!r} (must match tree-sitter-[a-z0-9-]+)." + ) + out[name] = {"hold": g.get("hold")} + return out + + +# Vendored set + holds, keyed by full grammar name (e.g. "tree-sitter-c"). +VENDORED: dict[str, dict] = load_vendored_manifest() +VENDORED_NAMES: frozenset[str] = frozenset(VENDORED) + + # ── Helpers ───────────────────────────────────────────────────────────── def _load_package_json() -> dict: @@ -134,6 +172,8 @@ def npm_view_json(pkg: str) -> dict | None: being available (it's a batch file on Windows which complicates subprocess calls). """ + if OFFLINE: + return None url = f"https://registry.npmjs.org/{pkg}/latest" try: req = urllib.request.Request(url, headers={"Accept": "application/json"}) @@ -190,6 +230,8 @@ def fetch_text(url: str, timeout: int = 8) -> str | None: Adds an Authorization header for github.com URLs when GITHUB_TOKEN is set (raises the rate limit from 60 to 5 000 requests/hour). """ + if OFFLINE: + return None headers: dict[str, str] = {} # Parse the URL and check the hostname rather than substring-matching # on the full URL string (CodeQL py/incomplete-url-substring-sanitization). @@ -231,14 +273,8 @@ def md_h(text: str, level: int = 2) -> str: def _first_sentence(text: str) -> str: - """Return the leading sentence of a free-form rationale string. - - Vendor package.json `_vendoredBy` fields often look like - ". . Do NOT ." — the - first sentence is what reviewers actually want to read; the rest is - noise in this context. Match a sentence-ending '.' followed by - whitespace; fall back to the whole string if nothing matches. - """ + """Return the leading sentence of a `_vendoredBy` rationale (the rest tails off + into install-script breadcrumbs); fall back to the whole string.""" text = text.strip() match = re.search(r"\.\s+[A-Z]", text) return text[: match.start() + 1] if match else text @@ -262,21 +298,26 @@ def range_includes(spec: str | None, version: str) -> bool: return spec.strip() == version.strip() -def is_vendored_pin(spec: str | None) -> bool: - return bool(spec) and spec.startswith(("file:", "git", "http")) +def vendored_abi_from_repo(name: str, parser_path: str) -> int | None: + """Read a vendored grammar's ABI directly from gitnexus/vendor/. + + Local-only (no network) — the offline half of ``vendored_drift_summary``, + factored out so the hermetic ``--assert-current`` gate can introspect vendored + ABIs without triggering the upstream-drift fetches it never uses (#858 review). + """ + vendor_dir = GITNEXUS_DIR / "vendor" / name + vendored_parser = vendor_dir / parser_path + if not vendored_parser.is_file(): + vendored_parser = vendor_dir / "src" / "parser.c" + return extract_language_version(vendored_parser) def vendored_drift_summary( name: str, upstream_repo: str, upstream_branch: str, parser_path: str ) -> dict: - """Inspect a vendored grammar under gitnexus/vendor/. - - Returns the vendored package.json's ``version`` and ``_vendoredBy`` - fields (which carry the human rationale for vendoring), the vendored - parser's ABI, and a comparison against upstream main. We deliberately - rely on ``_vendoredBy`` rather than a parallel registry in this - script: the rationale belongs next to the vendored sources, not in - a daily-running CI script. + """Inspect a vendored grammar under gitnexus/vendor/: returns its + package.json ``version`` + ``_vendoredBy`` (the rationale, kept next to the + sources), the vendored ABI, and a comparison against upstream main. """ vendor_dir = GITNEXUS_DIR / "vendor" / name pkg: dict = {} @@ -290,7 +331,7 @@ def vendored_drift_summary( vendored_parser = vendor_dir / parser_path if not vendored_parser.is_file(): vendored_parser = vendor_dir / "src" / "parser.c" - vendored_abi = extract_language_version(vendored_parser) + vendored_abi = vendored_abi_from_repo(name, parser_path) upstream_url = ( f"https://raw.githubusercontent.com/{upstream_repo}/" @@ -302,10 +343,13 @@ def vendored_drift_summary( sha_text = fetch_text( f"https://api.github.com/repos/{upstream_repo}/commits/{upstream_branch}" ) - upstream_sha = "?" + # Labeled fallback rather than a bare "?": in CI this fetch succeeds, but + # offline (or on a transient API miss) the report should say *why* it's + # blank instead of leaving a placeholder (#858). + upstream_sha = "unknown" if sha_text: try: - upstream_sha = json.loads(sha_text).get("sha", "?")[:12] + upstream_sha = json.loads(sha_text).get("sha", "unknown")[:12] except json.JSONDecodeError: pass @@ -321,7 +365,9 @@ def vendored_drift_summary( return { "name": name, - "vendored_version": pkg.get("version", "?"), + # Labeled fallback, never a bare "?": a vendor package.json should always + # carry a version, but if one is missing the report says so plainly (#858). + "vendored_version": pkg.get("version") or "unknown", "vendored_by": pkg.get("_vendoredBy"), "vendored_abi": vendored_abi, "upstream_repo": upstream_repo, @@ -336,27 +382,14 @@ def vendored_drift_summary( def assert_current() -> int: - """Assert every grammar's ABI is loadable by the CURRENT runtime. - - Unlike the readiness report (which probes the npm registry + upstream - main for the *target* runtime), this mode is hermetic and offline: it - reads only what's checked out / installed locally and asserts each - grammar's compiled ABI lies within the current runtime's - ``RUNTIME_ABI_RANGES`` window. It is the static half of the #1922 ABI - gate; the runtime load-smoke (`parser-loader-abi.test.ts`) is the - dynamic half. - - Coverage, reusing the existing helpers: - - npm-installed grammars: ABI from node_modules//. - - vendored grammars (dart/proto/swift): ABI via ``vendored_drift_summary``. - - Swift is prebuilt-only (no parser.c) → not introspectable here; - treated as "covered by the runtime load-smoke", not asserted. - - INTENTIONAL_PINS are honored: a pinned grammar is expected to sit at - an ABI the current runtime loads (that's *why* it's pinned), so it is - asserted like any other rather than skipped. + """Assert every grammar's compiled ABI loads on the CURRENT runtime. + The hermetic/offline static half of the #1922 ABI gate (the runtime + load-smoke is the dynamic half): reads only local files — npm ABIs from + node_modules/, vendored ABIs from gitnexus/vendor/ via + ``vendored_abi_from_repo`` (no network). A prebuilt-only vendor (no + parser.c) is skipped; INTENTIONAL_PINS are asserted like any other grammar. Returns 0 when every introspectable grammar is in range, 1 otherwise. - Prints a plain-text (non-Markdown) report so CI logs stay readable. """ current_runtime = read_current_runtime() abi_range = RUNTIME_ABI_RANGES.get(current_runtime) @@ -380,13 +413,20 @@ def assert_current() -> int: for name, (upstream_repo, upstream_branch, parser_path) in sorted(GRAMMARS.items()): pinned_spec = pinned_versions.get(name, "—") - pin_note = f" [intentional pin: {pinned_spec}]" if name in INTENTIONAL_PINS else "" + if name in VENDORED_NAMES and VENDORED[name].get("hold"): + pin_note = " [vendored, held]" + elif name in INTENTIONAL_PINS: + pin_note = f" [intentional pin: {pinned_spec}]" + else: + pin_note = "" - if is_vendored_pin(pinned_spec): - v = vendored_drift_summary(name, upstream_repo, upstream_branch, parser_path) - abi = v["vendored_abi"] + # Vendored grammars: ABI read locally from the repo via vendored_abi_from_repo + # (NOT vendored_drift_summary, which fetches upstream — this gate is hermetic), + # so the offline #1922 gate covers them instead of skipping them (#858/#2187). + if name in VENDORED_NAMES: + abi = vendored_abi_from_repo(name, parser_path) if abi is None: - # Prebuilt-only vendor (e.g. tree-sitter-swift): no parser.c to + # Prebuilt-only vendor (e.g. a binary-only grammar): no parser.c to # introspect. The runtime load-smoke covers it instead. skipped.append(f"{name} (vendored, prebuilt — covered by load-smoke)") continue @@ -453,32 +493,18 @@ def _classify_grammar( ) -> dict: """Decide a single primary disposition + a separate bump-now hint. - Buckets are mutually exclusive and ordered by what a reviewer should - look at first: - - fetch_failed : npm registry fetch failed (treat as blocker, but - surface separately so reviewers don't confuse it - with an upstream block) - - intentional : pinned in INTENTIONAL_PINS — explicit choice - - ready : npm-latest peer dep already accepts the target - runtime; nothing to do - - waiting : main has a fix (ABI 15 or relaxed peer) but no - published npm release yet - - blocked : peer dep too tight on both npm and main - - Independently of bucket, `bump_now` reports whether reviewers can - move the pin forward today without touching the runtime — we only - suggest it when npm-latest's peer dep also accepts our *current* - runtime, otherwise the bump would break `npm install`. + Mutually-exclusive buckets, ordered by reviewer priority: ``fetch_failed`` + (npm fetch failed — surfaced apart from upstream blocks), ``intentional`` + (in INTENTIONAL_PINS), ``ready`` (npm-latest peer accepts the target), + ``waiting`` (a fix on main, unpublished), ``blocked`` (peer too tight on + both). ``bump_now`` is independent: True only when npm-latest's peer also + accepts our *current* runtime (else the bump would break ``npm install``). """ - is_vendored = is_vendored_pin(pinned_spec) - behind_latest = ( - not is_vendored - and npm_version != "?" - and not range_includes(pinned_spec, npm_version) - ) - # Intentional pins must never appear as actionable bumps — by definition - # we're holding them back on purpose. The pin can only be lifted by - # editing INTENTIONAL_PINS and package.json together. + # Only npm-path grammars reach this function — vendored grammars are routed + # to the vendored branch in main() and `continue` before classification. + behind_latest = npm_version != "?" and not range_includes(pinned_spec, npm_version) + # Intentional pins are never actionable bumps (held on purpose; lifted only by + # editing INTENTIONAL_PINS + package.json together). bump_now = behind_latest and current_compat and name not in INTENTIONAL_PINS if fetch_failed: @@ -496,6 +522,9 @@ def _classify_grammar( "name": name, "pinned_spec": pinned_spec or "—", "npm_version": npm_version, + # Display form for the disposition prose, laundering a "?" (a malformed 200 + # npm response lacking `version`) so it never shows bare, like the matrix cell. + "npm_version_label": "unknown" if npm_version == "?" else npm_version, "peer_range": peer_range, "target_compat": target_compat, "current_compat": current_compat, @@ -503,15 +532,105 @@ def _classify_grammar( "behind_latest": behind_latest, "bump_now": bump_now, "bucket": bucket, - "is_vendored": is_vendored, } +def _render_vendored_section( + vendored_grammars: list[dict], + target_abi_range: tuple[int, int], + blockers: dict[str, str], +) -> list[str]: + """Render the 'Vendored parsers' prose block. Appends any runtime-side blocker + (upstream ABI beyond the target range) to ``blockers`` in place; returns the + markdown lines (empty when nothing is vendored). Extracted from main() so that + function coordinates named render phases rather than inlining them (#2187).""" + if not vendored_grammars: + return [] + # Hoisted out of the list literal below: an implicit string concatenation + # inside a list display trips CodeQL py/implicit-string-concatenation-in-list + # (it reads as a possibly-missing comma between elements). + intro = ( + "These grammars ship from `gitnexus/vendor/` rather than the npm " + "registry. Their compatibility is governed by the **vendored " + "ABI** (must lie in the target runtime's range), not by a peer-" + "dep negotiation. The rationale for each vendored copy lives in " + "its own `package.json` `_vendoredBy` field." + ) + lines = [md_h(f"Vendored parsers ({len(vendored_grammars)})", 2), intro, ""] + for v in sorted(vendored_grammars, key=lambda v: v["name"]): + sync_label = "in sync with upstream" if v["in_sync"] else "diverged from upstream" + if v["abi_state"] == "in_range": + abi_label = f"ABI `{v['vendored_abi']}` (in target range)" + elif v["abi_state"] == "prebuilt": + abi_label = "ABI `prebuilt` (binary-only vendor, source not introspectable)" + else: + abi_label = ( + f"ABI `{v['vendored_abi']}` (**outside** target range " + f"{target_abi_range[0]}..{target_abi_range[1]})" + ) + # Never a bare "?": when upstream parser.c can't be read (generated at build, + # or a transient fetch miss), use the neutral `n/a` token (#858). + upstream_abi_str = ( + f"ABI `{v['upstream_abi']}`" if v["upstream_abi"] is not None else "ABI `n/a`" + ) + lines.append( + f"- **`{v['name']}`** `{v['vendored_version']}` — {abi_label}, " + f"upstream `{v['upstream_repo']}@{v['upstream_sha']}` " + f"{upstream_abi_str} · {sync_label}" + ) + if v.get("hold"): + lines.append(f" - **Held:** {v['hold']}") + if v["vendored_by"]: + # First sentence only — vendor _vendoredBy fields tail off into noise. + lines.append(f" - **Why vendored:** {_first_sentence(v['vendored_by'])}") + # Action: regen iff upstream ABI exceeds vendored AND stays within target; + # beyond target is a runtime-side blocker. Prebuilt-only vendors get a + # manual-refresh action driven by the in-sync flag instead. + if v["abi_state"] == "prebuilt": + if not v["in_sync"]: + lines.append( + " - **Action:** check whether upstream has shipped a new " + "prebuilt release; this vendor ships binary-only artefacts." + ) + elif v["upstream_abi"] and v["vendored_abi"] and v["upstream_abi"] > v["vendored_abi"]: + if v["upstream_abi"] <= target_abi_range[1]: + lines.append( + f" - **Action:** after upgrading to tree-sitter@{TARGET_RUNTIME}, " + f"regenerate `parser.c` from upstream `{v['upstream_sha']}`." + ) + else: + lines.append( + f" - **Action:** wait for a runtime supporting ABI " + f"{v['upstream_abi']}; current target ({TARGET_RUNTIME}) only " + f"goes up to ABI {target_abi_range[1]}." + ) + blockers[f"vendored-{v['name']}-abi"] = ( + f"vendored {v['name']}: upstream ABI {v['upstream_abi']} outside target range" + ) + elif not v["in_sync"]: + lines.append( + " - **Action:** review upstream changes; vendored copy may " + "need a refresh (no ABI bump required)." + ) + lines.append("") + return lines + + def main() -> int: blockers: dict[str, str] = {} lines: list[str] = [] + # Label for npm/upstream values we couldn't determine: in --offline mode the + # fetch was deliberately skipped (not "failed"), so say so honestly. + miss_label = "offline" if OFFLINE else "fetch failed" lines.append(md_h("Tree-sitter 0.25 upgrade readiness", 1)) lines.append("") + if OFFLINE: + lines.append( + "> **Offline mode** — npm registry + upstream GitHub checks were skipped. " + "npm-installed grammars show as unverified; vendored-grammar ABIs are read " + "from `gitnexus/vendor/`." + ) + lines.append("") current_runtime = read_current_runtime() current_abi_range = RUNTIME_ABI_RANGES.get(current_runtime, (0, 0)) @@ -525,10 +644,9 @@ def main() -> int: ) lines.append("") - # First pass: gather raw data + classification per grammar. We render - # the human-friendly buckets first, then the raw matrix in a
- # block at the end. Status text in the matrix is preserved verbatim - # so the workflow's row-diff change-detection keeps working. + # First pass: gather + classify per grammar. Human buckets render first, then + # the raw matrix in a
block (Status text preserved verbatim so the + # workflow's row-diff change-detection keeps working). grammar_rows: list[dict] = [] raw_matrix: list[str] = [ "| Grammar | Pinned | npm latest | Peer dep | Satisfies 0.25? | ABI | Upstream ABI | Status |", @@ -540,17 +658,17 @@ def main() -> int: for name, (upstream_repo, upstream_branch, parser_path) in sorted(GRAMMARS.items()): pinned_spec = pinned_versions.get(name, "—") - # Vendored grammars don't have an "npm latest" we install from — - # we ship our own copy under gitnexus/vendor/. Treat them - # as a separate kind of artefact: their readiness for the runtime - # upgrade depends on the vendored ABI being in the target range, - # not on a peer-dep negotiation. - if is_vendored_pin(pinned_spec): + # Vendored grammars are classified by manifest membership (NOT a file: pin + # heuristic — they aren't in package.json at all, the #858 misrouting bug). + # Their readiness is governed by the vendored ABI, read from the repo, not a + # peer-dep negotiation. npm-latest columns get sentinels. + if name in VENDORED_NAMES: v = vendored_drift_summary(name, upstream_repo, upstream_branch, parser_path) v["pinned_spec"] = pinned_spec - # Three-state classification: in-range, out-of-range, or - # not-introspectable (e.g. tree-sitter-swift ships only - # prebuilt .node binaries, no parser.c — assume compatible). + hold = VENDORED[name].get("hold") + v["hold"] = hold + # Three-state ABI classification: in-range, out-of-range, or + # not-introspectable (e.g. a prebuilt-only vendor with no parser.c). if v["vendored_abi"] is None: v["target_compat"] = True v["abi_state"] = "prebuilt" @@ -567,13 +685,37 @@ def main() -> int: f"vendored `{name}`: ABI {v['vendored_abi']} outside target range " f"{target_abi_range[0]}..{target_abi_range[1]}" ) + # A held vendored grammar (e.g. tree-sitter-c, #1242/#858) is frozen below + # a runtime upgrade: in-range ABI or not, keep it a blocker until the hold + # (from the manifest) is lifted — same treatment as npm INTENTIONAL_PINS. + if hold: + v["target_compat"] = False + status = "Vendored — held" + # Compose with any out-of-range reason rather than overwriting it: + # both share the blockers[name] key, and the ABI-out-of-range + # detail would otherwise be lost from the blockers summary. + hold_reason = f"vendored `{name}` held: {hold}" + prior = blockers.get(name) + blockers[name] = f"{prior}; {hold_reason}" if prior else hold_reason + # Cell sentinels: never emit a bare "?". A vendored grammar's ABI is + # the real LANGUAGE_VERSION when introspectable, else a labeled token. + vendored_abi_cell = ( + str(v["vendored_abi"]) if v["vendored_abi"] is not None else "prebuilt" + ) + # A None upstream ABI means the upstream parser.c couldn't be read — + # either it is generated at build time (e.g. swift) or the fetch + # missed. We can't tell which here, so use a neutral label rather + # than asserting "generated at build". Never a bare "?". + upstream_abi_cell = ( + str(v["upstream_abi"]) if v["upstream_abi"] is not None else "n/a" + ) # Keep vendored grammars in the raw matrix so the workflow's - # row-diff change-detection picks up status transitions on - # them too. npm-only columns get sentinels. + # row-diff change-detection picks up status transitions on them too. + # npm-only columns get sentinels. raw_matrix.append( f"| `{name}` | {pinned_spec} | (vendored) | (vendored) | " f"{'Yes' if v['target_compat'] else '**No**'} | " - f"{v['vendored_abi'] or '?'} | {v['upstream_abi'] or '?'} | {status} |" + f"{vendored_abi_cell} | {upstream_abi_cell} | {status} |" ) vendored_grammars.append(v) continue @@ -593,7 +735,7 @@ def main() -> int: peer_optional = ts_meta.get("optional", False) if peer_range else True if fetch_failed: - peer_display = "? (fetch failed)" + peer_display = f"n/a ({miss_label})" target_compat = False current_compat = False else: @@ -609,7 +751,9 @@ def main() -> int: # Fallback to default location. installed_parser = GITNEXUS_DIR / "node_modules" / name / "src" / "parser.c" installed_abi = extract_language_version(installed_parser) - abi_display = str(installed_abi) if installed_abi else "?" + # Labeled sentinel, never a bare "?": CI's `npm ci` populates node_modules, + # but if it's absent say so plainly rather than leaving a placeholder (#858). + abi_display = str(installed_abi) if installed_abi else "n/a (not installed)" # Check upstream (main/master branch) ABI for unreleased work. upstream_url = ( @@ -618,23 +762,19 @@ def main() -> int: ) upstream_text = fetch_text(upstream_url) upstream_abi = extract_abi_from_text(upstream_text) if upstream_text else None - upstream_abi_display = str(upstream_abi) if upstream_abi else "?" + upstream_abi_display = str(upstream_abi) if upstream_abi else "n/a" # Status text + upstream-progress detection. The Status column # values are preserved as-is to keep the workflow's row-diff # change-detection working on the raw matrix below. upstream_progress: str | None = None if fetch_failed: - status = "Unknown (fetch failed)" - blockers[name] = f"`{name}`: npm registry fetch failed — could not verify peer dep" + status = f"Unknown ({miss_label})" + reason = "checks skipped (offline)" if OFFLINE else "npm registry fetch failed" + blockers[name] = f"`{name}`: {reason} — could not verify peer dep" elif name in INTENTIONAL_PINS: - # An intentional pin is, by definition, a held-back grammar: - # whatever npm-latest's peer dep says, our shipped version is - # the one whose ABI/peer must accept the target runtime, and - # the pin entry exists precisely because it does not. Treat - # it as a blocker until the pin is lifted (entry removed from - # INTENTIONAL_PINS), at which point this grammar falls back - # to standard classification on the next run. + # A held-back grammar: treated as a blocker until the pin is lifted + # (entry removed from INTENTIONAL_PINS), then reclassified next run. status = "Intentionally pinned" blockers[name] = ( f"`{name}` intentionally pinned at `{pinned_spec}` " @@ -675,8 +815,11 @@ def main() -> int: pinned_spec = pinned_versions.get(name, "—") compat_icon = "Yes" if target_compat else "**No**" + # "?" stays the internal fetch-failed sentinel (compared above); render a + # labeled token in the matrix so the report never shows a bare "?" (#858). + npm_version_cell = f"n/a ({miss_label})" if npm_version == "?" else npm_version raw_matrix.append( - f"| `{name}` | {pinned_spec} | {npm_version} | {peer_display} | " + f"| `{name}` | {pinned_spec} | {npm_version_cell} | {peer_display} | " f"{compat_icon} | {abi_display} | {upstream_abi_display} | {status} |" ) @@ -726,7 +869,8 @@ def main() -> int: lines.append(f"- {len(by_bucket['waiting'])} waiting on an upstream npm release") lines.append(f"- {len(by_bucket['blocked'])} blocked on upstream (no fix even on main)") if by_bucket['fetch_failed']: - lines.append(f"- {len(by_bucket['fetch_failed'])} could not be checked (npm registry unreachable)") + why = "checks skipped in offline mode" if OFFLINE else "npm registry unreachable" + lines.append(f"- {len(by_bucket['fetch_failed'])} could not be checked ({why})") if bump_now: lines.append( f"- **{len(bump_now)} bump candidate(s) you can take TODAY** (npm-latest " @@ -745,7 +889,7 @@ def main() -> int: lines.append("") for r in sorted(bump_now, key=lambda r: r["name"]): lines.append( - f"- `{r['name']}`: `{r['pinned_spec']}` → `{r['npm_version']}` " + f"- `{r['name']}`: `{r['pinned_spec']}` → `{r['npm_version_label']}` " f"(peer `{r['peer_range'] or 'none'}`)" ) lines.append("") @@ -768,7 +912,7 @@ def main() -> int: "These grammars' npm-latest peer dep already accepts the target runtime. No action needed for the upgrade.", by_bucket["ready"], lambda r: ( - f"- `{r['name']}` — pinned `{r['pinned_spec']}`, npm latest `{r['npm_version']}`" + f"- `{r['name']}` — pinned `{r['pinned_spec']}`, npm latest `{r['npm_version_label']}`" + (" _(also a bump candidate — see above)_" if r["bump_now"] else "") ), ) @@ -784,7 +928,7 @@ def main() -> int: reason = INTENTIONAL_PINS.get(r["name"], "(no rationale recorded)") lines.append( f"- `{r['name']}` pinned at `{r['pinned_spec']}` " - f"(npm latest `{r['npm_version']}`)\n {reason}" + f"(npm latest `{r['npm_version_label']}`)\n {reason}" ) lines.append("") @@ -794,7 +938,7 @@ def main() -> int: "We can move forward as soon as upstream cuts a release.", by_bucket["waiting"], lambda r: ( - f"- `{r['name']}@{r['npm_version']}` — peer `{r['peer_range'] or 'none'}`. " + f"- `{r['name']}@{r['npm_version_label']}` — peer `{r['peer_range'] or 'none'}`. " f"_{r['upstream_progress']}_" ), ) @@ -804,90 +948,23 @@ def main() -> int: "Peer dep is too tight on both the latest npm release and on upstream main. " "These need an upstream issue/PR before we can proceed.", by_bucket["blocked"], - lambda r: ( - f"- `{r['name']}@{r['npm_version']}` — peer `{r['peer_range'] or 'none'}`" - + (" _(vendored)_" if r["is_vendored"] else "") - ), + lambda r: f"- `{r['name']}@{r['npm_version_label']}` — peer `{r['peer_range'] or 'none'}`", ) _emit_bucket( "Could not check", - "npm registry fetch failed for these grammars. Re-run the workflow to retry.", + ( + "Checks were skipped because the report ran in `--offline` mode. " + "Re-run online to verify these grammars." + if OFFLINE + else "npm registry fetch failed for these grammars. Re-run the workflow to retry." + ), by_bucket["fetch_failed"], lambda r: f"- `{r['name']}` (pinned `{r['pinned_spec']}`)", ) # ── Vendored parsers ──────────────────────────────────────────── - if vendored_grammars: - lines.append(md_h(f"Vendored parsers ({len(vendored_grammars)})", 2)) - lines.append( - "These grammars ship from `gitnexus/vendor/` rather than the npm " - "registry. Their compatibility is governed by the **vendored " - "ABI** (must lie in the target runtime's range), not by a peer-" - "dep negotiation. The rationale for each vendored copy lives in " - "its own `package.json` `_vendoredBy` field." - ) - lines.append("") - for v in sorted(vendored_grammars, key=lambda v: v["name"]): - sync_label = ( - "in sync with upstream" if v["in_sync"] else "diverged from upstream" - ) - if v["abi_state"] == "in_range": - abi_label = f"ABI `{v['vendored_abi']}` (in target range)" - elif v["abi_state"] == "prebuilt": - abi_label = "ABI `prebuilt` (binary-only vendor, source not introspectable)" - else: - abi_label = ( - f"ABI `{v['vendored_abi']}` (**outside** target range " - f"{target_abi_range[0]}..{target_abi_range[1]})" - ) - upstream_abi_str = ( - f"ABI `{v['upstream_abi']}`" if v["upstream_abi"] else "ABI `?`" - ) - lines.append( - f"- **`{v['name']}`** `{v['vendored_version']}` — {abi_label}, " - f"upstream `{v['upstream_repo']}@{v['upstream_sha']}` " - f"{upstream_abi_str} · {sync_label}" - ) - if v["vendored_by"]: - # Show the first sentence — vendor package.json fields tend - # to start with the rationale and tail off into install- - # script breadcrumbs that aren't useful in this report. - rationale = _first_sentence(v["vendored_by"]) - lines.append(f" - **Why vendored:** {rationale}") - # Action computation: needs regen iff upstream ABI exceeds - # vendored AND is still within target range. If upstream ABI - # exceeds the target, that's a runtime-side blocker. For - # prebuilt-only vendors we can't drive this from source ABI; - # the action is a manual upstream-binary refresh, surfaced - # via the in-sync flag instead. - if v["abi_state"] == "prebuilt": - if not v["in_sync"]: - lines.append( - " - **Action:** check whether upstream has shipped a new " - "prebuilt release; this vendor ships binary-only artefacts." - ) - elif v["upstream_abi"] and v["vendored_abi"] and v["upstream_abi"] > v["vendored_abi"]: - if v["upstream_abi"] <= target_abi_range[1]: - lines.append( - f" - **Action:** after upgrading to tree-sitter@{TARGET_RUNTIME}, " - f"regenerate `parser.c` from upstream `{v['upstream_sha']}`." - ) - else: - lines.append( - f" - **Action:** wait for a runtime supporting ABI " - f"{v['upstream_abi']}; current target ({TARGET_RUNTIME}) only " - f"goes up to ABI {target_abi_range[1]}." - ) - blockers[f"vendored-{v['name']}-abi"] = ( - f"vendored {v['name']}: upstream ABI {v['upstream_abi']} outside target range" - ) - elif not v["in_sync"]: - lines.append( - " - **Action:** review upstream changes; vendored copy may " - "need a refresh (no ABI bump required)." - ) - lines.append("") + lines.extend(_render_vendored_section(vendored_grammars, target_abi_range, blockers)) # ── Raw matrix (for completeness + workflow row-diff) ──────────── lines.append(md_h("Full grammar matrix", 2)) @@ -911,6 +988,11 @@ if __name__ == "__main__": sys.stdout.reconfigure(encoding="utf-8") # type: ignore[attr-defined] except Exception: pass + # `--offline` skips all network so the readiness report renders hermetically + # (vendored ABIs from the repo; npm columns marked unverified). Useful for + # air-gapped runs and deterministic tests. + if "--offline" in sys.argv[1:]: + OFFLINE = True # `--assert-current` is the offline CI gate (#1922): assert every grammar's # ABI loads on the CURRENT runtime. Bare invocation keeps the original # target-runtime readiness report behaviour. diff --git a/.github/scripts/test_check_tree_sitter_upgrade_readiness.py b/.github/scripts/test_check_tree_sitter_upgrade_readiness.py new file mode 100644 index 000000000..c36e68d07 --- /dev/null +++ b/.github/scripts/test_check_tree_sitter_upgrade_readiness.py @@ -0,0 +1,394 @@ +#!/usr/bin/env python3 +"""Tests for check-tree-sitter-upgrade-readiness.py. + +Stdlib-only (``unittest`` + ``unittest.mock``) to match the script under test, +which is deliberately dependency-free so it runs on any vanilla runner. Run with: + + python3 -m unittest .github/scripts/test_check_tree_sitter_upgrade_readiness.py + +(pytest also discovers ``unittest.TestCase`` classes, so a future pytest CI job +picks these up unchanged.) + +These tests lock in the #858 fix: the 5 vendored grammars +(c/swift/kotlin/dart/proto) are classified from the shared manifest +(.github/vendored-grammars.json), their ABI is read from gitnexus/vendor/, +and the report never renders a bare ``?`` placeholder. All network is mocked. +""" +from __future__ import annotations + +import contextlib +import importlib.util +import io +import json +import pathlib +import re +from unittest import TestCase, main, mock + +# ── Load the hyphenated script as a module ─────────────────────────────── +_SCRIPTS_DIR = pathlib.Path(__file__).resolve().parent +_SCRIPT = _SCRIPTS_DIR / "check-tree-sitter-upgrade-readiness.py" +_REPO_ROOT = _SCRIPTS_DIR.parents[1] +_MANIFEST = _REPO_ROOT / ".github" / "vendored-grammars.json" + +_spec = importlib.util.spec_from_file_location("readiness_under_test", _SCRIPT) +readiness = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(readiness) # type: ignore[union-attr] + +# The exact row-diff regex the workflow's change-detection bot uses +# (.github/workflows/tree-sitter-upgrade-readiness.yml) — byte-identical so a matrix +# format change that would silently break change-detection fails here. Group 2 is +# ONLY the Status cell ([^|]+? before the final `|$`). +_ROW_DIFF_RE = re.compile(r"\| `(tree-sitter-[^`]+)` \|.*\| ([^|]+?) \|$", re.M) + + +def _physical_vendor_grammars() -> set[str]: + vendor = _REPO_ROOT / "gitnexus" / "vendor" + return { + p.name + for p in vendor.iterdir() + if p.is_dir() and p.name.startswith("tree-sitter-") + } + + +def _render_report() -> tuple[str, int]: + """Run main() with network mocked to mirror PRODUCTION; return (md, exit_code). + + - npm grammars resolve to a permissive "Ready" peer dep, so the ONLY blocker + left is the held vendored tree-sitter-c — letting us assert the hold is + load-bearing (exit code stays non-zero because of it). + - npm_view_json records its calls so we can prove vendored grammars are never + npm-queried. + - fetch_text mirrors the real workflow: upstream parser.c resolves to a real + ABI (committed upstream), commit endpoints return a sha — EXCEPT swift's + upstream, whose parser.c is generated at build time and so is unreachable + (None). That single miss exercises the labeled-sentinel path; every other + cell must be a real value, never a bare '?'. + """ + npm_calls: list[str] = [] + + def fake_npm_view_json(pkg: str): + npm_calls.append(pkg) + return {"version": "9.9.9", "peerDependencies": {"tree-sitter": "^0.25.0"}} + + def fake_fetch_text(url: str, timeout: int = 8): + if "parser.c" in url: + # swift's upstream parser.c is generated at build time → unreachable; + # the others ship a committed parser.c. + if "alex-pinkus" in url: + return None + return "#define LANGUAGE_VERSION 14\n#define STATE_COUNT 1\n" + if "/commits/" in url: + return json.dumps({"sha": "0123456789abcdef"}) + # package.json (relaxed-peer probe) etc. — not needed for these assertions. + return None + + buf = io.StringIO() + with mock.patch.object(readiness, "npm_view_json", side_effect=fake_npm_view_json), \ + mock.patch.object(readiness, "fetch_text", side_effect=fake_fetch_text), \ + contextlib.redirect_stdout(buf): + code = readiness.main() + report = buf.getvalue() + _render_report.last_npm_calls = npm_calls # type: ignore[attr-defined] + return report, code + + +class ManifestClassification(TestCase): + def test_manifest_matches_physical_vendor_dirs(self): + """Consistency guard: the manifest set == the gitnexus/vendor/tree-sitter-* + dirs. Vendoring a grammar without a manifest entry (or vice-versa) fails — + this is what keeps the two tree-sitter workflows aligned (#858).""" + manifest_names = { + g["name"] + for g in json.loads(_MANIFEST.read_text())["grammars"].values() + } + self.assertEqual(manifest_names, _physical_vendor_grammars()) + + def test_vendored_names_loaded_from_manifest(self): + self.assertEqual(set(readiness.VENDORED_NAMES), _physical_vendor_grammars()) + # npm-installed grammars must NOT be classified vendored. + self.assertNotIn("tree-sitter-cpp", readiness.VENDORED_NAMES) + self.assertNotIn("tree-sitter-go", readiness.VENDORED_NAMES) + + def test_c_carries_a_hold_cpp_does_not(self): + self.assertTrue(readiness.VENDORED["tree-sitter-c"]["hold"]) + self.assertNotIn("tree-sitter-c", readiness.INTENTIONAL_PINS) + # cpp stays an npm intentional pin. + self.assertIn("tree-sitter-cpp", readiness.INTENTIONAL_PINS) + + def test_vendored_names_are_a_subset_of_GRAMMARS(self): + # The report + --assert-current iterate the hardcoded GRAMMARS dict for + # upstream-drift coords. A vendored grammar present in the manifest but + # missing from GRAMMARS would be silently dropped from both — re-creating + # the cross-workflow divergence the manifest exists to kill (#858). Guard it. + missing = set(readiness.VENDORED_NAMES) - set(readiness.GRAMMARS) + self.assertEqual(missing, set(), f"manifest grammars missing from GRAMMARS: {missing}") + + def test_missing_manifest_raises_a_clear_error(self): + import pathlib + import tempfile + + with tempfile.TemporaryDirectory() as d: + with mock.patch.object(readiness, "REPO_ROOT", pathlib.Path(d)): + with self.assertRaises(SystemExit) as ctx: + readiness.load_vendored_manifest() + self.assertIn("vendored-grammars manifest", str(ctx.exception)) + + def test_malformed_manifest_raises_a_clear_error(self): + import pathlib + import tempfile + + with tempfile.TemporaryDirectory() as d: + gh = pathlib.Path(d) / ".github" + gh.mkdir() + (gh / "vendored-grammars.json").write_text("{ not valid json", encoding="utf-8") + with mock.patch.object(readiness, "REPO_ROOT", pathlib.Path(d)): + with self.assertRaises(SystemExit) as ctx: + readiness.load_vendored_manifest() + self.assertIn("not valid JSON", str(ctx.exception)) + + def test_path_traversal_grammar_name_is_rejected(self): + import pathlib + import tempfile + + bad = '{"grammars": {"evil": {"name": "../etc"}}}' + with tempfile.TemporaryDirectory() as d: + gh = pathlib.Path(d) / ".github" + gh.mkdir() + (gh / "vendored-grammars.json").write_text(bad, encoding="utf-8") + with mock.patch.object(readiness, "REPO_ROOT", pathlib.Path(d)): + with self.assertRaises(SystemExit) as ctx: + readiness.load_vendored_manifest() + self.assertIn("invalid grammar name", str(ctx.exception)) + + +class AssertCurrent(TestCase): + """The offline #1922 ABI gate (--assert-current) must stay hermetic — it reads + vendored ABIs from the repo, never the network. (Regression guard: a prior + revision routed vendored grammars through vendored_drift_summary, which fetches + upstream parser.c + commit sha, silently breaking the 'hermetic and offline' + contract — #858 review.)""" + + def _run_assert_current(self): + import urllib.request + + def explode(*a, **k): + raise AssertionError("--assert-current attempted a network call") + + buf = io.StringIO() + with mock.patch.object(urllib.request, "urlopen", side_effect=explode), \ + contextlib.redirect_stdout(buf): + code = readiness.assert_current() + return buf.getvalue(), code + + def test_assert_current_is_network_free_and_passes(self): + report, code = self._run_assert_current() # raises if any urlopen fires + self.assertEqual(code, 0) + # All 5 vendored grammars are introspected from the repo (ABI 14), not skipped. + for name in readiness.VENDORED_NAMES: + self.assertIn(f"{name}: vendored ABI", report) + + def test_assert_current_fails_an_out_of_range_vendored_abi(self): + # vendored_abi_from_repo is the local-read injection point: force one + # grammar out of the current runtime's ABI window and assert the gate trips. + real = readiness.vendored_abi_from_repo + + def fake(name, parser_path): + return 99 if name == "tree-sitter-dart" else real(name, parser_path) + + import urllib.request + buf = io.StringIO() + with mock.patch.object(readiness, "vendored_abi_from_repo", side_effect=fake), \ + mock.patch.object(urllib.request, "urlopen", side_effect=AssertionError("network")), \ + contextlib.redirect_stdout(buf): + code = readiness.assert_current() + self.assertEqual(code, 1) + self.assertIn("tree-sitter-dart", buf.getvalue()) + self.assertIn("outside current runtime range", buf.getvalue()) + + +class ReportRendering(TestCase): + @classmethod + def setUpClass(cls): + cls.report, cls.code = _render_report() + cls.rows = dict(_ROW_DIFF_RE.findall(cls.report)) + + def test_no_bare_question_mark_anywhere(self): + # The only legitimate '?' is the "Satisfies 0.25?" column header. + sanitized = self.report.replace("Satisfies 0.25?", "Satisfies 0.25") + self.assertNotIn("?", sanitized, "report still contains a bare '?' placeholder") + + def test_malformed_npm_version_renders_unknown_in_prose_not_bare_question(self): + # A successful (200) npm /latest response that omits `version` leaves + # npm_version == "?"; the grammar is still bucketed (fetch did not fail), so + # its disposition PROSE line must show the labeled sentinel, never a bare '?'. + def fake_npm(pkg: str): + if pkg == "tree-sitter-go": + return {"peerDependencies": {"tree-sitter": "^0.25.0"}} # no 'version' + return {"version": "9.9.9", "peerDependencies": {"tree-sitter": "^0.25.0"}} + + def fake_fetch(url: str, timeout: int = 8): + if "parser.c" in url and "alex-pinkus" not in url: + return "#define LANGUAGE_VERSION 14\n" + if "/commits/" in url: + return json.dumps({"sha": "0123456789abcdef"}) + return None + + buf = io.StringIO() + with mock.patch.object(readiness, "npm_view_json", side_effect=fake_npm), \ + mock.patch.object(readiness, "fetch_text", side_effect=fake_fetch), \ + contextlib.redirect_stdout(buf): + readiness.main() + report = buf.getvalue() + sanitized = report.replace("Satisfies 0.25?", "Satisfies 0.25") + self.assertNotIn("?", sanitized) + # The Ready bucket prose line for go shows the labeled 'unknown', not '?'. + self.assertRegex(report, r"`tree-sitter-go`.*npm latest `unknown`") + + def test_every_vendored_grammar_shows_numeric_abi_not_question_mark(self): + for name in readiness.VENDORED_NAMES: + row = self._matrix_row(name) + cells = [c.strip() for c in row.strip().strip("|").split("|")] + abi_cell = cells[5] # Grammar|Pinned|npm|Peer|Satisfies|ABI|UpstreamABI|Status + self.assertRegex( + abi_cell, r"^\d+$", + f"{name} ABI cell is '{abi_cell}', expected a number (read from vendor/)", + ) + + def test_proto_is_never_npm_queried(self): + # github-only vendored grammars must skip the npm peer-dep path entirely, + # which is what removes the old "? (fetch failed)" for tree-sitter-proto. + self.assertNotIn("tree-sitter-proto", _render_report.last_npm_calls) + self.assertNotIn("tree-sitter-dart", _render_report.last_npm_calls) + self.assertNotIn("Could not check", self.report) + self.assertNotIn("fetch failed", self.report) + + def test_held_c_renders_held_and_keeps_exit_nonzero(self): + # Status is the last matrix cell (the row-diff regex captures the whole + # tail, not just status, so read the cell directly). + cells = [c.strip() for c in self._matrix_row("tree-sitter-c").strip().strip("|").split("|")] + self.assertEqual(cells[-1], "Vendored — held") + self.assertIn("**Held:**", self.report) + # With every npm grammar mocked to "Ready", the ONLY remaining blocker is + # the held c — so a non-zero exit proves the hold is treated as a blocker. + self.assertEqual(self.code, 1) + + def test_upstream_abi_miss_uses_labeled_sentinel(self): + # swift's upstream parser.c is unreachable (mocked None), so its + # upstream-ABI cell is the labeled 'n/a' token, never a bare '?'. + cells = [c.strip() for c in self._matrix_row("tree-sitter-swift").strip().strip("|").split("|")] + self.assertEqual(cells[6], "n/a") # Upstream ABI column + + def test_row_diff_regex_captures_all_fifteen_grammar_statuses(self): + # The change-detection bot keys on this regex: group 1 = grammar name, + # group 2 = the Status cell ONLY (not the whole tail). It must match every + # row after the format change so status transitions keep being detected. + self.assertEqual(len(self.rows), 15) + for name in readiness.VENDORED_NAMES: + self.assertIn(name, self.rows) + # group 2 is the Status cell — held c renders exactly "Vendored — held", + # and no captured status contains a pipe (proves cell-scoped capture). + self.assertEqual(self.rows["tree-sitter-c"], "Vendored — held") + for status in self.rows.values(): + self.assertNotIn("|", status) + + def _matrix_row(self, name: str) -> str: + for line in self.report.splitlines(): + if line.startswith(f"| `{name}` |"): + return line + # Explicit terminating raise (not self.fail, which CodeQL doesn't model as + # NoReturn) so the function has no implicit fall-through return (CodeQL 754). + raise AssertionError(f"no matrix row for {name}") + + +class OfflineMode(TestCase): + """--offline must render the report touching ZERO network — vendored ABIs come + from the repo, npm columns are marked unverified. This is what makes the + network-dependent report deterministically testable in air-gapped CI.""" + + def _render_offline(self): + import urllib.request + + def explode(*a, **k): + raise AssertionError("network call attempted in --offline mode") + + buf = io.StringIO() + with mock.patch.object(readiness, "OFFLINE", True), \ + mock.patch.object(urllib.request, "urlopen", side_effect=explode), \ + contextlib.redirect_stdout(buf): + code = readiness.main() + return buf.getvalue(), code + + def test_offline_touches_no_network_and_still_renders(self): + report, code = self._render_offline() # raises if any urlopen fires + self.assertIn("Offline mode", report) + # Vendored grammars are introspected from the repo → real ABI 14, not a miss. + for name in readiness.VENDORED_NAMES: + row = next(l for l in report.splitlines() if l.startswith(f"| `{name}` |")) + cells = [c.strip() for c in row.strip().strip("|").split("|")] + self.assertRegex(cells[5], r"^\d+$", f"{name} vendored ABI missing offline") + + def test_offline_marks_npm_grammars_offline_not_fetch_failed(self): + report, _ = self._render_offline() + self.assertIn("(offline)", report) + self.assertNotIn("fetch failed", report) # honest: skipped, not failed + + def test_offline_report_has_no_bare_question_mark(self): + report, _ = self._render_offline() + sanitized = report.replace("Satisfies 0.25?", "Satisfies 0.25") + self.assertNotIn("?", sanitized) + + +class VendoredAbiBranches(TestCase): + """main()'s vendored-ABI classification reads through vendored_abi_from_repo + (the same local-read seam --assert-current uses), so a single patch drives the + out-of-range and prebuilt-only branches that no real vendor dir can trigger + today (all ship parser.c at ABI 14).""" + + def _render_with_vendored_abi(self, override): + """Render main() with the standard production-faithful network mock plus a + vendored_abi_from_repo override (dict: name -> int|None; others read real).""" + real = readiness.vendored_abi_from_repo + + def abi_seam(name, parser_path): + return override[name] if name in override else real(name, parser_path) + + def fake_npm(pkg): + return {"version": "9.9.9", "peerDependencies": {"tree-sitter": "^0.25.0"}} + + def fake_fetch(url, timeout=8): + if "parser.c" in url and "alex-pinkus" not in url: + return "#define LANGUAGE_VERSION 14\n" + if "/commits/" in url: + return json.dumps({"sha": "0123456789abcdef"}) + return None + + buf = io.StringIO() + with mock.patch.object(readiness, "vendored_abi_from_repo", side_effect=abi_seam), \ + mock.patch.object(readiness, "npm_view_json", side_effect=fake_npm), \ + mock.patch.object(readiness, "fetch_text", side_effect=fake_fetch), \ + contextlib.redirect_stdout(buf): + code = readiness.main() + return buf.getvalue(), code + + def _row(self, report, name): + line = next(l for l in report.splitlines() if l.startswith(f"| `{name}` |")) + return [c.strip() for c in line.strip().strip("|").split("|")] + + def test_out_of_range_vendored_abi_is_a_blocker(self): + # Force tree-sitter-dart's vendored ABI outside the target range (13–15). + report, code = self._render_with_vendored_abi({"tree-sitter-dart": 99}) + cells = self._row(report, "tree-sitter-dart") + self.assertEqual(cells[-1], "Vendored (ABI out of range)") + self.assertEqual(cells[5], "99") + self.assertEqual(code, 1) # out-of-range vendored grammar is a blocker + + def test_prebuilt_only_vendored_abi_renders_prebuilt_not_question(self): + # vendored_abi None (a future binary-only vendor with no parser.c). + report, _ = self._render_with_vendored_abi({"tree-sitter-dart": None}) + cells = self._row(report, "tree-sitter-dart") + self.assertEqual(cells[5], "prebuilt") # labeled, never a bare '?' + self.assertEqual(cells[4], "Yes") # prebuilt is assumed target-compatible + + +if __name__ == "__main__": + main() diff --git a/.github/scripts/update-vendored-grammars.mjs b/.github/scripts/update-vendored-grammars.mjs index 957200e77..769310bd7 100644 --- a/.github/scripts/update-vendored-grammars.mjs +++ b/.github/scripts/update-vendored-grammars.mjs @@ -42,17 +42,52 @@ const COMPATIBLE_ABI = new Set([13, 14]); // tree-sitter@0.21.1 LANGUAGE_VERSION // github grammars (no usable npm release) track the default branch HEAD. A `hold` // reason makes a grammar report-only: updates are detected + surfaced but never // auto-applied (c is ABI-pinned and must not move without a runtime upgrade). -const GRAMMARS = { - c: { - name: 'tree-sitter-c', - npm: 'tree-sitter-c', - hold: 'ABI-pinned at 0.21.4 (#1242/#858) — needs a tree-sitter runtime upgrade before bumping', - }, - swift: { name: 'tree-sitter-swift', npm: 'tree-sitter-swift' }, - kotlin: { name: 'tree-sitter-kotlin', npm: 'tree-sitter-kotlin' }, - dart: { name: 'tree-sitter-dart', github: 'UserNobody14/tree-sitter-dart' }, - proto: { name: 'tree-sitter-proto', github: 'coder3101/tree-sitter-proto' }, -}; +// +// The vendored set lives in .github/vendored-grammars.json — the SHARED source of +// truth this monitor and .github/scripts/check-tree-sitter-upgrade-readiness.py both +// read, so the two tree-sitter workflows can never disagree about which grammars are +// vendored or where their upstream lives. We reshape the manifest's +// `{ upstream: { npm | github } }` form into the flat `{ npm? , github? }` shape the +// rest of this script consumes. This is a local file read (import-safe, no network). +const MANIFEST = path.join(REPO_ROOT, '.github', 'vendored-grammars.json'); +// `raw` is injectable for testing; production reads the manifest file. +function loadManifestGrammars(raw = null) { + if (raw === null) { + // Fail loud with a pointer, not a bare ENOENT/SyntaxError: this runs at import. + try { + raw = JSON.parse(fs.readFileSync(MANIFEST, 'utf8')); + } catch (e) { + throw new Error( + `Could not load the vendored-grammars manifest at ${MANIFEST} ` + + `(shared source of truth — see CONTRIBUTING.md → CI automation contracts): ${e.message}`, + ); + } + } + return Object.fromEntries( + Object.entries(raw.grammars || {}).map(([key, g]) => { + if (!g.name) + throw new Error(`manifest entry '${key}' is missing a 'name' field (${MANIFEST})`); + // Defense-in-depth: `name` is joined into gitnexus/vendor/ paths (and + // apply() WRITES there), so reject anything that isn't a plain grammar name + // before it can traverse the filesystem (#2187). + if (!/^tree-sitter-[a-z0-9-]+$/.test(g.name)) + throw new Error( + `manifest entry '${key}' has an invalid grammar name '${g.name}' ` + + `(must match tree-sitter-[a-z0-9-]+)`, + ); + return [ + key, + { + name: g.name, + ...(g.upstream?.npm ? { npm: g.upstream.npm } : {}), + ...(g.upstream?.github ? { github: g.upstream.github } : {}), + ...(g.hold ? { hold: g.hold } : {}), + }, + ]; + }), + ); +} +const GRAMMARS = loadManifestGrammars(); const sh = (cmd, args, opts = {}) => execFileSync(cmd, args, { encoding: 'utf8', stdio: ['ignore', 'pipe', 'pipe'], ...opts }).trim(); @@ -62,6 +97,26 @@ const clean = (v) => .replace(/^[v^~]/, '') .trim(); +// Shared "is the candidate newer than what we ship?" check, used by BOTH detect() +// and apply() so they can never disagree. up.version is the comparable identity for +// both kinds: a plain semver for npm, and the `-g` provenance string for +// github (which apply() also writes to package.json). detect() previously compared +// the bare sha7 for github, so after the bot re-vendored a github grammar once it +// reported a perpetual false "update available" while apply() saw "already current" +// (#2187 review). Comparing up.version on both sides removes that asymmetry. +const isNewer = (up, have) => !have || up.version !== have; + +// apply() throws this (instead of calling process.exit) so its error branches are +// exercisable in-process by tests; the CLI entrypoint maps `.code` back to the +// original exit code, keeping the monitor's subprocess contract identical (#2187). +class ApplyExit extends Error { + constructor(message, code) { + super(message); + this.name = 'ApplyExit'; + this.code = code; + } +} + function vendoredVersion(g) { const p = path.join(VENDOR, g.name, 'package.json'); return clean(JSON.parse(fs.readFileSync(p, 'utf8')).version); @@ -136,22 +191,30 @@ function readAbi(srcRoot) { return null; // unknown (e.g. parser.c only generated at build time) } -function detect() { +// `deps` injects the network/filesystem seams (vendoredVersion / resolveUpstream / +// fetchSource / readAbi) so the classification logic — newer-detection, the ABI +// gate, and the policy-hold gate — can be unit-tested offline with fixtures, never +// touching live npm/GitHub. Production passes nothing and gets the real functions. +function detect(deps = {}) { + const getVendored = deps.vendoredVersion || vendoredVersion; + const resolveUp = deps.resolveUpstream || resolveUpstream; + const fetchSrc = deps.fetchSource || fetchSource; + const readAbiFn = deps.readAbi || readAbi; const report = []; for (const [key, g] of Object.entries(GRAMMARS)) { - const have = vendoredVersion(g); + const have = getVendored(g); let up; try { - up = resolveUpstream(g); + up = resolveUp(g); } catch (err) { report.push({ grammar: key, error: String(err.message || err) }); continue; } - const newer = up.kind === 'npm' ? up.version !== have : !have || up.ref.slice(0, 7) !== have; + const newer = isNewer(up, have); let abi = null; if (newer) { try { - abi = readAbi(fetchSource(g, up.ref)); + abi = readAbiFn(fetchSrc(g, up.ref)); } catch { /* fetch/abi best-effort; null = unknown */ } @@ -190,34 +253,48 @@ const copyFile = (srcRoot, dest, rel) => { * notice), LICENSE, and prebuilds/ (the build workflow refreshes those). Bumps the * stripped vendor package.json version + provenance — never re-introduces * scripts/dependencies (#836/#1728). Returns the new version. + * + * opts.dryRun resolves + ABI-validates the candidate but writes NOTHING — it logs + * what it would re-vendor and returns the version, so the flow can be rehearsed + * (locally or in CI) without mutating gitnexus/vendor/. opts.deps injects the + * network/fs seams for offline testing (same shape as detect()). */ -function apply(key) { +function apply(key, opts = {}) { + const dryRun = opts.dryRun || false; + const deps = opts.deps || {}; + const getVendored = deps.vendoredVersion || vendoredVersion; + const resolveUp = deps.resolveUpstream || resolveUpstream; + const fetchSrc = deps.fetchSource || fetchSource; + const readAbiFn = deps.readAbi || readAbi; const g = GRAMMARS[key]; - if (!g) { - console.error(`unknown grammar '${key}'`); - process.exit(2); - } - if (g.hold) { - console.error( + if (!g) throw new ApplyExit(`unknown grammar '${key}'`, 2); + if (g.hold) + throw new ApplyExit( `${key}: report-only (${g.hold}); not auto-applied. Re-vendor manually if intended.`, + 3, ); - process.exit(3); - } - const have = vendoredVersion(g); - const up = resolveUpstream(g); - const newer = up.kind === 'npm' ? up.version !== have : !have || up.version !== have; + const have = getVendored(g); + const up = resolveUp(g); + const newer = isNewer(up, have); if (!newer) { + // Already current: nothing to apply. Return (exit 0 via the CLI) — NOT an error. console.error(`${key}: already current (${have}); nothing to apply.`); - process.exit(0); + return have; } - const srcRoot = fetchSource(g, up.ref); - const abi = readAbi(srcRoot); - if (abi == null || !COMPATIBLE_ABI.has(abi)) { - console.error( + const srcRoot = fetchSrc(g, up.ref); + const abi = readAbiFn(srcRoot); + if (abi == null || !COMPATIBLE_ABI.has(abi)) + throw new ApplyExit( `${key}: candidate ${up.version} is ABI ${abi ?? 'unknown'} — not tree-sitter@0.21.1 ` + `compatible (need 13/14); refusing to re-vendor. Handle manually.`, + 3, ); - process.exit(3); + + if (dryRun) { + console.log( + `${key}: [dry-run] would re-vendor ${g.name} → ${up.version} (ABI ${abi}); no files written.`, + ); + return up.version; } const dest = path.join(VENDOR, g.name); @@ -256,11 +333,31 @@ function apply(key) { // makes live network calls, so importing must be side-effect-free. const isMain = process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href; if (isMain) { - if (process.argv[2] === '--apply') { - apply(process.argv[3]); + const args = process.argv.slice(2); + const dryRun = args.includes('--dry-run'); + if (args[0] === '--apply') { + // `--apply [--dry-run]` — --dry-run previews without writing. + // Map apply()'s thrown ApplyExit back to the original exit codes (0/2/3) so + // the monitor workflow's subprocess (which only distinguishes zero vs non-zero) + // sees identical behavior. + try { + apply(args[1], { dryRun }); + } catch (e) { + console.error(e.message); + process.exit(e instanceof ApplyExit ? e.code : 1); + } } else { process.stdout.write(JSON.stringify(detect(), null, 2) + '\n'); } } -export { detect, apply, resolveUpstream, readAbi, vendoredVersion, GRAMMARS, COMPATIBLE_ABI }; +export { + detect, + apply, + resolveUpstream, + readAbi, + vendoredVersion, + loadManifestGrammars, + GRAMMARS, + COMPATIBLE_ABI, +}; diff --git a/.github/vendored-grammars.json b/.github/vendored-grammars.json new file mode 100644 index 000000000..2777e26a0 --- /dev/null +++ b/.github/vendored-grammars.json @@ -0,0 +1,26 @@ +{ + "_comment": "Single source of truth for the VENDORED SET + policy holds, read by BOTH .github/scripts/update-vendored-grammars.mjs (weekly auto-PR bot) and .github/scripts/check-tree-sitter-upgrade-readiness.py (daily readiness report -> issue #858). The monitor also resolves each grammar's upstream from the `upstream` field here; the readiness report reads vendored ABIs from gitnexus/vendor//src/parser.c and keeps its own upstream-drift coords. A consistency-guard test asserts this set equals the gitnexus/vendor/tree-sitter-* directories. See CONTRIBUTING.md.", + "grammars": { + "c": { + "name": "tree-sitter-c", + "upstream": { "npm": "tree-sitter-c" }, + "hold": "ABI-pinned at 0.21.4 (#1242/#858) — needs a tree-sitter runtime upgrade before bumping" + }, + "swift": { + "name": "tree-sitter-swift", + "upstream": { "npm": "tree-sitter-swift" } + }, + "kotlin": { + "name": "tree-sitter-kotlin", + "upstream": { "npm": "tree-sitter-kotlin" } + }, + "dart": { + "name": "tree-sitter-dart", + "upstream": { "github": "UserNobody14/tree-sitter-dart" } + }, + "proto": { + "name": "tree-sitter-proto", + "upstream": { "github": "coder3101/tree-sitter-proto" } + } + } +} diff --git a/.github/workflows/grammar-update-monitor.yml b/.github/workflows/grammar-update-monitor.yml index ee14c093c..f6166de82 100644 --- a/.github/workflows/grammar-update-monitor.yml +++ b/.github/workflows/grammar-update-monitor.yml @@ -14,6 +14,13 @@ name: Vendored grammar update monitor # never auto-bumped — a maintainer re-vendors it deliberately after a runtime # upgrade. # +# The vendored set + per-grammar upstream coords + the tree-sitter-c hold live in +# .github/vendored-grammars.json — the SHARED source of truth this monitor and +# tree-sitter-upgrade-readiness.yml both read, so the two workflows can never +# disagree about which grammars are vendored (#858). This monitor additionally +# resolves each grammar's upstream from it; the readiness report reads vendored +# ABIs from gitnexus/vendor/ and keeps its own upstream-drift coords. +# # Concurrency convention: see CONTRIBUTING.md -> "GitHub Actions — Concurrency Convention". on: diff --git a/.github/workflows/tree-sitter-upgrade-readiness.yml b/.github/workflows/tree-sitter-upgrade-readiness.yml index 1eca8861d..bd887319f 100644 --- a/.github/workflows/tree-sitter-upgrade-readiness.yml +++ b/.github/workflows/tree-sitter-upgrade-readiness.yml @@ -1,12 +1,21 @@ name: Tree-sitter Upgrade Readiness # Monitors readiness for upgrading tree-sitter to 0.25.x. Tracks: -# 1. Peer-dep compatibility — can each grammar install cleanly with -# tree-sitter@0.25.0 without --legacy-peer-deps? -# 2. Vendored proto drift — has coder3101/tree-sitter-proto moved -# ahead of our vendored snapshot? +# 1. Peer-dep compatibility — can each NPM-installed grammar install cleanly +# with tree-sitter@0.25.0 without --legacy-peer-deps? +# 2. Vendored grammars — each grammar in .github/vendored-grammars.json +# (c/swift/kotlin/dart/proto) is classified by its vendored ABI, read +# straight from gitnexus/vendor//src/parser.c (NOT node_modules, +# which is never populated for vendored grammars — that mismatch is why +# the report used to render bare "?" placeholders, #858). # See .github/scripts/check-tree-sitter-upgrade-readiness.py for the logic. # +# .github/vendored-grammars.json is the SHARED source of truth for the vendored +# SET + policy holds: this readiness report and grammar-update-monitor.yml both +# read it, so the two workflows can never disagree about which grammars are +# vendored. (The monitor also resolves upstreams from it; this report keeps its +# own upstream-drift coords and reads vendored ABIs from gitnexus/vendor/.) +# # Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention". on: @@ -18,6 +27,8 @@ on: pull_request: paths: - '.github/scripts/check-tree-sitter-upgrade-readiness.py' + - '.github/scripts/test_check_tree_sitter_upgrade_readiness.py' + - '.github/vendored-grammars.json' - '.github/workflows/tree-sitter-upgrade-readiness.yml' concurrency: @@ -28,14 +39,18 @@ permissions: contents: read jobs: - readiness: + report: name: Check upgrade readiness runs-on: ubuntu-latest timeout-minutes: 10 + # Least privilege: rendering the report needs no write. The issue mutation + # lives in the schedule-only `upsert-issue` job below, so PR runs (incl. forks) + # never receive `issues: write` (#2187 review). permissions: contents: read - # Needed to open/update the tracking issue on scheduled runs. - issues: write + outputs: + report: ${{ steps.readiness.outputs.report }} + exit_code: ${{ steps.readiness.outputs.exit_code }} steps: - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 @@ -43,6 +58,17 @@ jobs: with: build: 'false' + # Guard the readiness script's logic (vendored classification, no bare "?", + # the manifest⇄vendor-dir consistency guard). Stdlib-only, so no extra deps; + # node_modules is populated by setup-gitnexus above, which the npm-path ABI + # reads need. Runs only on validation events (PR / manual), not the daily + # scheduled report. + - name: Run readiness script unit tests + if: github.event_name != 'schedule' + shell: bash + working-directory: .github/scripts + run: python3 -m unittest test_check_tree_sitter_upgrade_readiness -v + - name: Run upgrade readiness check id: readiness shell: bash @@ -54,10 +80,15 @@ jobs: code=$? set -e echo "exit_code=$code" >> "$GITHUB_OUTPUT" + # Unguessable per-run heredoc delimiter: the report includes the manifest's + # `hold` field, which a fork PR can edit — a fixed delimiter (e.g. DRIFT_EOF) + # in a hold value could close the heredoc early and inject $GITHUB_OUTPUT keys. + # A random hex delimiter the report cannot contain neutralizes that. + DELIM="DRIFT_EOF_$(openssl rand -hex 16)" { - echo 'report<> "$GITHUB_OUTPUT" echo "=== Report ===" cat drift-report.md @@ -69,13 +100,22 @@ jobs: run: | echo "::warning::Tree-sitter 0.25 upgrade has blockers. See job output for the full readiness report." - - name: Upsert tracking issue on scheduled runs - if: > - github.event_name == 'schedule' && - steps.readiness.outputs.exit_code != '0' + # Issue mutation is isolated here so `issues: write` is only ever granted on the + # scheduled run (never on PRs). Consumes the report + exit_code via job outputs. + upsert-issue: + name: Upsert tracking issue + needs: report + if: github.event_name == 'schedule' + runs-on: ubuntu-latest + timeout-minutes: 5 + permissions: + issues: write + steps: + - name: Upsert tracking issue on blockers + if: needs.report.outputs.exit_code != '0' uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 env: - REPORT: ${{ steps.readiness.outputs.report }} + REPORT: ${{ needs.report.outputs.report }} with: script: | const title = 'Tree-sitter 0.25 upgrade readiness'; @@ -105,7 +145,11 @@ jobs: // | `tree-sitter-foo` | ... | Blocking | const parseRows = (md) => { const map = {}; - for (const m of md.matchAll(/\| `(tree-sitter-[^`]+)` \|.*?\| (\S+(?:\s\S+)*?) \|$/gm)) { + // Group 2 captures ONLY the Status cell ([^|]+? before the final + // `|$`), so change-detection fires on status transitions, not on + // unrelated cell drift (e.g. an upstream-ABI bump). Mirror this in + // _ROW_DIFF_RE in test_check_tree_sitter_upgrade_readiness.py. + for (const m of md.matchAll(/\| `(tree-sitter-[^`]+)` \|.*\| ([^|]+?) \|$/gm)) { map[m[1]] = m[2].trim(); } return map; @@ -152,10 +196,8 @@ jobs: core.info(`Opened issue #${created.number}`); } - - name: Close tracking issue on clean scheduled runs - if: > - github.event_name == 'schedule' && - steps.readiness.outputs.exit_code == '0' + - name: Close tracking issue on clean runs + if: needs.report.outputs.exit_code == '0' uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 with: script: | diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 848884be4..278dd72d2 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -144,6 +144,15 @@ Re-invoking `/autofix` after a successful apply is a safe no-op — the workflow **Sensitive paths.** The apply workflow refuses any patch that touches `.github/` (workflow files, CODEOWNERS, dependabot config). A malicious PR could ship a custom prettier or ESLint config that reformats workflow YAML; if accepted, those edits would be pushed under `contents: write` without human review. Apply formatter changes to files under `.github/` manually in a normal commit so they get the same review every other workflow change gets. +### Vendored tree-sitter grammars + +`.github/vendored-grammars.json` is the **single source of truth** for the vendored tree-sitter grammar **set** and each grammar's policy `hold` (the ones shipped from `gitnexus/vendor/` rather than installed from npm). It lists each grammar's name, upstream coords (`npm` or `github`), and any `hold`. The monitor resolves upstreams from it; the readiness report keeps its own upstream-drift coords and reads vendored ABIs from `gitnexus/vendor/`. Two workflows read it: + +- `grammar-update-monitor.yml` (`.github/scripts/update-vendored-grammars.mjs`) — weekly; opens auto-PRs re-vendoring ABI-compatible upstream updates. +- `tree-sitter-upgrade-readiness.yml` (`.github/scripts/check-tree-sitter-upgrade-readiness.py`) — daily; renders the tree-sitter-0.25 readiness report (issue #858), reading each vendored grammar's ABI from `gitnexus/vendor//src/parser.c`. + +Sharing the manifest keeps the two aligned: a consistency-guard test asserts the manifest set equals the `gitnexus/vendor/tree-sitter-*` directories. **When you vendor a new grammar (or remove one), update `.github/vendored-grammars.json` in the same change** — otherwise that guard fails CI and the readiness report regresses to `?` placeholders. + ## AI-assisted contributions If you use coding agents, follow project context files (e.g. `AGENTS.md`, `CLAUDE.md`) and avoid drive-by refactors unrelated to the issue. Prefer incremental, test-backed changes. diff --git a/gitnexus/test/unit/grammar-update-monitor.test.ts b/gitnexus/test/unit/grammar-update-monitor.test.ts index 43078e02b..4461f8498 100644 --- a/gitnexus/test/unit/grammar-update-monitor.test.ts +++ b/gitnexus/test/unit/grammar-update-monitor.test.ts @@ -1,5 +1,5 @@ import { describe, it, expect, beforeAll, afterAll } from 'vitest'; -import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from 'node:fs'; +import { mkdtempSync, mkdirSync, writeFileSync, rmSync, readFileSync, readdirSync } from 'node:fs'; import { tmpdir } from 'node:os'; import path from 'node:path'; import { fileURLToPath, pathToFileURL } from 'node:url'; @@ -20,10 +20,21 @@ const MOD = pathToFileURL( ), ).href; +type Grammar = { name: string; npm?: string; github?: string; hold?: string }; +type Upstream = { version: string; ref: string; kind: 'npm' | 'github' }; +type DetectDeps = { + vendoredVersion?: (g: Grammar) => string; + resolveUpstream?: (g: Grammar) => Upstream; + fetchSource?: (g: Grammar, ref: string) => string; + readAbi?: (root: string) => number | null; +}; let mod: { readAbi: (root: string) => number | null; COMPATIBLE_ABI: Set; - GRAMMARS: Record; + GRAMMARS: Record; + detect: (deps?: DetectDeps) => Array>; + apply: (key: string, opts?: { dryRun?: boolean; deps?: DetectDeps }) => string; + loadManifestGrammars: (raw?: unknown) => Record; }; let tmp: string; @@ -76,3 +87,270 @@ describe('GRAMMARS registry', () => { } }); }); + +describe('shared vendored-grammars manifest', () => { + // The vendored set is sourced from .github/vendored-grammars.json — the single + // source of truth shared with check-tree-sitter-upgrade-readiness.py. This guards + // against the loader silently skewing from the manifest file (#858 alignment). + const manifestPath = path.resolve( + path.dirname(fileURLToPath(import.meta.url)), + '../../../.github/vendored-grammars.json', + ); + const manifest: { + grammars: Record< + string, + { name: string; upstream: { npm?: string; github?: string }; hold?: string } + >; + } = JSON.parse(readFileSync(manifestPath, 'utf8')); + + it('reshapes every manifest entry into the GRAMMARS shape, losing no information', () => { + expect(Object.keys(mod.GRAMMARS).sort()).toEqual(Object.keys(manifest.grammars).sort()); + for (const [key, g] of Object.entries(manifest.grammars)) { + const entry = mod.GRAMMARS[key]; + expect(entry.name).toBe(g.name); + expect(entry.hold).toBe(g.hold); + // Assert the absent upstream field is explicitly undefined, not just + // matching the manifest's absent property (avoids an undefined===undefined + // pass that would miss the loader mis-mapping a github coord into `npm`). + if (g.upstream.npm) { + expect(entry.npm).toBe(g.upstream.npm); + expect(entry.github).toBeUndefined(); + } else { + expect(entry.github).toBe(g.upstream.github); + expect(entry.npm).toBeUndefined(); + } + } + }); + + it('each grammar has exactly one upstream source (npm xor github)', () => { + for (const g of Object.values(mod.GRAMMARS)) { + expect(Boolean(g.npm) !== Boolean(g.github)).toBe(true); + } + }); + + it('the manifest grammar set equals the physical vendor/tree-sitter-* dirs (#858)', () => { + // Monitor-side mirror of the Python consistency guard. The monitor is the side + // that WRITES files from manifest `name`, so vendoring a grammar (or removing + // one) without updating the manifest must fail CI here too. + const vendorDir = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../../vendor'); + const physical = readdirSync(vendorDir, { withFileTypes: true }) + .filter((d) => d.isDirectory() && d.name.startsWith('tree-sitter-')) + .map((d) => d.name) + .sort(); + const manifestNames = Object.values(manifest.grammars) + .map((g) => g.name) + .sort(); + expect(manifestNames).toEqual(physical); + }); + + it('rejects a path-traversal grammar name at load (defense-in-depth)', () => { + // `name` is joined into vendor/ paths and apply() writes there. + expect(() => mod.loadManifestGrammars({ grammars: { evil: { name: '../etc' } } })).toThrow( + /invalid grammar name/, + ); + }); +}); + +// Narrowing accessor: throws a clear error instead of a non-null assertion (`!`), +// which @typescript-eslint/no-non-null-assertion forbids. +function must(value: T | undefined, message: string): T { + if (value === undefined) throw new Error(message); + return value; +} + +describe('detect() classification (offline, injected deps)', () => { + // Drive the real detect() loop with faked network/fs seams so the load-bearing + // gates — newer-detection, the ABI gate, and the policy hold — are exercised + // deterministically without touching live npm/GitHub. + const baseResolveUpstream = (g: Grammar): Upstream => + g.npm + ? { version: '9.9.9', ref: '9.9.9', kind: 'npm' } + : { version: '1.0.0-gabc1234', ref: 'abc1234def0', kind: 'github' }; + const deps: DetectDeps = { + vendoredVersion: (g) => (g.name === 'tree-sitter-kotlin' ? '9.9.9' : '0.0.0'), + resolveUpstream: baseResolveUpstream, + fetchSource: (g) => g.name, // pass the name through to the fake readAbi + readAbi: (name) => (name === 'tree-sitter-swift' ? 15 : 14), + }; + let report: Array>; + const byKey = (k: string) => + must( + report.find((r) => r.grammar === k), + `no detect row for ${k}`, + ); + beforeAll(() => { + report = mod.detect(deps); + }); + + it('flags newer npm + github grammars as updates', () => { + expect(byKey('swift').update).toBe(true); // npm 9.9.9 != vendored 0.0.0 + expect(byKey('dart').update).toBe(true); // github sha differs from vendored + }); + + it('does not flag a same-version grammar, and skips its ABI fetch', () => { + expect(byKey('kotlin').update).toBe(false); // vendored == upstream 9.9.9 + expect(byKey('kotlin').abi).toBeNull(); + expect(byKey('kotlin').applicable).toBe(false); + }); + + it('holds tree-sitter-c: update detected, ABI-compatible, but never applicable', () => { + const c = byKey('c'); + expect(c.update).toBe(true); + expect(c.abi).toBe(14); + expect(c.abiCompatible).toBe(true); + expect(c.hold).toBeTruthy(); + expect(c.applicable).toBe(false); // policy-hold gate + }); + + it('refuses an ABI-incompatible candidate (15) — not applicable', () => { + const s = byKey('swift'); + expect(s.abi).toBe(15); + expect(s.abiCompatible).toBe(false); + expect(s.applicable).toBe(false); // ABI gate + }); + + it('marks a newer, un-held, ABI-14 grammar applicable', () => { + const d = byKey('dart'); + expect(d.abi).toBe(14); + expect(d.applicable).toBe(true); + }); + + it('records a per-grammar error entry when resolveUpstream throws, without skipping siblings', () => { + const report2 = mod.detect({ + ...deps, + resolveUpstream: (g) => { + if (g.name === 'tree-sitter-dart') throw new Error('gh api 503'); + return baseResolveUpstream(g); + }, + }); + const dart = must( + report2.find((r) => r.grammar === 'dart'), + 'no detect row for dart', + ); + expect(dart.error).toContain('gh api 503'); + expect(dart.update).toBeUndefined(); // error entry, not a classification + // The throw on one grammar must not drop the rest. + const swift = must( + report2.find((r) => r.grammar === 'swift'), + 'no detect row for swift', + ); + expect(swift.update).toBe(true); + expect(report2).toHaveLength(Object.keys(mod.GRAMMARS).length); + }); +}); + +describe('detect()/apply() agree on "newer" for github grammars', () => { + // github grammars carry up.version = `-g` (the provenance string apply() + // writes). detect() must compare the same up.version (not the bare sha7) so it stops + // reporting a false "update" once the bot has re-vendored once (#2187 review). + const PROV = '1.0.0-gabc1234'; + // deps for dart (github); other grammars get a harmless npm-shaped upstream so the + // detect() loop completes — we only inspect dart. + const dartDeps = (vendored: string): DetectDeps => ({ + vendoredVersion: (g) => (g.name === 'tree-sitter-dart' ? vendored : '0.0.0'), + resolveUpstream: (g) => + g.name === 'tree-sitter-dart' + ? { version: PROV, ref: 'abc1234def0', kind: 'github' } + : { version: '9.9.9', ref: '9.9.9', kind: 'npm' }, + fetchSource: (g) => g.name, + readAbi: () => 14, + }); + const dartRow = (vendored: string) => + must( + mod.detect(dartDeps(vendored)).find((r) => r.grammar === 'dart'), + 'no detect row for dart', + ); + + it('equal provenance → update:false (the asymmetry that is fixed)', () => { + expect(dartRow(PROV).update).toBe(false); + }); + + it('first-vendoring (plain version vs provenance) → update:true (not suppressed)', () => { + // vendored is the plain pre-bot version; up.version is `-g` → still newer. + expect(dartRow('1.0.0').update).toBe(true); + }); + + it('upstream sha advanced → update:true', () => { + expect(dartRow('1.0.0-g0000000').update).toBe(true); + }); + // The detect⇄apply agreement on the equal-provenance (already-current) case is + // asserted in U12's apply() tests — apply()'s not-newer path currently calls + // process.exit(0), which can't be exercised in-process until U12 makes it return. +}); + +describe('apply(--dry-run): resolves + validates but writes nothing', () => { + it('returns the candidate version without mutating the vendored package.json', () => { + const pkgPath = path.resolve( + path.dirname(fileURLToPath(import.meta.url)), + '../../vendor/tree-sitter-dart/package.json', + ); + const before = readFileSync(pkgPath, 'utf8'); + const version = mod.apply('dart', { + dryRun: true, + deps: { + vendoredVersion: () => '0.0.0', + resolveUpstream: () => ({ version: '9.9.9', ref: '9.9.9abc', kind: 'github' }), + fetchSource: () => 'unused', + readAbi: () => 14, + }, + }); + expect(version).toBe('9.9.9'); + expect(readFileSync(pkgPath, 'utf8')).toBe(before); // untouched + }); +}); + +describe('apply() error branches throw ApplyExit (CLI maps to exit codes)', () => { + const dartPkg = path.resolve( + path.dirname(fileURLToPath(import.meta.url)), + '../../vendor/tree-sitter-dart/package.json', + ); + // catch + return the thrown error's exit code (apply() throws instead of calling + // process.exit, so the error branches are exercisable in-process). + const codeOf = (fn: () => unknown): number => { + try { + fn(); + } catch (e) { + return (e as { code?: number }).code ?? -1; + } + throw new Error('expected apply() to throw'); + }; + + it('unknown grammar key → exit code 2', () => { + expect(codeOf(() => mod.apply('nope', { deps: {} }))).toBe(2); + }); + + it('held grammar (c) → exit code 3 (short-circuits before the newer check)', () => { + expect(codeOf(() => mod.apply('c', { deps: {} }))).toBe(3); + }); + + it('ABI-incompatible candidate (15) → exit code 3', () => { + const code = codeOf(() => + mod.apply('dart', { + deps: { + vendoredVersion: () => '0.0.0', // newer than upstream → reaches the ABI gate + resolveUpstream: () => ({ version: '9.9.9', ref: '9.9.9abc', kind: 'github' }), + fetchSource: () => 'unused', + readAbi: () => 15, + }, + }), + ); + expect(code).toBe(3); + }); + + it('not-newer (already current) → returns the current version, no throw, no write', () => { + const before = readFileSync(dartPkg, 'utf8'); + const deps = { + vendoredVersion: () => '9.9.9-gabc1234', + resolveUpstream: () => ({ + version: '9.9.9-gabc1234', + ref: 'abc1234', + kind: 'github' as const, + }), + fetchSource: () => 'unused', + readAbi: () => 14, + }; + // No dryRun: the not-newer path returns `have` before any fetch/copy. + expect(mod.apply('dart', { deps })).toBe('9.9.9-gabc1234'); + expect(readFileSync(dartPkg, 'utf8')).toBe(before); // untouched + }); +}); From 7c3d4e6862a2fced91c060bfc1ffab5edae3eea7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sat, 13 Jun 2026 18:49:03 +0100 Subject: [PATCH 14/16] =?UTF-8?q?feat(pdg):=20control=20dependence=20?= =?UTF-8?q?=E2=80=94=20post-dominators=20+=20CDG=20(Ferrante)=20[M5=20#208?= =?UTF-8?q?5]=20(#2188)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(pdg): add CDG + POST_DOMINATE edge types (M5 #2085) * feat(pdg): post-dominator tree on reverse CFG (M5 #2085) * feat(pdg): Ferrante control-dependence over the post-dom tree (M5 #2085) * feat(pdg): emitFileCdg + optional POST_DOMINATE debug edges (M5 #2085) * feat(pdg): wire CDG emission in-phase + pdgModeMismatch CDG-cap stamp (M5 #2085) * test(pdg): CDG snapshot + end-to-end pipeline answerability (M5 #2085) * fix(review): apply autofix feedback (M5 #2085) * fix(pdg): label CDG edges by controller arm sense, not edge kind (#2188 F1/F2/F4) Tri-review (with Codex as the independent engine) found the CDG 'T'/'F' label was wrong for the commonest control flow: the M1 TS visitor wires a condition's fall-through FALSE arm as `seq`/`loop-back`, but `branchSense` mapped both to 'T', so guard clauses, if-no-else, and loop `break` got 'T' instead of 'F' (F1, P1). The structural CDG edges were correct; only the label — the AC3 "under what condition does X run?" answer — was wrong. - F1: replace edge-kind `branchSense` with controller-arm-sense `labelFor`. An ambiguous fall-through edge (seq/loop-back) takes the COMPLEMENT of its source block's explicit cond-true/cond-false sibling arm. This correctly handles do/while (loop-back = TRUE arm) and inner-if-in-loop (loop-back = FALSE arm) — the ambiguity a kind→label table cannot resolve. Adds real-parser regression tests (the hand-built tests used a fictional cond-false edge and missed it). - F2: correct the false "sound over-approximation that never drops a real dependence" claim in post-dominators.ts — exit-unreachable regions both drop and invent control dependences (latent for the current TS visitor, which keeps EXIT reverse-reachable). Reframe the exit-less-loop test to characterize, not bless, the degenerate behavior. - F4: make the AC2 property-test reference compute post-dominance INDEPENDENTLY (node-removal reachability, no shared code with post-dominators.ts), so a post-dom direction bug can no longer pass both the impl and the reference. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(ci): root-prettier format + run-analyze pdg stamp gains maxCdgEdgesPerFunction (#2085) Two deterministic CI failures from the M5 CDG work: - quality/format: basicblock-roundtrip.test.ts failed CI's root `prettier --check .` (the pre-commit hook uses the gitnexus-local prettier config, which differs); reformatted with the root config. - tests/ubuntu/coverage: run-analyze.test.ts pinned the resolved RepoMeta.pdg shape (DEFAULTS) and the all-zero cap override without the new maxCdgEdgesPerFunction key (default 5000); added it so resolvePdgConfig toEqual and pdgModeMismatch(DEFAULTS) pass. (The stale-test sweep missed this file in PR #2188 — same trap M2 hit.) Co-Authored-By: Claude Opus 4.8 (1M context) * feat(mcp): add pdg_query tool definition (controls/flows modes) [M6 #2086] * feat(mcp): pdg_query backend — controls (CDG) + flows (REACHING_DEF) + e2e test [M6 #2086] * feat(mcp): document PDG edges + pdg_query (schema, cypher, skill, --pdg-gated ai-context) [M6 #2086] * fix(mcp): correct pdg_query symbol-anchor lower bound + harden inputs [PR #2188 review] Tri-review (Codex + adversarial + correctness lanes) of the M6 pdg_query surface found the symbol-anchor window over-includes a neighbor function's block. The upper bound was widened to the 1-based BasicBlock basis (symEnd+1) but the lower bound was left 0-based, so a block on the line directly above the target function leaked into the result. Shift both bounds +1 ([symStart+1, symEnd+1]) so the window is the function's true block span. Also from the same review: - pdg_query no longer throws on a no-arguments MCP call: the dispatch passes raw `params`, so default it to {} → a clean mode-validation error instead of a TypeError. (`explain` shares this latent pattern — pre-existing follow-up.) - tools.ts: the controls-mode description no longer hard-codes the 'F' branch sense for guards — `if (!ok) return;` rides the predicate's 'T' arm; the guard:true flag is label-agnostic (regex on the dependent block text). Tests: a hand-seeded adjacency regression (verified failing without the lower-bound +1) + a no-arguments validation test. Skill doc updated to document the two-sided [symStart+1, symEnd+1] window. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(mcp): drop always-true anchor conditional in pdg_query [CodeQL #2188] CodeQL alert 756 flagged `...(anchor ? { anchor } : {})` in _pdgQueryImpl as a useless conditional: `anchor` is unconditionally assigned in both the file-path and symbol branches before the return (the not-found/ambiguous/no-layer paths return earlier), so it is always truthy. Drop `| undefined` from the declaration (TypeScript definite-assignment holds across both branches) and emit `anchor` directly. No runtime change — the `anchor` field was already present on every result. Co-Authored-By: Claude Opus 4.8 (1M context) * test(cli): add hasPdg to the noStats bridge expectation [#2188] The M6 work threaded `hasPdg: options.pdg === true` into the AIContextOptions passed to generateAIContextFiles on the --skills regeneration path, but this test's strict .toEqual expectation predated it (4 keys vs 3 → CI failure). Add `hasPdg: false` (the value on this non---pdg path). The assertion stays strict; the #1477 noStats bridging it guards is unchanged. Co-Authored-By: Claude Opus 4.8 (1M context) * refactor(cli): collapse generateGitNexusContent params to an options bag [#2188] The function had grown to 9 positional params; reaching `hasPdg` meant passing six `undefined`s (the M6 review's maintainability flag). Collapse params 3-9 (generatedSkills, groupNames, noStats, skipSkills, runnerPath, defaultBranch, hasPdg) into a `GitNexusContentOptions` object with the defaults moved to destructuring. The body is unchanged (same local names); the single production caller and the test calls become self-documenting named fields. Pure refactor — generated AGENTS.md/CLAUDE.md content is byte-identical. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(cfg): skip CDG for exit-unreachable CFGs (unsound post-dominance) [#2188] M5 review P2: computePostDominators roots only at cfg.exitIndex and nothing enforced that EXIT is reachable from every block. For an entry-reachable region that cannot reach EXIT (a non-terminating loop, or a multi-terminal CFG a future visitor might emit) the EXIT-rooted reverse walk degenerates — it both drops real control dependences and invents spurious ones. Add a pure precondition predicate `isExitReachableFromAllBlocks` (co-located with the algorithm it guards) and gate it in emitFileCdg: a CFG that violates it is skipped for CDG (counted as skippedUnsoundFunctions + one onWarn), while its CFG and REACHING_DEF projections — which do not depend on post-dominance — are kept. A CDG-specific gate, not a widening of isEmitSafeCfg, so the blast radius is exactly the unsound CDG. The current TS visitor always satisfies the precondition (every loop gets a structural header→loopExit edge), so CDG output for real fixtures is unchanged. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(cfg): bound computeControlDependence materialization (heap parity) [#2188] M5 review P2: unlike computeReachingDefs (maxFacts) and the emit-side edge cap, computeControlDependence materialized the full deduped seen/out before emitFileCdg's per-function cap could trim it — O(edges × post-dom depth) heap for a deeply nested function. Add a `maxEdges` ceiling (default 0 = unbounded) returning {edges, truncated}, mirroring computeReachingDefs's {facts, truncated}. The ceiling is checked before pushing a new unique edge, so `truncated` means a genuine overflow (not merely "reached cap"). emitFileCdg passes a FIXED materialization ceiling (8× the default edge cap) — deliberately NOT derived from the runtime edge cap, because CDG's materialization IS the deduped-edge quantity the cap reports on (deriving it would pre-truncate that set and lose the exact dropped count). A ceiling hit is surfaced via onWarn + the truncated flag — never silent. Co-Authored-By: Claude Opus 4.8 (1M context) * refactor(mcp): share resolveBlockAnchor; fix explain's anchor off-by-one [#2188] M6 review P2 (duplication) + the flagged pre-existing _explainImpl correctness follow-up. _pdgQueryImpl and _explainImpl each carried a near-identical symbol↔block anchor resolver that had DRIFTED: pdg_query used the corrected [symStart+1, symEnd+1] window (BasicBlock startLine is 1-based, the symbol span 0-based) while _explainImpl still used [symStart, symEnd] — dropping a taint source on the function's final line AND leaking a neighbor's block on the line directly above. Extract one `resolveBlockAnchor` helper, used by both, that applies the correct window and a single (bare) clause convention (callers compose their own WHERE). This removes ~50 duplicated lines and fixes explain's anchor in one place. A hand-seeded characterization test (taint-explain Block 4) pins both bounds — verified to FAIL on the pre-fix window (it returned the line-10 neighbor instead of the line-15 final-line source). Existing taint-explain + pdg-query suites are unchanged (their fixtures have interior sources/sinks). Co-Authored-By: Claude Opus 4.8 (1M context) * fix(mcp): pdg_query reports "status unknown" when the layer can't be confirmed [#2188] M6 review P3 (Codex): when meta is UNREADABLE and the bounded global existence probe returns zero rows of the edge type, _pdgQueryImpl asserted "no PDG layer" — but a genuinely edge-free layer (all-linear functions) is indistinguishable from a missing one via that probe. Soften only that fallback path to an inconclusive "PDG layer status unknown — was this repo indexed with --pdg?" note. The meta-stamped path (stamp present, cap absent ⇒ layer truly missing) keeps the definitive "no PDG layer" wording. Co-Authored-By: Claude Opus 4.8 (1M context) * test(mcp): cover pdg_query ambiguous / pagination / Windows-path gaps [#2188] M6 review test-gap follow-ups, all hand-seeded with controlled data: - ambiguous symbol name → status:'ambiguous' + ranked candidates shape (uid/name/filePath/score), never a silent guess; - total/truncated page boundary in both directions (limit below the match count sets truncated with the full total; limit above it omits truncated); - a Windows-style filePath containing ':' resolves and fnLineOf decodes the function-line segment correctly (split-from-right past the drive letter). Co-Authored-By: Claude Opus 4.8 (1M context) * docs(skills): ship gitnexus-pdg-query skill mirrors + add pdg_query to the guide [#2086] M6 bundled pdg_query into this PR, but the skill shipped only in the canonical gitnexus/skills/ root. Mirror it (byte-identical) to the two hand-maintained roots the sibling taint skill uses — .claude/skills/gitnexus/ and the plugin — so Claude Code + plugin users get it too. Also extend the gitnexus-guide tool reference (all 3 copies, now byte-identical): add a `pdg_query` row + a "Control & data dependence" section mirroring the taint/`explain` section, and reconcile the pre-existing drift where only the .claude copy carried the `check` tool row (a real registered tool) — all three now list it. Co-Authored-By: Claude Opus 4.8 (1M context) * docs(architecture): refresh CFG/PDG section for the full M1–M6 stack [#2086] The PR body had deferred the "ARCHITECTURE docs refresh" to #2086; now that M6 ships here, do it: - MCP tools table gains `explain` and `pdg_query` (were absent). - "Optional CFG/PDG emission" was M1-only; rewrite to cover the whole opt-in stack — M1 CFG, M2 REACHING_DEF, M3/M4 taint, M5 CDG (Ferrante over CHK post-dominators, with the exit-unreachable skip), M6 read surface (pdg_query + explain, anchored + LIMIT-bounded, shared resolveBlockAnchor) — and note the no-Function→BasicBlock-edge join. - LadybugDB schema notes the `--pdg` additions: the `BasicBlock` node table and the CFG/REACHING_DEF/CDG/TAINTED/SANITIZES/TAINT_PATH relation types, kept out of the default VALID_RELATION_TYPES / web schema. Co-Authored-By: Claude Opus 4.8 (1M context) --------- Co-authored-by: Claude Opus 4.8 (1M context) --- .../skills/gitnexus/gitnexus-guide/SKILL.md | 12 +- .../gitnexus/gitnexus-pdg-query/SKILL.md | 89 ++++ ARCHITECTURE.md | 16 +- .../skills/gitnexus-guide/SKILL.md | 11 + .../skills/gitnexus-pdg-query/SKILL.md | 89 ++++ gitnexus-shared/src/graph/types.ts | 16 +- gitnexus-shared/src/lbug/schema-constants.ts | 6 + gitnexus/skills/gitnexus-guide.md | 11 + gitnexus/skills/gitnexus-pdg-query.md | 89 ++++ gitnexus/src/cli/ai-context.ts | 76 ++- gitnexus/src/cli/analyze.ts | 1 + .../core/ingestion/cfg/control-dependence.ts | 172 +++++++ gitnexus/src/core/ingestion/cfg/emit.ts | 188 +++++++ .../src/core/ingestion/cfg/post-dominators.ts | 218 ++++++++ gitnexus/src/core/ingestion/pipeline.ts | 9 + .../scope-resolution/pipeline/phase.ts | 1 + .../scope-resolution/pipeline/run.ts | 25 + gitnexus/src/core/run-analyze.ts | 14 + gitnexus/src/mcp/local/local-backend.ts | 373 +++++++++++--- gitnexus/src/mcp/resources.ts | 11 + gitnexus/src/mcp/tools.ts | 61 ++- gitnexus/src/storage/repo-manager.ts | 8 + .../integration/basicblock-roundtrip.test.ts | 30 +- .../__snapshots__/cdg-snapshot.test.ts.snap | 93 ++++ .../test/integration/cfg/cdg-snapshot.test.ts | 58 +++ .../test/integration/cfg/cfg-emit.test.ts | 167 +++++- .../cfg/fixtures/pdg-repo/guards.ts | 24 + .../test/integration/cfg/pipeline-pdg.test.ts | 45 +- gitnexus/test/integration/pdg-query.test.ts | 485 ++++++++++++++++++ .../test/integration/taint-explain.test.ts | 81 +++ gitnexus/test/unit/ai-context.test.ts | 50 +- .../test/unit/analyze-no-stats-bridge.test.ts | 2 + .../test/unit/cfg/control-dependence.test.ts | 413 +++++++++++++++ .../test/unit/cfg/post-dominators.test.ts | 230 +++++++++ gitnexus/test/unit/pdg-mode-flip.test.ts | 38 ++ gitnexus/test/unit/run-analyze.test.ts | 3 + gitnexus/test/unit/schema.test.ts | 6 + gitnexus/test/unit/security.test.ts | 9 + gitnexus/test/unit/tools.test.ts | 5 +- 39 files changed, 3097 insertions(+), 138 deletions(-) create mode 100644 .claude/skills/gitnexus/gitnexus-pdg-query/SKILL.md create mode 100644 gitnexus-claude-plugin/skills/gitnexus-pdg-query/SKILL.md create mode 100644 gitnexus/skills/gitnexus-pdg-query.md create mode 100644 gitnexus/src/core/ingestion/cfg/control-dependence.ts create mode 100644 gitnexus/src/core/ingestion/cfg/post-dominators.ts create mode 100644 gitnexus/test/integration/cfg/__snapshots__/cdg-snapshot.test.ts.snap create mode 100644 gitnexus/test/integration/cfg/cdg-snapshot.test.ts create mode 100644 gitnexus/test/integration/cfg/fixtures/pdg-repo/guards.ts create mode 100644 gitnexus/test/integration/pdg-query.test.ts create mode 100644 gitnexus/test/unit/cfg/control-dependence.test.ts create mode 100644 gitnexus/test/unit/cfg/post-dominators.test.ts diff --git a/.claude/skills/gitnexus/gitnexus-guide/SKILL.md b/.claude/skills/gitnexus/gitnexus-guide/SKILL.md index a70279222..7f90f4e6d 100644 --- a/.claude/skills/gitnexus/gitnexus-guide/SKILL.md +++ b/.claude/skills/gitnexus/gitnexus-guide/SKILL.md @@ -36,10 +36,11 @@ For any task involving code understanding, debugging, impact analysis, or refact | `context` | 360-degree symbol view — categorized refs, processes it participates in | | `impact` | Symbol blast radius — what breaks at depth 1/2/3 with confidence | | `detect_changes` | Git-diff impact — what do your current changes affect | -| `check` | Check graph invariants such as circular imports | | `rename` | Multi-file coordinated rename with confidence-tagged edits | | `cypher` | Raw graph queries (read `gitnexus://repo/{name}/schema` first) | | `explain` | Persisted taint findings — source→sink data flows (needs `analyze --pdg`) | +| `pdg_query` | Control/data dependence — what gates X (CDG) / where Y flows (REACHING_DEF); needs `analyze --pdg` | +| `check` | Check graph invariants such as circular imports | | `list_repos` | Discover indexed repos (paginated — `limit`/`offset`) | ### Paginating `list_repos` @@ -83,6 +84,15 @@ Notes: `offset` ≥ `total` returns an empty page (with `total` still reported). A repo indexed without `--pdg` returns a clear "no taint layer" note. Caveats: findings are intra-procedural only — cross-function, closure/callback, property/field, and implicit flows are not modeled, so the absence of a finding is **not** proof of safety. `SANITIZES` (sanitizer-kill) edges are queryable via `cypher`. +### Control & data dependence (`pdg_query`) + +`pdg_query` reads the control/data-dependence layers `gitnexus analyze --pdg` records (CDG + REACHING_DEF, basic-block granular) — the control/data analog of `explain`. It is **always anchored** (a `target` file path or symbol, resolved like `context`) and has two modes: + +- `pdg_query { mode: "controls", target: "..." }` — CDG: "under what condition does X run?". Each edge is a controlling predicate block → dependent block with the branch sense (`'T'`/`'F'`) in `reason`; an edge into an early `return`/`throw` is flagged `guard: true` (guard-clause discovery — the sense depends on the predicate, so don't filter guards by a fixed label). +- `pdg_query { mode: "flows", target: "...", variable?: "..." }` — REACHING_DEF def→use edges within the function; pass `variable` to trace one binding. + +A repo indexed without `--pdg` returns a "no PDG layer" note (or "status unknown" when the layer can't be confirmed). Intra-procedural only — cross-function flow is taint's domain (`explain`). The raw CDG/REACHING_DEF edges are also queryable via `cypher`. See the `gitnexus-pdg-query` skill for the full query surface. + ## Resources Reference Lightweight reads (~100-500 tokens) for navigation: diff --git a/.claude/skills/gitnexus/gitnexus-pdg-query/SKILL.md b/.claude/skills/gitnexus/gitnexus-pdg-query/SKILL.md new file mode 100644 index 000000000..f2fcd7d3b --- /dev/null +++ b/.claude/skills/gitnexus/gitnexus-pdg-query/SKILL.md @@ -0,0 +1,89 @@ +--- +name: gitnexus-pdg-query +description: "Use when querying or extending GitNexus's PDG control/data-dependence surface (the `pdg_query` MCP tool, CDG/REACHING_DEF edges), or reasoning about \"what controls X\" / \"where does Y flow\" / guard clauses. Examples: \"what guards this statement?\", \"trace this variable within the function\", \"why is the pdg_query result empty?\", \"add a CDG query\"." +--- + +# PDG query surface with GitNexus + +Expert knowledge for the `pdg_query` MCP tool and the control/data-dependence +edges it reads — the opt-in `--pdg` program-dependence layers. Read this before +touching `gitnexus/src/mcp/local/local-backend.ts` (`_pdgQueryImpl`) or the +`pdg_query` tool def, or when explaining a `pdg_query` result. + +## When to Use + +- "Under what condition does this statement run?" (guarding predicates). +- "Where does this variable flow inside the function?" (def→use). +- Guard-clause discovery (early-return guards — subsumes the #559 heuristic). +- Extending or reviewing `pdg_query` / the CDG / REACHING_DEF read path. +- Debugging an empty or surprising `pdg_query` result. + +## The layered substrate (build order) + +`pdg_query` runs **on** the same graph taint runs on. Each layer is opt-in +behind `--pdg`; a default `analyze` run records none of them (byte-identical). + +``` +L1 CFG per-function basic blocks + control-flow edges (M1 #2081) +L2 REACHING_DEF GEN/KILL def→use data dependence (pure solver) (M2 #2082) +L5 CDG Ferrante control dependence (post-dominators) (M5 #2085) +``` + +All three are `BasicBlock → BasicBlock` edges in the single `CodeRelation` table +(keyed by the `type` property). There is **no** `Function → BasicBlock` edge. + +## The two modes + +- `pdg_query({ mode: 'controls', target })` — CDG. For the anchored function, + each edge: controlling predicate block → dependent block + branch sense in + `label` (`'T'` = predicate's true/taken arm, `'F'` = false/fall-through). An + edge into an early-return/throw block is flagged `guard: true`. +- `pdg_query({ mode: 'flows', target, variable? })` — REACHING_DEF def→use + edges; `variable` filters to one binding. + +`target` is **required** — a file path or a symbol/function name (resolved like +`context()`). There is no anchorless mode (see below). + +## The corrected guard-clause Cypher + +The RFC #567 §2 form (`[:CDG {label:'F'}]`) does **not** run as written. Edges +are values of the single `CodeRelation` table's `type` property, and the branch +sense is in `reason`, NOT a `label` column: + +```cypher +MATCH (pred:BasicBlock)-[r:CodeRelation {type: 'CDG'}]->(dep:BasicBlock) +WHERE dep.text STARTS WITH 'return' OR dep.text STARTS WITH 'throw' +RETURN pred.startLine, r.reason AS branch, dep.startLine, dep.text +``` + +`r.reason` is the sense the predicate took to reach the early exit. For +`if (!ok) return;` the return rides the predicate's **true** arm (`'T'`) and the +protected body rides the **false** arm (`'F'`) — polarity depends on the guard, +so don't hard-code one sense. + +## Gotchas (the load-bearing ones) + +- **Always anchored + LIMIT-bounded.** LadybugDB has no rel-property index, so + an unanchored `[:CDG*]`/`[:REACHING_DEF*]` path scan is unbounded. `pdg_query` + requires `target` and bounds the page; raw `cypher` callers must anchor on a + file id-prefix or symbol span themselves. +- **BasicBlock↔symbol join is reconstructed.** No `Function→BasicBlock` edge: + the block is matched by its id-prefix (`BasicBlock:::…`) + plus `startLine` within the symbol's span. BasicBlock `startLine` is **1-based** + while the symbol node's `startLine`/`endLine` are **0-based**, so **both** bounds + are shifted `+1` (`[symStart+1, symEnd+1]`): the upper `+1` keeps a guard/def/use + on the function's **final line**, the lower `+1` excludes an adjacent function's + block on the line directly **above**. Same-line / nested functions anchor coarsely. +- **No PDG layer ⇒ a note, not an error.** If the repo wasn't indexed with + `--pdg` the tool returns `{ results: [], note: "no PDG layer …" }` (cheap meta + probe on `RepoMeta.pdg.maxCdgEdgesPerFunction` / `maxReachingDefEdgesPerFunction`). +- **CDG labels are binary in M5/M6.** Every `switch`-case arm is `'T'`; per-case + conditions are not yet distinguished. +- **Intra-procedural only.** Cross-function flow is taint's domain (`explain`). + +## Mirror, don't fork + +`_pdgQueryImpl` is the front half of `_explainImpl` (WAL wrapper, meta no-layer +probe, limit validation, `resolveSymbolCandidates` anchoring) with CDG/ +REACHING_DEF instead of TAINTED — and none of taint's path-codec / interproc +`TAINT_PATH` machinery. Reuse those shared helpers; do not re-implement them. diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index b3319f172..4aa854a88 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -41,6 +41,8 @@ Monorepo: **CLI/MCP** (`gitnexus/`) + **browser UI** (`gitnexus-web/`). | `route_map` | API route → handler → consumer mappings | | `tool_map` | MCP/RPC tool definitions and handlers | | `shape_check` | Response shape vs consumer property access mismatches | +| `explain` | Persisted taint findings (source→sink data flows) — needs `analyze --pdg` | +| `pdg_query` | Control/data dependence — CDG (`mode: controls`) / REACHING_DEF (`mode: flows`) — needs `analyze --pdg` | | `group_list` | List repo groups or details for one group | | `group_sync` | Rebuild group Contract Registry (`contracts.json`) and bridge graph | @@ -204,9 +206,17 @@ Language-agnostic scope-resolution resolver. This is the resolution path for eve Orchestrator: `runScopeResolution(input, provider)` in `scope-resolution/pipeline/run.ts`. Pipeline phase: `scopeResolutionPhase` in `scope-resolution/pipeline/phase.ts` — iterates the registered `SCOPE_RESOLVERS` over the worker-serialized `ParsedFile`s. (Per-language `emitScopeCaptures` hooks may reuse a cached Tree via the orchestrator's `treeCache`, but in worker-pool runs that cache is empty — Trees can't cross MessageChannels — so they consume the pre-extracted `ParsedFile` instead; § Performance notes.) -### Optional CFG/PDG emission (`--pdg`, #2081 M1) +### Optional CFG/PDG emission (`--pdg`, #2081–#2086) -On a `--pdg` run, the parse worker builds a per-function control-flow graph from the tree-sitter AST (`LanguageProvider.cfgVisitor`; TypeScript/JavaScript in M1) and serializes it onto `ParsedFile.cfgSideChannel` as plain data. Scope-resolution then emits `BasicBlock` nodes + `CFG` edges from that side-channel **inside Phase 4 of `runScopeResolution`, while the disk-backed ParsedFile store is still live** — the only window where the worker-built CFGs are loaded (the store is cleared right after the phase returns). A standalone post-`mro` phase would read an empty store, so the CFG emit deliberately lives in-phase, mirroring the `applyCaptureSideChannel` pattern. The opt-in is off by default (graph byte-identical), folded into the parse-cache key (a pdg-off warm cache is never reused on a `--pdg` run), and bounded by a per-function edge cap that logs any dropped edges. Edge *kind* (`seq`/`cond-true`/`loop-back`/…) rides in the `CFG` relationship's `reason` (CFG is a single `CodeRelation` type, not one type per kind). See `core/ingestion/cfg/`. +On a `--pdg` run the parse worker builds a per-function control-flow graph from the tree-sitter AST (`LanguageProvider.cfgVisitor`; TypeScript/JavaScript today) and serializes it onto `ParsedFile.cfgSideChannel` as plain data. Scope-resolution then emits the program-dependence layers from that side-channel **inside Phase 4 of `runScopeResolution`, while the disk-backed ParsedFile store is still live** — the only window where the worker-built CFGs are loaded (the store is cleared right after the phase returns). A standalone post-`mro` phase would read an empty store, so the emit deliberately lives in-phase, mirroring the `applyCaptureSideChannel` pattern. The opt-in is off by default (graph byte-identical), folded into the parse-cache key (a pdg-off warm cache is never reused on a `--pdg` run), and each layer is bounded by a per-function edge cap that logs any dropped edges. All layers are `BasicBlock → BasicBlock` edges in the single `CodeRelation` table, keyed by `type`; there is **no** `Function → BasicBlock` edge — the symbol↔block join is reconstructed from the BasicBlock id prefix + line span. The layers build on each other: + +- **M1 — CFG** (#2081): `BasicBlock` nodes + `CFG` edges. Edge *kind* (`seq`/`cond-true`/`loop-back`/…) rides the `reason` column (CFG is one `CodeRelation` type, not one per kind). +- **M2 — REACHING_DEF** (#2082): GEN/KILL def→use data dependence from a pure fixpoint solver; the variable name rides `reason`. +- **M3/M4 — TAINTED / SANITIZES / TAINT_PATH** (#2083–#2084): intra- and inter-procedural taint (source→sink) — the `explain` tool's data. +- **M5 — CDG** (#2085): Ferrante control dependence over a Cooper–Harvey–Kennedy post-dominator tree (the EXIT-rooted reverse CFG); branch sense (`'T'`/`'F'`) rides `reason`. A CFG whose EXIT is unreachable from some block is skipped for CDG (post-dominance would be unsound) while its CFG/REACHING_DEF layers are kept. +- **M6 — read surface** (#2086): the `pdg_query` MCP tool answers "what gates X?" (CDG, `mode: controls`) and "where does Y flow?" (REACHING_DEF, `mode: flows`); `explain` is the taint consumer. Both are always anchored + `LIMIT`-bounded (LadybugDB has no rel-property index) and share one `resolveBlockAnchor` helper. These PDG edge types are deliberately kept out of the default `VALID_RELATION_TYPES` / web schema. + +See `core/ingestion/cfg/` (emit + the pure CFG / post-dominator / control-dependence / reaching-defs / taint passes) and `mcp/local/local-backend.ts` (`_pdgQueryImpl`, `_explainImpl`, the shared `resolveBlockAnchor`). ### `ScopeResolver` contract @@ -383,6 +393,8 @@ Defined in `lbug/schema.ts`. Separate node tables per type, single `CodeRelation **Relation types** (`CodeRelation.type`): CONTAINS, DEFINES, CALLS, IMPORTS, EXTENDS, IMPLEMENTS, HAS_METHOD, HAS_PROPERTY, ACCESSES, METHOD_OVERRIDES, METHOD_IMPLEMENTS, MEMBER_OF, STEP_IN_PROCESS, HANDLES_ROUTE, FETCHES, HANDLES_TOOL, ENTRY_POINT_OF. +**Optional `--pdg` additions** (off by default, opt-in via `gitnexus analyze --pdg`; see _Optional CFG/PDG emission_ above): a `BasicBlock` node table, plus the PDG relation types `CFG`, `REACHING_DEF`, `CDG`, `TAINTED`, `SANITIZES`, and `TAINT_PATH` on the same `CodeRelation` table. These are deliberately kept out of the default `VALID_RELATION_TYPES` / web graph schema — query them via `cypher`, `explain`, or `pdg_query`. + ## Embeddings and search **Embeddings** (`src/core/embeddings/`): Snowflake arctic-embed-xs (384D). Embeddable: File, Function, Class, Method, Interface. Incremental via SHA1 content hash. Separate `Embedding` table. diff --git a/gitnexus-claude-plugin/skills/gitnexus-guide/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-guide/SKILL.md index a71429f32..7f90f4e6d 100644 --- a/gitnexus-claude-plugin/skills/gitnexus-guide/SKILL.md +++ b/gitnexus-claude-plugin/skills/gitnexus-guide/SKILL.md @@ -39,6 +39,8 @@ For any task involving code understanding, debugging, impact analysis, or refact | `rename` | Multi-file coordinated rename with confidence-tagged edits | | `cypher` | Raw graph queries (read `gitnexus://repo/{name}/schema` first) | | `explain` | Persisted taint findings — source→sink data flows (needs `analyze --pdg`) | +| `pdg_query` | Control/data dependence — what gates X (CDG) / where Y flows (REACHING_DEF); needs `analyze --pdg` | +| `check` | Check graph invariants such as circular imports | | `list_repos` | Discover indexed repos (paginated — `limit`/`offset`) | ### Paginating `list_repos` @@ -82,6 +84,15 @@ Notes: `offset` ≥ `total` returns an empty page (with `total` still reported). A repo indexed without `--pdg` returns a clear "no taint layer" note. Caveats: findings are intra-procedural only — cross-function, closure/callback, property/field, and implicit flows are not modeled, so the absence of a finding is **not** proof of safety. `SANITIZES` (sanitizer-kill) edges are queryable via `cypher`. +### Control & data dependence (`pdg_query`) + +`pdg_query` reads the control/data-dependence layers `gitnexus analyze --pdg` records (CDG + REACHING_DEF, basic-block granular) — the control/data analog of `explain`. It is **always anchored** (a `target` file path or symbol, resolved like `context`) and has two modes: + +- `pdg_query { mode: "controls", target: "..." }` — CDG: "under what condition does X run?". Each edge is a controlling predicate block → dependent block with the branch sense (`'T'`/`'F'`) in `reason`; an edge into an early `return`/`throw` is flagged `guard: true` (guard-clause discovery — the sense depends on the predicate, so don't filter guards by a fixed label). +- `pdg_query { mode: "flows", target: "...", variable?: "..." }` — REACHING_DEF def→use edges within the function; pass `variable` to trace one binding. + +A repo indexed without `--pdg` returns a "no PDG layer" note (or "status unknown" when the layer can't be confirmed). Intra-procedural only — cross-function flow is taint's domain (`explain`). The raw CDG/REACHING_DEF edges are also queryable via `cypher`. See the `gitnexus-pdg-query` skill for the full query surface. + ## Resources Reference Lightweight reads (~100-500 tokens) for navigation: diff --git a/gitnexus-claude-plugin/skills/gitnexus-pdg-query/SKILL.md b/gitnexus-claude-plugin/skills/gitnexus-pdg-query/SKILL.md new file mode 100644 index 000000000..f2fcd7d3b --- /dev/null +++ b/gitnexus-claude-plugin/skills/gitnexus-pdg-query/SKILL.md @@ -0,0 +1,89 @@ +--- +name: gitnexus-pdg-query +description: "Use when querying or extending GitNexus's PDG control/data-dependence surface (the `pdg_query` MCP tool, CDG/REACHING_DEF edges), or reasoning about \"what controls X\" / \"where does Y flow\" / guard clauses. Examples: \"what guards this statement?\", \"trace this variable within the function\", \"why is the pdg_query result empty?\", \"add a CDG query\"." +--- + +# PDG query surface with GitNexus + +Expert knowledge for the `pdg_query` MCP tool and the control/data-dependence +edges it reads — the opt-in `--pdg` program-dependence layers. Read this before +touching `gitnexus/src/mcp/local/local-backend.ts` (`_pdgQueryImpl`) or the +`pdg_query` tool def, or when explaining a `pdg_query` result. + +## When to Use + +- "Under what condition does this statement run?" (guarding predicates). +- "Where does this variable flow inside the function?" (def→use). +- Guard-clause discovery (early-return guards — subsumes the #559 heuristic). +- Extending or reviewing `pdg_query` / the CDG / REACHING_DEF read path. +- Debugging an empty or surprising `pdg_query` result. + +## The layered substrate (build order) + +`pdg_query` runs **on** the same graph taint runs on. Each layer is opt-in +behind `--pdg`; a default `analyze` run records none of them (byte-identical). + +``` +L1 CFG per-function basic blocks + control-flow edges (M1 #2081) +L2 REACHING_DEF GEN/KILL def→use data dependence (pure solver) (M2 #2082) +L5 CDG Ferrante control dependence (post-dominators) (M5 #2085) +``` + +All three are `BasicBlock → BasicBlock` edges in the single `CodeRelation` table +(keyed by the `type` property). There is **no** `Function → BasicBlock` edge. + +## The two modes + +- `pdg_query({ mode: 'controls', target })` — CDG. For the anchored function, + each edge: controlling predicate block → dependent block + branch sense in + `label` (`'T'` = predicate's true/taken arm, `'F'` = false/fall-through). An + edge into an early-return/throw block is flagged `guard: true`. +- `pdg_query({ mode: 'flows', target, variable? })` — REACHING_DEF def→use + edges; `variable` filters to one binding. + +`target` is **required** — a file path or a symbol/function name (resolved like +`context()`). There is no anchorless mode (see below). + +## The corrected guard-clause Cypher + +The RFC #567 §2 form (`[:CDG {label:'F'}]`) does **not** run as written. Edges +are values of the single `CodeRelation` table's `type` property, and the branch +sense is in `reason`, NOT a `label` column: + +```cypher +MATCH (pred:BasicBlock)-[r:CodeRelation {type: 'CDG'}]->(dep:BasicBlock) +WHERE dep.text STARTS WITH 'return' OR dep.text STARTS WITH 'throw' +RETURN pred.startLine, r.reason AS branch, dep.startLine, dep.text +``` + +`r.reason` is the sense the predicate took to reach the early exit. For +`if (!ok) return;` the return rides the predicate's **true** arm (`'T'`) and the +protected body rides the **false** arm (`'F'`) — polarity depends on the guard, +so don't hard-code one sense. + +## Gotchas (the load-bearing ones) + +- **Always anchored + LIMIT-bounded.** LadybugDB has no rel-property index, so + an unanchored `[:CDG*]`/`[:REACHING_DEF*]` path scan is unbounded. `pdg_query` + requires `target` and bounds the page; raw `cypher` callers must anchor on a + file id-prefix or symbol span themselves. +- **BasicBlock↔symbol join is reconstructed.** No `Function→BasicBlock` edge: + the block is matched by its id-prefix (`BasicBlock:::…`) + plus `startLine` within the symbol's span. BasicBlock `startLine` is **1-based** + while the symbol node's `startLine`/`endLine` are **0-based**, so **both** bounds + are shifted `+1` (`[symStart+1, symEnd+1]`): the upper `+1` keeps a guard/def/use + on the function's **final line**, the lower `+1` excludes an adjacent function's + block on the line directly **above**. Same-line / nested functions anchor coarsely. +- **No PDG layer ⇒ a note, not an error.** If the repo wasn't indexed with + `--pdg` the tool returns `{ results: [], note: "no PDG layer …" }` (cheap meta + probe on `RepoMeta.pdg.maxCdgEdgesPerFunction` / `maxReachingDefEdgesPerFunction`). +- **CDG labels are binary in M5/M6.** Every `switch`-case arm is `'T'`; per-case + conditions are not yet distinguished. +- **Intra-procedural only.** Cross-function flow is taint's domain (`explain`). + +## Mirror, don't fork + +`_pdgQueryImpl` is the front half of `_explainImpl` (WAL wrapper, meta no-layer +probe, limit validation, `resolveSymbolCandidates` anchoring) with CDG/ +REACHING_DEF instead of TAINTED — and none of taint's path-codec / interproc +`TAINT_PATH` machinery. Reuse those shared helpers; do not re-implement them. diff --git a/gitnexus-shared/src/graph/types.ts b/gitnexus-shared/src/graph/types.ts index 86abc9eba..085c27d03 100644 --- a/gitnexus-shared/src/graph/types.ts +++ b/gitnexus-shared/src/graph/types.ts @@ -157,7 +157,21 @@ export type RelationshipType = | 'SANITIZES' /** Materialized source→sink taint path. Working name — final name/representation * is confirmed when M3/M4 emits it; no persisted edge exists before then. */ - | 'TAINT_PATH'; + | 'TAINT_PATH' + /** Control-dependence edge (PDG, issue #2085 M5): block `dependent` (target) + * executes only because the branch at block `controller` (source) took a + * given side. The branch sense (`'T'` | `'F'`) rides the relation's existing + * `reason` column — mirroring how `CFG` stores its edge kind there — since + * the single `CodeRelation` table has no dedicated label column. */ + | 'CDG' + /** Debug-only post-dominator-tree edge (#2085 M5): a block → its immediate + * post-dominator, emitted behind the `GITNEXUS_PDG_EMIT_POST_DOMINATE` env + * flag for inspection. Never emitted in a normal `--pdg` run. Note: as a + * member of this exported union it is a forward-compatibility commitment — + * removing it later is a breaking schema change — and it is deliberately + * excluded from `VALID_RELATION_TYPES` so it never enters impact-style + * symbol-space traversal (same posture as the taint substrate edges). */ + | 'POST_DOMINATE'; export interface GraphNode { id: string; diff --git a/gitnexus-shared/src/lbug/schema-constants.ts b/gitnexus-shared/src/lbug/schema-constants.ts index d022ba5c4..875f74d2e 100644 --- a/gitnexus-shared/src/lbug/schema-constants.ts +++ b/gitnexus-shared/src/lbug/schema-constants.ts @@ -77,6 +77,12 @@ export const REL_TYPES = [ 'TAINTED', 'SANITIZES', 'TAINT_PATH', + // Control dependence (PDG, issue #2085 M5) — CDG carries its 'T'|'F' branch + // label in the relation's `reason` column; POST_DOMINATE is debug-only + // (behind GITNEXUS_PDG_EMIT_POST_DOMINATE). Both are BasicBlock→BasicBlock, + // reusing the existing FROM BasicBlock TO BasicBlock pair in RELATION_SCHEMA. + 'CDG', + 'POST_DOMINATE', ] as const; export type RelType = (typeof REL_TYPES)[number]; diff --git a/gitnexus/skills/gitnexus-guide.md b/gitnexus/skills/gitnexus-guide.md index a71429f32..7f90f4e6d 100644 --- a/gitnexus/skills/gitnexus-guide.md +++ b/gitnexus/skills/gitnexus-guide.md @@ -39,6 +39,8 @@ For any task involving code understanding, debugging, impact analysis, or refact | `rename` | Multi-file coordinated rename with confidence-tagged edits | | `cypher` | Raw graph queries (read `gitnexus://repo/{name}/schema` first) | | `explain` | Persisted taint findings — source→sink data flows (needs `analyze --pdg`) | +| `pdg_query` | Control/data dependence — what gates X (CDG) / where Y flows (REACHING_DEF); needs `analyze --pdg` | +| `check` | Check graph invariants such as circular imports | | `list_repos` | Discover indexed repos (paginated — `limit`/`offset`) | ### Paginating `list_repos` @@ -82,6 +84,15 @@ Notes: `offset` ≥ `total` returns an empty page (with `total` still reported). A repo indexed without `--pdg` returns a clear "no taint layer" note. Caveats: findings are intra-procedural only — cross-function, closure/callback, property/field, and implicit flows are not modeled, so the absence of a finding is **not** proof of safety. `SANITIZES` (sanitizer-kill) edges are queryable via `cypher`. +### Control & data dependence (`pdg_query`) + +`pdg_query` reads the control/data-dependence layers `gitnexus analyze --pdg` records (CDG + REACHING_DEF, basic-block granular) — the control/data analog of `explain`. It is **always anchored** (a `target` file path or symbol, resolved like `context`) and has two modes: + +- `pdg_query { mode: "controls", target: "..." }` — CDG: "under what condition does X run?". Each edge is a controlling predicate block → dependent block with the branch sense (`'T'`/`'F'`) in `reason`; an edge into an early `return`/`throw` is flagged `guard: true` (guard-clause discovery — the sense depends on the predicate, so don't filter guards by a fixed label). +- `pdg_query { mode: "flows", target: "...", variable?: "..." }` — REACHING_DEF def→use edges within the function; pass `variable` to trace one binding. + +A repo indexed without `--pdg` returns a "no PDG layer" note (or "status unknown" when the layer can't be confirmed). Intra-procedural only — cross-function flow is taint's domain (`explain`). The raw CDG/REACHING_DEF edges are also queryable via `cypher`. See the `gitnexus-pdg-query` skill for the full query surface. + ## Resources Reference Lightweight reads (~100-500 tokens) for navigation: diff --git a/gitnexus/skills/gitnexus-pdg-query.md b/gitnexus/skills/gitnexus-pdg-query.md new file mode 100644 index 000000000..f2fcd7d3b --- /dev/null +++ b/gitnexus/skills/gitnexus-pdg-query.md @@ -0,0 +1,89 @@ +--- +name: gitnexus-pdg-query +description: "Use when querying or extending GitNexus's PDG control/data-dependence surface (the `pdg_query` MCP tool, CDG/REACHING_DEF edges), or reasoning about \"what controls X\" / \"where does Y flow\" / guard clauses. Examples: \"what guards this statement?\", \"trace this variable within the function\", \"why is the pdg_query result empty?\", \"add a CDG query\"." +--- + +# PDG query surface with GitNexus + +Expert knowledge for the `pdg_query` MCP tool and the control/data-dependence +edges it reads — the opt-in `--pdg` program-dependence layers. Read this before +touching `gitnexus/src/mcp/local/local-backend.ts` (`_pdgQueryImpl`) or the +`pdg_query` tool def, or when explaining a `pdg_query` result. + +## When to Use + +- "Under what condition does this statement run?" (guarding predicates). +- "Where does this variable flow inside the function?" (def→use). +- Guard-clause discovery (early-return guards — subsumes the #559 heuristic). +- Extending or reviewing `pdg_query` / the CDG / REACHING_DEF read path. +- Debugging an empty or surprising `pdg_query` result. + +## The layered substrate (build order) + +`pdg_query` runs **on** the same graph taint runs on. Each layer is opt-in +behind `--pdg`; a default `analyze` run records none of them (byte-identical). + +``` +L1 CFG per-function basic blocks + control-flow edges (M1 #2081) +L2 REACHING_DEF GEN/KILL def→use data dependence (pure solver) (M2 #2082) +L5 CDG Ferrante control dependence (post-dominators) (M5 #2085) +``` + +All three are `BasicBlock → BasicBlock` edges in the single `CodeRelation` table +(keyed by the `type` property). There is **no** `Function → BasicBlock` edge. + +## The two modes + +- `pdg_query({ mode: 'controls', target })` — CDG. For the anchored function, + each edge: controlling predicate block → dependent block + branch sense in + `label` (`'T'` = predicate's true/taken arm, `'F'` = false/fall-through). An + edge into an early-return/throw block is flagged `guard: true`. +- `pdg_query({ mode: 'flows', target, variable? })` — REACHING_DEF def→use + edges; `variable` filters to one binding. + +`target` is **required** — a file path or a symbol/function name (resolved like +`context()`). There is no anchorless mode (see below). + +## The corrected guard-clause Cypher + +The RFC #567 §2 form (`[:CDG {label:'F'}]`) does **not** run as written. Edges +are values of the single `CodeRelation` table's `type` property, and the branch +sense is in `reason`, NOT a `label` column: + +```cypher +MATCH (pred:BasicBlock)-[r:CodeRelation {type: 'CDG'}]->(dep:BasicBlock) +WHERE dep.text STARTS WITH 'return' OR dep.text STARTS WITH 'throw' +RETURN pred.startLine, r.reason AS branch, dep.startLine, dep.text +``` + +`r.reason` is the sense the predicate took to reach the early exit. For +`if (!ok) return;` the return rides the predicate's **true** arm (`'T'`) and the +protected body rides the **false** arm (`'F'`) — polarity depends on the guard, +so don't hard-code one sense. + +## Gotchas (the load-bearing ones) + +- **Always anchored + LIMIT-bounded.** LadybugDB has no rel-property index, so + an unanchored `[:CDG*]`/`[:REACHING_DEF*]` path scan is unbounded. `pdg_query` + requires `target` and bounds the page; raw `cypher` callers must anchor on a + file id-prefix or symbol span themselves. +- **BasicBlock↔symbol join is reconstructed.** No `Function→BasicBlock` edge: + the block is matched by its id-prefix (`BasicBlock:::…`) + plus `startLine` within the symbol's span. BasicBlock `startLine` is **1-based** + while the symbol node's `startLine`/`endLine` are **0-based**, so **both** bounds + are shifted `+1` (`[symStart+1, symEnd+1]`): the upper `+1` keeps a guard/def/use + on the function's **final line**, the lower `+1` excludes an adjacent function's + block on the line directly **above**. Same-line / nested functions anchor coarsely. +- **No PDG layer ⇒ a note, not an error.** If the repo wasn't indexed with + `--pdg` the tool returns `{ results: [], note: "no PDG layer …" }` (cheap meta + probe on `RepoMeta.pdg.maxCdgEdgesPerFunction` / `maxReachingDefEdgesPerFunction`). +- **CDG labels are binary in M5/M6.** Every `switch`-case arm is `'T'`; per-case + conditions are not yet distinguished. +- **Intra-procedural only.** Cross-function flow is taint's domain (`explain`). + +## Mirror, don't fork + +`_pdgQueryImpl` is the front half of `_explainImpl` (WAL wrapper, meta no-layer +probe, limit validation, `resolveSymbolCandidates` anchoring) with CDG/ +REACHING_DEF instead of TAINTED — and none of taint's path-codec / interproc +`TAINT_PATH` machinery. Reuse those shared helpers; do not re-implement them. diff --git a/gitnexus/src/cli/ai-context.ts b/gitnexus/src/cli/ai-context.ts index 6700c020b..8cbfe2b6d 100644 --- a/gitnexus/src/cli/ai-context.ts +++ b/gitnexus/src/cli/ai-context.ts @@ -35,6 +35,12 @@ export interface AIContextOptions { * plain caller that omits it gets "main", preserving prior behavior. */ defaultBranch?: string; + /** + * Whether the index was built with `--pdg` (#2086 M6). Gates the `pdg_query` + * line in the generated block — without the PDG layer the tool only returns a + * "no PDG layer" note, so advertising it on a non-`--pdg` index is noise. + */ + hasPdg?: boolean; } const GITNEXUS_START_MARKER = ''; @@ -105,26 +111,45 @@ export function markdownSafeBranch(branch: string): string { return branch.replace(/`/g, ''); } +/** Options for {@link generateGitNexusContent} (collapsed from positional + * params, #2188 review — six `undefined`s to reach `hasPdg` was the smell). */ +export interface GitNexusContentOptions { + generatedSkills?: GeneratedSkillInfo[]; + groupNames?: string[]; + noStats?: boolean; + skipSkills?: boolean; + /** Project-relative path to the runner `gitnexus analyze` drops next to the + * index (#1945). Referenced by docs so a single CLI-neutral command resolves + * the available runner (global `gitnexus` → `pnpm dlx` → `npx`) at call time. */ + runnerPath?: string; + /** Default branch for the regression-compare example (#243). Configurable so + * projects on `develop`/`master`/etc. don't get `base_ref: "main"` rewritten + * back over their fix on every analyze. The value is embedded inside a + * Markdown inline-code span: validateBranchName rejects backticks upstream, + * and `markdownSafeBranch` strips any remaining backtick here as defense in + * depth, so JSON.stringify's quote/escape handling is sufficient and the + * branch cannot break out of the span (#1996 tri-review P1). */ + defaultBranch?: string; + /** Whether the index was built with `--pdg` (#2086 M6). Gates the pdg_query + * line below — false (default) omits it, so a non-pdg index doesn't advertise + * a tool that only returns a "no PDG layer" note. */ + hasPdg?: boolean; +} + export function generateGitNexusContent( projectName: string, stats: RepoStats, - generatedSkills?: GeneratedSkillInfo[], - groupNames?: string[], - noStats?: boolean, - skipSkills?: boolean, - // Project-relative path to the runner `gitnexus analyze` drops next to the - // index (#1945). Referenced by docs so a single CLI-neutral command resolves - // the available runner (global `gitnexus` → `pnpm dlx` → `npx`) at call time. - runnerPath: string = '.gitnexus/run.cjs', - // Default branch for the regression-compare example (#243). Configurable so - // projects on `develop`/`master`/etc. don't get `base_ref: "main"` rewritten - // back over their fix on every analyze. The value is embedded inside a - // Markdown inline-code span: validateBranchName rejects backticks upstream, - // and `markdownSafeBranch` strips any remaining backtick here as defense in - // depth, so JSON.stringify's quote/escape handling is sufficient and the - // branch cannot break out of the span (#1996 tri-review P1). - defaultBranch: string = 'main', + opts: GitNexusContentOptions = {}, ): string { + const { + generatedSkills, + groupNames, + noStats, + skipSkills, + runnerPath = '.gitnexus/run.cjs', + defaultBranch = 'main', + hasPdg = false, + } = opts; const generatedRows = generatedSkills && generatedSkills.length > 0 ? generatedSkills @@ -179,7 +204,11 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s - **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits. - When exploring unfamiliar code, use \`query({search_query: "concept"})\` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance. - When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use \`context({name: "symbolName"})\`. -- For security review, \`explain({target: "fileOrSymbol"})\` lists taint findings (source→sink flows; needs \`analyze --pdg\`). +- For security review, \`explain({target: "fileOrSymbol"})\` lists taint findings (source→sink flows; needs \`analyze --pdg\`).${ + hasPdg + ? `\n- For control/data dependence, \`pdg_query({mode: "controls", target: "fileOrSymbol"})\` answers "under what condition does X run?" (CDG, incl. guard clauses) and \`pdg_query({mode: "flows", target, variable})\` traces "where does variable Y flow?" (REACHING_DEF). \`--pdg\` layer.` + : '' + } ## Never Do @@ -447,16 +476,15 @@ export async function generateAIContextFiles( logger.warn(`Could not write GitNexus runner to ${runnerPath}: ${String(err)}`); } - const content = generateGitNexusContent( - projectName, - stats, + const content = generateGitNexusContent(projectName, stats, { generatedSkills, groupNames, - options?.noStats, - options?.skipSkills, + noStats: options?.noStats, + skipSkills: options?.skipSkills, runnerPath, - options?.defaultBranch ?? 'main', - ); + defaultBranch: options?.defaultBranch ?? 'main', + hasPdg: options?.hasPdg ?? false, + }); const createdFiles: string[] = []; if (!options?.skipAgentsMd) { diff --git a/gitnexus/src/cli/analyze.ts b/gitnexus/src/cli/analyze.ts index da768c689..0cf17ab13 100644 --- a/gitnexus/src/cli/analyze.ts +++ b/gitnexus/src/cli/analyze.ts @@ -1403,6 +1403,7 @@ const analyzeCommandImpl = async ( // Mirror runFullAnalysis `noStats` bridge (#1477) — same expression; // exercised on the `--skills` path by analyze-no-stats-bridge.test.ts. noStats: options.stats === false, + hasPdg: options.pdg === true, }, ); } diff --git a/gitnexus/src/core/ingestion/cfg/control-dependence.ts b/gitnexus/src/core/ingestion/cfg/control-dependence.ts new file mode 100644 index 000000000..033075327 --- /dev/null +++ b/gitnexus/src/core/ingestion/cfg/control-dependence.ts @@ -0,0 +1,172 @@ +/** + * Control dependence (#2085 M5 U3) — Ferrante, Ottenstein & Warren §3.1.1 over + * the post-dominator tree. A block `dependent` is control-dependent on a branch + * block `controller` when `controller` decides whether `dependent` executes: + * formally, there is a CFG edge `controller → B` such that `dependent` + * post-dominates `B` but does NOT strictly post-dominate `controller`. + * + * Construction (§3.1.1): for each CFG edge `(A, B)` where `B` does NOT + * post-dominate `A`, walk UP the post-dom tree from `B` to (but not including) + * `ipdom(A)`; every block on that path is control-dependent on `A`. The branch + * SENSE of the edge ('T' | 'F') becomes the edge label (KTD4 / KTD3 — it rides + * the persisted relation's `reason` column). + * + * PURE AND DETERMINISTIC (mirrors post-dominators.ts / reaching-defs.ts): no + * graph, no logger, importable outside the worker; output is deduped per + * (controller, dependent, label) and sorted, so snapshot tests and + * content-derived edge ids are stable. The loop header legitimately appears as + * control-dependent on ITSELF (`controller === dependent`) — the loop predicate + * gates its own re-execution; this is standard PDG behavior, not a bug. + */ +import { + computePostDominators, + postDominates, + NO_IPDOM, + type PostDomTree, +} from './post-dominators.js'; +import type { CfgEdgeKind, FunctionCfg } from './types.js'; + +export type CdgLabel = 'T' | 'F'; + +export interface ControlDepEdge { + /** The branch block whose outcome controls `dependentBlock`. */ + readonly controllerBlock: number; + /** The block that executes only because `controllerBlock` took `label`. */ + readonly dependentBlock: number; + /** Branch sense of the controlling CFG edge — see {@link branchSense}. */ + readonly label: CdgLabel; +} + +export interface ControlDepResult { + /** Deduped, sorted (controller, dependent, label) control-dependence edges. */ + readonly edges: readonly ControlDepEdge[]; + /** + * True when the `maxEdges` ceiling was reached; `edges` is then a + * deterministic prefix (CFG-edge iteration order, sorted), never a silent + * drop. Mirrors {@link computeReachingDefs}'s `truncated`. + */ + readonly truncated: boolean; +} + +/** + * Per-controller branch-arm senses, derived from the controller block's OUTGOING + * edge kinds. The CFG edge kind alone cannot name a branch sense: the M1 visitor + * emits an explicit `cond-true`/`cond-false` only for a `then`/`else` arm, but a + * condition's FALL-THROUGH false arm (no-`else`, or a guard's `if (!ok) return;`) + * is wired as `seq`, and an `if` ending a loop body falls through as `loop-back` + * — while a `do/while` bottom-test's TRUE arm is also a `loop-back`. So `seq` + * and `loop-back` are genuinely ambiguous in isolation (issue #2188 F1). + * + * The fix reads the sense from the CONTROLLER's structure: a 2-way branch emits + * exactly one explicitly-sensed arm (`cond-true`/`switch-case` ⇒ true, or + * `cond-false` ⇒ false), and its other (ambiguous) arm is the COMPLEMENT. This + * map records which explicit senses each block emits so {@link labelFor} can + * resolve an ambiguous edge against its sibling. + */ +interface ArmSenses { + hasTrueArm: boolean; // emits a cond-true or switch-case edge + hasFalseArm: boolean; // emits a cond-false edge +} + +function buildArmSenses(cfg: FunctionCfg): ArmSenses[] { + const n = cfg.blocks.length; + const senses: ArmSenses[] = Array.from({ length: n }, () => ({ + hasTrueArm: false, + hasFalseArm: false, + })); + for (const e of cfg.edges) { + if (e.from < 0 || e.from >= n) continue; + if (e.kind === 'cond-true' || e.kind === 'switch-case') senses[e.from].hasTrueArm = true; + else if (e.kind === 'cond-false') senses[e.from].hasFalseArm = true; + } + return senses; +} + +/** + * The CDG label ('T'|'F') for a control-dependence edge, given the controlling + * block's arm senses. An explicitly-sensed edge is taken at face value; an + * ambiguous fall-through edge (`seq`/`loop-back`/`fallthrough`/jump) is the + * COMPLEMENT of the controller's explicit sibling arm. Per-case `switch` value + * labels are deferred to #2086 — every `switch-case` is 'T' in M5. + */ +function labelFor(kind: CfgEdgeKind, controller: ArmSenses): CdgLabel { + if (kind === 'cond-true' || kind === 'switch-case') return 'T'; + if (kind === 'cond-false') return 'F'; + // Ambiguous structural kind: take the complement of the controller's explicit + // arm. A block with a true arm reaches here via its false fall-through; a + // do/while bottom-test (false arm = cond-false) reaches here via its true + // loop-back. With neither explicit arm (a degenerate / exit-unreachable + // region — see #2188 F2, where the dependence itself is unsound) the sense is + // indeterminate; default 'F' since fall-through is the common case. + if (controller.hasTrueArm) return 'F'; + if (controller.hasFalseArm) return 'T'; + return 'F'; +} + +/** + * Compute control-dependence edges for one function's CFG. `postDom` may be + * supplied to reuse an already-built tree; otherwise it is computed. See the + * module doc for the purity/determinism contract. + */ +export function computeControlDependence( + cfg: FunctionCfg, + postDom?: PostDomTree, + // Heap-safety ceiling on materialized edges, mirroring computeReachingDefs' + // `maxFacts` (#2188 review): the pre-dedup walk is O(edges × post-dom depth), + // so bound it before it can spike. `0` ⇒ unbounded. On overflow `edges` is a + // deterministic prefix and `truncated` is set — never a silent drop. + maxEdges: number = 0, +): ControlDepResult { + const tree = postDom ?? computePostDominators(cfg); + const { ipdom } = tree; + const n = cfg.blocks.length; + const armSenses = buildArmSenses(cfg); + const cap = maxEdges > 0 ? maxEdges : Infinity; + + const out: ControlDepEdge[] = []; + const seen = new Set(); + let truncated = false; + + scan: for (const e of cfg.edges) { + const a = e.from; + const b = e.to; + if (a < 0 || a >= n || b < 0 || b >= n) continue; + // No control dependence when B post-dominates A — every path leaving A + // through this edge still reaches B, so A does not decide B's execution. + // This guard is exactly AC2: a dependence exists IFF post-dominance fails. + if (postDominates(tree, b, a)) continue; + + // Sense is read from the CONTROLLER's arms, not this edge's kind alone — + // seq/loop-back fall-through false arms would otherwise mislabel as 'T' + // (#2188 F1). + const label = labelFor(e.kind, armSenses[a]); + const stop = ipdom[a]; // walk up to ipdom(A), EXCLUSIVE (NO_IPDOM ⇒ to root) + let cur = b; + let steps = 0; + // `steps <= n` is defensive — the ipdom chain is a finite tree. + while (cur !== NO_IPDOM && cur !== stop && steps <= n) { + const key = `${a}:${cur}:${label}`; + if (!seen.has(key)) { + // Check BEFORE pushing so `truncated` means a genuine overflow (a new + // unique edge had to be dropped), not merely "reached the ceiling" — + // exactly `cap` edges is a full, non-truncated result. + if (out.length >= cap) { + truncated = true; + break scan; + } + seen.add(key); + out.push({ controllerBlock: a, dependentBlock: cur, label }); + } + cur = ipdom[cur]; + steps += 1; + } + } + + out.sort( + (x, y) => + x.controllerBlock - y.controllerBlock || + x.dependentBlock - y.dependentBlock || + (x.label < y.label ? -1 : x.label > y.label ? 1 : 0), + ); + return { edges: out, truncated }; +} diff --git a/gitnexus/src/core/ingestion/cfg/emit.ts b/gitnexus/src/core/ingestion/cfg/emit.ts index dc682d5b0..246baa1d4 100644 --- a/gitnexus/src/core/ingestion/cfg/emit.ts +++ b/gitnexus/src/core/ingestion/cfg/emit.ts @@ -21,6 +21,12 @@ import type { KnowledgeGraph } from '../../graph/types.js'; import { generateId } from '../../../lib/utils.js'; import { computeReachingDefs } from './reaching-defs.js'; +import { computeControlDependence } from './control-dependence.js'; +import { + computePostDominators, + isExitReachableFromAllBlocks, + NO_IPDOM, +} from './post-dominators.js'; import type { BindingEntry, FunctionCfg } from './types.js'; /** @@ -41,6 +47,38 @@ export const DEFAULT_MAX_CFG_EDGES_PER_FUNCTION = 5000; */ export const DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION = 4000; +/** + * Default per-function CDG edge cap (#2085 M5). CDG edge count is bounded by + * (blocks × control-nesting-depth) — comparable to the CFG edge count — so it + * reuses the CFG default of 5000. Counts DEDUPED (controller, dependent, label) + * edges (the pure {@link computeControlDependence} already dedups). `0` ⇒ + * unlimited; `undefined` ⇒ this default. Folded into the `RepoMeta.pdg` stamp + * (U5) so introducing CDG forces a full writeback for pre-CDG `--pdg` indexes. + */ +export const DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION = 5000; + +/** + * Heap-safety ceiling on {@link computeControlDependence}'s pre-dedup + * materialization (#2188 review). The walk is O(edges × post-dom depth), and its + * `out` IS the deduped-edge quantity the per-function cap trims — so, UNLIKE + * REACHING_DEF's facts ceiling, this is deliberately NOT derived from the + * runtime edge cap (doing so would pre-truncate the very set the cap reports on, + * losing the exact dropped count). A fixed, generous multiple of the default + * edge cap: far above any real function — a catastrophe backstop only. When hit, + * the per-function cap reporting plus the `truncated` flag keep it observable + * (never a silent drop). + */ +export const DEFAULT_PDG_MAX_CDG_MATERIALIZATION_PER_FUNCTION = + 8 * DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION; + +/** + * Env flag that additionally emits diagnostic `POST_DOMINATE` edges + * (block → its immediate post-dominator) alongside CDG (#2085 M5 KTD8). Off in + * every normal `--pdg` run — these are for inspecting the post-dom tree, not a + * queryable product surface. Accepts `1`/`true` (case-insensitive). + */ +export const POST_DOMINATE_DEBUG_ENV = 'GITNEXUS_PDG_EMIT_POST_DOMINATE'; + /** * Fact-materialization headroom over the edge cap (#2082 M2 U3/F3): facts are * O(defs×uses) BY SPEC in merge-heavy code, and the edge cap alone bounds the @@ -417,3 +455,153 @@ export function emitFileReachingDefs( return result; } + +export interface CdgEmitResult { + /** Deduped (controller, dependent, label) CDG edges persisted. */ + edges: number; + /** CDG edges dropped by the per-function edge cap. */ + droppedEdges: number; + /** Functions that hit the CDG edge cap. */ + cappedFunctions: number; + /** Diagnostic POST_DOMINATE edges emitted (0 unless the debug env is set). */ + postDominateEdges: number; + /** + * Functions skipped because EXIT was not reachable from every entry-reachable + * block — post-dominance would be unsound (#2188 review). CFG/REACHING_DEF for + * those functions are kept; only their CDG projection is omitted. + */ + skippedUnsoundFunctions: number; +} + +/** Whether the POST_DOMINATE debug env flag is enabled (`1`/`true`). */ +const postDominateDebugEnabled = (): boolean => { + const v = process.env[POST_DOMINATE_DEBUG_ENV]; + return v === '1' || v?.toLowerCase() === 'true'; +}; + +/** + * Compute control dependence per function and persist the bounded CDG + * projection (#2085 M5 U4). Mirrors {@link emitFileReachingDefs}: the pure + * {@link computeControlDependence} already dedups to (controller, dependent, + * label), so the per-function cap applies to deduped edges and overflow logs + * one unconditional `onWarn` naming the dropped count — no silent truncation + * (R6/R7). The branch label ('T'|'F') rides the `reason` column (KTD3), + * mirroring how CFG stores its edge kind. + * + * When {@link POST_DOMINATE_DEBUG_ENV} is set, also emits diagnostic + * `POST_DOMINATE` edges (block → its immediate post-dominator). These are NOT + * capped or counted against the CDG budget — they exist only for inspecting the + * post-dom tree and never appear in a normal run. + */ +export function emitFileCdg( + graph: KnowledgeGraph, + cfgs: readonly FunctionCfg[], + maxEdgesPerFunction: number = DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION, + onWarn?: (message: string) => void, +): CdgEmitResult { + const result: CdgEmitResult = { + edges: 0, + droppedEdges: 0, + cappedFunctions: 0, + postDominateEdges: 0, + skippedUnsoundFunctions: 0, + }; + const cap = maxEdgesPerFunction > 0 ? maxEdgesPerFunction : Infinity; + const emitPostDom = postDominateDebugEnabled(); + + for (const cfg of cfgs) { + const { filePath, functionStartLine, functionStartColumn } = cfg; + // Sound post-dominance requires EXIT reachable from every entry-reachable + // block (#2188 review). A CFG that violates it — a future visitor's + // multi-terminal / non-terminating shape — would yield a CDG that both + // drops real and invents spurious dependences, so skip CDG for it. CFG and + // REACHING_DEF (emitted elsewhere, independent of post-dominance) are kept. + if (!isExitReachableFromAllBlocks(cfg)) { + result.skippedUnsoundFunctions++; + onWarn?.( + `[cdg] ${filePath}:${functionStartLine}: EXIT not reachable from all ` + + `blocks — CDG skipped for this function (CFG/REACHING_DEF unaffected)`, + ); + continue; + } + // Compute the post-dom tree once and feed it to the control-dependence + // pass (avoids recomputing it) and to the optional POST_DOMINATE emit. + const tree = computePostDominators(cfg); + // Bound the pre-dedup materialization (heap parity with REACHING_DEF). The + // fixed ceiling is a catastrophe backstop; the per-function edge cap below + // remains the reporting authority. A ceiling hit is surfaced, not silent. + const { edges: cdgEdges, truncated } = computeControlDependence( + cfg, + tree, + DEFAULT_PDG_MAX_CDG_MATERIALIZATION_PER_FUNCTION, + ); + if (truncated) { + onWarn?.( + `[cdg] ${filePath}:${functionStartLine}: control-dependence materialization ` + + `ceiling (${DEFAULT_PDG_MAX_CDG_MATERIALIZATION_PER_FUNCTION}) reached — ` + + `edge counts for this function are a floor`, + ); + } + + let emittedForFn = 0; + for (const edge of cdgEdges) { + if (emittedForFn >= cap) { + const dropped = cdgEdges.length - emittedForFn; + result.droppedEdges += dropped; + result.cappedFunctions++; + onWarn?.( + `[cdg] ${filePath}:${functionStartLine}: per-function CDG edge cap ` + + `(${maxEdgesPerFunction}) reached — dropped ${dropped} of ${cdgEdges.length} edges`, + ); + break; + } + const sourceId = basicBlockId( + filePath, + functionStartLine, + functionStartColumn, + edge.controllerBlock, + ); + const targetId = basicBlockId( + filePath, + functionStartLine, + functionStartColumn, + edge.dependentBlock, + ); + graph.addRelationship({ + id: generateId( + 'CDG', + `${filePath}:${functionStartLine}:${functionStartColumn}:` + + `${edge.controllerBlock}->${edge.dependentBlock}:${edge.label}`, + ), + type: 'CDG', + sourceId, + targetId, + confidence: 1.0, + reason: edge.label, // 'T' | 'F' — queryable, mirrors CFG's kind-in-reason + }); + result.edges++; + emittedForFn++; + } + + if (emitPostDom) { + for (let b = 0; b < tree.ipdom.length; b++) { + const ip = tree.ipdom[b]; + if (ip === NO_IPDOM) continue; + graph.addRelationship({ + id: generateId( + 'POST_DOMINATE', + `${filePath}:${functionStartLine}:${functionStartColumn}:${b}->${ip}`, + ), + type: 'POST_DOMINATE', + sourceId: basicBlockId(filePath, functionStartLine, functionStartColumn, b), + targetId: basicBlockId(filePath, functionStartLine, functionStartColumn, ip), + confidence: 1.0, + reason: '', + }); + result.postDominateEdges++; + } + } + } + + return result; +} diff --git a/gitnexus/src/core/ingestion/cfg/post-dominators.ts b/gitnexus/src/core/ingestion/cfg/post-dominators.ts new file mode 100644 index 000000000..336a9e303 --- /dev/null +++ b/gitnexus/src/core/ingestion/cfg/post-dominators.ts @@ -0,0 +1,218 @@ +/** + * Post-dominators (#2085 M5 U2) — the immediate-post-dominator tree of one + * function's CFG, the substrate the Ferrante control-dependence pass walks. + * + * A block `p` post-dominates a block `b` iff every path from `b` to the + * function EXIT passes through `p`. Post-dominators are exactly the DOMINATORS + * of the REVERSE CFG rooted at EXIT, so this is the Cooper–Harvey–Kennedy + * "A Simple, Fast Dominance Algorithm" run over reversed edges. KTD2 of the M5 + * plan picks CHK over Lengauer–Tarjan: per-function CFGs are small and + * line-capped, CHK is near-linear in practice, and its iterative shape matches + * the reaching-defs fixpoint already in this module. + * + * PURE AND DETERMINISTIC (load-bearing, mirrors reaching-defs.ts): no graph, no + * logger, importable outside the worker; predecessors/successors are sorted and + * iteration is reverse-postorder so the `ipdom` array is identical across runs + * (snapshot tests and content-derived edge ids depend on it). + * + * The single-EXIT invariant the M1 TS visitor preserves (visitors/typescript.ts) + * makes EXIT the unique reverse-CFG root. Blocks that cannot reach EXIT in the + * forward CFG (an exit-less infinite loop) are not reverse-reachable from it and + * have NO post-dominator: their `ipdom` is {@link NO_IPDOM}. The control- + * dependence pass treats "no post-dominator" as "does not post-dominate" (KTD5). + * + * NOTE (issue #2188 F2): this is NOT a fully sound over-approximation. Inside a + * region where NO block reaches EXIT, every `ipdom` is `NO_IPDOM`, so the + * Ferrante walk degenerates to one edge per control point — it can both DROP a + * real control dependence and INVENT a spurious one. This does not arise for the + * current TS visitor (every loop is given a structural `header → loopExit` + * `cond-false` edge, so EXIT stays reverse-reachable), but it is unsound for + * hand-built CFGs and any future language visitor lacking that exit edge. + * Nontermination-sensitive post-dominance (a virtual root over the + * non-terminating SCCs) would be the correct treatment — tracked for follow-up. + */ +import type { FunctionCfg } from './types.js'; + +/** + * Sentinel `ipdom` value: the block has no immediate post-dominator. True for + * the EXIT block itself (the reverse-CFG root) and for any block that cannot + * reach EXIT. Chosen as -1 so the {@link postDominates} climb terminates + * naturally instead of self-looping on the root. + */ +export const NO_IPDOM = -1; + +export interface PostDomTree { + /** + * `ipdom[b]` = the index of `b`'s immediate post-dominator, or + * {@link NO_IPDOM} when `b` has none (EXIT, or a block that cannot reach EXIT). + */ + readonly ipdom: readonly number[]; +} + +/** + * Compute the immediate-post-dominator tree for one function's CFG. See the + * module doc for the purity/determinism contract and EXIT-root assumptions. + */ +export function computePostDominators(cfg: FunctionCfg): PostDomTree { + const n = cfg.blocks.length; + const exit = cfg.exitIndex; + if (n === 0 || exit < 0 || exit >= n) { + return { ipdom: new Array(n).fill(NO_IPDOM) }; + } + + // Forward adjacency (sorted for deterministic intersect order). The reverse + // CFG, on which we compute dominators, flips these: a node's reverse-CFG + // successors are its CFG predecessors, and its reverse-CFG predecessors + // (the "preds" CHK intersects over) are its CFG successors. + const cfgPreds: number[][] = Array.from({ length: n }, () => []); + const cfgSuccs: number[][] = Array.from({ length: n }, () => []); + for (const e of cfg.edges) { + if (e.from < 0 || e.from >= n || e.to < 0 || e.to >= n) continue; + cfgSuccs[e.from].push(e.to); + cfgPreds[e.to].push(e.from); + } + for (const l of cfgPreds) l.sort((a, b) => a - b); + for (const l of cfgSuccs) l.sort((a, b) => a - b); + + // Postorder of the reverse CFG from EXIT (traversing CFG-predecessor edges). + // Iterative DFS with an explicit phase stack; children pushed in sorted order + // for determinism. postNum is the CHK comparison key: higher = closer to root. + const postNum = new Array(n).fill(-1); + const postorder: number[] = []; + const visited = new Array(n).fill(false); + const stack: { node: number; childIdx: number }[] = [{ node: exit, childIdx: 0 }]; + visited[exit] = true; + while (stack.length) { + const top = stack[stack.length - 1]; + const revSuccs = cfgPreds[top.node]; // reverse-CFG successors + if (top.childIdx < revSuccs.length) { + const next = revSuccs[top.childIdx]; + top.childIdx += 1; + if (!visited[next]) { + visited[next] = true; + stack.push({ node: next, childIdx: 0 }); + } + } else { + postNum[top.node] = postorder.length; + postorder.push(top.node); + stack.pop(); + } + } + const rpo = [...postorder].reverse(); + + // CHK fixpoint. ipdom[exit] = exit DURING computation (the root dominates + // itself, so the intersect climb has a common terminus); it is reset to + // NO_IPDOM before returning so callers' climbs terminate at the root. + const ipdom = new Array(n).fill(NO_IPDOM); + ipdom[exit] = exit; + + const intersect = (a: number, b: number): number => { + let f1 = a; + let f2 = b; + while (f1 !== f2) { + while (postNum[f1] < postNum[f2]) f1 = ipdom[f1]; + while (postNum[f2] < postNum[f1]) f2 = ipdom[f2]; + } + return f1; + }; + + let changed = true; + while (changed) { + changed = false; + for (const b of rpo) { + if (b === exit) continue; + // CHK "predecessors in the reverse CFG" = this block's CFG successors. + // Fold only those already processed (ipdom assigned); RPO guarantees at + // least one for every block reverse-reachable from EXIT. + let newIpdom = NO_IPDOM; + for (const s of cfgSuccs[b]) { + if (ipdom[s] !== NO_IPDOM) { + newIpdom = newIpdom === NO_IPDOM ? s : intersect(s, newIpdom); + } + } + if (newIpdom !== NO_IPDOM && ipdom[b] !== newIpdom) { + ipdom[b] = newIpdom; + changed = true; + } + } + } + + ipdom[exit] = NO_IPDOM; // root: no post-dominator above it + return { ipdom }; +} + +/** + * Does block `p` post-dominate block `b`? Climbs the post-dom tree from `b` + * toward EXIT and tests membership of `p`. Reflexive: a block post-dominates + * itself. A block with no post-dominator (EXIT, or one that cannot reach EXIT) + * is post-dominated only by itself. The step guard is purely defensive — the + * `ipdom` chain is a tree and always terminates at {@link NO_IPDOM}. + */ +export function postDominates(tree: PostDomTree, p: number, b: number): boolean { + const { ipdom } = tree; + const n = ipdom.length; + if (p < 0 || b < 0 || p >= n || b >= n) return false; + let cur = b; + let steps = 0; + while (cur !== NO_IPDOM && steps <= n) { + if (cur === p) return true; + cur = ipdom[cur]; + steps += 1; + } + return false; +} + +/** + * Precondition for SOUND post-dominance (#2188 review): EXIT must be reachable + * (forward) from every block that is itself reachable from ENTRY. When it + * fails — an entry-reachable region that cannot reach EXIT, e.g. a + * non-terminating loop or a multi-terminal CFG a future language visitor might + * emit — the EXIT-rooted reverse walk degenerates (every such block gets + * {@link NO_IPDOM}), which both DROPS real control dependences and INVENTS + * spurious ones (the unsoundness documented in the module header). Consumers + * ({@link emitFileCdg}) check this and skip CDG for the function rather than + * persist an unsound projection — CFG and REACHING_DEF, which do not depend on + * post-dominance, are unaffected. + * + * The current TS visitor always satisfies this (every loop is given a + * structural `header → loopExit` edge, keeping EXIT reverse-reachable), so this + * is a guard for future visitors and hand-built CFGs, not a behavior change + * today. Pure and O(V+E). + */ +export function isExitReachableFromAllBlocks(cfg: FunctionCfg): boolean { + const n = cfg.blocks.length; + if (n === 0) return true; + const { entryIndex, exitIndex } = cfg; + if (entryIndex < 0 || entryIndex >= n || exitIndex < 0 || exitIndex >= n) return false; + + const succ: number[][] = Array.from({ length: n }, () => []); + const pred: number[][] = Array.from({ length: n }, () => []); + for (const e of cfg.edges) { + if (e.from < 0 || e.from >= n || e.to < 0 || e.to >= n) continue; + succ[e.from].push(e.to); + pred[e.to].push(e.from); + } + + const reach = (start: number, adj: readonly number[][]): Uint8Array => { + const seen = new Uint8Array(n); + const stack = [start]; + seen[start] = 1; + while (stack.length > 0) { + const b = stack.pop() as number; + for (const next of adj[b]) { + if (!seen[next]) { + seen[next] = 1; + stack.push(next); + } + } + } + return seen; + }; + + const fromEntry = reach(entryIndex, succ); // forward-reachable from ENTRY + const canReachExit = reach(exitIndex, pred); // can reach EXIT (reverse from EXIT) + for (let i = 0; i < n; i++) { + if (fromEntry[i] && !canReachExit[i]) return false; + } + return true; +} diff --git a/gitnexus/src/core/ingestion/pipeline.ts b/gitnexus/src/core/ingestion/pipeline.ts index 269724e0d..334c022ac 100644 --- a/gitnexus/src/core/ingestion/pipeline.ts +++ b/gitnexus/src/core/ingestion/pipeline.ts @@ -83,6 +83,15 @@ export interface PipelineOptions { * programmatic / server path only, like the M1 caps. */ pdgMaxReachingDefEdgesPerFunction?: number; + /** + * Per-function CDG (control-dependence) edge cap for the scope-resolution + * emit step (#2085 M5). `undefined` ⇒ `DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION` + * (5000); `0` ⇒ no cap (unlimited). Emit-time-only — NOT folded into the + * parse-cache chunk key; recorded resolved in `RepoMeta.pdg` so introducing + * CDG (an absent stamp key) forces a full writeback for pre-CDG `--pdg` + * indexes. No CLI flag — programmatic / server path only. + */ + pdgMaxCdgEdgesPerFunction?: number; /** * Per-function taint findings cap for the scope-resolution taint pass * (#2083 M3). `undefined` ⇒ `DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION` diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts index 6b748dd97..6bf86dc84 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/phase.ts @@ -375,6 +375,7 @@ export const scopeResolutionPhase: PipelinePhase = { pdg: ctx.options?.pdg === true, pdgMaxEdgesPerFunction: ctx.options?.pdgMaxEdgesPerFunction, pdgMaxReachingDefEdgesPerFunction: ctx.options?.pdgMaxReachingDefEdgesPerFunction, + pdgMaxCdgEdgesPerFunction: ctx.options?.pdgMaxCdgEdgesPerFunction, pdgMaxTaintFindingsPerFunction: ctx.options?.pdgMaxTaintFindingsPerFunction, pdgMaxTaintHops: ctx.options?.pdgMaxTaintHops, recordResolutionOutcome: (outcome) => { diff --git a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts index faa29b17c..733f76be3 100644 --- a/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts +++ b/gitnexus/src/core/ingestion/scope-resolution/pipeline/run.ts @@ -37,10 +37,12 @@ import { buildGraphNodeLookup } from '../graph-bridge/node-lookup.js'; import { emitFileCfgs, emitFileReachingDefs, + emitFileCdg, isEmitSafeCfg, DEFAULT_MAX_CFG_EDGES_PER_FUNCTION, DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, DEFAULT_PDG_MAX_REACHING_DEF_FACTS_PER_FUNCTION, + DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION, REACHING_DEF_FACTS_PER_EDGE_CAP, } from '../../cfg/emit.js'; import { @@ -289,6 +291,9 @@ interface RunScopeResolutionInput { /** Per-function REACHING_DEF edge cap (#2082 M2). `undefined` ⇒ * {@link DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION}; `0` ⇒ no cap. */ readonly pdgMaxReachingDefEdgesPerFunction?: number; + /** Per-function CDG (control-dependence) edge cap (#2085 M5). `undefined` ⇒ + * {@link DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION}; `0` ⇒ no cap. */ + readonly pdgMaxCdgEdgesPerFunction?: number; /** Per-function taint findings cap (#2083 M3, consumed by the U4 taint * emit step in the pdg window). `undefined` ⇒ * `DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION` (200); `0` ⇒ no cap. */ @@ -771,6 +776,8 @@ export function runScopeResolution( let rdDropped = 0; let rdFacts = 0; let rdTruncated = 0; + let cdgEdges = 0; + let cdgDropped = 0; // ── M3 taint setup (#2083 U4) ──────────────────────────────────────── // Explicit model-registration seam (idempotent, cheap) — the registry // stays empty on non-pdg runs, preserving default-run parity. The @@ -877,6 +884,22 @@ export function runScopeResolution( rdFacts += rd.facts; rdTruncated += rd.truncatedFunctions; + // M5 (#2085 U5): control dependence over the SAME validated CFGs. + // Independent of taint — runs for every `--pdg` language (post-dom + + // Ferrante are language-agnostic, no source/sink model needed). Pure + // compute; the bounded (controller, dependent, label) projection is + // persisted and its time folds into the `pdg=` PROF segment next to RD. + const tCdg = PROF ? performance.now() : 0; + const cdg = emitFileCdg( + graph, + wellFormed, + input.pdgMaxCdgEdgesPerFunction ?? DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION, + (message) => logger.warn(message), // unconditional — R6, no silent truncation + ); + if (PROF) pdgMs += performance.now() - tCdg; + cdgEdges += cdg.edges; + cdgDropped += cdg.droppedEdges; + // M3 (#2083 U4): taint over the SAME validated CFGs, inside the SAME // per-file try (a taint throw costs this file's taint layer only — // its CFG/REACHING_DEF edges above are already in the graph). Skipped @@ -950,6 +973,8 @@ export function runScopeResolution( `; ${rdEdges} REACHING_DEF edges (${rdFacts} facts)` + (rdDropped > 0 ? `, ${rdDropped} REACHING_DEF edges dropped (per-function cap)` : '') + (rdTruncated > 0 ? `, ${rdTruncated} function(s) hit the fact limit` : '') + + `; ${cdgEdges} CDG edges` + + (cdgDropped > 0 ? `, ${cdgDropped} CDG edges dropped (per-function cap)` : '') + // M3 volume telemetry — only for languages with a registered model. (taintSpec !== undefined ? `; taint: ${taintTotals.findings} TAINTED, ${taintTotals.kills} SANITIZES ` + diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index 8de20925b..7910dde60 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -49,6 +49,7 @@ import { DEFAULT_PDG_MAX_FUNCTION_LINES } from './ingestion/cfg/collect.js'; import { DEFAULT_MAX_CFG_EDGES_PER_FUNCTION, DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, + DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION, } from './ingestion/cfg/emit.js'; import { DEFAULT_PDG_MAX_TAINT_FINDINGS_PER_FUNCTION, @@ -152,6 +153,10 @@ export interface AnalyzeOptions { /** Per-function REACHING_DEF edge cap (#2082 M2). Forwarded to * `PipelineOptions.pdgMaxReachingDefEdgesPerFunction`. */ pdgMaxReachingDefEdgesPerFunction?: number; + /** Per-function CDG edge cap (#2085 M5). Forwarded to + * `PipelineOptions.pdgMaxCdgEdgesPerFunction`. No CLI flag or rc key — + * programmatic / server path only, like the other pdg caps. */ + pdgMaxCdgEdgesPerFunction?: number; /** Per-function taint findings cap (#2083 M3). Forwarded to * `PipelineOptions.pdgMaxTaintFindingsPerFunction`. No CLI flag or rc key * (KTD8) — programmatic / server path only, like the other pdg caps. */ @@ -371,6 +376,7 @@ type PdgOptions = Pick< | 'pdgMaxFunctionLines' | 'pdgMaxEdgesPerFunction' | 'pdgMaxReachingDefEdgesPerFunction' + | 'pdgMaxCdgEdgesPerFunction' | 'pdgMaxTaintFindingsPerFunction' | 'pdgMaxTaintHops' | 'pdgMaxInterprocFindings' @@ -386,6 +392,12 @@ export const resolvePdgConfig = (options: PdgOptions): RepoMeta['pdg'] => maxReachingDefEdgesPerFunction: options.pdgMaxReachingDefEdgesPerFunction ?? DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION, + // #2085 M5: control-dependence cap. Absent on any pre-M5 (M2/M3/M4-era) + // stamp → the key-union pdgModeMismatch trips the first CDG-aware run + // over an existing `--pdg` index and forces the full writeback that + // materialises CDG edges for every file without `--force`. + maxCdgEdgesPerFunction: + options.pdgMaxCdgEdgesPerFunction ?? DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION, // #2083 M3: taint caps + model identity. The key-union comparator in // pdgModeMismatch picks these up structurally — an M2-era stamp lacks // all three, so the first M3 run over an M2 `--pdg` index trips a full @@ -802,6 +814,7 @@ export async function runFullAnalysis( pdgMaxFunctionLines: options.pdgMaxFunctionLines, pdgMaxEdgesPerFunction: options.pdgMaxEdgesPerFunction, pdgMaxReachingDefEdgesPerFunction: options.pdgMaxReachingDefEdgesPerFunction, + pdgMaxCdgEdgesPerFunction: options.pdgMaxCdgEdgesPerFunction, pdgMaxTaintFindingsPerFunction: options.pdgMaxTaintFindingsPerFunction, pdgMaxTaintHops: options.pdgMaxTaintHops, pdgMaxInterprocFindings: options.pdgMaxInterprocFindings, @@ -1423,6 +1436,7 @@ export async function runFullAnalysis( skipSkills: options.skipSkills, noStats: options.noStats, defaultBranch: options.defaultBranch, + hasPdg: options.pdg === true, }, ); } catch { diff --git a/gitnexus/src/mcp/local/local-backend.ts b/gitnexus/src/mcp/local/local-backend.ts index d50bd67ac..6526a4bef 100644 --- a/gitnexus/src/mcp/local/local-backend.ts +++ b/gitnexus/src/mcp/local/local-backend.ts @@ -59,6 +59,8 @@ import { LIST_REPOS_MAX_LIMIT, EXPLAIN_DEFAULT_LIMIT, EXPLAIN_MAX_LIMIT, + PDG_QUERY_DEFAULT_LIMIT, + PDG_QUERY_MAX_LIMIT, } from '../tools.js'; import { findImportCycles } from '../../core/graph/import-cycles.js'; import { decodeTaintPath } from '../../core/ingestion/taint/path-codec.js'; @@ -1293,6 +1295,8 @@ export class LocalBackend { return this.context(repo, params); case 'explain': return this.explain(repo, params); + case 'pdg_query': + return this.pdgQuery(repo, params); case 'impact': return this.impact(repo, params); case 'detect_changes': @@ -2794,6 +2798,103 @@ export class LocalBackend { }; } + /** + * Resolve a `target` (file path OR symbol/function name) into a BasicBlock + * SOURCE-block anchor, shared by `explain` (TAINTED) and `pdg_query` + * (CDG/REACHING_DEF) — both reconstruct the symbol↔block join the same way + * (there is no Function→BasicBlock edge). #2188 review: extracted from two + * near-identical copies that had DRIFTED — `_explainImpl` used a 0-based, + * un-widened span window that dropped a function's final-line block and could + * leak a neighbor's line-above block; this single resolver applies the correct + * `[symStart+1, symEnd+1]` window (1-based BasicBlock startLine vs 0-based + * symbol span) to BOTH callers. + * + * Returns a BARE `anchorClause` (no leading `AND`) so each caller composes its + * own `WHERE`; `early` carries the not-found/ambiguous payload (caller returns + * it verbatim). `target` / symbol names flow only through `queryParams` bind + * params — never interpolated into Cypher. + */ + private async resolveBlockAnchor( + repo: RepoHandle, + target: string, + toolName: 'explain' | 'pdg_query', + ): Promise<{ + anchorClause: string; + queryParams: Record; + anchor: { file: string; symbol?: string; startLine?: number; endLine?: number }; + early?: Record; + }> { + if (looksLikeFilePath(target)) { + return { + anchorClause: + '(a.id STARTS WITH $idPrefix OR a.filePath = $targetPath OR a.filePath ENDS WITH $targetSuffix)', + queryParams: { + idPrefix: `BasicBlock:${target}:`, + targetPath: target, + targetSuffix: `/${target}`, + }, + anchor: { file: target }, + }; + } + const outcome = await this.resolveSymbolCandidates(repo, { name: target }, {}); + if (outcome.kind === 'not_found') { + return { + anchorClause: '', + queryParams: {}, + anchor: { file: '' }, + early: { error: `Symbol '${target}' not found` }, + }; + } + if (outcome.kind === 'ambiguous') { + return { + anchorClause: '', + queryParams: {}, + anchor: { file: '' }, + early: { + status: 'ambiguous', + message: `Found ${outcome.candidates.length} symbols matching '${target}'. Re-call ${toolName} with the file path, or disambiguate via context() first.`, + candidates: outcome.candidates.map((c) => ({ + uid: c.id, + name: c.name, + kind: c.type, + filePath: c.filePath, + line: c.startLine, + score: Number(c.score.toFixed(2)), + })), + }, + }; + } + const sym = outcome.symbol; + const idPrefix = `BasicBlock:${sym.filePath}:`; + if ( + typeof sym.startLine === 'number' && + typeof sym.endLine === 'number' && + sym.endLine >= sym.startLine + ) { + // BasicBlock startLine is 1-based; the symbol span is 0-based. Shift BOTH + // bounds +1 so the window is the function's true block span: the lower +1 + // excludes a neighbor's block on the line directly above, the upper +1 + // keeps a guard/def/use on the final line (#2188 review). + return { + anchorClause: + 'a.id STARTS WITH $idPrefix AND a.startLine >= $symStart AND a.startLine <= $symEnd', + queryParams: { idPrefix, symStart: sym.startLine + 1, symEnd: sym.endLine + 1 }, + anchor: { + file: sym.filePath, + symbol: sym.name, + startLine: sym.startLine, + endLine: sym.endLine, + }, + }; + } + // No usable span — degrade to the file-level filter (documented). + return { + anchorClause: 'a.id STARTS WITH $idPrefix', + queryParams: { idPrefix }, + anchor: { file: sym.filePath, symbol: sym.name }, + }; + } + /** * Explain tool (#2083 M3 U6) — persisted taint-finding explanation. * WAL-aware wrapper mirroring `context`. @@ -2876,64 +2977,9 @@ export class LocalBackend { // Resolve the optional anchor into a WHERE clause on the SOURCE block. const target = typeof params.target === 'string' ? params.target.trim() : ''; let anchorClause = ''; - const queryParams: Record = {}; + let queryParams: Record = {}; let anchor: { file: string; symbol?: string; startLine?: number; endLine?: number } | undefined; - // Build the anchor as a file filter (used only when `target` is path-ish). - const buildFileAnchor = (): void => { - // Exact path via the BasicBlock id-prefix template, OR a - // path-separator-aligned suffix so partial paths work like context()'s - // file_path hint ("vuln.ts" ⇒ "src/vuln.ts", never "devuln.ts"). - anchorClause = - 'AND (a.id STARTS WITH $idPrefix OR a.filePath = $targetPath OR a.filePath ENDS WITH $targetSuffix)'; - queryParams.idPrefix = `BasicBlock:${target}:`; - queryParams.targetPath = target; - queryParams.targetSuffix = `/${target}`; - anchor = { file: target as string }; - }; - - // Resolve `target` as a symbol into the anchor. Returns an early-return - // payload (not_found / ambiguous) or undefined on success. - const resolveSymbolAnchor = async (): Promise | undefined> => { - const outcome = await this.resolveSymbolCandidates(repo, { name: target as string }, {}); - if (outcome.kind === 'not_found') { - return { error: `Symbol '${target}' not found` }; - } - if (outcome.kind === 'ambiguous') { - return { - status: 'ambiguous', - message: `Found ${outcome.candidates.length} symbols matching '${target}'. Re-call explain with the file path, or disambiguate via context() first.`, - candidates: outcome.candidates.map((c) => ({ - uid: c.id, - name: c.name, - kind: c.type, - filePath: c.filePath, - line: c.startLine, - score: Number(c.score.toFixed(2)), - })), - }; - } - const sym = outcome.symbol; - queryParams.idPrefix = `BasicBlock:${sym.filePath}:`; - anchor = { file: sym.filePath, symbol: sym.name }; - if ( - typeof sym.startLine === 'number' && - typeof sym.endLine === 'number' && - sym.endLine >= sym.startLine - ) { - anchorClause = - 'AND a.id STARTS WITH $idPrefix AND a.startLine >= $symStart AND a.startLine <= $symEnd'; - queryParams.symStart = sym.startLine; - queryParams.symEnd = sym.endLine; - anchor.startLine = sym.startLine; - anchor.endLine = sym.endLine; - } else { - // No usable span — degrade to the file-level filter (documented). - anchorClause = 'AND a.id STARTS WITH $idPrefix'; - } - return undefined; - }; - // Bounded by construction: the BasicBlock→BasicBlock partition holds only // the sparse pdg layers, TAINTED rows are per-function-capped at analyze // time, and the page is LIMIT-bounded (the limit is a validated integer — @@ -2941,7 +2987,7 @@ export class LocalBackend { const runAnchoredQuery = async (): Promise<{ rows: unknown[]; totalFindings: number }> => { const matchClause = ` MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock) - WHERE r.type = 'TAINTED' ${anchorClause}`; + WHERE r.type = 'TAINTED'${anchorClause ? ` AND ${anchorClause}` : ''}`; const [qRows, countRows] = await Promise.all([ executeParameterized( repo.lbugPath, @@ -2966,14 +3012,14 @@ export class LocalBackend { }; if (target) { - if (looksLikeFilePath(target)) { - buildFileAnchor(); - } else { - // A bare or dotted symbol name (`UserController.create`) — resolve as a - // symbol rather than silently file-anchoring to an empty result. - const early = await resolveSymbolAnchor(); - if (early) return early; - } + // Shared symbol↔block anchor resolver (#2188): file id-prefix OR symbol + // span, with the corrected [symStart+1, symEnd+1] window. A bare/dotted + // symbol name resolves as a symbol rather than silently file-anchoring. + const resolved = await this.resolveBlockAnchor(repo, target, 'explain'); + if (resolved.early) return resolved.early; + anchorClause = resolved.anchorClause; + queryParams = resolved.queryParams; + anchor = resolved.anchor; } const { rows, totalFindings } = await runAnchoredQuery(); @@ -3147,6 +3193,203 @@ export class LocalBackend { }; } + private async pdgQuery( + repo: RepoHandle, + params: { mode?: string; target?: string; variable?: string; limit?: number }, + ): Promise { + try { + return await this._pdgQueryImpl(repo, params); + } catch (err: any) { + const msg = (err instanceof Error ? err.message : String(err)) || 'pdg_query failed'; + if (isWalCorruptionError(err)) { + return { error: msg, recoverySuggestion: WAL_RECOVERY_SUGGESTION }; + } + throw err; + } + } + + /** + * Query the persisted PDG (#2086 M6) — the control/data-dependence analog of + * `explain`. `controls` reads CDG ("under what condition does X run?", branch + * sense 'T'|'F' in `reason`); `flows` reads REACHING_DEF (def→use, variable + * name in `reason`). Intra-procedural, basic-block granular. + * + * Bounded by construction: the BasicBlock→BasicBlock partition holds only the + * sparse, per-function-capped pdg layers, the query is anchored to one file/ + * symbol, and the page is LIMIT-bounded (validated integer, interpolated + * because LadybugDB does not parameterize LIMIT). LadybugDB has no rel- + * property index, so the anchor IS the bound — there is no anchorless mode. + * + * Symbol↔block join: there is no Function→BasicBlock edge; the SOURCE block + * (`a` — controller for CDG, def for REACHING_DEF) is filtered by the + * BasicBlock id-prefix (`basicBlockId` template) plus its `startLine` within + * the symbol's span. BasicBlock `startLine` is 1-based while symbol-node + * `startLine`/`endLine` are 0-based, so BOTH bounds are shifted +1 + * (`[symStart+1, symEnd+1]`) onto the block basis: the upper +1 keeps a + * guard/def/use on the function's final line, and the lower +1 excludes an + * adjacent function's block on the line directly above (#2188 review). Both + * endpoints share the function (intra-procedural), so filtering the source + * endpoint suffices. + */ + private async _pdgQueryImpl( + repo: RepoHandle, + params: { mode?: string; target?: string; variable?: string; limit?: number } = {}, + ): Promise { + await this.ensureInitialized(repo); + + // Mode validation — the JSON-schema enum is advisory for MCP clients, so + // the backend enforces it (an unhandled mode would otherwise fall through). + const mode = params.mode; + if (mode !== 'controls' && mode !== 'flows') { + return { + error: `Invalid "mode": expected "controls" or "flows", got ${JSON.stringify(params.mode)}.`, + }; + } + + const rawLimit = params.limit ?? PDG_QUERY_DEFAULT_LIMIT; + if (!Number.isInteger(rawLimit) || rawLimit < 1 || rawLimit > PDG_QUERY_MAX_LIMIT) { + return { + error: `Invalid "limit": expected an integer in [1, ${PDG_QUERY_MAX_LIMIT}], got ${JSON.stringify(params.limit)}.`, + }; + } + const limit = rawLimit; + + // PDG queries are always anchored (no rel-property index ⇒ an unanchored + // basic-block path scan is unbounded). `target` is required. + const target = typeof params.target === 'string' ? params.target.trim() : ''; + if (!target) { + return { + error: + 'pdg_query requires a "target" (a file path or symbol/function name) — PDG queries are always anchored.', + }; + } + + const edgeType = mode === 'controls' ? 'CDG' : 'REACHING_DEF'; + // Definitive: the meta stamp says this layer was never recorded. + const NO_PDG_NOTE = `no PDG layer — run gitnexus analyze --pdg to record ${edgeType} edges for this repo`; + // Inconclusive: meta is unreadable AND a global probe found zero rows of this + // edge type — but a genuinely edge-free layer (all-linear functions) looks + // identical to a missing one, so don't assert absence (#2188 review). + const PDG_LAYER_UNKNOWN_NOTE = `no ${edgeType} edges found for this target; PDG layer status unknown — was this repo indexed with gitnexus analyze --pdg?`; + + // Cheap meta probe: the layer exists iff the pdg stamp carries the + // mode-relevant cap (maxCdgEdgesPerFunction for CDG, maxReachingDef… + // for REACHING_DEF). Absent ⇒ the no-layer hint without a DB scan. + let pdgStamped: boolean | undefined; + try { + const meta = await loadMeta(path.dirname(repo.lbugPath)); + if (meta) { + pdgStamped = + mode === 'controls' + ? meta.pdg?.maxCdgEdgesPerFunction !== undefined + : meta.pdg?.maxReachingDefEdgesPerFunction !== undefined; + } + } catch { + /* meta unreadable — decide from the DB below */ + } + if (pdgStamped === false) { + return { mode, results: [], total: 0, note: NO_PDG_NOTE }; + } + + // Resolve the anchor on the SOURCE block via the shared resolver also used + // by explain (#2188): file id-prefix OR symbol span on the corrected + // [symStart+1, symEnd+1] window. `target` is required, so the early cases + // (not-found/ambiguous) return here and `anchor`/`anchorClause` are always + // set below (anchor stays non-optional — no `| undefined` — #2188 CodeQL). + const resolved = await this.resolveBlockAnchor(repo, target, 'pdg_query'); + if (resolved.early) return resolved.early; + const { anchorClause, anchor } = resolved; + const queryParams = resolved.queryParams; + + // Optional variable filter (flows mode) — REACHING_DEF stores the variable + // name in `reason`. + let reasonClause = ''; + if (mode === 'flows' && typeof params.variable === 'string' && params.variable.trim()) { + reasonClause = ' AND r.reason = $variable'; + queryParams.variable = params.variable.trim(); + } + + // edgeType is a hardcoded per-mode literal (never user input); `target` / + // `variable` flow only through bind params (no Cypher interpolation). + const matchClause = ` + MATCH (a:BasicBlock)-[r:CodeRelation]->(b:BasicBlock) + WHERE r.type = '${edgeType}' AND ${anchorClause}${reasonClause}`; + const [rows, countRows] = await Promise.all([ + executeParameterized( + repo.lbugPath, + `${matchClause} + RETURN a.id AS srcId, a.startLine AS srcLine, b.startLine AS dstLine, b.text AS dstText, r.reason AS reason + ORDER BY srcId, dstLine, reason + LIMIT ${limit}`, + queryParams, + ), + executeParameterized( + repo.lbugPath, + `${matchClause}\n RETURN COUNT(*) AS total`, + queryParams, + ), + ]); + const total = Number((countRows[0] as any)?.total ?? (countRows[0] as any)?.[0] ?? 0); + + // Unreadable meta + anchored miss: one bounded probe distinguishes "no rows + // for this anchor" from "no rows of this edge type at all". With meta + // unreadable we cannot tell a missing layer from an edge-free one, so the + // note is the inconclusive "status unknown" form, not the definitive + // NO_PDG_NOTE (which is reserved for the meta-stamped absence above). + if (total === 0 && pdgStamped === undefined) { + const probe = await executeParameterized( + repo.lbugPath, + `MATCH (:BasicBlock)-[r:CodeRelation]->(:BasicBlock) WHERE r.type = '${edgeType}' RETURN r.reason AS reason LIMIT 1`, + {}, + ); + if (probe.length === 0) return { mode, results: [], total: 0, note: PDG_LAYER_UNKNOWN_NOTE }; + } + + // basicBlockId = `BasicBlock::::` — split + // from the RIGHT (filePath may contain ':'). + const fnLineOf = (id: string): number => { + const parts = id.split(':'); + return Number(parts[parts.length - 3]); + }; + + const results = + mode === 'controls' + ? rows.map((r: any) => { + const fnLine = fnLineOf(String(r.srcId ?? r[0] ?? '')); + const dstText = String(r.dstText ?? r[3] ?? ''); + // A CDG edge into an early-exit block is a guard clause (subsumes + // #559): the controller predicate gates the dependent via `label`. + const isGuardExit = /^\s*(return|throw|continue|break)\b/.test(dstText); + return { + ...(Number.isInteger(fnLine) ? { functionLine: fnLine } : {}), + controller: { line: (r.srcLine ?? r[1]) as number | undefined }, + dependent: { line: (r.dstLine ?? r[2]) as number | undefined, text: dstText }, + label: String(r.reason ?? r[4] ?? ''), + ...(isGuardExit ? { guard: true } : {}), + }; + }) + : rows.map((r: any) => { + const fnLine = fnLineOf(String(r.srcId ?? r[0] ?? '')); + return { + ...(Number.isInteger(fnLine) ? { functionLine: fnLine } : {}), + variable: String(r.reason ?? r[4] ?? ''), + def: { line: (r.srcLine ?? r[1]) as number | undefined }, + use: { + line: (r.dstLine ?? r[2]) as number | undefined, + text: String(r.dstText ?? r[3] ?? ''), + }, + }; + }); + + return { + mode, + anchor, + results, + total, + ...(total > results.length ? { truncated: true } : {}), + }; + } + /** * Legacy explore — kept for backwards compatibility with resources.ts. * Routes cluster/process types to direct graph queries. diff --git a/gitnexus/src/mcp/resources.ts b/gitnexus/src/mcp/resources.ts index ec37c196c..5bf81a772 100644 --- a/gitnexus/src/mcp/resources.ts +++ b/gitnexus/src/mcp/resources.ts @@ -472,6 +472,12 @@ relationships: - MEMBER_OF: Symbol belongs to community - STEP_IN_PROCESS: Symbol is step N in process +pdg_layers: "Recorded ONLY when indexed with 'gitnexus analyze --pdg'. Intra-procedural, basic-block granular; both endpoints are BasicBlock nodes. Prefer the pdg_query tool over raw Cypher." + - BasicBlock: "Basic-block node. Columns: id, filePath, startLine, endLine, text. id = 'BasicBlock::::'." + - CFG: "Control-flow edge BasicBlock->BasicBlock. Edge kind (seq/cond-true/cond-false/loop-back/...) is in reason." + - CDG: "Control-DEPENDENCE edge BasicBlock->BasicBlock — the source predicate gates the target's execution. Branch sense 'T'|'F' in reason. Query via pdg_query mode:'controls'." + - REACHING_DEF: "Data-dependence (def->use) edge BasicBlock->BasicBlock. Source-level variable name is in reason. Query via pdg_query mode:'flows'." + relationship_table: "All relationships use a single CodeRelation table with a 'type' property. Properties: type (STRING), confidence (DOUBLE), reason (STRING), step (INT32)" example_queries: @@ -489,6 +495,11 @@ example_queries: WHERE p.heuristicLabel = "LoginFlow" RETURN s.name, r.step ORDER BY r.step + + guard_clauses (--pdg only; prefer pdg_query mode:'controls'): | + MATCH (pred:BasicBlock)-[r:CodeRelation {type: 'CDG'}]->(dep:BasicBlock) + WHERE dep.text STARTS WITH 'return' OR dep.text STARTS WITH 'throw' + RETURN pred.startLine, r.reason AS branch, dep.startLine, dep.text `; } diff --git a/gitnexus/src/mcp/tools.ts b/gitnexus/src/mcp/tools.ts index ec3d76e04..e2ffe10a7 100644 --- a/gitnexus/src/mcp/tools.ts +++ b/gitnexus/src/mcp/tools.ts @@ -72,6 +72,11 @@ export const LIST_REPOS_MAX_LIMIT = 200; export const EXPLAIN_DEFAULT_LIMIT = 50; export const EXPLAIN_MAX_LIMIT = 200; +// pdg_query result-page bounds (#2086 M6). Mirror the EXPLAIN_* limits — the +// no-rel-index path means every page must be anchored + LIMIT-bounded. +export const PDG_QUERY_DEFAULT_LIMIT = 50; +export const PDG_QUERY_MAX_LIMIT = 200; + export const GITNEXUS_TOOLS: ToolDefinition[] = [ { name: 'list_repos', @@ -226,7 +231,8 @@ TIPS: - All relationships use single CodeRelation table — filter with {type: 'CALLS'} etc. - Community = auto-detected functional area (Leiden algorithm). Properties: heuristicLabel, cohesion, symbolCount, keywords, description, enrichedBy - Process = execution flow trace from entry point to terminal. Properties: heuristicLabel, processType, stepCount, communities, entryPointId, terminalId -- Use heuristicLabel (not label) for human-readable community/process names`, +- Use heuristicLabel (not label) for human-readable community/process names +- PDG layers (only when indexed with \`--pdg\`): BasicBlock nodes + CFG / CDG (control dependence, branch sense 'T'|'F' in reason) / REACHING_DEF (def→use, variable in reason) edges, all BasicBlock→BasicBlock. Prefer the \`pdg_query\` tool — it anchors + bounds these for you (raw \`[:CDG*]\`/\`[:REACHING_DEF*]\` path scans are unindexed and unbounded).`, annotations: READ_ONLY_TOOL_ANNOTATIONS, inputSchema: { type: 'object', @@ -582,6 +588,58 @@ Findings are deliberately NOT part of impact()'s traversal or the web schema — required: [], }, }, + { + name: 'pdg_query', + description: `Query the persisted Program Dependence Graph recorded by \`gitnexus analyze --pdg\` — control dependence (CDG) and data dependence (REACHING_DEF) at basic-block granularity. The control/data analog of \`explain\` (which is the taint consumer). + +MODES: +- \`controls\` — "under what condition does X run?". Returns, for the anchored function, each control-dependence edge: the controlling predicate block, the dependent block, and the branch sense ('T' = the predicate's true/taken arm, 'F' = its false/fall-through arm). An edge into an early return/throw block is flagged \`guard: true\` (subsumes the #559 guard heuristic); the branch sense of a guard depends on its predicate — \`if (!ok) return;\` rides the 'T' arm — so don't filter guards by a fixed label. +- \`flows\` — "where does variable Y flow?". Returns REACHING_DEF def→use edges for the anchored function; pass \`variable\` to filter to one binding. + +WHEN TO USE: comprehension ("what guards this statement?"), data-flow tracing within a function, guard-clause discovery. Requires \`gitnexus analyze --pdg\`; without that layer the tool returns a clear "no PDG layer" note, not an error. + +ANCHORING (required): \`target\` is a file path or a symbol/function name (resolved like context()). PDG queries are ALWAYS anchored — there is no whole-repo enumeration (an unanchored basic-block path scan is unbounded; LadybugDB has no rel-property index). A symbol target is line-range granular; an ambiguous name returns ranked candidates, unknown returns not-found. + +CONTRACT CAVEATS: +- CDG labels are binary 'T'/'F' in M5/M6; per-case \`switch\` arm conditions are not yet distinguished (every case dispatch is 'T'). +- Granularity is basic-block, reconstructed to the function via the BasicBlock id + line span (no Function→BasicBlock edge); deeply same-line-packed functions may anchor coarsely. +- Control/data dependence is intra-procedural (per function). Cross-function flow is taint's domain (\`explain\`). +- These edges are deliberately NOT part of impact()'s traversal — \`pdg_query\` is the dedicated consumer; raw edges are also queryable via \`cypher\`.`, + annotations: READ_ONLY_TOOL_ANNOTATIONS, + inputSchema: { + type: 'object', + properties: { + mode: { + type: 'string', + enum: ['controls', 'flows'], + description: + "'controls' = control dependence (CDG: what condition gates X); 'flows' = data dependence (REACHING_DEF: where variable Y flows).", + }, + target: { + type: 'string', + description: + 'Required anchor: a file path (e.g. "src/handlers/run.ts" — suffix match accepted) or a symbol/function name (resolved like context()).', + }, + variable: { + type: 'string', + description: + 'Optional (flows mode only): restrict REACHING_DEF results to this source-level variable name.', + }, + limit: { + type: 'integer', + description: `Max edges returned (default: ${PDG_QUERY_DEFAULT_LIMIT}, max: ${PDG_QUERY_MAX_LIMIT}). "total" reports the full matched count; "truncated" is set when the page is smaller.`, + default: PDG_QUERY_DEFAULT_LIMIT, + minimum: 1, + maximum: PDG_QUERY_MAX_LIMIT, + }, + repo: { + type: 'string', + description: 'Repository name or path. Omit if only one repo is indexed.', + }, + }, + required: ['mode', 'target'], + }, + }, { name: 'route_map', description: `Show API route mappings: which components/hooks fetch which API endpoints, and which handler files serve them. @@ -717,6 +775,7 @@ const BRANCH_SCOPED_TOOLS = new Set([ 'context', 'detect_changes', 'explain', + 'pdg_query', 'check', 'impact', 'rename', diff --git a/gitnexus/src/storage/repo-manager.ts b/gitnexus/src/storage/repo-manager.ts index 56714e5be..b03db9fc5 100644 --- a/gitnexus/src/storage/repo-manager.ts +++ b/gitnexus/src/storage/repo-manager.ts @@ -158,6 +158,14 @@ export interface RepoMeta { * type for that reason; resolved (always present) on every M2+ write. */ maxReachingDefEdgesPerFunction?: number; + /** + * Emit-side per-function CDG (control-dependence) edge cap, resolved + * (0 = unlimited; #2085 M5). ABSENT on any pre-M5 stamp — that absence is + * what trips `pdgModeMismatch` on the first CDG-aware run and forces the + * full writeback that materialises CDG edges. Optional for that upgrade + * reason; resolved (always present) on every M5+ write. + */ + maxCdgEdgesPerFunction?: number; /** * Per-function taint findings cap, resolved (0 = unlimited; #2083 M3). * ABSENT on an M1/M2-era stamp — like `maxReachingDefEdgesPerFunction`, diff --git a/gitnexus/test/integration/basicblock-roundtrip.test.ts b/gitnexus/test/integration/basicblock-roundtrip.test.ts index 88f9a60c9..ef6cfc86f 100644 --- a/gitnexus/test/integration/basicblock-roundtrip.test.ts +++ b/gitnexus/test/integration/basicblock-roundtrip.test.ts @@ -4,11 +4,13 @@ * * Exercises the real csv-generator → loadGraphToLbug → COPY → query path: * - a BasicBlock node (id/filePath/startLine/endLine/text) round-trips - * - one edge of each new type (CFG/REACHING_DEF/TAINTED/SANITIZES/TAINT_PATH) - * between two BasicBlocks round-trips (asserts the new FROM/TO DDL pair + - * REL_TYPES load through bulk COPY) + * - one edge of each new type (CFG/REACHING_DEF/TAINTED/SANITIZES/TAINT_PATH, + * plus CDG/POST_DOMINATE from #2085 M5) between two BasicBlocks round-trips + * (asserts the new FROM/TO DDL pair + REL_TYPES load through bulk COPY) * - REACHING_DEF carries its `variable` in the existing `reason` column * (M0/S1 storage decision) and a variable-filtered query returns it + * - CDG carries its branch label ('T'|'F') in the same `reason` column + * (#2085 M5) and a label-filtered query returns it * - the DDL (BASICBLOCK_SCHEMA wired into NODE_SCHEMA_QUERIES) loads on a * fresh DB — if BASICBLOCK_SCHEMA were not in SCHEMA_QUERIES, initLbug would * never create the table and these COPYs would fail (F1 guard, end-to-end) @@ -27,7 +29,15 @@ let dbPath: string; const BB1 = 'BasicBlock:src/a.ts:0'; const BB2 = 'BasicBlock:src/a.ts:1'; -const NEW_EDGE_TYPES = ['CFG', 'REACHING_DEF', 'TAINTED', 'SANITIZES', 'TAINT_PATH'] as const; +const NEW_EDGE_TYPES = [ + 'CFG', + 'REACHING_DEF', + 'TAINTED', + 'SANITIZES', + 'TAINT_PATH', + 'CDG', + 'POST_DOMINATE', +] as const; beforeAll(async () => { tmpBase = path.join(os.tmpdir(), `gitnexus-bb-roundtrip-${Date.now()}-${process.pid}`); @@ -65,7 +75,7 @@ beforeAll(async () => { sourceId: BB1, targetId: BB2, type, - reason: type === 'REACHING_DEF' ? 'x' : `${type.toLowerCase()}-edge`, + reason: type === 'REACHING_DEF' ? 'x' : type === 'CDG' ? 'T' : `${type.toLowerCase()}-edge`, })), ); @@ -151,4 +161,14 @@ describe('BasicBlock + taint/PDG edge round-trip (#2080)', () => { expect(rows[0].from).toBe(BB1); expect(rows[0].to).toBe(BB2); }); + + it('CDG carries its branch label in reason and is queryable by it (#2085 M5)', async () => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const rows = await adapter.executeQuery( + "MATCH (a:BasicBlock)-[r:CodeRelation {type: 'CDG', reason: 'T'}]->(b:BasicBlock) RETURN a.id AS from, b.id AS to", + ); + expect(rows).toHaveLength(1); + expect(rows[0].from).toBe(BB1); + expect(rows[0].to).toBe(BB2); + }); }); diff --git a/gitnexus/test/integration/cfg/__snapshots__/cdg-snapshot.test.ts.snap b/gitnexus/test/integration/cfg/__snapshots__/cdg-snapshot.test.ts.snap new file mode 100644 index 000000000..c7b07fdc3 --- /dev/null +++ b/gitnexus/test/integration/cfg/__snapshots__/cdg-snapshot.test.ts.snap @@ -0,0 +1,93 @@ +// Vitest Snapshot v1, https://vitest.dev/guide/snapshot.html + +exports[`AC1 — CDG snapshot on the M1 fixture > matches the committed control-dependence set for every fixture function 1`] = ` +[ + { + "cdg": [], + "startLine": 9, + }, + { + "cdg": [ + "2->3:T", + "2->4:F", + ], + "startLine": 14, + }, + { + "cdg": [ + "2->3:T", + "2->4:F", + "4->5:T", + "4->6:F", + ], + "startLine": 23, + }, + { + "cdg": [ + "2->2:T", + "2->4:T", + ], + "startLine": 34, + }, + { + "cdg": [ + "2->2:T", + "2->4:T", + "2->5:T", + ], + "startLine": 41, + }, + { + "cdg": [ + "2->2:T", + "2->4:T", + ], + "startLine": 48, + }, + { + "cdg": [ + "2->4:T", + "2->5:T", + "2->6:T", + "2->7:T", + "2->8:T", + ], + "startLine": 55, + }, + { + "cdg": [ + "5->3:F", + "5->4:F", + ], + "startLine": 69, + }, + { + "cdg": [ + "2->3:T", + "2->4:F", + ], + "startLine": 80, + }, + { + "cdg": [ + "2->2:T", + "2->4:T", + "4->5:T", + "4->6:F", + ], + "startLine": 87, + }, + { + "cdg": [ + "3->7:F", + "4->5:T", + "4->6:F", + ], + "startLine": 102, + }, + { + "cdg": [], + "startLine": 115, + }, +] +`; diff --git a/gitnexus/test/integration/cfg/cdg-snapshot.test.ts b/gitnexus/test/integration/cfg/cdg-snapshot.test.ts new file mode 100644 index 000000000..1b8a594eb --- /dev/null +++ b/gitnexus/test/integration/cfg/cdg-snapshot.test.ts @@ -0,0 +1,58 @@ +import { describe, it, expect } from 'vitest'; +import fs from 'fs'; +import path from 'path'; +import Parser from 'tree-sitter'; +import TypeScript from 'tree-sitter-typescript'; +import { collectFunctionCfgs } from '../../../src/core/ingestion/cfg/collect.js'; +import { computeControlDependence } from '../../../src/core/ingestion/cfg/control-dependence.js'; +import { computePostDominators } from '../../../src/core/ingestion/cfg/post-dominators.js'; +import { getProvider } from '../../../src/core/ingestion/languages/index.js'; +import { SupportedLanguages } from '../../../src/config/supported-languages.js'; +import type { FunctionCfg } from '../../../src/core/ingestion/cfg/types.js'; + +// #2085 M5 AC1 — a committed snapshot of the CDG edge set on the shared M1 +// fixture (the same `ten-functions.ts` the REACHING_DEF snapshot uses). The +// serialization is deterministic — sorted `controller->dependent:label` +// strings — so any post-dominator / Ferrante behavior change shows as a +// reviewable snapshot diff, never silent drift. + +const FIXTURES = path.join(__dirname, 'fixtures'); + +function cfgsOfFile(file: string): readonly FunctionCfg[] { + const visitor = getProvider(SupportedLanguages.TypeScript).cfgVisitor; + if (!visitor) throw new Error('no cfgVisitor'); + const source = fs.readFileSync(path.join(FIXTURES, file), 'utf8'); + const parser = new Parser(); + parser.setLanguage(TypeScript.typescript); + return collectFunctionCfgs(parser.parse(source).rootNode, visitor, file).cfgs; +} + +/** Deterministic rendering: startLine + sorted controller->dependent:label. */ +function serialize(cfg: FunctionCfg): Record { + const { edges } = computeControlDependence(cfg); + return { + startLine: cfg.functionStartLine, + cdg: edges.map((e) => `${e.controllerBlock}->${e.dependentBlock}:${e.label}`), + }; +} + +describe('AC1 — CDG snapshot on the M1 fixture', () => { + it('matches the committed control-dependence set for every fixture function', () => { + const cfgs = cfgsOfFile('ten-functions.ts'); + expect(cfgs).toHaveLength(12); + expect(cfgs.map(serialize)).toMatchSnapshot(); + }); + + it('every CDG edge references in-range blocks with a valid T/F label (AC2 sanity)', () => { + for (const cfg of cfgsOfFile('ten-functions.ts')) { + const tree = computePostDominators(cfg); + for (const e of computeControlDependence(cfg, tree).edges) { + expect(e.controllerBlock).toBeGreaterThanOrEqual(0); + expect(e.controllerBlock).toBeLessThan(cfg.blocks.length); + expect(e.dependentBlock).toBeGreaterThanOrEqual(0); + expect(e.dependentBlock).toBeLessThan(cfg.blocks.length); + expect(['T', 'F']).toContain(e.label); + } + } + }); +}); diff --git a/gitnexus/test/integration/cfg/cfg-emit.test.ts b/gitnexus/test/integration/cfg/cfg-emit.test.ts index 98efe908a..84459e0aa 100644 --- a/gitnexus/test/integration/cfg/cfg-emit.test.ts +++ b/gitnexus/test/integration/cfg/cfg-emit.test.ts @@ -2,10 +2,20 @@ import { describe, it, expect, vi } from 'vitest'; import Parser from 'tree-sitter'; import TypeScript from 'tree-sitter-typescript'; import { collectFunctionCfgs } from '../../../src/core/ingestion/cfg/collect.js'; -import { emitFileCfgs, emitFileReachingDefs } from '../../../src/core/ingestion/cfg/emit.js'; +import { + emitFileCfgs, + emitFileReachingDefs, + emitFileCdg, + POST_DOMINATE_DEBUG_ENV, +} from '../../../src/core/ingestion/cfg/emit.js'; import { getProvider } from '../../../src/core/ingestion/languages/index.js'; import { SupportedLanguages } from '../../../src/config/supported-languages.js'; -import type { CfgVisitor, FunctionCfg } from '../../../src/core/ingestion/cfg/types.js'; +import type { + BasicBlockData, + CfgEdgeData, + CfgVisitor, + FunctionCfg, +} from '../../../src/core/ingestion/cfg/types.js'; import type { SyntaxNode } from '../../../src/core/ingestion/utils/ast-helpers.js'; import type { KnowledgeGraph } from '../../../src/core/graph/types.js'; @@ -338,3 +348,156 @@ describe('U4 (#2082 M2) — emitFileReachingDefs', () => { expect(rels.slice(firstIds.length).map((e) => e.id)).toEqual(firstIds); }); }); + +describe('U4 (#2085 M5) — emitFileCdg', () => { + it('emits CDG edges between BasicBlocks with the branch label in reason', () => { + // if/else diamond → both arms control-dependent on the branch (T and F) + const cfgs = cfgsOf( + `function f(x: number) { if (x) { a(); } else { b(); } c(); }`, + 'src/cdg.ts', + ); + const { graph, rels } = recordingGraph(); + const r = emitFileCdg(graph, cfgs); + + expect(r.edges).toBe(rels.length); + expect(rels.length).toBeGreaterThan(0); + for (const e of rels) { + expect(e.type).toBe('CDG'); + expect(e.sourceId).toMatch(/^BasicBlock:src\/cdg\.ts:\d+:\d+:\d+$/); + expect(e.targetId).toMatch(/^BasicBlock:src\/cdg\.ts:\d+:\d+:\d+$/); + expect(['T', 'F']).toContain(e.reason); // label rides reason (KTD3) + } + const labels = new Set(rels.map((e) => e.reason)); + expect(labels.has('T')).toBe(true); + expect(labels.has('F')).toBe(true); + expect(r.postDominateEdges).toBe(0); // debug env off + }); + + it('a straight-line function has no control dependence', () => { + const cfgs = cfgsOf(`function f() { a(); b(); c(); }`, 'lin.ts'); + const { graph, rels } = recordingGraph(); + const r = emitFileCdg(graph, cfgs); + expect(rels).toHaveLength(0); + expect(r.edges).toBe(0); + }); + + it('deduped edge ids are unique and deterministic across runs', () => { + const cfgs = cfgsOf( + `function f(x: number, y: number) { if (x) { if (y) { a(); } else { b(); } } else { c(); } }`, + 'det.ts', + ); + const first = recordingGraph(); + emitFileCdg(first.graph, cfgs); + const ids = first.rels.map((e) => e.id); + expect(new Set(ids).size).toBe(ids.length); // no id collisions + const second = recordingGraph(); + emitFileCdg(second.graph, cfgs); + expect(second.rels.map((e) => e.id)).toEqual(ids); // deterministic + }); + + it('per-function edge cap stops at the cap, records the drop, and warns (R6)', () => { + const cfgs = cfgsOf( + `function f(x: number, y: number) { if (x) { if (y) { a(); } else { b(); } } else { c(); } }`, + 'cap.ts', + ); + const full = recordingGraph(); + const total = emitFileCdg(full.graph, cfgs).edges; + expect(total).toBeGreaterThan(1); + + const { graph, rels } = recordingGraph(); + const onWarn = vi.fn(); + const r = emitFileCdg(graph, cfgs, 1, onWarn); + expect(rels.length).toBe(1); // emitted exactly the cap + expect(r.droppedEdges).toBe(total - 1); + expect(r.cappedFunctions).toBe(1); + expect(onWarn).toHaveBeenCalledTimes(1); + expect(onWarn.mock.calls[0][0]).toContain('CDG edge cap'); + }); + + it('cap of 0 means unlimited (no warning)', () => { + const cfgs = cfgsOf(`function f(x: number) { if (x) { a(); } else { b(); } }`, 'u.ts'); + const { graph, rels } = recordingGraph(); + const onWarn = vi.fn(); + const r = emitFileCdg(graph, cfgs, 0, onWarn); + expect(rels.length).toBe(r.edges); + expect(r.droppedEdges).toBe(0); + expect(onWarn).not.toHaveBeenCalled(); + }); + + it('skips CDG for a CFG whose EXIT is unreachable from all blocks (#2188 unsound guard)', () => { + // Hand-built exit-less loop: 0=entry → 1 ⇄ 2 spin forever; 3=exit is + // disconnected. Post-dominance would be unsound there, so CDG is skipped — + // while a normal sibling function in the same batch still emits CDG. + const blocks: BasicBlockData[] = [0, 1, 2, 3].map((i) => ({ + index: i, + startLine: i + 1, + endLine: i + 1, + text: '', + kind: i === 0 ? 'entry' : i === 3 ? 'exit' : 'normal', + })); + const edges: CfgEdgeData[] = [ + { from: 0, to: 1, kind: 'seq' }, + { from: 1, to: 2, kind: 'seq' }, + { from: 2, to: 1, kind: 'seq' }, + ]; + const unsound: FunctionCfg = { + filePath: 'spin.ts', + functionStartLine: 1, + functionStartColumn: 0, + entryIndex: 0, + exitIndex: 3, + blocks, + edges, + }; + const sound = cfgsOf( + `function f(x: number) { if (x) { a(); } else { b(); } c(); }`, + 'sound.ts', + )[0]; + + const { graph, rels } = recordingGraph(); + const onWarn = vi.fn(); + const r = emitFileCdg(graph, [unsound, sound], 0, onWarn); + + expect(r.skippedUnsoundFunctions).toBe(1); + // No CDG edge originates from the unsound function... + expect(rels.some((e) => e.sourceId.startsWith('BasicBlock:spin.ts:'))).toBe(false); + // ...but the sound sibling still emitted CDG normally. + expect(rels.length).toBeGreaterThan(0); + expect(rels.every((e) => e.sourceId.startsWith('BasicBlock:sound.ts:'))).toBe(true); + expect(r.edges).toBe(rels.length); + expect(onWarn).toHaveBeenCalledTimes(1); + expect(onWarn.mock.calls[0][0]).toContain('EXIT not reachable'); + }); + + it('emits POST_DOMINATE debug edges only when the env flag is set (KTD8)', () => { + const cfgs = cfgsOf(`function f(x: number) { if (x) { a(); } else { b(); } c(); }`, 'pd.ts'); + + // flag unset → no POST_DOMINATE edges + const off = recordingGraph(); + const rOff = emitFileCdg(off.graph, cfgs); + expect(off.rels.some((e) => e.type === 'POST_DOMINATE')).toBe(false); + expect(rOff.postDominateEdges).toBe(0); + + // flag set → POST_DOMINATE edges appear (not counted against CDG cap) + const prev = process.env[POST_DOMINATE_DEBUG_ENV]; + process.env[POST_DOMINATE_DEBUG_ENV] = '1'; + try { + const on = recordingGraph(); + const rOn = emitFileCdg(on.graph, cfgs); + const pd = on.rels.filter((e) => e.type === 'POST_DOMINATE'); + expect(pd.length).toBeGreaterThan(0); + expect(rOn.postDominateEdges).toBe(pd.length); + // CDG edge count is unchanged by the debug flag + expect(rOn.edges).toBe(rOff.edges); + + // the case-insensitive 'true' OR-branch of postDominateDebugEnabled + process.env[POST_DOMINATE_DEBUG_ENV] = 'TRUE'; + const onTrue = recordingGraph(); + const rOnTrue = emitFileCdg(onTrue.graph, cfgs); + expect(rOnTrue.postDominateEdges).toBeGreaterThan(0); + } finally { + if (prev === undefined) delete process.env[POST_DOMINATE_DEBUG_ENV]; + else process.env[POST_DOMINATE_DEBUG_ENV] = prev; + } + }); +}); diff --git a/gitnexus/test/integration/cfg/fixtures/pdg-repo/guards.ts b/gitnexus/test/integration/cfg/fixtures/pdg-repo/guards.ts new file mode 100644 index 000000000..161332d6e --- /dev/null +++ b/gitnexus/test/integration/cfg/fixtures/pdg-repo/guards.ts @@ -0,0 +1,24 @@ +// Guard-clause + data-flow fixture for the pdg_query integration test (#2086). +// Kept free of taint sources/sinks so it adds no TAINTED findings to the shared +// pdg-repo fixture (taint-explain / pipeline-pdg assert on taint dynamically). + +export function guarded(ok: boolean, x: number): number { + // `if (!ok) return` — the early return is control-dependent on the guard + // predicate (the #559 guard-clause shape); the post-guard body is + // control-dependent on the complementary arm. + if (!ok) { + return -1; + } + const y = x * 2; + const z = y + 1; + return z; +} + +export function loopFlow(items: number[]): number { + // A loop-carried accumulator — exercises REACHING_DEF (def→use of `sum`). + let sum = 0; + for (const it of items) { + sum = sum + it; + } + return sum; +} diff --git a/gitnexus/test/integration/cfg/pipeline-pdg.test.ts b/gitnexus/test/integration/cfg/pipeline-pdg.test.ts index d5d8415b3..2177c761a 100644 --- a/gitnexus/test/integration/cfg/pipeline-pdg.test.ts +++ b/gitnexus/test/integration/cfg/pipeline-pdg.test.ts @@ -21,6 +21,7 @@ function counts(result: PipelineResult): { reachingDefs: number; tainted: number; sanitizes: number; + cdg: number; } { let basicBlocks = 0; result.graph.forEachNode((n) => { @@ -30,13 +31,15 @@ function counts(result: PipelineResult): { let reachingDefs = 0; let tainted = 0; let sanitizes = 0; + let cdg = 0; for (const rel of result.graph.iterRelationships()) { if (rel.type === 'CFG') cfgEdges++; if (rel.type === 'REACHING_DEF') reachingDefs++; if (rel.type === 'TAINTED') tainted++; if (rel.type === 'SANITIZES') sanitizes++; + if (rel.type === 'CDG') cdg++; } - return { basicBlocks, cfgEdges, reachingDefs, tainted, sanitizes }; + return { basicBlocks, cfgEdges, reachingDefs, tainted, sanitizes, cdg }; } const tmpDirs: string[] = []; @@ -138,13 +141,49 @@ describe('U7 — end-to-end --pdg pipeline', () => { expect(sawVulnFlow).toBe(true); // the req.body → exec flow, via `cmd` }, 60000); - it('with --pdg off (default): emits zero BasicBlock nodes and zero CFG edges', async () => { + // M5 (#2085 U6): control dependence rides the same `--pdg` gate. AC3 — the + // CDG edges make "under what condition does block X run?" answerable: each + // edge's source is the controlling branch block and its `reason` is the + // 'T'|'F' sense. (The dedicated `pdg_query` MCP tool is #2086; here the raw + // graph carries the answer.) + it('with --pdg on: emits CDG edges (controller→dependent, T/F label) — AC3 answerability', async () => { + const result = await runPipelineFromRepo(freshRepo(), () => {}, { pdg: true }); + const { cdg } = counts(result); + expect(cdg).toBeGreaterThan(0); + + const blockIds = new Set(); + result.graph.forEachNode((n) => { + if (n.label === 'BasicBlock') blockIds.add(n.id); + }); + + // "What controls block X?" — index CDG edges by dependent block. Every CDG + // edge connects two persisted BasicBlocks and carries a T/F label. + const controllersOf = new Map(); + for (const rel of result.graph.iterRelationships()) { + if (rel.type !== 'CDG') continue; + expect(blockIds.has(rel.sourceId)).toBe(true); + expect(blockIds.has(rel.targetId)).toBe(true); + expect(['T', 'F']).toContain(rel.reason); + const list = controllersOf.get(rel.targetId) ?? []; + list.push({ controller: rel.sourceId, label: rel.reason }); + controllersOf.set(rel.targetId, list); + } + // At least one block has its controlling branch + condition recoverable — + // the query "under what condition does this block run?" is answerable. + expect(controllersOf.size).toBeGreaterThan(0); + for (const [, controls] of controllersOf) { + expect(controls.length).toBeGreaterThan(0); + } + }, 60000); + + it('with --pdg off (default): emits zero BasicBlock nodes and zero CFG/CDG edges', async () => { const result = await runPipelineFromRepo(freshRepo(), () => {}); - const { basicBlocks, cfgEdges, reachingDefs, tainted, sanitizes } = counts(result); + const { basicBlocks, cfgEdges, reachingDefs, tainted, sanitizes, cdg } = counts(result); expect(basicBlocks).toBe(0); expect(cfgEdges).toBe(0); expect(reachingDefs).toBe(0); expect(tainted).toBe(0); expect(sanitizes).toBe(0); + expect(cdg).toBe(0); }, 60000); }); diff --git a/gitnexus/test/integration/pdg-query.test.ts b/gitnexus/test/integration/pdg-query.test.ts new file mode 100644 index 000000000..74bf232d0 --- /dev/null +++ b/gitnexus/test/integration/pdg-query.test.ts @@ -0,0 +1,485 @@ +/** + * Integration Tests: MCP `pdg_query` tool (#2086 M6) + * + * End-to-end against a REAL LadybugDB: the pdg-repo fixture is indexed by the + * real pipeline with `--pdg` (workers — requires `node scripts/build.js`), the + * resulting BasicBlock nodes + CDG/REACHING_DEF edges and the fixture's + * Function symbols are persisted into the test DB, and `pdg_query` is exercised + * through the full `callTool` dispatch: + * + * - controls mode: "under what condition does X run?" (CDG), incl. the + * guard-clause subset (early-return block, #559 subsumption / R1) + * - flows mode: "where does variable Y flow?" (REACHING_DEF def→use) / R2 + * - symbol + file anchoring; required-target / invalid-mode / bad-limit errors + * - a repo WITHOUT the pdg layer → the "no PDG layer" note, not an error + * + * Seeding via the real emit output (not hand-written rows) pins the format + * compatibility between the M5/M2 write path and the M6 read path — the + * BasicBlock id template + the 'T'/'F' / variable `reason` semantics. + */ +import { describe, it, expect, beforeAll, vi } from 'vitest'; +import fs from 'fs'; +import os from 'os'; +import path from 'path'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos, loadMeta } from '../../src/storage/repo-manager.js'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; +import { runPipelineFromRepo } from '../../src/core/ingestion/pipeline.js'; + +vi.mock('../../src/storage/repo-manager.js', async (importOriginal) => { + const actual = await importOriginal(); + return { + ...actual, + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), + // No meta.json for the seeded test DB — pdg_query's meta probe degrades to + // the row-existence probe (the seeded-DB reality, like taint-explain). + loadMeta: vi.fn().mockResolvedValue(null), + }; +}); + +const FIXTURE = path.join(__dirname, 'cfg', 'fixtures', 'pdg-repo'); + +// ─── Block 1: a --pdg index with real CDG + REACHING_DEF edges ─────── + +withTestLbugDB( + 'pdg-query', + (handle) => { + describe('pdg_query against a --pdg index', () => { + let backend: LocalBackend; + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) throw new Error('LocalBackend not initialized in afterSetup'); + backend = ext._backend; + }); + + it('controls mode answers "what controls X" and flags the guard clause (R1)', async () => { + const result = await backend.callTool('pdg_query', { mode: 'controls', target: 'guarded' }); + expect(result).not.toHaveProperty('error'); + expect(result.mode).toBe('controls'); + expect(result.anchor.symbol).toBe('guarded'); + expect(result.results.length).toBeGreaterThan(0); + // every edge has a 'T'/'F' branch label + for (const e of result.results) expect(['T', 'F']).toContain(e.label); + // the early `return -1` is control-dependent on the guard predicate → + // flagged guard:true (the #559 guard-clause subsumption) + const guardEdge = result.results.find((e: any) => e.guard === true); + expect(guardEdge, 'a guard-clause edge into an early-exit block').toBeDefined(); + expect(guardEdge.dependent.text).toMatch(/return/); + }); + + it('flows mode answers "where does variable Y flow" (R2)', async () => { + const result = await backend.callTool('pdg_query', { + mode: 'flows', + target: 'loopFlow', + variable: 'sum', + }); + expect(result).not.toHaveProperty('error'); + expect(result.mode).toBe('flows'); + expect(result.results.length).toBeGreaterThan(0); + for (const e of result.results) expect(e.variable).toBe('sum'); + }); + + it('flows mode without a variable filter returns all def→use edges for the anchor', async () => { + const result = await backend.callTool('pdg_query', { mode: 'flows', target: 'loopFlow' }); + expect(result).not.toHaveProperty('error'); + expect(result.results.length).toBeGreaterThan(0); + expect(result.results.some((e: any) => e.variable === 'sum')).toBe(true); + }); + + it('controls mode anchors by file path too', async () => { + const result = await backend.callTool('pdg_query', { + mode: 'controls', + target: 'guards.ts', + }); + expect(result).not.toHaveProperty('error'); + expect(result.results.length).toBeGreaterThan(0); + }); + + it('rejects a missing target (PDG queries are always anchored)', async () => { + const result = await backend.callTool('pdg_query', { mode: 'controls' }); + expect(result).toHaveProperty('error'); + expect(result.error).toMatch(/target/i); + }); + + it('rejects an invalid mode', async () => { + const result = await backend.callTool('pdg_query', { mode: 'slice', target: 'guarded' }); + expect(result).toHaveProperty('error'); + expect(result.error).toMatch(/mode/i); + }); + + it('rejects an out-of-bounds limit', async () => { + for (const limit of [0, -1, 1.5, 10_000, NaN]) { + const result = await backend.callTool('pdg_query', { + mode: 'controls', + target: 'guarded', + limit, + }); + expect(result).toHaveProperty('error'); + expect(result.error).toMatch(/limit/i); + } + }); + + it('an unknown symbol target mirrors context() not-found semantics', async () => { + const result = await backend.callTool('pdg_query', { + mode: 'controls', + target: 'nonexistentPdgFn999', + }); + expect(result).toHaveProperty('error'); + expect(result.error).toMatch(/not found/i); + }); + + it('a call with no arguments returns a clean validation error, not a crash (#2188)', async () => { + // An MCP client may send {"name":"pdg_query"} with no `arguments` field; + // the dispatch then hands `params: undefined` to the impl. It must + // default to {} and surface the mode-validation error, not a TypeError. + const result = await backend.callTool('pdg_query'); + expect(result).toHaveProperty('error'); + expect(result.error).toMatch(/mode/i); + }); + }); + }, + { + poolAdapter: true, + afterSetup: async (handle) => { + const repoDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-pdgq-')); + try { + fs.cpSync(FIXTURE, repoDir, { recursive: true }); + const pipelineResult = await runPipelineFromRepo(repoDir, () => {}, { pdg: true }); + + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const nodes: Array<{ label: string; props: Record }> = []; + pipelineResult.graph.forEachNode((n) => { + if (n.label === 'BasicBlock') { + nodes.push({ + label: 'BasicBlock', + props: { + id: n.id, + filePath: n.properties.filePath ?? '', + startLine: n.properties.startLine ?? 0, + endLine: n.properties.endLine ?? 0, + text: n.properties.text ?? '', + }, + }); + } else if (n.label === 'Function') { + nodes.push({ + label: 'Function', + props: { + id: n.id, + name: n.properties.name ?? '', + filePath: n.properties.filePath ?? '', + startLine: n.properties.startLine ?? 0, + endLine: n.properties.endLine ?? 0, + }, + }); + } + }); + for (const node of nodes) { + const assignments = Object.keys(node.props) + .map((k) => `${k}: $${k}`) + .join(', '); + await adapter.executePrepared( + `CREATE (n:${node.label} {${assignments}})`, + node.props as Record, + ); + } + let pdgEdges = 0; + for (const rel of pipelineResult.graph.iterRelationships()) { + if (rel.type !== 'CDG' && rel.type !== 'REACHING_DEF') continue; + await adapter.executePrepared( + `MATCH (a:BasicBlock {id: $src}), (b:BasicBlock {id: $dst}) + CREATE (a)-[:CodeRelation {type: '${rel.type}', confidence: $confidence, reason: $reason, step: 0}]->(b)`, + { + src: rel.sourceId, + dst: rel.targetId, + confidence: rel.confidence ?? 1.0, + reason: rel.reason ?? '', + }, + ); + pdgEdges++; + } + if (pdgEdges === 0) { + throw new Error('fixture produced no CDG/REACHING_DEF edges — pdg emit regressed?'); + } + } finally { + fs.rmSync(repoDir, { recursive: true, force: true }); + } + + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'pdg-repo', + path: '/pdg/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'abc123', + stats: { files: 4, nodes: 4, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); + +// ─── Block 2: a repo indexed WITHOUT --pdg ─────────────────────────── + +withTestLbugDB( + 'pdg-query-nopdg', + (handle) => { + describe('pdg_query without a PDG layer', () => { + let backend: LocalBackend; + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) throw new Error('LocalBackend not initialized in afterSetup'); + backend = ext._backend; + }); + + it('controls returns the status-unknown note when meta is unreadable + probe empty (#2188)', async () => { + // Meta is mocked unreadable (null) and the seed has no CDG rows. A + // missing layer is indistinguishable from an edge-free one here, so the + // note is inconclusive ("status unknown"), not the definitive absence. + const result = await backend.callTool('pdg_query', { mode: 'controls', target: 'plainFn' }); + expect(result).not.toHaveProperty('error'); + expect(result.results).toEqual([]); + expect(result.note).toMatch(/status unknown/i); + expect(result.note).not.toMatch(/no PDG layer/i); + expect(result.note).toContain('--pdg'); + }); + + it('flows returns the status-unknown note too when meta is unreadable', async () => { + const result = await backend.callTool('pdg_query', { mode: 'flows', target: 'plain.ts' }); + expect(result).not.toHaveProperty('error'); + expect(result.results).toEqual([]); + expect(result.note).toMatch(/status unknown/i); + }); + + it('a readable meta without a pdg stamp short-circuits to the DEFINITIVE no-layer note', async () => { + // Meta is readable but carries no CDG cap ⇒ the layer truly was never + // recorded; this path keeps the definitive "no PDG layer" wording. + vi.mocked(loadMeta).mockResolvedValueOnce({} as any); + const result = await backend.callTool('pdg_query', { mode: 'controls', target: 'plainFn' }); + expect(result.results).toEqual([]); + expect(result.note).toMatch(/no PDG layer/i); + }); + }); + }, + { + seed: [ + `CREATE (fn:Function {id: 'func:plainFn', name: 'plainFn', filePath: 'src/plain.ts', startLine: 1, endLine: 5, isExported: true, content: 'function plainFn() {}', description: 'no pdg layer here'})`, + ], + poolAdapter: true, + afterSetup: async (handle) => { + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'plain-repo', + path: '/plain/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'def456', + stats: { files: 1, nodes: 1, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); + +// ─── Block 3: symbol-anchor line-base off-by-one (#2188 review) ────── +// +// Hand-seeded with controlled line numbers (no parser dependency): `targetFn` +// occupies 0-based symbol lines 10–14, and a neighbor function sits directly +// above it with its last block on 1-based line 10 — the line right above +// targetFn's declaration (1-based line 11). BasicBlock startLine is 1-based +// while the symbol span is 0-based, so the anchor window must be [11,15] (both +// bounds shifted +1). The pre-fix window [10,15] (lower bound left 0-based) +// over-includes the neighbor's line-10 block. This pins the lower-bound +1. + +withTestLbugDB( + 'pdg-query-adjacency', + (handle) => { + describe('pdg_query symbol anchoring (#2188 lower-bound off-by-one)', () => { + let backend: LocalBackend; + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) throw new Error('LocalBackend not initialized in afterSetup'); + backend = ext._backend; + }); + + it('excludes a neighbor function block on the line directly above the target', async () => { + const result = await backend.callTool('pdg_query', { + mode: 'controls', + target: 'targetFn', + }); + expect(result).not.toHaveProperty('error'); + // Only targetFn's own control edge — the neighbor's line-10 edge is out + // of the [11,15] window after the lower-bound +1 fix. + expect(result.results).toHaveLength(1); + expect(result.results[0].dependent.text).toMatch(/doThing/); + expect(result.results[0].functionLine).toBe(11); + expect(result.results.some((e: any) => /aboveDep/.test(e.dependent.text))).toBe(false); + }); + }); + }, + { + poolAdapter: true, + afterSetup: async (handle) => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const nodeStmts = [ + `CREATE (fn:Function {id: 'func:targetFn', name: 'targetFn', filePath: 'src/adj.ts', startLine: 10, endLine: 14, isExported: true, content: 'function targetFn(x) {}', description: 'adjacency regression'})`, + // targetFn's blocks (fnStartLine segment '11', 1-based startLines 12/13) + `CREATE (b:BasicBlock {id: 'BasicBlock:src/adj.ts:11:0:0', filePath: 'src/adj.ts', startLine: 12, endLine: 12, text: 'if (x)'})`, + `CREATE (b:BasicBlock {id: 'BasicBlock:src/adj.ts:11:0:1', filePath: 'src/adj.ts', startLine: 13, endLine: 13, text: 'doThing();'})`, + // neighbor function's blocks (fnStartLine segment '9', 1-based startLine 10) + `CREATE (b:BasicBlock {id: 'BasicBlock:src/adj.ts:9:0:0', filePath: 'src/adj.ts', startLine: 10, endLine: 10, text: 'if (above)'})`, + `CREATE (b:BasicBlock {id: 'BasicBlock:src/adj.ts:9:0:1', filePath: 'src/adj.ts', startLine: 10, endLine: 10, text: 'aboveDep();'})`, + ]; + for (const s of nodeStmts) await adapter.executePrepared(s, {}); + const cdgEdge = (src: string, dst: string) => + adapter.executePrepared( + `MATCH (a:BasicBlock {id: $src}), (b:BasicBlock {id: $dst}) + CREATE (a)-[:CodeRelation {type: 'CDG', confidence: 1.0, reason: 'T', step: 0}]->(b)`, + { src, dst }, + ); + await cdgEdge('BasicBlock:src/adj.ts:11:0:0', 'BasicBlock:src/adj.ts:11:0:1'); + await cdgEdge('BasicBlock:src/adj.ts:9:0:0', 'BasicBlock:src/adj.ts:9:0:1'); + + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'adj-repo', + path: '/adj/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'adj789', + stats: { files: 1, nodes: 5, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); + +// ─── Block 4: coverage gaps — ambiguous, truncated, Windows-':' path (#2188) ── +// +// Hand-seeded edge cases the M6 review flagged as untested. + +withTestLbugDB( + 'pdg-query-gaps', + (handle) => { + describe('pdg_query coverage gaps (#2188)', () => { + let backend: LocalBackend; + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) throw new Error('LocalBackend not initialized in afterSetup'); + backend = ext._backend; + }); + + it('an ambiguous symbol name returns ranked candidates, not a guess', async () => { + const result = await backend.callTool('pdg_query', { mode: 'controls', target: 'dupFn' }); + expect(result.status).toBe('ambiguous'); + expect(Array.isArray(result.candidates)).toBe(true); + expect(result.candidates.length).toBeGreaterThanOrEqual(2); + for (const c of result.candidates) { + expect(c).toHaveProperty('uid'); + expect(c.name).toBe('dupFn'); + expect(c).toHaveProperty('filePath'); + expect(typeof c.score).toBe('number'); + } + }); + + it('paginates: results capped at limit, total reports the full count, truncated set', async () => { + const result = await backend.callTool('pdg_query', { + mode: 'controls', + target: 'busyFn', + limit: 2, + }); + expect(result).not.toHaveProperty('error'); + expect(result.results).toHaveLength(2); + expect(result.total).toBe(3); + expect(result.truncated).toBe(true); + }); + + it('does not set truncated when the page holds every match', async () => { + const result = await backend.callTool('pdg_query', { + mode: 'controls', + target: 'busyFn', + limit: 50, + }); + expect(result.results).toHaveLength(3); + expect(result.total).toBe(3); + expect(result).not.toHaveProperty('truncated'); + }); + + it("decodes functionLine for a Windows-style filePath containing ':' (split-from-right)", async () => { + const result = await backend.callTool('pdg_query', { mode: 'controls', target: 'winFn' }); + expect(result).not.toHaveProperty('error'); + expect(result.results.length).toBeGreaterThan(0); + // id = BasicBlock:C:/src/win.ts:6:0:0 ⇒ fnLine segment '6' despite the + // ':' in the drive letter (fnLineOf splits from the right). + expect(result.results[0].functionLine).toBe(6); + }); + }); + }, + { + poolAdapter: true, + afterSetup: async (handle) => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const fn = (id: string, name: string, filePath: string, startLine: number, endLine: number) => + adapter.executePrepared( + `CREATE (fn:Function {id: $id, name: $name, filePath: $filePath, startLine: $startLine, endLine: $endLine, isExported: true, content: 'x', description: 'gap fixture'})`, + { id, name, filePath, startLine, endLine }, + ); + const block = (id: string, filePath: string, startLine: number, text: string) => + adapter.executePrepared( + `CREATE (b:BasicBlock {id: $id, filePath: $filePath, startLine: $startLine, endLine: $startLine, text: $text})`, + { id, filePath, startLine, text }, + ); + const cdg = (src: string, dst: string) => + adapter.executePrepared( + `MATCH (a:BasicBlock {id: $src}), (b:BasicBlock {id: $dst}) + CREATE (a)-[:CodeRelation {type: 'CDG', confidence: 1.0, reason: 'T', step: 0}]->(b)`, + { src, dst }, + ); + + // (1) Ambiguous: two functions sharing a name in different files. + await fn('func:dupFn@a', 'dupFn', 'a.ts', 1, 3); + await fn('func:dupFn@b', 'dupFn', 'b.ts', 1, 3); + + // (2) Truncated: busyFn (0-based 10–20 ⇒ window [11,21]); one controller + // block (line 12) with three CDG dependents. + await fn('func:busyFn', 'busyFn', 'busy.ts', 10, 20); + await block('BasicBlock:busy.ts:11:0:0', 'busy.ts', 12, 'if (x)'); + await block('BasicBlock:busy.ts:11:0:1', 'busy.ts', 13, 'a();'); + await block('BasicBlock:busy.ts:11:0:2', 'busy.ts', 14, 'b();'); + await block('BasicBlock:busy.ts:11:0:3', 'busy.ts', 15, 'c();'); + await cdg('BasicBlock:busy.ts:11:0:0', 'BasicBlock:busy.ts:11:0:1'); + await cdg('BasicBlock:busy.ts:11:0:0', 'BasicBlock:busy.ts:11:0:2'); + await cdg('BasicBlock:busy.ts:11:0:0', 'BasicBlock:busy.ts:11:0:3'); + + // (3) Windows-style path with a ':' (drive letter) inside the block id. + await fn('func:winFn', 'winFn', 'C:/src/win.ts', 5, 8); + await block('BasicBlock:C:/src/win.ts:6:0:0', 'C:/src/win.ts', 7, 'if (y)'); + await block('BasicBlock:C:/src/win.ts:6:0:1', 'C:/src/win.ts', 7, 'd();'); + await cdg('BasicBlock:C:/src/win.ts:6:0:0', 'BasicBlock:C:/src/win.ts:6:0:1'); + + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'gaps-repo', + path: '/gaps/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'gap001', + stats: { files: 4, nodes: 12, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); diff --git a/gitnexus/test/integration/taint-explain.test.ts b/gitnexus/test/integration/taint-explain.test.ts index 8d999ccfc..7f500679b 100644 --- a/gitnexus/test/integration/taint-explain.test.ts +++ b/gitnexus/test/integration/taint-explain.test.ts @@ -479,3 +479,84 @@ withTestLbugDB( }, }, ); + +// ─── Block 4: symbol-anchor window correctness (#2188 _explainImpl off-by-one) ── +// +// Hand-seeded with controlled line numbers (no parser dependency). `tailFn` +// occupies 0-based symbol lines 10–14, so its BasicBlocks land on 1-based lines +// 11–15 and the correct anchor window is [symStart+1, symEnd+1] = [11,15]. The +// pre-fix _explainImpl used [symStart, symEnd] = [10,14], which both DROPPED a +// taint source on the function's final line (1-based 15) and LEAKED a neighbor's +// block on the line directly above (1-based 10). One query proves both bounds — +// and FAILS on the pre-fix window (it would return the line-10 neighbor instead). + +withTestLbugDB( + 'taint-explain-anchor-window', + (handle) => { + describe('explain symbol anchoring (#2188 [symStart+1, symEnd+1] window)', () => { + let backend: LocalBackend; + beforeAll(() => { + const ext = handle as typeof handle & { _backend?: LocalBackend }; + if (!ext._backend) throw new Error('LocalBackend not initialized'); + backend = ext._backend; + }); + + it('includes the final-line taint source and excludes the neighbor-above block', async () => { + const result = (await backend.callTool('explain', { target: 'tailFn' })) as { + findings: Array<{ source?: { line?: number } }>; + error?: string; + }; + expect(result).not.toHaveProperty('error'); + // Only tailFn's own final-line (1-based 15) taint source survives; the + // neighbor's line-10 block is below the [11,15] window (lower-bound +1). + expect(result.findings).toHaveLength(1); + expect(result.findings[0].source?.line).toBe(15); + expect(result.findings.some((f) => f.source?.line === 10)).toBe(false); + }); + }); + }, + { + poolAdapter: true, + afterSetup: async (handle) => { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + // tailFn: 0-based span 10–14 ⇒ 1-based blocks on 11–15, window [11,15]. + await adapter.executePrepared( + `CREATE (fn:Function {id: 'func:tailFn', name: 'tailFn', filePath: 'anchor.ts', startLine: 10, endLine: 14, isExported: true, content: 'function tailFn() {}', description: 'anchor-window regression'})`, + {}, + ); + const block = (id: string, startLine: number, text: string) => + adapter.executePrepared( + `CREATE (b:BasicBlock {id: $id, filePath: 'anchor.ts', startLine: $startLine, endLine: $startLine, text: $text})`, + { id, startLine, text }, + ); + // tailFn's source/sink on its FINAL line (1-based 15 = endLine 14 + 1). + await block('BasicBlock:anchor.ts:11:0:5', 15, 'const x = req.body;'); + await block('BasicBlock:anchor.ts:11:0:6', 15, 'exec(x);'); + // a neighbor function's block on the line directly ABOVE tailFn (1-based 10). + await block('BasicBlock:anchor.ts:9:0:0', 10, 'const y = other();'); + await block('BasicBlock:anchor.ts:9:0:1', 10, 'use(y);'); + const tainted = (src: string, dst: string, reason: string) => + adapter.executePrepared( + `MATCH (a:BasicBlock {id: $src}), (b:BasicBlock {id: $dst}) + CREATE (a)-[:CodeRelation {type: 'TAINTED', confidence: 1.0, reason: $reason, step: 0}]->(b)`, + { src, dst, reason }, + ); + await tainted('BasicBlock:anchor.ts:11:0:5', 'BasicBlock:anchor.ts:11:0:6', 'tail'); + await tainted('BasicBlock:anchor.ts:9:0:0', 'BasicBlock:anchor.ts:9:0:1', 'neighbor'); + + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'anchor-repo', + path: '/anchor/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'aw0001', + stats: { files: 1, nodes: 5, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as any)._backend = backend; + }, + }, +); diff --git a/gitnexus/test/unit/ai-context.test.ts b/gitnexus/test/unit/ai-context.test.ts index 747bcec1a..30a9441c1 100644 --- a/gitnexus/test/unit/ai-context.test.ts +++ b/gitnexus/test/unit/ai-context.test.ts @@ -138,8 +138,7 @@ describe('generateAIContextFiles', () => { const content = generateGitNexusContent( 'TestProject', { nodes: 50, edges: 100, processes: 5 }, - undefined, - ['TeamGroup'], + { groupNames: ['TeamGroup'] }, ); expect(content).toContain('## Cross-Repo Groups'); expect(content).toContain('node .gitnexus/run.cjs group list'); @@ -150,6 +149,20 @@ describe('generateAIContextFiles', () => { expect(content).not.toMatch(/npx gitnexus group/); }); + it('gates the pdg_query line on hasPdg (#2086 M6 — no existing taint gate to mirror)', () => { + const stats = { nodes: 50, edges: 100, processes: 5 }; + // hasPdg=true → the pdg_query line is present. + const withPdg = generateGitNexusContent('PdgProject', stats, { hasPdg: true }); + expect(withPdg).toContain('pdg_query'); + expect(withPdg).toContain('under what condition does X run'); + // hasPdg omitted (default false) → no pdg_query line; a non-pdg index must + // not advertise a tool that only returns a "no PDG layer" note. + const withoutPdg = generateGitNexusContent('PlainProject', stats); + expect(withoutPdg).not.toContain('pdg_query'); + // the unconditional explain line stays regardless of the pdg flag. + expect(withoutPdg).toContain('explain('); + }); + it('degrades gracefully when the runner copy fails (#1945)', async () => { // A read-only/full-disk storage dir must not abort generation. The copy is // best-effort + logged; the generated docs still carry the inline bootstrap @@ -893,16 +906,7 @@ Indexed as **placeholder** (1 symbols, 1 relationships, 1 execution flows). Cust it('generated regression-compare example uses the configured default branch (#243)', () => { const stats = { nodes: 50, edges: 100, processes: 5 }; - const develop = generateGitNexusContent( - 'P', - stats, - undefined, - undefined, - undefined, - undefined, - undefined, - 'develop', - ); + const develop = generateGitNexusContent('P', stats, { defaultBranch: 'develop' }); expect(develop).toContain('base_ref: "develop"'); expect(develop).not.toContain('base_ref: "main"'); }); @@ -930,32 +934,14 @@ Indexed as **placeholder** (1 symbols, 1 relationships, 1 execution flows). Cust it('JSON-escapes a markdown/quote-bearing branch so it cannot break the code span (#243)', () => { // A branch name with a double-quote must be JSON-escaped, not concatenated // raw, so it stays inside the inline code span. - const content = generateGitNexusContent( - 'P', - { nodes: 1 }, - undefined, - undefined, - undefined, - undefined, - undefined, - 'we"ird', - ); + const content = generateGitNexusContent('P', { nodes: 1 }, { defaultBranch: 'we"ird' }); expect(content).toContain('base_ref: "we\\"ird"'); }); it('a backtick branch cannot break the generated Markdown code span (#1996 P1)', () => { // The branch is embedded inside a backtick inline-code span; a stray // backtick would close it early. markdownSafeBranch strips it at the sink. - const content = generateGitNexusContent( - 'P', - { nodes: 1 }, - undefined, - undefined, - undefined, - undefined, - undefined, - 'main`evil', - ); + const content = generateGitNexusContent('P', { nodes: 1 }, { defaultBranch: 'main`evil' }); const line = content.split('\n').find((l) => l.includes('base_ref'))!; // Even backtick count ⇒ every span is balanced (the regression line opens // and closes exactly one). diff --git a/gitnexus/test/unit/analyze-no-stats-bridge.test.ts b/gitnexus/test/unit/analyze-no-stats-bridge.test.ts index 22e58a061..629bfac02 100644 --- a/gitnexus/test/unit/analyze-no-stats-bridge.test.ts +++ b/gitnexus/test/unit/analyze-no-stats-bridge.test.ts @@ -168,6 +168,8 @@ describe('analyzeCommand commander → runFullAnalysis noStats bridge (#1477)', // #243: resolved default branch threaded into the --skills regen path. defaultBranch: 'main', noStats: true, + // #2086 M6: the --pdg gate is threaded too; false here (no --pdg flag). + hasPdg: false, }); } finally { exitSpy.mockRestore(); diff --git a/gitnexus/test/unit/cfg/control-dependence.test.ts b/gitnexus/test/unit/cfg/control-dependence.test.ts new file mode 100644 index 000000000..e307fb69b --- /dev/null +++ b/gitnexus/test/unit/cfg/control-dependence.test.ts @@ -0,0 +1,413 @@ +import { describe, it, expect } from 'vitest'; +import { + computeControlDependence, + type ControlDepEdge, + type CdgLabel, +} from '../../../src/core/ingestion/cfg/control-dependence.js'; +import { + computePostDominators, + postDominates, +} from '../../../src/core/ingestion/cfg/post-dominators.js'; +import type { + BasicBlockData, + CfgEdgeData, + CfgEdgeKind, + FunctionCfg, +} from '../../../src/core/ingestion/cfg/types.js'; +import Parser from 'tree-sitter'; +import TypeScript from 'tree-sitter-typescript'; +import { collectFunctionCfgs } from '../../../src/core/ingestion/cfg/collect.js'; +import { getProvider } from '../../../src/core/ingestion/languages/index.js'; +import { SupportedLanguages } from '../../../src/config/supported-languages.js'; + +// U3 (#2085 M5) — Ferrante §3.1.1 control dependence over the post-dom tree. +// Hand-built CFG literals plus real-parser regression tests. The labelled +// expected edge sets ARE the spec; the property test (AC2) cross-checks the +// tree-walk against a reference that computes post-dominance INDEPENDENTLY (by +// node-removal reachability, sharing NO code with post-dominators.ts), so a +// post-dominator *direction* bug cannot pass both (#2188 F4). + +// ── hand-built CFG helper (edges carry a kind so labels can be asserted) ───── + +function mkCfg( + blockCount: number, + edges: [number, number, CfgEdgeKind][], + opts: { entry?: number; exit?: number } = {}, +): FunctionCfg { + const entry = opts.entry ?? 0; + const exit = opts.exit ?? blockCount - 1; + const blocks: BasicBlockData[] = Array.from({ length: blockCount }, (_, i) => ({ + index: i, + startLine: i + 1, + endLine: i + 1, + text: '', + kind: i === entry ? 'entry' : i === exit ? 'exit' : 'normal', + })); + const cfgEdges: CfgEdgeData[] = edges.map(([from, to, kind]) => ({ from, to, kind })); + return { + filePath: 't.ts', + functionStartLine: 1, + functionStartColumn: 0, + entryIndex: entry, + exitIndex: exit, + blocks, + edges: cfgEdges, + }; +} + +const ser = (e: ControlDepEdge): string => `${e.controllerBlock}->${e.dependentBlock}:${e.label}`; +const serAll = (edges: readonly ControlDepEdge[]): string[] => edges.map(ser); + +/** + * Build a successor adjacency list for a CFG (in-range edges only). + */ +function succsOf(cfg: FunctionCfg): number[][] { + const n = cfg.blocks.length; + const succs: number[][] = Array.from({ length: n }, () => []); + for (const e of cfg.edges) + if (e.from >= 0 && e.from < n && e.to >= 0 && e.to < n) succs[e.from].push(e.to); + return succs; +} + +/** + * INDEPENDENT post-dominance via node-removal reachability — shares NO code with + * post-dominators.ts (that is the whole point: it must catch a CHK *direction* + * bug that a shared-substrate reference would mirror, #2188 F4). `p` + * post-dominates `b` iff every path from `b` to EXIT passes through `p`: + * reflexive (`p === b`), else true exactly when EXIT is unreachable from `b` + * once `p` is removed (AND `b` can reach EXIT at all). Defined only for the + * exit-reachable fixtures used below — the exit-unreachable case is the known + * unsound region (#2188 F2) and is deliberately excluded from the AC2 set. + */ +function independentPostDom(cfg: FunctionCfg, succs: number[][], p: number, b: number): boolean { + if (p === b) return true; + const exit = cfg.exitIndex; + const reach = (avoid: number): boolean => { + if (b === avoid) return false; + const seen = new Set([b]); + const stack = [b]; + while (stack.length) { + const x = stack.pop()!; + if (x === exit) return true; + for (const y of succs[x]) { + if (y === avoid || seen.has(y)) continue; + seen.add(y); + stack.push(y); + } + } + return false; + }; + // p post-dominates b ⇔ b reaches EXIT, but cannot reach it with p removed. + return reach(-1) && !reach(p); +} + +/** + * Reference control-dependence pairs from the Ferrante definition, using the + * INDEPENDENT post-dominance above: N is control-dependent on A iff some CFG + * edge A→B has N post-dominating B while N does NOT strictly post-dominate A. + * Label-agnostic (distinct "A->N" pairs) — the pure definition has no sense. + */ +function referencePairs(cfg: FunctionCfg): Set { + const n = cfg.blocks.length; + const succs = succsOf(cfg); + const pd = (x: number, y: number): boolean => independentPostDom(cfg, succs, x, y); + const pairs = new Set(); + for (let a = 0; a < n; a++) { + for (const b of succs[a]) { + if (pd(b, a)) continue; // edge is not a control point + for (let nn = 0; nn < n; nn++) { + const nPostDomB = pd(nn, b); + const nStrictlyPostDomA = nn !== a && pd(nn, a); + if (nPostDomB && !nStrictlyPostDomA) pairs.add(`${a}->${nn}`); + } + } + } + return pairs; +} + +describe('computeControlDependence — Ferrante §3.1.1', () => { + it('diamond: each arm is control-dependent on the branch with its own T/F label', () => { + // 0(branch) → 1(then, T), 2(else, F); 1,2 → 3(join) → 4(exit) + const cfg = mkCfg(5, [ + [0, 1, 'cond-true'], + [0, 2, 'cond-false'], + [1, 3, 'seq'], + [2, 3, 'seq'], + [3, 4, 'seq'], + ]); + const { edges } = computeControlDependence(cfg); + expect(serAll(edges).sort()).toEqual(['0->1:T', '0->2:F']); + // the join (3) post-dominates the branch, so it depends on nothing + expect(edges.some((e) => e.dependentBlock === 3)).toBe(false); + }); + + it('guard clause: the post-guard body is control-dependent on the guard (the #559/#2086 case)', () => { + // function f(x){ if(!ok(x)) return; use(x); } + // 0(entry) → 1(guard); 1 → 2(return, T) , 1 → 3(use, F); 2,3 → 4(exit) + const cfg = mkCfg(5, [ + [0, 1, 'seq'], + [1, 2, 'cond-true'], // !ok(x) → return + [1, 3, 'cond-false'], // else → use(x) + [2, 4, 'return'], + [3, 4, 'seq'], + ]); + const { edges } = computeControlDependence(cfg); + // use(x) (block 3) runs only when the guard condition is false → label 'F' + expect(serAll(edges).sort()).toEqual(['1->2:T', '1->3:F']); + }); + + it('straight-line function (no branches) has no control dependence', () => { + const cfg = mkCfg(3, [ + [0, 1, 'seq'], + [1, 2, 'seq'], + ]); + expect(computeControlDependence(cfg).edges).toEqual([]); + }); + + it('while loop: the body depends on the header, and the header is control-dependent on itself', () => { + // 0(entry) → 1(header); 1 → 2(body, T) , 1 → 3(exit, F); 2 → 1 (back-edge) + const cfg = mkCfg(4, [ + [0, 1, 'seq'], + [1, 2, 'cond-true'], + [2, 1, 'loop-back'], + [1, 3, 'cond-false'], + ]); + const { edges } = computeControlDependence(cfg); + // body(2) control-dep on header(1); header(1) control-dep on itself (the + // loop predicate gates its own re-execution — standard PDG behavior). + expect(serAll(edges).sort()).toEqual(['1->1:T', '1->2:T']); + }); + + it('switch: every case body is control-dependent on the dispatch (all T in M5)', () => { + // 0(entry) → 1(dispatch); 1 → 2,3,4 (cases); 2,3,4 → 5(exit) + const cfg = mkCfg(6, [ + [0, 1, 'seq'], + [1, 2, 'switch-case'], + [1, 3, 'switch-case'], + [1, 4, 'switch-case'], + [2, 5, 'break'], + [3, 5, 'break'], + [4, 5, 'break'], + ]); + const { edges } = computeControlDependence(cfg); + expect(serAll(edges).sort()).toEqual(['1->2:T', '1->3:T', '1->4:T']); + }); + + it('exit-less loop (KTD5): terminates and stays in-range, but the result is KNOWN-UNSOUND (#2188 F2)', () => { + // No block can reach EXIT (block 3), so every ipdom is NO_IPDOM and the + // Ferrante walk degenerates to one edge per control point. The termination / + // in-range invariants MUST hold (the walk hits NO_IPDOM immediately). The + // emitted dependence SET, however, is NOT a sound over-approximation: it both + // drops real dependences and invents spurious ones in exit-unreachable + // regions (#2188 F2). This test pins the degenerate output to document that + // behavior, NOT to bless it; the labels here are likewise indeterminate + // (no controller carries an explicit cond-true/cond-false arm, so the + // fall-through complement resolves to 'F'). The current TS visitor never + // produces such a region (every loop gets a structural header→loopExit edge). + const cfg = mkCfg(4, [ + [0, 1, 'seq'], + [1, 2, 'seq'], + [2, 1, 'loop-back'], + ]); + const { edges } = computeControlDependence(cfg); + expect(serAll(edges).sort()).toEqual(['0->1:F', '1->2:F', '2->1:F']); + for (const e of edges) { + expect(e.controllerBlock).toBeGreaterThanOrEqual(0); + expect(e.controllerBlock).toBeLessThan(cfg.blocks.length); + expect(e.dependentBlock).toBeGreaterThanOrEqual(0); + expect(e.dependentBlock).toBeLessThan(cfg.blocks.length); + } + }); + + it('is deterministic (stable sorted order across runs)', () => { + const make = (): FunctionCfg => + mkCfg(5, [ + [0, 1, 'cond-true'], + [0, 2, 'cond-false'], + [1, 3, 'seq'], + [2, 3, 'seq'], + [3, 4, 'seq'], + ]); + expect(serAll(computeControlDependence(make()).edges)).toEqual( + serAll(computeControlDependence(make()).edges), + ); + }); + + describe('AC2 — a control dependence exists iff post-dominance fails for the branch', () => { + const fixtures: Record = { + diamond: mkCfg(5, [ + [0, 1, 'cond-true'], + [0, 2, 'cond-false'], + [1, 3, 'seq'], + [2, 3, 'seq'], + [3, 4, 'seq'], + ]), + guard: mkCfg(5, [ + [0, 1, 'seq'], + [1, 2, 'cond-true'], + [1, 3, 'cond-false'], + [2, 4, 'return'], + [3, 4, 'seq'], + ]), + loop: mkCfg(4, [ + [0, 1, 'seq'], + [1, 2, 'cond-true'], + [2, 1, 'loop-back'], + [1, 3, 'cond-false'], + ]), + // NOTE: the exit-unreachable case is deliberately NOT an AC2 fixture — its + // dependence set is unsound (#2188 F2), so asserting walk == independent + // reference would (correctly) fail. It has its own characterization test + // above that documents the degenerate behavior. + // nested if: outer branch (0) → inner branch (1) or outer-else (5); + // inner branch → 2/3 → inner join (4); 4 and 5 → outer join (6, exit). + nestedIf: mkCfg( + 7, + [ + [0, 1, 'cond-true'], + [0, 5, 'cond-false'], + [1, 2, 'cond-true'], + [1, 3, 'cond-false'], + [2, 4, 'seq'], + [3, 4, 'seq'], + [4, 6, 'seq'], + [5, 6, 'seq'], + ], + { entry: 0, exit: 6 }, + ), + switchStmt: mkCfg(6, [ + [0, 1, 'seq'], + [1, 2, 'switch-case'], + [1, 3, 'switch-case'], + [1, 4, 'switch-case'], + [2, 5, 'break'], + [3, 5, 'break'], + [4, 5, 'break'], + ]), + }; + + it.each(Object.keys(fixtures))( + '%s: tree-walk pair set equals the brute-force reference', + (name) => { + const cfg = fixtures[name]; + const { edges } = computeControlDependence(cfg); + const walkPairs = new Set(edges.map((e) => `${e.controllerBlock}->${e.dependentBlock}`)); + expect(walkPairs).toEqual(referencePairs(cfg)); + }, + ); + + it.each(Object.keys(fixtures))( + '%s: for every CFG edge, it yields a dependent IFF the target does not post-dominate the source', + (name) => { + const cfg = fixtures[name]; + const tree = computePostDominators(cfg); + const { edges } = computeControlDependence(cfg); + for (const e of cfg.edges) { + const failsPostDom = !postDominates(tree, e.to, e.from); + // does THIS edge's source appear as a controller with at least one + // dependent reachable from its target? Equivalent statement of AC2: + // post-dominance failing for (from→to) ⇔ `from` is a control point. + const fromIsControlPoint = edges.some((c) => c.controllerBlock === e.from); + if (failsPostDom) { + expect( + fromIsControlPoint, + `${name}: edge ${e.from}->${e.to} should make ${e.from} a control point`, + ).toBe(true); + } + // and a self-post-dominating edge (to post-dominates from) can never + // be the SOLE reason a block is a control point: if from has only + // post-dominating successors it controls nothing. + } + }, + ); + }); +}); + +describe('computeControlDependence — maxEdges materialization ceiling (#2188)', () => { + // A switch dispatch yields three deduped CDG edges (1->2/3/4, all 'T'). + const switchCfg = (): FunctionCfg => + mkCfg(6, [ + [0, 1, 'seq'], + [1, 2, 'switch-case'], + [1, 3, 'switch-case'], + [1, 4, 'switch-case'], + [2, 5, 'break'], + [3, 5, 'break'], + [4, 5, 'break'], + ]); + + it('stops at the ceiling and reports truncated (deterministic prefix)', () => { + const r = computeControlDependence(switchCfg(), undefined, 2); + expect(r.truncated).toBe(true); + expect(r.edges).toHaveLength(2); + // the prefix is still sorted/deduped, a valid subset of the full result + for (const e of r.edges) expect(['T', 'F']).toContain(e.label); + }); + + it('maxEdges of 0 means unbounded (full result, not truncated)', () => { + const r = computeControlDependence(switchCfg(), undefined, 0); + expect(r.truncated).toBe(false); + expect(serAll(r.edges).sort()).toEqual(['1->2:T', '1->3:T', '1->4:T']); + }); + + it('a ceiling at/above the true count is not truncated', () => { + const r = computeControlDependence(switchCfg(), undefined, 3); + expect(r.truncated).toBe(false); + expect(r.edges).toHaveLength(3); + }); +}); + +describe('#2188 F1 — branch-label correctness on the REAL TS visitor (regression)', () => { + // The label is the AC3 "under what condition does X run?" answer. The bug: + // branchSense inferred it from the edge KIND alone, but the M1 visitor wires a + // condition's fall-through FALSE arm as `seq`/`loop-back` (not `cond-false`), + // so guard clauses / loop break got 'T' instead of 'F'. These tests run the + // REAL parser+visitor (the hand-built tests above used a fictional `cond-false` + // edge and could not catch the regression). + const tsVisitor = getProvider(SupportedLanguages.TypeScript).cfgVisitor; + const parser = new Parser(); + if (tsVisitor) parser.setLanguage(TypeScript.typescript); + + function cdgOf(code: string): { cfg: FunctionCfg; edges: readonly ControlDepEdge[] } { + if (!tsVisitor) throw new Error('no cfgVisitor'); + const cfgs = collectFunctionCfgs(parser.parse(code).rootNode, tsVisitor, 't.ts').cfgs; + expect(cfgs.length).toBe(1); + return { cfg: cfgs[0], edges: computeControlDependence(cfgs[0]).edges }; + } + const labelOf = ( + edges: readonly ControlDepEdge[], + controller: number, + dependent: number, + ): CdgLabel | undefined => + edges.find((e) => e.controllerBlock === controller && e.dependentBlock === dependent)?.label; + + it("guard clause: post-guard body runs on the guard's FALSE (seq) arm → 'F'", () => { + const { cfg, edges } = cdgOf(`function f(x){ if (!ok(x)) return; use(x); }`); + const guard = cfg.blocks.find((b) => b.text.includes('ok(x)'))!; + const use = cfg.blocks.find((b) => b.text.includes('use(x)'))!; + expect(labelOf(edges, guard.index, use.index)).toBe('F'); + }); + + it("do/while: body runs on the bottom-test's TRUE (loop-back) arm → 'T'", () => { + const { cfg, edges } = cdgOf(`function f(){ do { body(); } while (c()); }`); + const test = cfg.blocks.find((b) => b.text.includes('c()'))!; + const body = cfg.blocks.find((b) => b.text.includes('body()'))!; + expect(labelOf(edges, test.index, body.index)).toBe('T'); + }); + + it("while+break: post-break tail runs on the if's FALSE (seq) arm → 'F'", () => { + const { cfg, edges } = cdgOf(`function f(o,i){ while (o) { if (i) break; tail(); } }`); + const ifCond = cfg.blocks.find((b) => b.text === '(i)')!; + const tail = cfg.blocks.find((b) => b.text.includes('tail()'))!; + expect(labelOf(edges, ifCond.index, tail.index)).toBe('F'); + }); + + it("if/else still labels both arms correctly (no regression) → then 'T', else 'F'", () => { + const { cfg, edges } = cdgOf(`function f(x){ if (x) { a(); } else { b(); } }`); + const cond = cfg.blocks.find((b) => b.text === '(x)')!; + const thenB = cfg.blocks.find((b) => b.text.includes('a()'))!; + const elseB = cfg.blocks.find((b) => b.text.includes('b()'))!; + expect(labelOf(edges, cond.index, thenB.index)).toBe('T'); + expect(labelOf(edges, cond.index, elseB.index)).toBe('F'); + }); +}); diff --git a/gitnexus/test/unit/cfg/post-dominators.test.ts b/gitnexus/test/unit/cfg/post-dominators.test.ts new file mode 100644 index 000000000..abc257b41 --- /dev/null +++ b/gitnexus/test/unit/cfg/post-dominators.test.ts @@ -0,0 +1,230 @@ +import { describe, it, expect } from 'vitest'; +import { + computePostDominators, + isExitReachableFromAllBlocks, + postDominates, +} from '../../../src/core/ingestion/cfg/post-dominators.js'; +import type { + BasicBlockData, + CfgEdgeData, + FunctionCfg, +} from '../../../src/core/ingestion/cfg/types.js'; + +// U2 (#2085 M5) — post-dominators on the EXIT-rooted reverse CFG. Pinned on +// hand-built FunctionCfg literals with zero tree-sitter dependency, mirroring +// reaching-defs.test.ts / cfg-builder.test.ts. Post-dominance has crisp, +// well-known expected outputs per topology, so the expected ipdom values ARE +// the spec for the Cooper–Harvey–Kennedy iterative dominators implementation. + +// ── hand-built CFG helper ─────────────────────────────────────────────────── + +function mkCfg( + blockCount: number, + edges: [number, number][], + opts: { entry?: number; exit?: number } = {}, +): FunctionCfg { + const entry = opts.entry ?? 0; + const exit = opts.exit ?? blockCount - 1; + const blocks: BasicBlockData[] = Array.from({ length: blockCount }, (_, i) => ({ + index: i, + startLine: i + 1, + endLine: i + 1, + text: '', + kind: i === entry ? 'entry' : i === exit ? 'exit' : 'normal', + })); + const cfgEdges: CfgEdgeData[] = edges.map(([from, to]) => ({ from, to, kind: 'seq' })); + return { + filePath: 't.ts', + functionStartLine: 1, + functionStartColumn: 0, + entryIndex: entry, + exitIndex: exit, + blocks, + edges: cfgEdges, + }; +} + +const NONE = -1; + +describe('computePostDominators — ipdom on the reverse CFG', () => { + it('linear chain: each block is post-dominated by its successor', () => { + // 0(entry) → 1 → 2 → 3(exit) + const cfg = mkCfg(4, [ + [0, 1], + [1, 2], + [2, 3], + ]); + const tree = computePostDominators(cfg); + expect(tree.ipdom[3]).toBe(NONE); // exit (root) has no post-dominator above it + expect(tree.ipdom[2]).toBe(3); + expect(tree.ipdom[1]).toBe(2); + expect(tree.ipdom[0]).toBe(1); + + expect(postDominates(tree, 3, 0)).toBe(true); + expect(postDominates(tree, 2, 0)).toBe(true); + expect(postDominates(tree, 0, 2)).toBe(false); + expect(postDominates(tree, 1, 1)).toBe(true); // reflexive + }); + + it('diamond (if/else with join): the join post-dominates the branch, arms do not', () => { + // 0(branch) → 1(then), 2(else); 1,2 → 3(join) → 4(exit) + const cfg = mkCfg(5, [ + [0, 1], + [0, 2], + [1, 3], + [2, 3], + [3, 4], + ]); + const tree = computePostDominators(cfg); + expect(tree.ipdom[4]).toBe(NONE); + expect(tree.ipdom[3]).toBe(4); + expect(tree.ipdom[1]).toBe(3); + expect(tree.ipdom[2]).toBe(3); + expect(tree.ipdom[0]).toBe(3); // join post-dominates the branch, not an arm + + expect(postDominates(tree, 3, 0)).toBe(true); + expect(postDominates(tree, 1, 0)).toBe(false); // `then` does NOT post-dominate the branch + expect(postDominates(tree, 2, 0)).toBe(false); + expect(postDominates(tree, 3, 1)).toBe(true); + }); + + it('while loop: the header post-dominates the body; back-edge does not break the tree', () => { + // 0(entry) → 1(header) → 2(body) → 1; header → 3(exit) + const cfg = mkCfg(4, [ + [0, 1], + [1, 2], + [2, 1], + [1, 3], + ]); + const tree = computePostDominators(cfg); + expect(tree.ipdom[3]).toBe(NONE); + expect(tree.ipdom[1]).toBe(3); + expect(tree.ipdom[2]).toBe(1); // every path from body to exit goes through the header + expect(tree.ipdom[0]).toBe(1); + + expect(postDominates(tree, 1, 2)).toBe(true); + expect(postDominates(tree, 3, 2)).toBe(true); + expect(postDominates(tree, 2, 1)).toBe(false); + }); + + it('multiple returns collapsing to a single EXIT: exit post-dominates everything', () => { + // 0(entry) → 1(cond) → 2(return), 3(return); 2,3 → 4(exit) + const cfg = mkCfg(5, [ + [0, 1], + [1, 2], + [1, 3], + [2, 4], + [3, 4], + ]); + const tree = computePostDominators(cfg); + expect(tree.ipdom[4]).toBe(NONE); + expect(tree.ipdom[1]).toBe(4); + expect(tree.ipdom[2]).toBe(4); + expect(tree.ipdom[3]).toBe(4); + expect(tree.ipdom[0]).toBe(1); + for (const b of [0, 1, 2, 3]) expect(postDominates(tree, 4, b)).toBe(true); + }); + + it('exit-less infinite loop: blocks that cannot reach EXIT have no post-dominator (KTD5)', () => { + // 0(entry) → 1 → 2 → 1 (no edge ever reaches exit block 3) + const cfg = mkCfg(4, [ + [0, 1], + [1, 2], + [2, 1], + ]); + const tree = computePostDominators(cfg); + expect(tree.ipdom[3]).toBe(NONE); // exit itself + expect(tree.ipdom[0]).toBe(NONE); // cannot reach exit + expect(tree.ipdom[1]).toBe(NONE); + expect(tree.ipdom[2]).toBe(NONE); + // No post-dominator means only reflexive post-dominance, and the climb must + // terminate (no infinite loop) even on the cycle. + expect(postDominates(tree, 3, 0)).toBe(false); + expect(postDominates(tree, 0, 0)).toBe(true); + expect(postDominates(tree, 1, 2)).toBe(false); + }); + + it('trivial single-block function (entry === exit) does not crash', () => { + const cfg = mkCfg(1, [], { entry: 0, exit: 0 }); + const tree = computePostDominators(cfg); + expect(tree.ipdom[0]).toBe(NONE); + expect(postDominates(tree, 0, 0)).toBe(true); + }); + + it('is deterministic across runs', () => { + const make = (): FunctionCfg => + mkCfg(5, [ + [0, 1], + [0, 2], + [1, 3], + [2, 3], + [3, 4], + ]); + const a = computePostDominators(make()); + const b = computePostDominators(make()); + expect(a.ipdom).toEqual(b.ipdom); + }); +}); + +// ── post-dominance soundness precondition (#2188 review) ──────────────────── +// EXIT must be reachable (forward) from every block reachable from ENTRY, else +// the EXIT-rooted reverse walk degenerates and CDG is unsound. The current TS +// visitor always satisfies this; the guard protects future / hand-built CFGs. +describe('isExitReachableFromAllBlocks', () => { + it('holds for a normal single-EXIT diamond (every block reaches EXIT)', () => { + const cfg = mkCfg(5, [ + [0, 1], + [0, 2], + [1, 3], + [2, 3], + [3, 4], + ]); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('holds for a loop whose header has a structural edge to EXIT', () => { + // 0=entry → 1=header; header → 2=body → back to header; header → 3=exit. + const cfg = mkCfg(4, [ + [0, 1], + [1, 2], + [2, 1], + [1, 3], + ]); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('fails when an entry-reachable region cannot reach EXIT (exit-less loop)', () => { + // 0=entry → 1; 1↔2 spin forever with no edge to 3=exit. EXIT is unreachable + // from the {1,2} region → post-dominance would be unsound there. + const cfg = mkCfg(4, [ + [0, 1], + [1, 2], + [2, 1], + ]); + expect(isExitReachableFromAllBlocks(cfg)).toBe(false); + }); + + it('fails for the review counterexample (A→B, A→L, B→X→L→A; EXIT disconnected)', () => { + // Indices: 0=A(entry), 1=B, 2=X, 3=L, 4=EXIT (disconnected). The A/B/X/L + // cycle never reaches EXIT, so the precondition must reject it. + const cfg = mkCfg(5, [ + [0, 1], + [0, 3], + [1, 2], + [2, 3], + [3, 0], + ]); + expect(isExitReachableFromAllBlocks(cfg)).toBe(false); + }); + + it('ignores blocks unreachable from ENTRY (they need not reach EXIT)', () => { + // 0=entry → 1=exit directly; 2 is an island unreachable from entry. The + // island does not violate the precondition (it is never analyzed). + const cfg = mkCfg(3, [[0, 1]], { entry: 0, exit: 1 }); + expect(isExitReachableFromAllBlocks(cfg)).toBe(true); + }); + + it('holds for the single-block CFG (entry === exit)', () => { + expect(isExitReachableFromAllBlocks(mkCfg(1, [], { entry: 0, exit: 0 }))).toBe(true); + }); +}); diff --git a/gitnexus/test/unit/pdg-mode-flip.test.ts b/gitnexus/test/unit/pdg-mode-flip.test.ts index 572007a1c..5e8dbe2f9 100644 --- a/gitnexus/test/unit/pdg-mode-flip.test.ts +++ b/gitnexus/test/unit/pdg-mode-flip.test.ts @@ -140,6 +140,42 @@ describe('pdgModeMismatch — M3→M4 interproc-cap stamp upgrade (#2084 review }); }); +describe('pdgModeMismatch — pre-M5→M5 CDG-cap stamp upgrade (#2085 M5, pure)', () => { + it('resolvePdgConfig stamps the resolved CDG cap', async () => { + const { resolvePdgConfig } = await import('../../src/core/run-analyze.js'); + const stamp = resolvePdgConfig({ pdg: true }); + expect(stamp?.maxCdgEdgesPerFunction).toBe(5000); + }); + + it('a pre-M5 stamp (no CDG key) mismatches a CDG-aware request — upgrade forces full writeback', async () => { + const { pdgModeMismatch } = await import('../../src/core/run-analyze.js'); + // What an M4-era run wrote: every cap through the interproc set + model + // digest, but NO maxCdgEdgesPerFunction. The key-union comparator sees + // 5000 !== undefined and trips the full writeback that materialises CDG + // edges for every file without --force. + const m4Stamp = { + maxFunctionLines: 2000, + maxEdgesPerFunction: 5000, + maxReachingDefEdgesPerFunction: 4000, + maxTaintFindingsPerFunction: 200, + maxTaintHops: 32, + maxInterprocFindings: 2000, + maxInterprocHops: 32, + maxInterprocEdges: 1000, + taintModelVersion, + }; + expect(pdgModeMismatch(m4Stamp, { pdg: true })).toBe(true); + }); + + it('a CDG cap change alone trips the mismatch', async () => { + const { pdgModeMismatch, resolvePdgConfig } = await import('../../src/core/run-analyze.js'); + const stamp = resolvePdgConfig({ pdg: true }); + expect(pdgModeMismatch(stamp, { pdg: true, pdgMaxCdgEdgesPerFunction: 10 })).toBe(true); + // explicit default ≡ default (resolution before comparison) + expect(pdgModeMismatch(stamp, { pdg: true, pdgMaxCdgEdgesPerFunction: 5000 })).toBe(false); + }); +}); + describe('detect_changes BasicBlock exclusion (#2082 U7)', () => { it('the symbol-overlap id-prefix filter excludes exactly the BasicBlock rows', async () => { const repo = await setupMiniRepo(); @@ -215,6 +251,7 @@ describe('runFullAnalysis — pdg-mode flip (#2099 F1)', () => { maxFunctionLines: 2000, maxEdgesPerFunction: 5000, maxReachingDefEdgesPerFunction: 4000, + maxCdgEdgesPerFunction: 5000, maxTaintFindingsPerFunction: 200, maxTaintHops: 32, maxInterprocFindings: 2000, @@ -270,6 +307,7 @@ describe('runFullAnalysis — pdg-mode flip (#2099 F1)', () => { maxFunctionLines: 2000, maxEdgesPerFunction: 1, maxReachingDefEdgesPerFunction: 4000, + maxCdgEdgesPerFunction: 5000, maxTaintFindingsPerFunction: 200, maxTaintHops: 32, maxInterprocFindings: 2000, diff --git a/gitnexus/test/unit/run-analyze.test.ts b/gitnexus/test/unit/run-analyze.test.ts index e1f442321..d89114360 100644 --- a/gitnexus/test/unit/run-analyze.test.ts +++ b/gitnexus/test/unit/run-analyze.test.ts @@ -340,6 +340,7 @@ describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { maxFunctionLines: 2000, maxEdgesPerFunction: 5000, maxReachingDefEdgesPerFunction: 4000, + maxCdgEdgesPerFunction: 5000, maxTaintFindingsPerFunction: 200, maxTaintHops: 32, maxInterprocFindings: 2000, @@ -365,6 +366,7 @@ describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { pdgMaxFunctionLines: 0, pdgMaxEdgesPerFunction: 0, pdgMaxReachingDefEdgesPerFunction: 0, + pdgMaxCdgEdgesPerFunction: 0, pdgMaxTaintFindingsPerFunction: 0, pdgMaxTaintHops: 0, pdgMaxInterprocFindings: 0, @@ -375,6 +377,7 @@ describe('pdgModeMismatch / resolvePdgConfig (#2099 F1)', () => { maxFunctionLines: 0, maxEdgesPerFunction: 0, maxReachingDefEdgesPerFunction: 0, + maxCdgEdgesPerFunction: 0, maxTaintFindingsPerFunction: 0, maxTaintHops: 0, maxInterprocFindings: 0, diff --git a/gitnexus/test/unit/schema.test.ts b/gitnexus/test/unit/schema.test.ts index 33f18ed01..0dd55c4f2 100644 --- a/gitnexus/test/unit/schema.test.ts +++ b/gitnexus/test/unit/schema.test.ts @@ -101,6 +101,12 @@ describe('LadybugDB Schema', () => { expect(REL_TYPES).toContain(t); } }); + + it('includes the control-dependence edge types (issue #2085 M5)', () => { + for (const t of ['CDG', 'POST_DOMINATE']) { + expect(REL_TYPES).toContain(t); + } + }); }); describe('node schema DDL', () => { diff --git a/gitnexus/test/unit/security.test.ts b/gitnexus/test/unit/security.test.ts index 5f1934068..37cb9f6f8 100644 --- a/gitnexus/test/unit/security.test.ts +++ b/gitnexus/test/unit/security.test.ts @@ -65,6 +65,15 @@ describe('VALID_RELATION_TYPES', () => { expect(VALID_RELATION_TYPES.has('TAINT_PATH')).toBe(false); expect(VALID_RELATION_TYPES.size).toBe(16); }); + + it('CDG control-dependence edge types stay OUT of the impact allow-list (#2085 M5)', () => { + // CDG and POST_DOMINATE are BasicBlock→BasicBlock (block space), like the + // taint substrate — they must not enter impact()'s symbol-space BFS. Pinned + // explicitly (not just via the size==16 guard) so a future "add all emitted + // types" sweep can't drag them in, mirroring the TAINTED/TAINT_PATH pins. + expect(VALID_RELATION_TYPES.has('CDG')).toBe(false); + expect(VALID_RELATION_TYPES.has('POST_DOMINATE')).toBe(false); + }); }); // ─── Valid node labels ─────────────────────────────────────────────── diff --git a/gitnexus/test/unit/tools.test.ts b/gitnexus/test/unit/tools.test.ts index 156cc7a20..0cd25fb70 100644 --- a/gitnexus/test/unit/tools.test.ts +++ b/gitnexus/test/unit/tools.test.ts @@ -21,8 +21,8 @@ const MUTATING_TOOLS = new Set(['rename', 'group_sync']); const OPEN_WORLD_READ_ONLY_TOOLS = new Set(['query']); describe('GITNEXUS_TOOLS', () => { - it('exports all tools (8 base + 1 explain + 3 route/tool/shape + 1 api_impact + 2 group)', () => { - expect(GITNEXUS_TOOLS).toHaveLength(15); + it('exports all tools (8 base + 1 explain + 1 pdg_query + 3 route/tool/shape + 1 api_impact + 2 group)', () => { + expect(GITNEXUS_TOOLS).toHaveLength(16); }); it('contains all expected tool names', () => { @@ -38,6 +38,7 @@ describe('GITNEXUS_TOOLS', () => { 'rename', 'impact', 'explain', + 'pdg_query', 'api_impact', ]), ); From fb068a94809eede1a8e16e1cbfe0e098faa34f87 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sat, 13 Jun 2026 20:11:15 +0100 Subject: [PATCH 15/16] fix(group): pin repos during sync so large groups resolve cross-links (#2191) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(lbug): pin repos to exempt them from automatic pool eviction [#2189] Add a pinnedRepos set and pinRepo/unpinRepo to the LadybugDB pool adapter. evictLRU and the idle-timeout sweep skip pinned repos; closeOne clears the pin on teardown so explicit close always wins and pins never leak across operations. Behavior is byte-identical when nothing is pinned. Bounded multi-repo callers (group sync) can now keep more than MAX_POOL_SIZE repos resident through deferred cross-repo resolution. * fix(group): pin repos during sync so >MAX_POOL_SIZE groups resolve [#2189] syncGroup now pins each repo immediately after initLbug and releases the pin (unpin then close) in the finally. This keeps every group member resident through the deferred manifest/workspace resolution that runs after the init loop, so cross-links anchor to real graph symbols instead of falling back to synthetic UIDs when a group has more than MAX_POOL_SIZE repos. Release is unpin-before-close plus closeOne's own pin-clear, so pins never leak across syncs in the long-lived MCP server even on error. * style(test): apply prettier formatting to #2189 test files * fix(review): apply autofix feedback Clarify the pinRepo docstring: the pin does not survive teardown (closeOne clears it) and the repoId must match the key passed to initLbug. Addresses a code-review finding that the prior 'or later holds' wording contradicted closeOne's unconditional pin-clear. * refactor(lbug): reference-count pool pins so overlapping holders are safe [#2189] Change pinnedRepos from Set to Map. pinRepo increments the lease count; unpinRepo decrements and deletes the key at 0 (flooring at zero, unknown-id no-op). evictLRU, the idle sweep, and closeOne are transparent to the swap (has()/delete() keep their semantics: skip while count>=1, force-clear on teardown). A boolean Set could not represent two simultaneous holders, so the first release wrongly cleared a pin another holder still needed — the concurrent overlapping group_sync teardown race from the PR #2191 review (Finding 1). Reference counts let two windows of one sync, or two concurrent syncs sharing a repo, coexist safely: the repo stays exempt until the last lease releases. * refactor(lbug): pinRepo returns a leak-proof release disposer [#2189] pinRepo now returns a release() disposer (mirroring addPoolCloseListener) that releases its own lease exactly once — a double-call is a guarded no-op, so it can never over-decrement a sibling holder's reference count. Callers can use the leak-proof pattern `const release = pinRepo(id); try { … } finally { release(); }`. unpinRepo stays exported for explicit pairing. Addresses the PR #2191 review's P3: the exported pin primitive had no built-in pairing, so a caller that forgot to unpin would disable eviction for a repo permanently. * refactor(group): windowed manifest resolution bounds sync pool residency [#2189] Replace whole-sync pinning with windowed deferred resolution. The init loop extracts contracts without pinning (repos evict naturally); manifest links are pre-sorted and partitioned into windows whose referenced in-group repos number <= getMaxResidentRepos(), and each window re-inits + leases only its own repos, resolves, then RELEASES the leases (not closeLbug — released repos stay evictable for the LRU, which avoids stomping a concurrent MCP reader). Peak per-sync pool residency is now bounded by getMaxResidentRepos() distinct repos regardless of group size, removing the unbounded-mmap crash risk the PR #2191 review flagged (Finding 3) — without a new magic-number threshold (it reuses MAX_POOL_SIZE via an intent-named accessor). #2189 stays fixed: each window resolves against live, freshly-leased pools, so cross-links anchor to real graph symbols. partitionManifestWindows is a pure, unit-tested function (every link in exactly one window — the contract-dedup invariant). New sync-windowed-resolution.test.ts asserts the partition bound and, through the real pool, that concurrently-open Databases never exceed the resident cap for a group larger than it. Rewrote the sync.test.ts pinning block (init loop no longer pins; per-window lease/release; release-not-close). --- gitnexus/src/core/group/sync.ts | Bin 11940 -> 17612 bytes gitnexus/src/core/lbug/pool-adapter.ts | 91 ++++++- .../group/sync-windowed-resolution.test.ts | 243 ++++++++++++++++++ gitnexus/test/unit/group/sync.test.ts | 232 +++++++++++++++-- gitnexus/test/unit/lbug-pool-pinning.test.ts | 223 ++++++++++++++++ 5 files changed, 764 insertions(+), 25 deletions(-) create mode 100644 gitnexus/test/unit/group/sync-windowed-resolution.test.ts create mode 100644 gitnexus/test/unit/lbug-pool-pinning.test.ts diff --git a/gitnexus/src/core/group/sync.ts b/gitnexus/src/core/group/sync.ts index 5c535f29f4ccd18cef47e2414eee822370f4623a..a329500be5a0edb4a4588be0960b53d4c79b71aa 100644 GIT binary patch literal 17612 zcmeHP-EJI5cFwh)qJ$V`)8;fSZ({6@RuW{1UMrC-fs~;Hf)Msh*UYrl(>>eOBZ{F3 zL;W+>Uaxiqnn$?mFCr_TRhr&w?5rZo#^7EQf2 z2UVTh>B1dv@K@oiJGgtR&?TF!UG*37@9yoD`Fgup*v4G<*UaQuZ`CbhimGUz&bG@V z{BGad`L?w$vL;*G);7gI+MLffMRjU7HGf^&_F4A+)VdMcm?yf?t-fvB&1ntNoxB%_bw~L0*oJ5Bc!hNnAllkVd^!U4hh{VX z2$Ut3+_?Ycswo!jMU`j>f}8{=~>p! zSMuj8EShIcE>C(#t+}G8mLs<3(6@PeW}A1#+yAw3T(PYDiyj=OOD8j3 zdpS`E^*;45k1jVW+nk7c4`g%hRp7?UUNl8(gQ5gY?W3UzI=YSQ`noCdrR}Yq$%9Eg z8?06hHXk^nr*XmhU7508!WMMUVS40=!nW_ljusUx&;krMXP4Fd`KB%E${CuAsD95O z7r@ItnBu81Kbj<=v6+g^Pz!Qs?%g+FI4$t+ho-1o1 zaGu)3?_hPO29^=xab%V>XfI&Z!=nNC4gyfFwq<#r;N9C|^NJh;*cptl%_=??P>+iS zPZc^JW@TAloV8gwS`2#3YBzdlDcsMi@^T26=bfx)HqU{w>%Y6cb!#sV;5oKsD=Y}I z=cXfj{Q3>?n+J=G_t84*t8FD(Wb^$y!6yfNNFwaj8|y}LHuwR&I<@n<$$uXx&JE!$ z$k|BsQV;O-~*dW0#Xi&3xm#nWzL#= zMqs5CgdlC}r}c$x9^plDD8zu=wn2JwET+3py)-EnDIdw=1yno%b+Myvq4%a<^iVpK zli~OmCXv^Bjx_h~-BYwa{d9Pg?vSI*Z7lLAD_k@-sQL49m&yKO3S{>c>=UQa%`BIN6H62+Kn{LBCItGGn? z78#b?MFx3;)KTqJos$o-M~LIWWe!H;=d0&5+K64{qy@PWl$AW>cZWv0D5^XGfkee^ zB^(hR^i6}O3Hf&09)>$p_mvn5nX%UsO31FP_65@e^EXx0AQR9i1;8+fKjieZvqj-C z73qb%vLbikxsTVvtKOfw;+m-9AL!QcXP@0NpP9!t-y#SUZJ(6lm)lj=0u2~%4PGI` z#z8Gaw6<#qw5@V|-_|C)B`}1^;+4Av)P=Bu`z1U}!6{1zY7u7_jcX^)+RDHd0Jam5 zgf<6<@IZFer2}+3w+MUTP7<3g(<4yFITOMUqu2|;!xR9HZ}sGmw~S`YC9-y4*kBpr z``2%J=0R|~XDWMP&TQ+g)&0I*KtMU%e9x%8>sbOGNJAh82W?f6d225rFHkYQ6J3R6 z+0@t5Y5$SCT+izA`y#*o(LZkM(Kmgn+PAK|{~X6sf$MUmv55x~nJU*;i<7{V>yd&H%S6?`0Y2ssiNnj;UIh?~SS zxDAK}jAz9an(3GesAzqWlB+w8EySs>WR0-S^LHncP`*I+&6DAQiJne|NuKnVFyd<-c&Nn2QN&G=<{zdXf^TL6Jsk*B0W+M)ZKP)&4+76|fn|rg!eB z{OZ>J(xLsOkWW69jC8&@pd2wp~^21!}5J%Wb_45(TVr=Xt%J6&2uL6sROL7Qv#kld5hf^Afs2fOG`_2)|)g z+nGXzTe$?FtXZ2(jJB9p<(b3pU{}P`K#3CEp0EVGfmHjG@*{(@jGTqv)#}2?y>C6G zAON8@VHO!|w5SF^3Pa3EM_^u*i_OX3J$&@?>EHLH$Rv=RW(#@-!H1cJ9wRnFC1Op) za&bV>^g92*4P4S}+bX)=gCH>j#+fo0C$|?hVFfZ=)Y~#Qc8&YP?jx#!tg9bE80`p& zDWJ(_VDZe+S8(^KSRPUZ5YOLMKvN!_nc`Ivni1j<*>Xpd30IW5)?$ZgOx4P`tP4S?$A9v!Wu&SVT*{8 zE(1Q8mjElh@~OA;b}>4=Nb_I=UWt8l3GBKE?R2-&^7;(}h3@W%@H?N!S3RpUMdE|x zqdB?-rVh^-I8bVXd%l2=OUgY*#*iGF?%fxk^STr_cw|aMs&5dUm9SMv3?)A6kQ1NP z9Et~0S4x%rTXCl!K8o?6M=L2q;EVradvcimfY>KF_=|%PU@#1hJxx;~-Wy$=WUz~Z zmp?nn4%4zm?1Hp?4Hj51S&&L*heQ<@R9wBV?jc2?Cgy3^eoWVmW zq8eK2i7*TfyW+(ml5#~&MsQMC9#DjQi?_;3*|u;aR@}y(+Y*Ghz>BoB)v{e_pb!E3 zHR23Dr`iOt_xl3X9dCjvYn(yU0z}}pZF+ZSjKc6E{44Rf+(p-+R6(mgMM#Fi5JeGk zJC5ZT6B7isD0fHkjWInEOF~ClB2>b9ftfNwQY48Ckq{g_lZV&vQNMpD@qyP}O?jdQ zGXatihUp|Kt6qB^X_?j3o0ds`ueJhRiiHyFJ(X-hAoVrZ6ozD@uERf7yirdq`C{NAfra+nkmI9(O3}zsp4!mMeHM6Ma zFk79_ZJN)$$K4=<(S3)hfoa0-*`(_8lEaO zuQbvw>v|I-3LhoGYwFcK(UKT7Mt6017i;MY!CCx=?va6CU=W+Cz1UjzPhk> zLno>v^T7+lT`00`5ZEDD16R?MX;HwDupqCus1hMYi!o1z{B>dBGAH$>BO77?9C)hB z0_EuL60zs+>LM4q;AnP6`j*tzAhf7ox>LziOF|2PRjA+C+S^2_D16Zwxq-R>UUrc& zV8gD=8O0$g^PpkEs;B0kFnY$85FAC0=|@)yL;Wj4?e~4c)*wDCeCgWJZoNrZCbA9& zg^T;YPowae2M@@{-GYxlAewNsGHk}2yazZ6>~05Vj3v7+arvN2GhV6PmAH1)k=a?T z%@&Wx%R%d9yw*2NBIt3Oh3g?L{OS3N*P^pjI69yu%gk?>5>`~uXS^7E>=Zp#X`l4i20|PrM+4}@>fGYVVLWc zuqMdW@EdZw4cr#M&MX9hz+in~DTV;iC}=Wf(bY+}B$^S3sCj(q7pQWGN6u8?^KNRwi{-c}aA=y6E3gU=a46YbV= z0WyuUuGsBqtk4R#fyoI3({vV=%ey`oeBTH^DQO4|JD(`Er_QhT4i(lQ^KlN@BWee~ z{T=hp_=Edu_drNXL7*3?P+XCmJbLOLL=x7c*oV|}o+%uX)`3)6gXcsZ;ytFh?lhdZ zs0sv1AbuG(`jGJmuUM%>PagrJPjJFLbmy?+AH{mGmB#gI@9Dk+*|+vLbTbpLbi6We zIJ}F*tP)wKx<93b9@>rW#-dA-5vfwDH4ok}2_!i(qcg*UfvyhB<(#(Bd(|$32^2%) ztRH2n@#<`4ScWWZ9*W}OnHJ~vRMWc!zNdXZbpNyI?jzZsP5Bu4{!05_NPCrK%=2aR8Xu2=xe`M(2nq-0Y#a&)Puye|F(*KM zj(OfU@K8xO@W>o&+r{L!Lug_58hiid?D=XbF0D$S$fE9o0HSL36a!;)OB^Qn1#`T^n|GachtEeGZ@nH%A z?Krl_vWpDzvD=rSV-ZpZv}@7$(`f;$d213TJP7kAnDB9zuC-|O;0M5Bxda`fWyRbm zpPl%@IEPk;c0?*`0Un75BYReuv0SIpJ$*Y)+H5q%Z8-S zEu@H9mQSS(sI`8mFAA@|=d=Rn>u;nHf!`h# zVmNK`OX@oR@5KjJeMTt!5}lD^7%!oDhZ`qaI>Zu{EJnymz6*D%Q){6_7Q`?LrZ8+! zK+#e^EiPJxu{0b{AVLAGFZgTUf?0Dv8bJ}VFAd?(BZCBE2;fZ@@e{fh2mLT#AL1a_ zVXxdoIOy%RJ>j7=pi`$~l|{9v*|_WfoTE(=$7Lya=&gDUMP~NNm6NMILf5{9Rp2}y z=C@oLb#gf8cCC#P)ncy0G!F|@e*a+SHAj8;(wz@&b)+!y3L58R!Ep zrB9oMy4dj|gF)jI0xCWjjG-k`cv9c#z{!BC10AHDyvzy?{V8;g2cS_eooF6_E^#Pa zp)AwMq}IedNEP_OrG2kTreeR98lX!FI?=u->D*Iy_6;agfv$bYQ{b_c4g1E`q0`D= zcbl!ZJ6uACZz$&h++GF}IWNDMSg3~bd#>@pG5ZQ{I#(h$2op}6T?$iBKwt8KYp}() zMZ%1V@7XKiwSqc@{72lhp^H%H@LFdtL>G7xXEM-_XYU1gz-yuXZqR-o%5_hwgG`TX zRPcncH2lmD#h-+)55~1*Em6@W+}NiNOg!qT0SdjlpMZ!gbW#a($#Kj697detnSy%5 z4wVf4&|fP?VelvDHIzo2EZ}jFMu>(lM>s-sdh+z-;n|5v;v@+tr!eclQ5JNGd}~GZ zC{zOh58bXB zYcGAe0V!4bUN|7--VR>Iz7N&8wU}pOH;d|2>3A@V3xo`vT}-XD=Gmhcz%%32;yOd! z5%WKo+mRpK^2VVjXYQC!&2M5ov!}kw0&xn6Nb8G%H&Ff6C829IH4mF6yX?2DAW_=P zRZPZ-Lmx^Eys&57HXN2RdjSf;DiGmw{sK&Uf@n=a^u|RxiRYHyEn}~tyQpt_>OQ5lzIJ#9RPGN`c_@?r#1g|6N@?`AFL-p+xWN?7{je0)vZ0%g=XxZ5lqvybm8NwU^(nF++9=qQn4?Y-b-w-;VbBawrS!0wC ze(zC9bAb>uYn&h-gnn`re6o*sLeQLias?j5OfkOA>BczF`Z-P-?KKl0Dh@J`>s}0x zVvL0C*sCN)L@(o2*Pi5wJM0DEk!T^z2Ej&Lgau^dPe(0exB|Tn4=gYWnK>MNH^+gv zRi1cx&>X!IQwVU#;n$cCTX6a%3|68^y241KbY4^QlCvIChe3@1Gn-<8bq-QifrnXu zrq@Ii(O@XRH&2O8ki{meVq8H&-8cmj0PY(G(K!%=Crl^69i|9=5>_G&FM~dA;ferl z7_#ifDTzc6obp5mSaO1Xyr8jy_xt-bX{Z!TnT1@cJ0kUJagcs>-#@}X?tT=R49vz+ z5Vi3AqsWTFx`H~ONQQ5Xk(KygAvkcl{jVy=@CF!s2s{F8`4UqB4!+Aw;(0K?{Tl&0 zkX(dy{CP@O1LBp`>S+i)FE*U>g|rS3*HK{w7jzw8YkIwubb~%#+UGJdj5Lt+c!}g= zdLUok9dw_DKvz)3g)>)x@?A{$90nmTK5Rj5B2jQS%OK@+oHpRpx3tL@B{OHt53&7^ z4`i4{R-6sMg=Kso!zYo@l^z-a*a4>#P|)mVs7VD*1Y;J8K_~=#bXZlH-Z^klWjHPJZ-7kac6Ck1z1S{CP zhJP8|BL7BR8UJ?vMxs1?)X5RA+pPAa@$1hJv}vfe_Y+>J52XQy%mv9e`QC&>5c=Qn zvK+C6ol0Qw-WpCRLa3p!>&)PDiMoOI*hX##F~b-{!M^z#2gJI>TP3XMygT z3I_^E&jA6d(>Ze<52-Z$$;beWGV15{pgDDg`T7!t+iipR8PNa+{xRi)kAG0lLUIQI z$Dlp)CH(+7K*{a0LGA=^gE47*xx#S=NdMUE=czf{>LzD8@xg+_|L+Jx`@nBx1V|_(?X3`9aUZ%pE_{j&Rh!DlWz-sT-CgM5r~mlR z|NWnT9k4))cy~KDp&|}Bg2AQ4k=H7dk~&eQMPwOaOyEPz`LbvsS(u*^O%-0AQ$R7n x-CU-zPZdZ6NfIAD_^k7T3>q^hIz&j5_nSzK*C;H4YI@yWChc$&w(9!U{{hUcIw=4E delta 737 zcmYjP&ubGw6lQafko2ZCjqTw<5AI@|q*7`cjfg?2P*56gA;SK6=}t49S$1YkLM_&t zJz1HP=%ELXg6M)*!L#@$=wBdSJ&Cj1reRs3Z7*8pIYx;-1R`h9~DPcOw}JRfDb)F$M;wV%IlyCl*p} zQa*(Awe_j2Cl`Me80wu0nG3wTWAqc#AY1zH>e3uMg@gwS_xjL=UOzp5sZr1`7ta*}Pyb%6r=6u2R%HfHqC|#vlsD^RxS-cyoMGQy zuNLu?JUxcP(lfIR?7Ex|&@kdSS5F0pIN`<_Ndp8GQVAka7=wx@B`{yT4jvQ!5Tt;a zb#e!j7$ulruqO!awri#+VAK?u9tk#;GC^2gX*RAo8)Q1~QHWq1a2cC`hfIu4;*h6Z z?jYySr1)YtVyFy@EsUfv79*Mdx$?s*&qzssUj1O{gT~^*18)a?MTR($wlUV^NMaMC zOy4)YSjFk?y4@rThCGW*)@GM-Zj|)FJ44`Mma~wezl|2)r(|2Gmw+O%;xm5jsq&5v4;=9TZ WZQFtNE!f*PQnPgS(>FI)ivIv^t_Cdt diff --git a/gitnexus/src/core/lbug/pool-adapter.ts b/gitnexus/src/core/lbug/pool-adapter.ts index e5338e3d0..11b1f7690 100644 --- a/gitnexus/src/core/lbug/pool-adapter.ts +++ b/gitnexus/src/core/lbug/pool-adapter.ts @@ -103,6 +103,31 @@ const IDLE_TIMEOUT_MS = 5 * 60 * 1000; // 5 minutes /** Max connections per repo (caps concurrent queries per repo) */ const MAX_CONNS_PER_REPO = 8; +/** + * Repos exempt from AUTOMATIC eviction (LRU + idle timeout) until explicitly + * unpinned. Used by bounded multi-repo operations like `group sync`, which + * initializes one pool per repo and then resolves cross-repo manifest/workspace + * links against ALL of those pools after the init loop. Without pinning, a + * group larger than MAX_POOL_SIZE would LRU-evict the earliest repos before + * resolution runs, leaving the deferred executor closures pointing at dead pool + * entries (issue #2189). + * + * Pins are REFERENCE-COUNTED: the map holds repoId → active lease count. This + * lets overlapping holders (two windows of one sync, or two concurrent + * `group sync` calls sharing a repo) coexist safely — the repo stays exempt + * until the LAST holder releases. A boolean Set could not represent "two + * holders," so the first release would wrongly clear a pin another holder still + * needs (PR #2191 review, Finding 1). + * + * Pins block only automatic eviction (LRU + idle). Explicit teardown + * (closeOne / closeLbug) always closes the entry and force-clears its count — + * teardown is authoritative. A present key always means count ≥ 1. While every + * pooled repo is pinned, evictLRU finds no eligible victim and the pool may + * transiently exceed MAX_POOL_SIZE — the same soft-cap behavior that already + * occurs when every entry is checked out. + */ +const pinnedRepos = new Map(); + // Behavior-neutral RSS tracing for the FTS evict→reload memory repro // (gitnexus/scripts/bench/fts-evict-reload-rss.mjs). Two invariants keep it safe // in the pool init/close hot path: it writes ONLY to stderr (stdout is the MCP @@ -145,6 +170,7 @@ function ensureIdleTimer(): void { idleTimer = setInterval(() => { const now = Date.now(); for (const [repoId, entry] of pool) { + if (pinnedRepos.has(repoId)) continue; if (now - entry.lastUsed > IDLE_TIMEOUT_MS && entry.checkedOut === 0) { closeOne(repoId); } @@ -167,7 +193,64 @@ export const touchRepo = (repoId: string): void => { }; /** - * Evict the least-recently-used repo if pool is at capacity + * Acquire one eviction-exemption lease on a repo (LRU + idle timeout) by + * incrementing its reference count. The repoId must match the key passed to + * initLbug (e.g. group sync leases by handle.id — the same id it inits with). + * Leasing a repoId before it enters the pool is allowed and protects the entry + * once it is created, but the lease does NOT survive a teardown: closeOne + * force-clears the count, so a later re-init of the same repoId starts + * unpinned. Each pinRepo MUST be balanced by exactly one release (the repo + * stays exempt until the last lease is released). See the pinnedRepos docstring + * for the full contract. + * + * Returns a `release` disposer (mirroring addPoolCloseListener) that releases + * THIS lease exactly once — calling it twice is a no-op, so it can never + * over-decrement a sibling holder's count. Prefer the disposer + * (`const release = pinRepo(id); try { … } finally { release(); }`) so the + * pin/release pair is leak-proof; unpinRepo remains available for callers that + * pair explicitly. + */ +export const pinRepo = (repoId: string): (() => void) => { + pinnedRepos.set(repoId, (pinnedRepos.get(repoId) ?? 0) + 1); + let released = false; + return () => { + if (released) return; + released = true; + unpinRepo(repoId); + }; +}; + +/** + * Release one eviction-exemption lease on a repo. The repo becomes eligible for + * automatic eviction again only once its count reaches 0 (the key is deleted). + * Idempotent at the floor: releasing a repo with no active lease is a no-op (no + * negative counts). Does NOT close the repo's pool. + */ +export const unpinRepo = (repoId: string): void => { + const count = pinnedRepos.get(repoId); + if (count === undefined) return; + if (count <= 1) { + pinnedRepos.delete(repoId); + } else { + pinnedRepos.set(repoId, count - 1); + } +}; + +/** + * Maximum number of repos a bounded multi-repo operation (e.g. group sync's + * windowed manifest resolution) should hold resident at once. Equals + * MAX_POOL_SIZE today, but exposed under an intent-named accessor so callers + * size their working set against "max repos a bounded op should hold" rather + * than coupling to the LRU eviction-cap constant, which may be tuned + * independently. + */ +export const getMaxResidentRepos = (): number => MAX_POOL_SIZE; + +/** + * Evict the least-recently-used repo if pool is at capacity. + * Pinned repos are never chosen as the eviction victim — when every eligible + * entry is pinned, no eviction occurs and the pool transiently exceeds + * MAX_POOL_SIZE (see the pinnedRepos docstring). */ function evictLRU(): void { if (pool.size < MAX_POOL_SIZE) return; @@ -175,6 +258,7 @@ function evictLRU(): void { let oldestId: string | null = null; let oldestTime = Infinity; for (const [id, entry] of pool) { + if (pinnedRepos.has(id)) continue; if (entry.checkedOut === 0 && entry.lastUsed < oldestTime) { oldestTime = entry.lastUsed; oldestId = id; @@ -244,6 +328,11 @@ function closeOne(repoId: string): void { pool.delete(repoId); + // Clear any eviction pin — the entry is gone, so the pin is meaningless and + // would otherwise leak across operations in a long-lived process. Teardown + // is authoritative: an explicit close always wins over a pin. + pinnedRepos.delete(repoId); + // Notify listeners AFTER the pool entry is gone so any cache-invalidation // they perform is consistent with `isLbugReady(repoId) === false`. for (const listener of poolCloseListeners) { diff --git a/gitnexus/test/unit/group/sync-windowed-resolution.test.ts b/gitnexus/test/unit/group/sync-windowed-resolution.test.ts new file mode 100644 index 000000000..d7cdb4090 --- /dev/null +++ b/gitnexus/test/unit/group/sync-windowed-resolution.test.ts @@ -0,0 +1,243 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; +import * as os from 'node:os'; +import * as path from 'node:path'; +import { mkdtempSync, writeFileSync, mkdirSync, rmSync } from 'node:fs'; +import type { GroupConfig, GroupManifestLink } from '../../../src/core/group/types.js'; + +// Two test surfaces for the windowed manifest resolution (issue #2189 / PR #2191 +// review, Finding 3 — bound peak pool residency to MAX_POOL_SIZE regardless of +// group size): +// +// 1. partitionManifestWindows — a pure function; the bounded-residency logic +// lives here (every window references <= maxResident repos, every link in +// exactly one window). Tested directly, no pool. +// 2. A real-pool integration test that drives syncGroup through the actual +// pool (native LadybugDB layer mocked, as in lbug-pool-pinning.test.ts) and +// asserts the count of concurrently-open Databases never exceeds the +// resident cap — the end-to-end residency bound the review flagged as +// missing. + +// ── Surface 1: pure partition function ────────────────────────────────────── + +describe('partitionManifestWindows (issue #2189 windowed resolution)', () => { + const link = (from: string, to: string): GroupManifestLink => ({ + from, + to, + type: 'http', + contract: `GET::/${from}-${to}`, + role: 'consumer', + }); + + it('keeps every window within maxResident repos and places every link exactly once', async () => { + const { partitionManifestWindows } = await import('../../../src/core/group/sync.js'); + const repos = ['r1', 'r2', 'r3', 'r4', 'r5', 'r6', 'r7', 'r8']; + const known = new Set(repos); + // A star: every leaf links to the hub r1, plus a few leaf-leaf links. + const links = [ + link('r1', 'r2'), + link('r1', 'r3'), + link('r1', 'r4'), + link('r1', 'r5'), + link('r1', 'r6'), + link('r7', 'r8'), + link('r2', 'r3'), + ]; + const maxResident = 5; + const windows = partitionManifestWindows(links, known, maxResident); + + // Bounded residency: no window references more than maxResident repos. + for (const w of windows) expect(w.repos.size).toBeLessThanOrEqual(maxResident); + + // True partition: every link appears in exactly one window. + const placed = windows.flatMap((w) => w.links); + expect(placed).toHaveLength(links.length); + const placedKeys = placed.map((l) => `${l.from}->${l.to}`).sort(); + const inputKeys = links.map((l) => `${l.from}->${l.to}`).sort(); + expect(placedKeys).toEqual(inputKeys); + // No link appears twice (the contract-dedup invariant — KTD-4). + expect(new Set(placedKeys).size).toBe(placedKeys.length); + }); + + it('counts only in-group repos toward a window; dangling links consume no budget', async () => { + const { partitionManifestWindows } = await import('../../../src/core/group/sync.js'); + const known = new Set(['r1']); + const links = [ + link('r1', 'external-a'), // 1 in-group repo + link('external-b', 'external-c'), // 0 in-group repos (fully dangling) + ]; + const windows = partitionManifestWindows(links, known, 5); + // Both links are still placed (so they yield synthetic-UID contracts)... + expect(windows.flatMap((w) => w.links)).toHaveLength(2); + // ...but the only repo counted is r1. + const allRepos = new Set(windows.flatMap((w) => [...w.repos])); + expect(allRepos).toEqual(new Set(['r1'])); + }); + + it('returns no windows for an empty link set', async () => { + const { partitionManifestWindows } = await import('../../../src/core/group/sync.js'); + expect(partitionManifestWindows([], new Set(['r1']), 5)).toEqual([]); + }); + + it('splits links across multiple windows when referenced repos exceed maxResident', async () => { + const { partitionManifestWindows } = await import('../../../src/core/group/sync.js'); + const known = new Set(['r1', 'r2', 'r3', 'r4', 'r5', 'r6']); + // 3 disjoint repo-pairs = 6 distinct repos; maxResident 2 forces ≥3 windows. + const links = [link('r1', 'r2'), link('r3', 'r4'), link('r5', 'r6')]; + const windows = partitionManifestWindows(links, known, 2); + expect(windows.length).toBeGreaterThanOrEqual(3); + for (const w of windows) expect(w.repos.size).toBeLessThanOrEqual(2); + expect(windows.flatMap((w) => w.links)).toHaveLength(3); + }); +}); + +// ── Surface 2: real-pool residency bound through syncGroup ─────────────────── + +const { loadFTSExtensionMock, openCounter } = vi.hoisted(() => ({ + loadFTSExtensionMock: vi.fn(), + openCounter: { live: 0, peak: 0 }, +})); + +vi.mock('@ladybugdb/core', () => ({ + default: { + Database: vi.fn(), + Connection: vi.fn(function (this: any) { + this.query = vi.fn().mockResolvedValue({ + getAll: vi.fn().mockResolvedValue([]), + close: vi.fn(), + }); + // executeParameterized's prepare/execute path (manifest resolveSymbol). + this.prepare = vi.fn().mockResolvedValue({ + isSuccess: () => true, + getErrorMessage: vi.fn().mockResolvedValue(''), + }); + this.execute = vi.fn().mockResolvedValue({ + getAll: vi.fn().mockResolvedValue([]), + close: vi.fn(), + }); + this.close = vi.fn().mockResolvedValue(undefined); + }), + }, +})); + +vi.mock('../../../src/core/lbug/lbug-adapter.js', () => ({ + isReadOnlyDbError: vi.fn(() => false), + loadFTSExtension: loadFTSExtensionMock, +})); + +vi.mock('../../../src/core/lbug/lbug-config.js', () => ({ + // Track concurrently-open Databases: a fresh fake per open, decrement on close. + createLbugDatabase: vi.fn(() => { + openCounter.live += 1; + openCounter.peak = Math.max(openCounter.peak, openCounter.live); + return { + init: vi.fn().mockResolvedValue(undefined), + close: vi.fn().mockImplementation(async () => { + openCounter.live -= 1; + }), + }; + }), + toNativeSafePath: vi.fn((p: string) => p), + isWalCorruptionError: vi.fn(() => false), + WAL_RECOVERY_SUGGESTION: '', +})); + +vi.mock('../../../src/core/lbug/sidecar-recovery.js', () => ({ + preflightLbugSidecars: vi.fn().mockResolvedValue(undefined), + isMissingFsError: vi.fn(() => false), + isMissingShadowSidecarError: vi.fn(() => false), + isReadOnlyShadowReplayError: vi.fn(() => false), + quarantineWalForMissingShadow: vi.fn().mockResolvedValue(''), + renameFailureMessage: vi.fn((p: string) => `rename failed for ${p}`), + statIfExists: vi.fn().mockResolvedValue(null), +})); + +// readRegistry is called in syncGroup's else branch; resolveRepoHandle is +// supplied, so an empty registry is fine (only the meta.json fallback reads it). +vi.mock('../../../src/storage/repo-manager.js', () => ({ + readRegistry: vi.fn().mockResolvedValue([]), +})); + +const { syncGroup } = await import('../../../src/core/group/sync.js'); +const { closeLbug, getMaxResidentRepos } = await import('../../../src/core/lbug/pool-adapter.js'); + +describe('syncGroup windowed resolution bounds pool residency (real pool, #2189)', () => { + let tmpRoot: string; + + beforeEach(() => { + tmpRoot = mkdtempSync(path.join(os.tmpdir(), 'gn-window-resid-')); + loadFTSExtensionMock.mockResolvedValue(true); + openCounter.live = 0; + openCounter.peak = 0; + }); + + afterEach(async () => { + await closeLbug().catch(() => {}); + rmSync(tmpRoot, { recursive: true, force: true }); + }); + + it('never holds more than getMaxResidentRepos() Databases open for a large group', async () => { + const maxResident = getMaxResidentRepos(); + const repoCount = maxResident + 4; // exceed the cap so windowing must split + + const repos: Record = {}; + const links: GroupManifestLink[] = []; + for (let i = 1; i <= repoCount; i++) { + const gp = `app/repo-${i}`; + repos[gp] = `repo-${i}`; + // Star topology: every repo links to repo-1 → many windows reference repo-1. + if (i > 1) { + links.push({ + from: gp, + to: 'app/repo-1', + type: 'http', + contract: `GET::/api/${i}`, + role: 'consumer', + }); + } + } + + const config: GroupConfig = { + version: 1, + name: 'test', + description: '', + repos, + links, + packages: {}, + // All detection off → init loop just opens pools (no extractor file reads). + detect: { + http: false, + grpc: false, + thrift: false, + topics: false, + shared_libs: false, + embedding_fallback: false, + workspace_deps: false, + }, + matching: { bm25_threshold: 0.7, embedding_threshold: 0.65, max_candidates_per_step: 3 }, + }; + + await syncGroup(config, { + resolveRepoHandle: async (_name, groupPath) => { + // Each repo gets a real storage dir with a fake lbug file so fs.stat in + // doInitLbug succeeds; distinct paths → distinct Databases. + const storagePath = path.join(tmpRoot, groupPath); + mkdirSync(storagePath, { recursive: true }); + writeFileSync(path.join(storagePath, 'lbug'), ''); + return { + id: groupPath.replace(/\//g, '-'), + path: groupPath, + repoPath: storagePath, + storagePath, + }; + }, + skipWrite: true, + }); + + // The init loop (no pin) keeps the pool at the LRU cap; windowed resolution + // leases <= maxResident repos per window and releases them. Peak concurrent + // open Databases must stay within the resident cap (+ at most one transient + // overshoot at an init boundary — the pool's documented soft cap). + expect(openCounter.peak).toBeGreaterThan(0); + expect(openCounter.peak).toBeLessThanOrEqual(maxResident + 1); + }); +}); diff --git a/gitnexus/test/unit/group/sync.test.ts b/gitnexus/test/unit/group/sync.test.ts index 4fc320076..061f5cfc4 100644 --- a/gitnexus/test/unit/group/sync.test.ts +++ b/gitnexus/test/unit/group/sync.test.ts @@ -166,20 +166,19 @@ describe('syncGroup', () => { expect(result).toBeDefined(); }); - it('test_syncGroup_closes_only_opened_pools', async () => { + it('test_syncGroup_does_not_force_close_pools (release-not-close, #2191 review)', async () => { + // Post windowed-resolution refactor, syncGroup releases its eviction leases + // and lets the pool's LRU reclaim repos — it does NOT call closeLbug. This + // avoids tearing down a pool entry a concurrent MCP reader may share. const config = makeConfig({ 'app/backend': 'backend-repo', 'app/frontend': 'frontend-repo', }); - const closedIds: string[] = []; - const { vi } = await import('vitest'); const poolAdapter = await import('../../../src/core/lbug/pool-adapter.js'); const initSpy = vi.spyOn(poolAdapter, 'initLbug').mockResolvedValue(undefined); - const closeSpy = vi.spyOn(poolAdapter, 'closeLbug').mockImplementation(async (id?: string) => { - if (id) closedIds.push(id); - }); + const closeSpy = vi.spyOn(poolAdapter, 'closeLbug').mockResolvedValue(undefined); try { await syncGroup(config, { @@ -192,19 +191,8 @@ describe('syncGroup', () => { skipWrite: true, }).catch(() => {}); - // closeLbug must have been called at least once with specific pool ids - expect(closeSpy.mock.calls.length).toBeGreaterThan(0); - expect(closedIds).toContain('app-backend'); - expect(closedIds).toContain('app-frontend'); - - // Every call must have a truthy string id - for (const id of closedIds) { - expect(id).toBeTruthy(); - expect(typeof id).toBe('string'); - } - // No blanket close (no-arg or empty-string or undefined) - const blanketCalls = closeSpy.mock.calls.filter((args) => args.length === 0 || !args[0]); - expect(blanketCalls).toHaveLength(0); + // No closeLbug — repos are left evictable for the LRU to reclaim. + expect(closeSpy.mock.calls.length).toBe(0); } finally { initSpy.mockRestore(); closeSpy.mockRestore(); @@ -534,7 +522,9 @@ service OrderService { }, }); expect(initSpy).toHaveBeenCalledWith('billing-repo', path.join(storageDir, 'lbug')); - expect(closeSpy).toHaveBeenCalledWith('billing-repo'); + // syncGroup no longer force-closes pools (release-not-close, #2191 review); + // repos are left evictable for the LRU. Assert no teardown call here. + expect(closeSpy).not.toHaveBeenCalled(); } finally { initSpy.mockRestore(); closeSpy.mockRestore(); @@ -1044,18 +1034,20 @@ service OrderService { skipWrite: true, }); - // Manifest symbol resolution must run while pools are still open + // Manifest symbol resolution runs against live (leased) pools. expect(manifestResolvedWhilePoolOpen).toBe(true); - expect(closeLbugCalled).toBe(true); // The manifest cross-link must use the real UID from the DB, not synthetic + // — the #2189 fix, now via windowed resolution (the svc/orders↔svc/payments + // link forms one window whose repos are re-inited + leased for resolution). const manifestLinks = result.crossLinks.filter((cl) => cl.matchType === 'manifest'); expect(manifestLinks).toHaveLength(1); expect(manifestLinks[0].to.symbolUid).toBe('real-uid-checkout'); expect(manifestLinks[0].to.symbolUid).not.toContain('manifest::'); - // closeLbug must fire exactly twice (one per repo) - expect(closeSpy).toHaveBeenCalledTimes(2); + // syncGroup no longer force-closes pools (release-not-close, #2191 review). + expect(closeLbugCalled).toBe(false); + expect(closeSpy).not.toHaveBeenCalled(); } finally { initSpy.mockRestore(); closeSpy.mockRestore(); @@ -1105,6 +1097,198 @@ service OrderService { }); }); +// Lifecycle wiring for issue #2189: syncGroup must pin every repo it +// initializes (so a group larger than MAX_POOL_SIZE survives deferred +// manifest/workspace resolution) and release those pins on completion AND on +// error. The eviction-survival MECHANISM itself is proven against real +// evictLRU in test/unit/lbug-pool-pinning.test.ts; these tests prove the sync +// loop drives that mechanism correctly. (A full end-to-end proof through the +// real pool — real symbolUid instead of synthetic after >5 repos — would +// require a real or fully-native-mocked LadybugDB stack; mechanism + wiring +// coverage stands in for it here.) +describe('syncGroup windowed manifest resolution (issue #2189 / PR #2191 review)', () => { + const groupConfig = (count: number, links: GroupManifestLink[] = []): GroupConfig => { + const repos: Record = {}; + for (let i = 1; i <= count; i++) repos[`app/repo-${i}`] = `repo-${i}`; + return { + version: 1, + name: 'test', + description: '', + repos, + links, + packages: {}, + detect: { + http: true, + grpc: false, + thrift: false, + topics: false, + shared_libs: false, + embedding_fallback: false, + workspace_deps: false, + }, + matching: { bm25_threshold: 0.7, embedding_threshold: 0.65, max_candidates_per_step: 3 }, + }; + }; + + const okHandle = async (_name: string, groupPath: string): Promise => ({ + id: groupPath.replace(/\//g, '-'), + path: groupPath, + repoPath: '/tmp/' + groupPath, + storagePath: '/tmp/' + groupPath + '/.gitnexus', + }); + + const httpLink = (from: string, to: string): GroupManifestLink => ({ + from, + to, + type: 'http', + contract: 'GET::/api/x', + role: 'consumer', + }); + + // pinRepo now returns a release disposer; the spy returns a tracked spy fn so + // tests can assert every acquired lease was released. + const setupPoolSpies = async () => { + const poolAdapter = await import('../../../src/core/lbug/pool-adapter.js'); + const releaseSpies: Array> = []; + const initSpy = vi.spyOn(poolAdapter, 'initLbug').mockResolvedValue(undefined); + const execSpy = vi.spyOn(poolAdapter, 'executeParameterized').mockResolvedValue([]); + const pinSpy = vi.spyOn(poolAdapter, 'pinRepo').mockImplementation(() => { + const release = vi.fn(); + releaseSpies.push(release); + return release; + }); + const restore = () => { + initSpy.mockRestore(); + execSpy.mockRestore(); + pinSpy.mockRestore(); + }; + return { releaseSpies, initSpy, execSpy, pinSpy, restore }; + }; + + it('pins only the repos referenced by manifest links, not the whole group', async () => { + const { pinSpy, restore } = await setupPoolSpies(); + try { + await syncGroup(groupConfig(8, [httpLink('app/repo-1', 'app/repo-2')]), { + resolveRepoHandle: okHandle, + skipWrite: true, + }); + const pinnedIds = pinSpy.mock.calls.map((c) => c[0]).sort(); + // Only the windowed (link-referenced) repos are leased — bounded residency, + // not the whole 8-repo group. + expect(pinnedIds).toEqual(['app-repo-1', 'app-repo-2']); + expect(pinnedIds).not.toContain('app-repo-3'); + } finally { + restore(); + } + }); + + it('does not pin during the init loop when there are no manifest links', async () => { + const { pinSpy, restore } = await setupPoolSpies(); + try { + await syncGroup(groupConfig(8, []), { resolveRepoHandle: okHandle, skipWrite: true }); + // The init loop extracts contracts without pinning; with no links there + // are no resolution windows, so nothing is ever pinned. + expect(pinSpy.mock.calls.length).toBe(0); + } finally { + restore(); + } + }); + + it('releases every window lease on successful completion', async () => { + const { releaseSpies, restore } = await setupPoolSpies(); + try { + await syncGroup( + groupConfig(8, [ + httpLink('app/repo-1', 'app/repo-2'), + httpLink('app/repo-7', 'app/repo-8'), + ]), + { resolveRepoHandle: okHandle, skipWrite: true }, + ); + expect(releaseSpies.length).toBeGreaterThan(0); + for (const release of releaseSpies) expect(release).toHaveBeenCalled(); + } finally { + restore(); + } + }); + + it('releases the window leases even when resolution throws mid-window', async () => { + const { ManifestExtractor } = + await import('../../../src/core/group/extractors/manifest-extractor.js'); + const { releaseSpies, restore } = await setupPoolSpies(); + const manifestSpy = vi + .spyOn(ManifestExtractor.prototype, 'extractFromManifest') + .mockRejectedValue(new Error('resolution boom')); + try { + await expect( + syncGroup(groupConfig(8, [httpLink('app/repo-1', 'app/repo-2')]), { + resolveRepoHandle: okHandle, + skipWrite: true, + }), + ).rejects.toThrow('resolution boom'); + // The window's finally released its acquired leases despite the throw. + expect(releaseSpies.length).toBeGreaterThan(0); + for (const release of releaseSpies) expect(release).toHaveBeenCalled(); + } finally { + restore(); + manifestSpy.mockRestore(); + } + }); + + it('does not pin a repo that fails to resolve (no pool handle)', async () => { + const { pinSpy, restore } = await setupPoolSpies(); + try { + await syncGroup(groupConfig(3, [httpLink('app/repo-1', 'app/repo-2')]), { + resolveRepoHandle: async (_name, groupPath) => + groupPath === 'app/repo-2' ? null : okHandle(_name, groupPath), + skipWrite: true, + }); + const pinnedIds = pinSpy.mock.calls.map((c) => c[0]); + // repo-2 has no handle (resolve returned null) → not in knownRepos → + // never windowed, never leased; repo-1 (resolved) is. + expect(pinnedIds).toContain('app-repo-1'); + expect(pinnedIds).not.toContain('app-repo-2'); + } finally { + restore(); + } + }); + + it('releases an already-acquired lease when a later init in the same window throws', async () => { + const poolAdapter = await import('../../../src/core/lbug/pool-adapter.js'); + const releaseSpies: Array> = []; + // Throw on the SECOND init of app-repo-2 — the first is the init-loop + // extraction; the second is the window re-init. This isolates the failure + // to window setup, after app-repo-1's lease was already acquired. + const initCounts = new Map(); + const initSpy = vi.spyOn(poolAdapter, 'initLbug').mockImplementation(async (id: string) => { + const n = (initCounts.get(id) ?? 0) + 1; + initCounts.set(id, n); + if (id === 'app-repo-2' && n === 2) throw new Error('window init boom'); + }); + const execSpy = vi.spyOn(poolAdapter, 'executeParameterized').mockResolvedValue([]); + const pinSpy = vi.spyOn(poolAdapter, 'pinRepo').mockImplementation(() => { + const release = vi.fn(); + releaseSpies.push(release); + return release; + }); + try { + await expect( + syncGroup(groupConfig(2, [httpLink('app/repo-1', 'app/repo-2')]), { + resolveRepoHandle: okHandle, + skipWrite: true, + }), + ).rejects.toThrow('window init boom'); + // Exactly one lease was acquired (app-repo-1) before app-repo-2's init + // threw, and the window finally released it — no leaked lease. + expect(releaseSpies.length).toBe(1); + expect(releaseSpies[0]).toHaveBeenCalled(); + } finally { + initSpy.mockRestore(); + execSpy.mockRestore(); + pinSpy.mockRestore(); + } + }); +}); + describe('stableRepoPoolId', () => { it('returns lowercase name when no collision', () => { const entry: RegistryEntry = { diff --git a/gitnexus/test/unit/lbug-pool-pinning.test.ts b/gitnexus/test/unit/lbug-pool-pinning.test.ts new file mode 100644 index 000000000..914647b09 --- /dev/null +++ b/gitnexus/test/unit/lbug-pool-pinning.test.ts @@ -0,0 +1,223 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; +import * as os from 'node:os'; +import * as path from 'node:path'; +import { mkdtempSync, writeFileSync, rmSync } from 'node:fs'; + +// Drive the REAL initLbug path (which calls evictLRU) rather than +// initLbugWithDb (which bypasses eviction entirely). The native LadybugDB +// open/connect/FTS stack is mocked exactly as in lbug-pool-fts-load.test.ts, +// plus sidecar-recovery so openReadOnlyDatabase's preflight is a no-op. fs is +// NOT mocked — each repo uses a real temp file so the fs.stat existence check +// in doInitLbug succeeds naturally. +// +// Covers issue #2189: a group sync larger than MAX_POOL_SIZE must keep every +// repo resident through deferred manifest/workspace resolution. Pinning makes +// that resident set survive automatic (LRU + idle) eviction. + +const { loadFTSExtensionMock } = vi.hoisted(() => ({ + loadFTSExtensionMock: vi.fn(), +})); + +vi.mock('@ladybugdb/core', () => ({ + default: { + Database: vi.fn(), + Connection: vi.fn(function (this: any) { + // probeDatabaseForShadowReplay() runs a probe query during the + // read-only open; the result must expose getAll()/close(). + this.query = vi.fn().mockResolvedValue({ + getAll: vi.fn().mockResolvedValue([]), + close: vi.fn(), + }); + this.close = vi.fn().mockResolvedValue(undefined); + }), + }, +})); + +vi.mock('../../src/core/lbug/lbug-adapter.js', () => ({ + isReadOnlyDbError: vi.fn(() => false), + loadFTSExtension: loadFTSExtensionMock, +})); + +vi.mock('../../src/core/lbug/lbug-config.js', () => ({ + // A fresh fake Database per call so distinct dbPaths get distinct entries + // and closeOne's db.close() resolves per repo. + createLbugDatabase: vi.fn(() => ({ + init: vi.fn().mockResolvedValue(undefined), + close: vi.fn().mockResolvedValue(undefined), + })), + toNativeSafePath: vi.fn((p: string) => p), + isWalCorruptionError: vi.fn(() => false), + WAL_RECOVERY_SUGGESTION: '', +})); + +vi.mock('../../src/core/lbug/sidecar-recovery.js', () => ({ + preflightLbugSidecars: vi.fn().mockResolvedValue(undefined), + isMissingFsError: vi.fn(() => false), + isMissingShadowSidecarError: vi.fn(() => false), + isReadOnlyShadowReplayError: vi.fn(() => false), + quarantineWalForMissingShadow: vi.fn().mockResolvedValue(''), + renameFailureMessage: vi.fn((p: string) => `rename failed for ${p}`), + statIfExists: vi.fn().mockResolvedValue(null), +})); + +const { initLbug, closeLbug, isLbugReady, pinRepo, unpinRepo } = + await import('../../src/core/lbug/pool-adapter.js'); + +describe('pool-adapter repo pinning (issue #2189)', () => { + let tmpDir: string; + // Track every repoId touched so afterEach can fully reset module-global state. + const touched = new Set(); + + const dbPathFor = (repoId: string): string => { + const p = path.join(tmpDir, `${repoId}.lbug`); + writeFileSync(p, ''); // real file so fs.stat() in doInitLbug succeeds + return p; + }; + + const init = async (repoId: string): Promise => { + touched.add(repoId); + await initLbug(repoId, dbPathFor(repoId)); + }; + + beforeEach(() => { + tmpDir = mkdtempSync(path.join(os.tmpdir(), 'gn-pin-test-')); + loadFTSExtensionMock.mockResolvedValue(true); + }); + + afterEach(async () => { + vi.useRealTimers(); + await closeLbug().catch(() => {}); + for (const id of touched) unpinRepo(id); + touched.clear(); + loadFTSExtensionMock.mockReset(); + rmSync(tmpDir, { recursive: true, force: true }); + }); + + // MAX_POOL_SIZE is 5; the 6th init triggers evictLRU. + it('characterization: WITHOUT pinning, the earliest repo is LRU-evicted past the cap', async () => { + for (let i = 1; i <= 6; i++) await init(`repo-${i}`); + + // repo-1 had the oldest lastUsed and nothing was checked out, so it is the + // eviction victim — exactly the stale-executor scenario from #2189. + expect(isLbugReady('repo-1')).toBe(false); + // The most recently initialized repo survives. + expect(isLbugReady('repo-6')).toBe(true); + }); + + it('FIX: pinning each repo keeps all of them resident past the cap', async () => { + for (let i = 1; i <= 6; i++) { + await init(`repo-${i}`); + pinRepo(`repo-${i}`); + } + + for (let i = 1; i <= 6; i++) { + expect(isLbugReady(`repo-${i}`)).toBe(true); + } + }); + + it('explicit close beats the pin AND clears it (no cross-operation leak)', async () => { + pinRepo('repo-x'); + await init('repo-x'); + expect(isLbugReady('repo-x')).toBe(true); + + // Explicit teardown closes a pinned repo without needing an unpin first. + await closeLbug('repo-x'); + expect(isLbugReady('repo-x')).toBe(false); + + // The pin must have been cleared on close: re-init repo-x FIRST (oldest + // lastUsed) and fill past the cap. If the pin had leaked, repo-x would be + // un-evictable; instead it is evicted as the LRU victim. + await init('repo-x'); + for (let i = 1; i <= 5; i++) await init(`fresh-${i}`); + expect(isLbugReady('repo-x')).toBe(false); + }); + + it('idle-timeout sweep skips pinned repos but still evicts idle unpinned ones', async () => { + vi.useFakeTimers(); + + await init('pinned-idle'); + pinRepo('pinned-idle'); + await init('unpinned-idle'); + + expect(isLbugReady('pinned-idle')).toBe(true); + expect(isLbugReady('unpinned-idle')).toBe(true); + + // The idle timer runs every 60s and closes entries idle past + // IDLE_TIMEOUT_MS (5 min) with no checked-out connections. advance past + // both thresholds, flushing microtasks between fires. + await vi.advanceTimersByTimeAsync(5 * 60 * 1000 + 60 * 1000); + + expect(isLbugReady('pinned-idle')).toBe(true); + expect(isLbugReady('unpinned-idle')).toBe(false); + }); + + it('unpinRepo re-enables eviction for that repo', async () => { + // Pin five repos and fill the pool; a sixth init evicts nothing (all pinned). + for (let i = 1; i <= 5; i++) { + await init(`p-${i}`); + pinRepo(`p-${i}`); + } + await init('p-6'); // unpinned; pool now holds 6 (soft-cap exceeded) + for (let i = 1; i <= 6; i++) expect(isLbugReady(`p-${i}`)).toBe(true); + + // Unpin the oldest, then init a 7th repo — the now-unpinned p-1 is the LRU + // victim. + unpinRepo('p-1'); + await init('p-7'); + expect(isLbugReady('p-1')).toBe(false); + expect(isLbugReady('p-7')).toBe(true); + }); + + it('reference-counts leases: two pins need two unpins before eviction (Finding 1)', async () => { + // Fill the pool to capacity, all leased. + for (let i = 1; i <= 4; i++) { + await init(`rc-${i}`); + pinRepo(`rc-${i}`); + } + await init('rc-shared'); + pinRepo('rc-shared'); // lease 1 + pinRepo('rc-shared'); // lease 2 (two holders) + + // Release ONE lease — a holder remains, so rc-shared stays exempt even + // under eviction pressure. + unpinRepo('rc-shared'); + await init('rc-extra'); // evictLRU finds no unpinned victim → pool grows + expect(isLbugReady('rc-shared')).toBe(true); + + // Release the LAST lease — now rc-shared (oldest unpinned) is evictable. + unpinRepo('rc-shared'); + await init('rc-extra2'); + expect(isLbugReady('rc-shared')).toBe(false); + }); + + it('unpinRepo floors at zero and tolerates unknown repoIds', () => { + expect(() => { + unpinRepo('never-touched'); // unknown repoId → no-op + pinRepo('floor-x'); + unpinRepo('floor-x'); // count 0 → key deleted + unpinRepo('floor-x'); // already gone → no-op, never a negative count + }).not.toThrow(); + }); + + it('pinRepo returns a disposer that releases exactly once and composes with refcount', async () => { + for (let i = 1; i <= 4; i++) { + await init(`d-${i}`); + pinRepo(`d-${i}`); + } + await init('d-shared'); + const release1 = pinRepo('d-shared'); // lease 1 + const release2 = pinRepo('d-shared'); // lease 2 + + release1(); + release1(); // double-call is a no-op — must NOT decrement lease 2 + + // lease 2 still held → d-shared survives eviction pressure. + await init('d-extra'); + expect(isLbugReady('d-shared')).toBe(true); + + // Release the last lease via its own disposer → now evictable. + release2(); + await init('d-extra2'); + expect(isLbugReady('d-shared')).toBe(false); + }); +}); From 1967512211cfecb4b65d54ef76f21c3a924691da Mon Sep 17 00:00:00 2001 From: Parafee41 Date: Sun, 14 Jun 2026 15:42:16 +0800 Subject: [PATCH 16/16] fix(cli): preserve trailing spaces in git roots (#2192) --- gitnexus/src/storage/git.ts | 32 +++++++++++++++------------- gitnexus/test/unit/git-utils.test.ts | 14 ++++++++++++ gitnexus/test/unit/git.test.ts | 6 +++--- 3 files changed, 34 insertions(+), 18 deletions(-) diff --git a/gitnexus/src/storage/git.ts b/gitnexus/src/storage/git.ts index 25b0cf2f0..e539d30c9 100644 --- a/gitnexus/src/storage/git.ts +++ b/gitnexus/src/storage/git.ts @@ -4,6 +4,8 @@ import path from 'path'; // Git utilities for repository detection, commit tracking, and diff analysis +const chompGitOutput = (value: Buffer): string => value.toString().replace(/\r?\n$/, ''); + export const isGitRepo = (repoPath: string): boolean => { try { execSync('git rev-parse --is-inside-work-tree', { @@ -102,14 +104,14 @@ export const getRemoteUrl = (repoPath: string): string | undefined => { */ export const getGitRoot = (fromPath: string): string | null => { try { - const raw = execSync('git rev-parse --show-toplevel', { - cwd: fromPath, - // Suppress stderr -- see getCurrentCommit comment and #1172. - stdio: ['ignore', 'pipe', 'ignore'], - windowsHide: true, - }) - .toString() - .trim(); + const raw = chompGitOutput( + execSync('git rev-parse --show-toplevel', { + cwd: fromPath, + // Suppress stderr -- see getCurrentCommit comment and #1172. + stdio: ['ignore', 'pipe', 'ignore'], + windowsHide: true, + }), + ); // On Windows, git returns /d/Projects/Foo — path.resolve normalizes to D:\Projects\Foo return path.resolve(raw); } catch { @@ -146,13 +148,13 @@ export const getGitRoot = (fromPath: string): string | null => { */ export const getCanonicalRepoRoot = (fromPath: string): string | null => { try { - const commonDir = execSync('git rev-parse --path-format=absolute --git-common-dir', { - cwd: fromPath, - stdio: ['ignore', 'pipe', 'ignore'], - windowsHide: true, - }) - .toString() - .trim(); + const commonDir = chompGitOutput( + execSync('git rev-parse --path-format=absolute --git-common-dir', { + cwd: fromPath, + stdio: ['ignore', 'pipe', 'ignore'], + windowsHide: true, + }), + ); if (!commonDir) return null; // Common dir is `/.git` for both the main checkout and all // linked worktrees. Its parent is the canonical repo root. diff --git a/gitnexus/test/unit/git-utils.test.ts b/gitnexus/test/unit/git-utils.test.ts index dba2b8886..14a51a385 100644 --- a/gitnexus/test/unit/git-utils.test.ts +++ b/gitnexus/test/unit/git-utils.test.ts @@ -146,6 +146,20 @@ describe('getGitRoot', () => { fs.rmSync(tmpDir, { recursive: true, force: true }); } }); + + it('preserves a trailing-space repository directory name (#2190)', async () => { + const { getGitRoot } = await import('../../src/storage/git.js'); + const parentDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-space-root-')); + const repoDir = path.join(parentDir, 'repo '); + try { + fs.mkdirSync(repoDir); + execSync('git init -q', { cwd: repoDir }); + + expect(getGitRoot(repoDir)).toBe(path.resolve(repoDir)); + } finally { + fs.rmSync(parentDir, { recursive: true, force: true }); + } + }); }); // ─── getRemoteUrl ───────────────────────────────────────────────────────── diff --git a/gitnexus/test/unit/git.test.ts b/gitnexus/test/unit/git.test.ts index a46bd0579..cfbf7d74e 100644 --- a/gitnexus/test/unit/git.test.ts +++ b/gitnexus/test/unit/git.test.ts @@ -159,11 +159,11 @@ describe('git utilities', () => { ); }); - it('trims output before resolving path', () => { - mockExecSync.mockReturnValueOnce(Buffer.from(' /repo \n')); + it('preserves path whitespace while removing the trailing newline', () => { + mockExecSync.mockReturnValueOnce(Buffer.from('/repo \n')); const result = getGitRoot('/repo/src'); expect(result).not.toBeNull(); - expect(result!.trim()).toBe(result); + expect(result).toBe(path.resolve('/repo ')); }); });