mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-07 02:58:02 +00:00
Merge remote-tracking branch's main sync into feat/factory-droid-integration
This commit is contained in:
commit
0d056e8888
93 changed files with 3896 additions and 395 deletions
5
.github/actionlint.yaml
vendored
Normal file
5
.github/actionlint.yaml
vendored
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
# Custom self-hosted runner labels actionlint can't discover on its own.
|
||||
# gitnexus-evolution: the skill-evolution EC2 runner (infra/gitnexus-evolution/).
|
||||
self-hosted-runner:
|
||||
labels:
|
||||
- gitnexus-evolution
|
||||
2
.github/claude-canary-runtime/package-lock.json
generated
vendored
2
.github/claude-canary-runtime/package-lock.json
generated
vendored
|
|
@ -11,7 +11,7 @@
|
|||
"@anthropic-ai/claude-code": "2.1.214"
|
||||
},
|
||||
"engines": {
|
||||
"node": "22.16.0"
|
||||
"node": "22.18.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@anthropic-ai/claude-code": {
|
||||
|
|
|
|||
2
.github/claude-canary-runtime/package.json
vendored
2
.github/claude-canary-runtime/package.json
vendored
|
|
@ -3,7 +3,7 @@
|
|||
"version": "0.0.0",
|
||||
"private": true,
|
||||
"engines": {
|
||||
"node": "22.16.0"
|
||||
"node": "22.18.0"
|
||||
},
|
||||
"dependencies": {
|
||||
"@anthropic-ai/claude-code": "2.1.214"
|
||||
|
|
|
|||
2
.github/gitnexus-review-runtime/package-lock.json
generated
vendored
2
.github/gitnexus-review-runtime/package-lock.json
generated
vendored
|
|
@ -11,7 +11,7 @@
|
|||
"gitnexus": "1.6.9"
|
||||
},
|
||||
"engines": {
|
||||
"node": "22.16.0"
|
||||
"node": "22.18.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@emnapi/runtime": {
|
||||
|
|
|
|||
2
.github/gitnexus-review-runtime/package.json
vendored
2
.github/gitnexus-review-runtime/package.json
vendored
|
|
@ -3,7 +3,7 @@
|
|||
"private": true,
|
||||
"version": "1.0.0",
|
||||
"engines": {
|
||||
"node": "22.16.0"
|
||||
"node": "22.18.0"
|
||||
},
|
||||
"dependencies": {
|
||||
"gitnexus": "1.6.9"
|
||||
|
|
|
|||
25
.github/workflows/ci-tests.yml
vendored
25
.github/workflows/ci-tests.yml
vendored
|
|
@ -378,15 +378,16 @@ jobs:
|
|||
"$PREFIX/bin/gitnexus" --version
|
||||
fi
|
||||
|
||||
# Node engines-floor gate (#2372). The embedding resolvers statically named
|
||||
# `module.registerHooks`, which only exists on Node >= 22.15 / >= 23.5, so on
|
||||
# the supported floor (engines: >=22.0.0) those ESM modules failed to LINK —
|
||||
# a class vitest/tsx transforms structurally mask, and the default
|
||||
# `node-version: 22` (resolves to latest) never hits. Build the dist on 22.x,
|
||||
# then import-link every module R1 names as a load surface on a pinned 22.14
|
||||
# so a regression fails here instead of shipping to users on that Node range.
|
||||
# Node engines-floor gate (#2372). A module that statically names an API
|
||||
# newer than the supported floor (e.g. `module.registerHooks`, added in
|
||||
# 22.15) fails to LINK on the floor — a class vitest/tsx transforms
|
||||
# structurally mask, and the default `node-version: 22` (resolves to latest)
|
||||
# never hits. Build the dist on 22.x, then import-link every module R1 names
|
||||
# as a load surface on the pinned engines floor (22.18.0, per package.json
|
||||
# `engines: ^22.18.0 || >=24.11.0`) so a regression fails here instead of
|
||||
# shipping to users on the minimum supported Node.
|
||||
node-floor-compat:
|
||||
name: node floor compat (22.14)
|
||||
name: node floor compat (22.18)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
|
|
@ -415,14 +416,14 @@ jobs:
|
|||
# (so no package-manager cache is needed).
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: '22.14.0'
|
||||
node-version: '22.18.0'
|
||||
package-manager-cache: false
|
||||
- name: Import-link the built dist on Node 22.14
|
||||
- name: Import-link the built dist on Node 22.18
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
node --version
|
||||
node --version | grep -q '^v22\.14\.' || { echo "expected Node 22.14.x" >&2; exit 1; }
|
||||
node --version | grep -q '^v22\.18\.' || { echo "expected Node 22.18.x" >&2; exit 1; }
|
||||
for m in \
|
||||
core/embeddings/runtime-install \
|
||||
core/embeddings/onnxruntime-node-resolver \
|
||||
|
|
@ -556,7 +557,7 @@ jobs:
|
|||
persist-credentials: false
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: '22.16.0'
|
||||
node-version: '22.18.0'
|
||||
cache: npm
|
||||
cache-dependency-path: |
|
||||
gitnexus/package-lock.json
|
||||
|
|
|
|||
10
.github/workflows/gitnexus-review-agent.yml
vendored
10
.github/workflows/gitnexus-review-agent.yml
vendored
|
|
@ -325,7 +325,7 @@ jobs:
|
|||
if: steps.context.outputs.ready == 'true'
|
||||
uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: '22.16.0'
|
||||
node-version: '22.18.0'
|
||||
|
||||
- name: Install and preflight Claude subprocess isolation
|
||||
id: isolation
|
||||
|
|
@ -377,7 +377,7 @@ jobs:
|
|||
.github/claude-canary-runtime/package-lock.json \
|
||||
"${runtime_dir}/package-lock.json"
|
||||
printf '%s\n' 'registry=https://registry.npmjs.org/' 'audit=false' 'fund=false' > "${npmrc}"
|
||||
test "$(node --version)" = 'v22.16.0'
|
||||
test "$(node --version)" = 'v22.18.0'
|
||||
test "$(uname -m)" = 'x86_64'
|
||||
|
||||
# The trusted lock and these independent receipts pin both the thin
|
||||
|
|
@ -398,7 +398,7 @@ jobs:
|
|||
if (
|
||||
lock.lockfileVersion !== 3 ||
|
||||
lock.packages?.['']?.dependencies?.['@anthropic-ai/claude-code'] !== '2.1.214' ||
|
||||
lock.packages?.['']?.engines?.node !== '22.16.0'
|
||||
lock.packages?.['']?.engines?.node !== '22.18.0'
|
||||
) {
|
||||
throw new Error('Claude runtime lock root is not exact');
|
||||
}
|
||||
|
|
@ -506,7 +506,7 @@ jobs:
|
|||
install -m 0600 .github/gitnexus-review-runtime/package.json "${runtime_dir}/package.json"
|
||||
install -m 0600 .github/gitnexus-review-runtime/package-lock.json "${runtime_dir}/package-lock.json"
|
||||
printf '%s\n' 'registry=https://registry.npmjs.org/' 'audit=false' 'fund=false' > "${npmrc}"
|
||||
test "$(node --version)" = 'v22.16.0'
|
||||
test "$(node --version)" = 'v22.18.0'
|
||||
npm ci \
|
||||
--prefix "${runtime_dir}" \
|
||||
--userconfig "${npmrc}" \
|
||||
|
|
@ -1241,7 +1241,7 @@ jobs:
|
|||
CLAUDE_CONFIG_DIR: ${{ runner.temp }}/gitnexus-review-claude-config
|
||||
CLAUDE_WORKING_DIR: ${{ runner.temp }}/gitnexus-review-control
|
||||
NPM_CONFIG_IGNORE_SCRIPTS: 'true'
|
||||
NODE_VERSION: '22.16.0'
|
||||
NODE_VERSION: '22.18.0'
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
path_to_claude_code_executable: ${{ runner.temp }}/gitnexus-review-claude-runtime/node_modules/@anthropic-ai/claude-code/bin/claude.exe
|
||||
|
|
|
|||
30
.github/workflows/gitnexus-skill-evolution.yml
vendored
30
.github/workflows/gitnexus-skill-evolution.yml
vendored
|
|
@ -12,12 +12,34 @@
|
|||
# App that opens the promotion PR). The Mint-App-Token step hard-fails
|
||||
# without them once a promotion is detected. Verify the App installation
|
||||
# is scoped to this repo with only Contents: RW + Pull requests: RW.
|
||||
# [ ] Create the protected Environment `gitnexus-evolution` with a
|
||||
# [x] Create the protected Environment `gitnexus-evolution` with a
|
||||
# deployment-branch rule restricting it to `main`, and ideally scope the
|
||||
# three secrets above to that Environment. workflow_dispatch runs this
|
||||
# workflow (and eval/workflow_bench/evolve.py) from the *dispatched ref*,
|
||||
# so this server-side rule — not a code-side guard the branch could edit
|
||||
# away — is what stops a non-main branch from running with the secrets.
|
||||
# [x] Register a self-hosted runner labeled `gitnexus-evolution` (a dedicated
|
||||
# EC2 box works well). GitHub-hosted runners hard-cap job execution at 6
|
||||
# hours, non-configurable — too short once a benchmark session actually
|
||||
# invokes Skill/MCP tools for real. Self-hosted runners cap at 5 days
|
||||
# instead. This job only ever runs on schedule/workflow_dispatch, never
|
||||
# on fork-PR content, so the usual public-repo self-hosted-runner risk
|
||||
# doesn't apply — still keep the box dedicated to this workflow, with
|
||||
# outbound-only network access, and prefer on-demand over Spot (a Spot
|
||||
# reclaim mid-run loses the same way a 6-hour timeout does). Instance,
|
||||
# security group, and IAM setup are documented privately, not in this
|
||||
# repo — publishing the exact topology of a real, live AWS account
|
||||
# isn't safe to do in a public repo even without literal secrets.
|
||||
# Accepted tradeoff: the box is stopped between runs (an EventBridge
|
||||
# schedule starts it ~15min before the Saturday cron and stops it 24h
|
||||
# later) but is not destroyed/recreated per run, so it isn't fully
|
||||
# ephemeral — a compromise between the review-flagged ideal (re-image
|
||||
# between runs, bounding how long the injected model API key could
|
||||
# matter if the box were ever compromised some other way) and the added
|
||||
# complexity of per-job ephemeral provisioning for a job that runs at
|
||||
# most weekly. Revisit if run frequency increases or the threat model
|
||||
# changes; stopping already bounds the exposure window to the job's own
|
||||
# runtime on 1 day out of 7.
|
||||
# [ ] Run workflow_dispatch once and confirm: containment preflight passes,
|
||||
# the benchmark completes inside the job timeout, the results artifact
|
||||
# uploads, and a promotion (if any) opens a well-formed PR.
|
||||
|
|
@ -77,13 +99,13 @@ jobs:
|
|||
github.event_name == 'workflow_dispatch' ||
|
||||
vars.GITNEXUS_EVOLUTION_ENABLED == 'true'
|
||||
)
|
||||
runs-on: ubuntu-latest
|
||||
runs-on: [self-hosted, linux, x64, gitnexus-evolution]
|
||||
# Gate promotion runs on a protected Environment. An admin must attach a
|
||||
# deployment-branch rule (main only) and ideally scope the three secrets to
|
||||
# it — server-side enforcement a dispatched non-main ref cannot bypass by
|
||||
# editing its own workflow copy. See the activation checklist above.
|
||||
environment: gitnexus-evolution
|
||||
timeout-minutes: 355 # ceiling just under GitHub's 360-minute hard cap
|
||||
timeout-minutes: 1440 # self-hosted ceiling is 5 days (7200min); 24h is a generous margin over a single-generation serial run
|
||||
permissions:
|
||||
contents: read # The promotion PR uses a short-lived App token minted below.
|
||||
env:
|
||||
|
|
@ -110,7 +132,7 @@ jobs:
|
|||
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: '22.16.0'
|
||||
node-version: '22.18.0'
|
||||
cache: npm
|
||||
cache-dependency-path: |
|
||||
gitnexus/package-lock.json
|
||||
|
|
|
|||
3
.gitignore
vendored
3
.gitignore
vendored
|
|
@ -68,9 +68,8 @@ gitnexus-web/test-results/
|
|||
eval/.coverage
|
||||
eval/.hypothesis/
|
||||
|
||||
# Local docs (docs/plans/ stays tracked — gitnexus-plan output travels with the work)
|
||||
# Local docs — planning output (gitnexus-plan / gitnexus-work) stays local, not tracked
|
||||
docs/*
|
||||
!docs/plans/
|
||||
|
||||
gitnexus/test/fixtures/mini-repo/*.md
|
||||
gitnexus/test/fixtures/mini-repo/.claude
|
||||
|
|
|
|||
|
|
@ -13,7 +13,7 @@ This project uses the [PolyForm Noncommercial License 1.0.0](https://polyformpro
|
|||
|
||||
## Development setup
|
||||
|
||||
**Prerequisites:** Node.js — `gitnexus/` requires `>=22.0.0` and `gitnexus-web/` requires `^20.19.0 || >=22.12.0` (enforced via the `engines` field in each package). Use `nvm install` to match the local version.
|
||||
**Prerequisites:** Node.js — `gitnexus/` requires `^22.18.0 || >=24.11.0` and `gitnexus-web/` requires `^20.19.0 || >=22.12.0` (enforced via the `engines` field in each package). Use `nvm install` to match the local version.
|
||||
|
||||
1. Clone the repository.
|
||||
2. **Shared package:** `cd gitnexus-shared && npm install && npm run build`
|
||||
|
|
|
|||
|
|
@ -504,6 +504,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max
|
|||
| `PROF_LBUG_LOAD` | unset | When `1`, emits one `[lbug-load prof]` summary line per `loadGraphToLbug` call breaking the graph-DB persistence wall into stages (`csv-emit` / `copy-nodes` / `copy-rels` / `fallback` / `total`) plus node & edge counts. Zero-cost when unset. | Attributing large-repo analyze wall time across CSV generation vs. LadybugDB `COPY` (issue #2203) — the analyze "emit" timing is the scope-resolution bucket, not this DB-write path. |
|
||||
| `GITNEXUS_MAX_FILE_SIZE` | `512` (KB) | Walker skip threshold in KB. Hard cap is `32768` (tree-sitter buffer ceiling). Equivalent to `--max-file-size <kb>`. | Indexing repos with intentionally-large source files (generated parsers, vendored bundles) that should still be parsed. |
|
||||
| `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS` | `30000` | Worker idle timeout in milliseconds before retry/fallback. Equivalent to `--worker-timeout <seconds>` × 1000. | Slow-parsing files (large minified JS, deeply-nested TS types) that legitimately need more than 30s. |
|
||||
| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget in milliseconds for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. | Slow or heavily loaded hosts where a full pool cold-starting concurrently needs more than 5s, and analyze aborts with "did not report ready within 5000ms". |
|
||||
| `GITNEXUS_FTS_STEMMER` | `porter` | Stemmer used when rebuilding BM25/FTS indexes. Use `none` for CJK-heavy repositories, or a language stemmer such as `german`, `french`, or `spanish` for matching repository comments. Re-run `gitnexus analyze --repair-fts` after changing it. | Keyword search quality is poor for non-English comments or identifiers under English stemming. |
|
||||
| `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold in bytes. Equivalent to `--wal-checkpoint-threshold <bytes>`. `-1` keeps LadybugDB's stock threshold (~16 MiB). Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | You need a larger or smaller WAL auto-checkpoint threshold for your analyze workload. |
|
||||
| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling in bytes for every GitNexus database (analyze, MCP server, serve, group bridges). `0` restores LadybugDB's native unbounded default of 80% of system RAM; invalid values warn and fall back to the default (#2557). | A long-lived `gitnexus mcp` or a big incremental `analyze` uses too much memory, or a huge repo's working set genuinely needs a pool larger than 2 GiB. |
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ from __future__ import annotations
|
|||
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import stat
|
||||
import subprocess
|
||||
import sys
|
||||
|
|
@ -19,10 +20,13 @@ from workflow_bench.process_control import ManagedProcessResult, run_managed
|
|||
from workflow_bench.proposer_sandbox import (
|
||||
MAX_BUNDLE_BYTES,
|
||||
MAX_EVIDENCE_FILE_BYTES,
|
||||
SANDBOX_NODE,
|
||||
SANDBOX_PYTHON3,
|
||||
SANDBOX_SHELL_PREFIX,
|
||||
SANDBOX_USER_SKILLS,
|
||||
ReadOnlyMount,
|
||||
SandboxError,
|
||||
_runtime_mount_args,
|
||||
build_claude_settings,
|
||||
build_sandbox_environment,
|
||||
prepare_sandbox,
|
||||
|
|
@ -58,6 +62,11 @@ def test_environment_is_allowlisted_and_shell_children_are_credential_free(monke
|
|||
assert settings["sandbox"]["failIfUnavailable"] is True
|
||||
assert settings["sandbox"]["allowUnsandboxedCommands"] is False
|
||||
assert settings["sandbox"]["network"]["deniedDomains"] == ["*"]
|
||||
# ENV_SCRUB forces "default" mode; the proposer's tools (Bash writes the
|
||||
# overlay) run headless only because they are explicitly pre-approved.
|
||||
# Requesting a non-default defaultMode would merely warn, so it must be gone.
|
||||
assert settings["permissions"]["allow"] == ["Read", "Grep", "Glob", "Bash"]
|
||||
assert "defaultMode" not in settings["permissions"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
|
|
@ -157,6 +166,27 @@ def test_sandbox_command_has_minimal_mounts_and_no_host_root_bind(tmp_path: Path
|
|||
)
|
||||
assert probe.returncode == 0, probe.stderr
|
||||
assert probe.stdout == "/home/agent|/opt/claude:/usr/local/bin:/usr/bin:/bin"
|
||||
|
||||
# The evidence-provenance.mjs plan-writer's PATH-scan trusts a Python 3
|
||||
# candidate only if it (and its directory) is owned by root or by the
|
||||
# current process — real /usr/bin/python3 is root-owned on the host,
|
||||
# which surfaces as the kernel's overflow uid inside this
|
||||
# --unshare-user sandbox (root itself is never mapped in). This wrapper
|
||||
# is freshly created by the host process instead, so it's trusted, and
|
||||
# it must still exec through to a real, working Python 3.
|
||||
python3_index = argv.index(SANDBOX_PYTHON3)
|
||||
assert argv[python3_index - 2] == "--ro-bind"
|
||||
python3_wrapper = Path(argv[python3_index - 1])
|
||||
assert stat.S_IMODE(python3_wrapper.stat().st_mode) == 0o500
|
||||
version = subprocess.run(
|
||||
[str(python3_wrapper), "-I", "-S", "-c", "import sys; print(sys.version_info[0])"],
|
||||
text=True,
|
||||
capture_output=True,
|
||||
check=False,
|
||||
)
|
||||
assert version.returncode == 0, version.stderr
|
||||
assert version.stdout.strip() == "3"
|
||||
|
||||
assert SANDBOX_USER_SKILLS in argv
|
||||
user_skills_index = argv.index(SANDBOX_USER_SKILLS)
|
||||
assert argv[user_skills_index - 2] == "--ro-bind"
|
||||
|
|
@ -164,6 +194,36 @@ def test_sandbox_command_has_minimal_mounts_and_no_host_root_bind(tmp_path: Path
|
|||
assert not private_root.exists()
|
||||
|
||||
|
||||
def test_runtime_mounts_bind_the_resolved_node_to_a_fresh_sandbox_path(monkeypatch) -> None:
|
||||
# sanitized_graph.py and runner_sessions.py invoke the sandboxed graph CLI
|
||||
# via SANDBOX_NODE. node's real host location varies (GitHub-hosted
|
||||
# runner images happen to have one under /usr/local/bin; a self-hosted
|
||||
# runner's actions/setup-node installs into its own tool-cache directory
|
||||
# instead), so this must bind to a FRESH sandbox path like /opt/claude/...
|
||||
# rather than anywhere under /usr, /bin, /lib, or /lib64: those are
|
||||
# already read-only bound by this same function, and bwrap can't create
|
||||
# a new mount-point file inside an already-read-only tree when the real
|
||||
# path doesn't already exist there on the host (observed empirically:
|
||||
# "bwrap: Can't create file at /usr/local/bin/node: Read-only file
|
||||
# system" when this bind first targeted that path on a self-hosted
|
||||
# runner where node isn't really there).
|
||||
monkeypatch.setattr(
|
||||
"workflow_bench.proposer_sandbox.shutil.which",
|
||||
lambda name: "/opt/hostedtoolcache/node/22.18.0/x64/bin/node" if name == "node" else None,
|
||||
)
|
||||
args = _runtime_mount_args()
|
||||
node_index = args.index("/opt/hostedtoolcache/node/22.18.0/x64/bin/node")
|
||||
assert args[node_index - 1] == "--ro-bind"
|
||||
assert args[node_index + 1] == SANDBOX_NODE
|
||||
assert not any(SANDBOX_NODE.startswith(bound + "/") for bound in ("/usr", "/bin", "/lib", "/lib64"))
|
||||
|
||||
|
||||
def test_runtime_mounts_skip_the_node_bind_when_node_is_unresolvable(monkeypatch) -> None:
|
||||
monkeypatch.setattr("workflow_bench.proposer_sandbox.shutil.which", lambda name: None)
|
||||
args = _runtime_mount_args()
|
||||
assert SANDBOX_NODE not in args
|
||||
|
||||
|
||||
def test_stricter_prefix_freezes_evaluated_skills_and_can_unshare_network(tmp_path: Path) -> None:
|
||||
clone = tmp_path / "clone"
|
||||
skill = clone / ".claude" / "skills" / "gitnexus-work"
|
||||
|
|
@ -192,6 +252,43 @@ def test_stricter_prefix_freezes_evaluated_skills_and_can_unshare_network(tmp_pa
|
|||
assert prefix[user_index - 2] == "--ro-bind"
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
|
||||
reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job",
|
||||
)
|
||||
def test_real_bubblewrap_runs_node_from_outside_the_bound_trees(tmp_path: Path, monkeypatch) -> None:
|
||||
# Reproduces the self-hosted-runner failure directly: node resolved from
|
||||
# a path outside /usr, /bin, /lib, /lib64 (actions/setup-node's own
|
||||
# tool-cache convention) must still be reachable inside the sandbox at
|
||||
# SANDBOX_NODE. A real node copied to a fresh, non-system location stands
|
||||
# in for the tool-cache install; argv-construction tests alone can't
|
||||
# catch a bwrap-level "Can't create file ...: Read-only file system"
|
||||
# (the actual error this fix resolves), only a real bwrap invocation can.
|
||||
real_node = shutil.which("node")
|
||||
if not real_node:
|
||||
pytest.skip("no node on PATH to relocate for this canary")
|
||||
toolcache = tmp_path / "toolcache"
|
||||
toolcache.mkdir()
|
||||
relocated_node = toolcache / "node"
|
||||
shutil.copy2(real_node, relocated_node)
|
||||
relocated_node.chmod(0o755)
|
||||
# Only fake "node"'s resolution -- prepare_sandbox's own bwrap/claude
|
||||
# lookups (_resolve_executable) also go through shutil.which, and must
|
||||
# keep resolving for real or preflight fails before the sandbox is even
|
||||
# built.
|
||||
real_which = shutil.which
|
||||
monkeypatch.setattr(
|
||||
"workflow_bench.proposer_sandbox.shutil.which",
|
||||
lambda name: str(relocated_node) if name == "node" else real_which(name),
|
||||
)
|
||||
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
with prepare_sandbox(clone=clone, claude_bin=Path(sys.executable), preflight=True) as sandbox:
|
||||
result = sandbox.run([SANDBOX_NODE, "--version"], timeout=10)
|
||||
assert result.ok, result.stderr_tail
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
|
||||
reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job",
|
||||
|
|
@ -716,8 +813,10 @@ for line in sys.stdin:
|
|||
"--strict-mcp-config",
|
||||
"--mcp-config",
|
||||
mcp_config,
|
||||
"--permission-mode",
|
||||
"dontAsk",
|
||||
# No --permission-mode: mirrors production (run_proposer).
|
||||
# ENV_SCRUB forces "default"; Bash runs only because
|
||||
# settings permissions.allow pre-approves it. This is the
|
||||
# authoritative empirical gate for that behavior.
|
||||
"--model",
|
||||
"claude-canary-20260718",
|
||||
"--allowedTools",
|
||||
|
|
@ -743,4 +842,3 @@ for line in sys.stdin:
|
|||
assert bash_result.get("is_error") is not True, bash_result
|
||||
assert (clone / "bash-called").read_text() == "canary"
|
||||
assert (clone / "mcp-called").read_text() == "ok"
|
||||
|
||||
|
|
|
|||
|
|
@ -262,3 +262,37 @@ def test_phase_workspace_accepts_new_regular_review_output(tmp_path):
|
|||
artifact.write_text("new review")
|
||||
|
||||
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
|
||||
|
||||
|
||||
def test_phase_workspace_ignores_claude_sandbox_bootstrap_noise(tmp_path):
|
||||
# Reproduced empirically: Claude Code's own enableWeakerNestedSandbox
|
||||
# bootstrap creates this exact set of paths on every session regardless
|
||||
# of task or model output (a trivial "say OK" prompt was enough). None
|
||||
# of it is something the model decided to write, so it must not read as
|
||||
# an unauthorized planning-phase change.
|
||||
before = runner_artifacts.workspace_snapshot(tmp_path)
|
||||
(tmp_path / ".claude" / "agents").mkdir(parents=True)
|
||||
(tmp_path / ".claude" / "commands").mkdir(parents=True)
|
||||
(tmp_path / ".claude" / ".cc-writes").write_text("{}")
|
||||
(tmp_path / ".env").write_text("")
|
||||
(tmp_path / ".env.development.local").write_text("")
|
||||
(tmp_path / ".npmrc").write_text("")
|
||||
(tmp_path / "package.json").write_text("{}")
|
||||
(tmp_path / "node_modules").mkdir()
|
||||
(tmp_path / "node_modules" / ".bin").mkdir()
|
||||
artifact = tmp_path / "review-output.md"
|
||||
artifact.write_text("new review")
|
||||
|
||||
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
|
||||
|
||||
|
||||
def test_phase_workspace_still_rejects_a_genuinely_unauthorized_change(tmp_path):
|
||||
# The bootstrap-noise exclusion must stay narrow: an actual source-file
|
||||
# edit outside the allowed artifact still has to be caught.
|
||||
before = runner_artifacts.workspace_snapshot(tmp_path)
|
||||
(tmp_path / "src.py").write_text("changed")
|
||||
artifact = tmp_path / "review-output.md"
|
||||
artifact.write_text("new review")
|
||||
|
||||
with pytest.raises(ValueError, match="unauthorized workspace path"):
|
||||
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
|
||||
|
|
|
|||
|
|
@ -113,6 +113,27 @@ def test_small_assets_use_a_bounded_buffered_fallback(monkeypatch, tmp_path: Pat
|
|||
assert (clone / "second").read_bytes() == b"def"
|
||||
|
||||
|
||||
def test_default_buffered_fallback_budget_covers_a_realistic_large_asset(
|
||||
monkeypatch,
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
# 20 MiB exceeds the old 16 MiB default but must fit comfortably under
|
||||
# the current default, proving the real (non-monkeypatched) budget
|
||||
# constant is sized for a realistic large sandbox_copy asset such as the
|
||||
# harness's own pre-built graph index, not just tiny fixtures.
|
||||
payload = os.urandom(20 * 1024 * 1024)
|
||||
repo, task = _repo_and_task(tmp_path, {"large": payload})
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
monkeypatch.setattr(task_assets, "_try_reflink", lambda *_args: False)
|
||||
|
||||
with TaskAssetCache(tmp_path / "cache") as cache:
|
||||
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
|
||||
snapshot.materialize(clone)
|
||||
|
||||
assert (clone / "large").read_bytes() == payload
|
||||
|
||||
|
||||
def test_large_asset_without_reflink_fails_before_publish_and_cleans_staging(
|
||||
monkeypatch,
|
||||
tmp_path: Path,
|
||||
|
|
|
|||
|
|
@ -10,6 +10,7 @@ import yaml
|
|||
|
||||
from workflow_bench.runner import (
|
||||
aggregate,
|
||||
broken_incumbent_arms,
|
||||
build_parser,
|
||||
infra_error_record,
|
||||
normalized_model_identifier,
|
||||
|
|
@ -64,6 +65,7 @@ def test_aggregate_takes_medians_and_counts_resolved():
|
|||
"valid_runs": 3,
|
||||
"excluded_runs": 0,
|
||||
"transcripts_missing": 0,
|
||||
"error_kinds": {},
|
||||
}
|
||||
|
||||
|
||||
|
|
@ -172,7 +174,7 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs():
|
|||
}
|
||||
assert containment["timeout-minutes"] == 20
|
||||
assert containment_node_setup["with"] == {
|
||||
"node-version": "22.16.0",
|
||||
"node-version": "22.18.0",
|
||||
"cache": "npm",
|
||||
"cache-dependency-path": "gitnexus/package-lock.json\ngitnexus-shared/package-lock.json\n",
|
||||
}
|
||||
|
|
@ -333,6 +335,61 @@ def test_render_report_surfaces_excluded_and_unverified_runs():
|
|||
assert "no locatable session transcript" in report
|
||||
|
||||
|
||||
def test_render_report_surfaces_why_each_row_failed():
|
||||
results = {
|
||||
"t": {
|
||||
"workflow": aggregate(
|
||||
[record(resolved=False, error_kind="plan-evidence-invalid")],
|
||||
),
|
||||
}
|
||||
}
|
||||
report = render_report(results)
|
||||
assert "plan-evidence-invalid×1" in report
|
||||
|
||||
|
||||
def test_broken_incumbent_arms_flags_an_incumbent_that_resolved_nothing():
|
||||
results = {
|
||||
"t1": {"workflow": aggregate([record(resolved=False, error_kind="plan-evidence-invalid")])},
|
||||
"t2": {"workflow": aggregate([record(resolved=False, error_kind="plan-evidence-invalid")])},
|
||||
}
|
||||
assert broken_incumbent_arms(results, {"workflow"}) == ["workflow"]
|
||||
|
||||
|
||||
def test_broken_incumbent_arms_ignores_a_merely_underperforming_candidate():
|
||||
# The incumbent works fine; only the candidate arm fails. That's a normal,
|
||||
# expected "bad candidate" outcome and must not read as a broken harness.
|
||||
results = {
|
||||
"t1": {
|
||||
"workflow": aggregate([record(resolved=True)]),
|
||||
"candidate_workflow": aggregate([record(resolved=False, error_kind="verify-failed")]),
|
||||
},
|
||||
}
|
||||
assert broken_incumbent_arms(results, {"workflow"}) == []
|
||||
|
||||
|
||||
def test_broken_incumbent_arms_flags_an_incumbent_with_zero_valid_runs():
|
||||
# Every run excluded via an excluded-but-non-systemic error_kind
|
||||
# ("evidence-unverified"): valid_runs == 0 for every task, which the old
|
||||
# `valid_runs > 0` guard let sail through silently, and which the outage
|
||||
# streak breaker also doesn't catch (it resets rather than accumulates
|
||||
# on this exact error_kind -- see test_systemic_outage_streak_resets_on_non_outage).
|
||||
results = {
|
||||
"t1": {"workflow": aggregate([record(resolved=False, error_kind="evidence-unverified")])},
|
||||
"t2": {"workflow": aggregate([record(resolved=False, error_kind="evidence-unverified")])},
|
||||
}
|
||||
assert results["t1"]["workflow"]["valid_runs"] == 0
|
||||
assert broken_incumbent_arms(results, {"workflow"}) == ["workflow"]
|
||||
|
||||
|
||||
def test_broken_incumbent_arms_ignores_partial_incumbent_failure():
|
||||
# Resolved in at least one task — struggling, not broken.
|
||||
results = {
|
||||
"t1": {"workflow": aggregate([record(resolved=False, error_kind="verify-failed")])},
|
||||
"t2": {"workflow": aggregate([record(resolved=True)])},
|
||||
}
|
||||
assert broken_incumbent_arms(results, {"workflow"}) == []
|
||||
|
||||
|
||||
def test_infra_error_record_captures_the_failure_and_is_excluded():
|
||||
exc = subprocess.TimeoutExpired(cmd="claude -p", timeout=5)
|
||||
rec = infra_error_record(exc)
|
||||
|
|
|
|||
|
|
@ -167,6 +167,56 @@ def test_run_claude_forwards_the_named_model_to_every_session(monkeypatch, tmp_p
|
|||
assert captured[captured.index("--model") + 1] == "claude-sonnet-4-20250514"
|
||||
|
||||
|
||||
def test_run_claude_restricts_tools_via_tools_flag_outside_bare(monkeypatch, tmp_path):
|
||||
# Outside --bare, the built-in toolset defaults to everything (subagents,
|
||||
# WebFetch, Task, ...) and --allowedTools only pre-approves within that —
|
||||
# it does not narrow it. --tools is what actually restricts the set, so a
|
||||
# non-bare arm session must pass it or it silently gets a far wider
|
||||
# toolset than intended.
|
||||
captured: list[str] = []
|
||||
|
||||
def fake_run(command, **kwargs):
|
||||
captured.extend(command)
|
||||
return fake_cli_result(VALID_REPORT)
|
||||
|
||||
monkeypatch.setattr(runner_sessions, "run_managed", fake_run)
|
||||
runner.run_claude(
|
||||
"task",
|
||||
tmp_path,
|
||||
claude_bin="claude",
|
||||
timeout=5,
|
||||
bare=False,
|
||||
allowed_tools=["Read", "Edit", "Bash", "Skill"],
|
||||
)
|
||||
tools_idx = captured.index("--tools")
|
||||
assert captured[tools_idx + 1 : tools_idx + 5] == ["Read", "Edit", "Bash", "Skill"]
|
||||
allowed_idx = captured.index("--allowedTools")
|
||||
assert captured[allowed_idx + 1 : allowed_idx + 5] == ["Read", "Edit", "Bash", "Skill"]
|
||||
|
||||
|
||||
def test_run_claude_omits_tools_flag_under_bare(monkeypatch, tmp_path):
|
||||
# --bare already hard-restricts to Bash/Edit/Read on its own (a Claude
|
||||
# Code design choice, not something --tools/--allowedTools can widen or
|
||||
# narrow further), so bare sessions must not also pass --tools.
|
||||
captured: list[str] = []
|
||||
|
||||
def fake_run(command, **kwargs):
|
||||
captured.extend(command)
|
||||
return fake_cli_result(VALID_REPORT)
|
||||
|
||||
monkeypatch.setattr(runner_sessions, "run_managed", fake_run)
|
||||
runner.run_claude(
|
||||
"task",
|
||||
tmp_path,
|
||||
claude_bin="claude",
|
||||
timeout=5,
|
||||
bare=True,
|
||||
allowed_tools=["Read", "Edit", "Bash", "Skill"],
|
||||
)
|
||||
assert "--tools" not in captured
|
||||
assert "--allowedTools" in captured
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("proc", "expected_kind"),
|
||||
[
|
||||
|
|
@ -192,11 +242,24 @@ def test_run_claude_keeps_raw_subtype_and_stderr_tail(monkeypatch, tmp_path):
|
|||
"returncode": 1,
|
||||
"process_state": "exited",
|
||||
"stderr_tail": "rate limit hit",
|
||||
"stdout_tail": VALID_REPORT,
|
||||
"process_detail": None,
|
||||
"event_stream_error": None,
|
||||
}
|
||||
|
||||
|
||||
def test_run_claude_surfaces_stdout_tail_on_empty_stderr(monkeypatch, tmp_path):
|
||||
# A session can exit non-zero with an EMPTY stderr (e.g. a pre-flight
|
||||
# sandbox failure before any model turn ever runs) -- stdout_tail is then
|
||||
# the only place the actual event stream is visible, so it must not be
|
||||
# dropped just because stderr had nothing to say.
|
||||
proc = fake_cli_result(VALID_REPORT, returncode=1, stderr="")
|
||||
monkeypatch.setattr(runner_sessions, "run_managed", lambda *a, **k: proc)
|
||||
rec = runner.run_claude("task", tmp_path, claude_bin="claude", timeout=5)
|
||||
assert rec["error_detail"]["stderr_tail"] == ""
|
||||
assert rec["error_detail"]["stdout_tail"] == VALID_REPORT
|
||||
|
||||
|
||||
def test_run_arm_labels_completed_but_unverified_runs_verify_failed(monkeypatch, tmp_path):
|
||||
monkeypatch.setattr(runner, "run_claude", lambda *a, **k: session_record())
|
||||
monkeypatch.setattr(runner, "run_verify", lambda *a, **k: (False, "failed"))
|
||||
|
|
@ -269,6 +332,15 @@ def test_agent_tool_grants_are_exact_and_nomcp_has_no_graph_tools(monkeypatch, t
|
|||
assert captured[3]["mcp_config_json"] == '{"mcpServers":{}}'
|
||||
assert captured[3]["disallowed_tools"] == ["Skill", "mcp__gitnexus"]
|
||||
|
||||
# --bare hard-disables the Skill tool and every mcp__* tool regardless of
|
||||
# --allowedTools (a Claude Code design choice, not something the harness
|
||||
# can override) -- every arm here except baseline_nomcp needs Skill
|
||||
# and/or MCP tools, so only baseline_nomcp may still run under --bare.
|
||||
assert captured[0]["bare"] is False # workflow: planning session
|
||||
assert captured[1]["bare"] is False # review
|
||||
assert captured[2]["bare"] is False # workflow_direct
|
||||
assert captured[3]["bare"] is True # baseline_nomcp
|
||||
|
||||
|
||||
def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tmp_path):
|
||||
runtime = tmp_path / "gitnexus"
|
||||
|
|
@ -277,10 +349,12 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm
|
|||
runtime / "dist" / "cli",
|
||||
runtime / "node_modules",
|
||||
runtime / "vendor",
|
||||
runtime / "hooks" / "claude",
|
||||
shared / "dist",
|
||||
):
|
||||
directory.mkdir(parents=True)
|
||||
(runtime / "dist" / "cli" / "index.js").write_text("")
|
||||
(runtime / "hooks" / "claude" / "resolve-analyze-cmd.cjs").write_text("")
|
||||
(runtime / "package.json").write_text(json.dumps({"version": runner.PINNED_GITNEXUS_VERSION}))
|
||||
(runtime / "node_modules" / "gitnexus-shared").symlink_to(shared, target_is_directory=True)
|
||||
(shared / "package.json").write_text(json.dumps({"name": "gitnexus-shared"}))
|
||||
|
|
@ -303,6 +377,7 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm
|
|||
(runtime / "vendor", f"{runner.SANDBOX_GITNEXUS}/vendor"),
|
||||
(shared / "dist", f"{runner.SANDBOX_GITNEXUS_SHARED}/dist"),
|
||||
(shared / "package.json", f"{runner.SANDBOX_GITNEXUS_SHARED}/package.json"),
|
||||
(runtime / "hooks" / "claude", f"{runner.SANDBOX_GITNEXUS}/hooks/claude"),
|
||||
]
|
||||
package = json.loads((runtime / "package.json").read_text())
|
||||
assert package["version"] == runner.PINNED_GITNEXUS_VERSION
|
||||
|
|
@ -317,6 +392,12 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm
|
|||
assert shared / forbidden not in mounted_sources
|
||||
assert f"{runner.SANDBOX_GITNEXUS_SHARED}/{forbidden}" not in mounted_targets
|
||||
|
||||
# Only hooks/claude is exposed, not the whole hooks/ directory (which also
|
||||
# has an unrelated hooks/antigravity/ tree) and not the runtime root itself.
|
||||
assert runtime / "hooks" not in mounted_sources
|
||||
assert runtime / "hooks" / "antigravity" not in mounted_sources
|
||||
assert f"{runner.SANDBOX_GITNEXUS}/hooks" not in mounted_targets
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
|
||||
|
|
@ -334,6 +415,7 @@ def test_real_bubblewrap_runtime_mount_imports_cli_without_exposing_checkout(tmp
|
|||
f"{runner.SANDBOX_GITNEXUS}/vendor",
|
||||
f"{runner.SANDBOX_GITNEXUS_SHARED}/dist/index.js",
|
||||
f"{runner.SANDBOX_GITNEXUS_SHARED}/package.json",
|
||||
f"{runner.SANDBOX_GITNEXUS}/hooks/claude/resolve-analyze-cmd.cjs",
|
||||
]
|
||||
forbidden = [
|
||||
f"{runner.SANDBOX_GITNEXUS}/{relative}"
|
||||
|
|
@ -356,16 +438,26 @@ def test_real_bubblewrap_runtime_mount_imports_cli_without_exposing_checkout(tmp
|
|||
preflight=True,
|
||||
) as sandbox:
|
||||
visibility = sandbox.run(
|
||||
["/usr/local/bin/node", "-e", visibility_script],
|
||||
[runner.SANDBOX_NODE, "-e", visibility_script],
|
||||
timeout=10,
|
||||
)
|
||||
imported = sandbox.run(
|
||||
["/usr/local/bin/node", runner.SANDBOX_GITNEXUS_ENTRYPOINT, "--version"],
|
||||
[runner.SANDBOX_NODE, runner.SANDBOX_GITNEXUS_ENTRYPOINT, "--version"],
|
||||
timeout=10,
|
||||
)
|
||||
# --version never reaches the `analyze` command, which is loaded via a
|
||||
# lazy dynamic import and is the only path that pulls in
|
||||
# resolve-invocation.ts's module-load-time require of hooks/claude/
|
||||
# resolve-analyze-cmd.cjs. Require the compiled analyze module
|
||||
# directly so this canary actually exercises that chain.
|
||||
analyze_imported = sandbox.run(
|
||||
[runner.SANDBOX_NODE, "-e", f"require('{runner.SANDBOX_GITNEXUS}/dist/cli/analyze.js')"],
|
||||
timeout=10,
|
||||
)
|
||||
|
||||
assert visibility.ok, visibility.stderr_tail
|
||||
assert imported.ok, imported.stderr_tail
|
||||
assert analyze_imported.ok, analyze_imported.stderr_tail
|
||||
assert imported.stdout_tail.strip() == runner.PINNED_GITNEXUS_VERSION
|
||||
|
||||
|
||||
|
|
@ -1012,3 +1104,55 @@ def test_review_phase_rejects_workspace_or_skill_mutation(
|
|||
assert rec["resolved"] is False
|
||||
assert rec["error_kind"] == "review-evidence-invalid"
|
||||
assert expected_detail in rec["error_detail"]
|
||||
|
||||
|
||||
def _git(repo, *args, check=True):
|
||||
return subprocess.run(["git", "-C", str(repo), *args], check=check, capture_output=True, text=True)
|
||||
|
||||
|
||||
def _git_commit(repo, message):
|
||||
_git(
|
||||
repo,
|
||||
"-c",
|
||||
"user.name=test",
|
||||
"-c",
|
||||
"user.email=test@invalid",
|
||||
"commit",
|
||||
"--quiet",
|
||||
"--allow-empty",
|
||||
"-m",
|
||||
message,
|
||||
)
|
||||
return _git(repo, "rev-parse", "HEAD").stdout.strip()
|
||||
|
||||
|
||||
def test_make_worktree_clone_has_no_tags_but_keeps_all_branches(tmp_path):
|
||||
# oracle_assets.MAX_CLONE_REFS refuses to sanitize a clone with more than
|
||||
# 1024 refs; this repo's own history has 1000+ release-candidate tags, so
|
||||
# a plain `git clone` of it (inheriting every tag) trips that cap on every
|
||||
# benchmark session. make_worktree must not carry tags into its throwaway
|
||||
# clone, but callers pass a bare SHA or "HEAD" as `ref` (never a branch
|
||||
# name -- see evolve.py:476, runner.py:1037, sanitized_graph.py:345), so
|
||||
# branch-fetching itself must stay untouched: a commit reachable only from
|
||||
# a non-default branch must still resolve via the existing
|
||||
# checkout(ref) -> checkout(origin/{ref}) fallback.
|
||||
repo = tmp_path / "repo"
|
||||
repo.mkdir()
|
||||
_git(repo, "init", "--quiet")
|
||||
_git(repo, "checkout", "--quiet", "-b", "main")
|
||||
_git_commit(repo, "base")
|
||||
_git(repo, "tag", "v1.0.0-rc.1")
|
||||
|
||||
_git(repo, "checkout", "--quiet", "-b", "other")
|
||||
other_sha = _git_commit(repo, "only on other")
|
||||
_git(repo, "checkout", "--quiet", "main")
|
||||
|
||||
clones = tmp_path / "clones"
|
||||
clones.mkdir()
|
||||
target = runner.make_worktree(repo, other_sha, clones)
|
||||
|
||||
tags = _git(target, "tag").stdout.split()
|
||||
assert tags == [], f"clone must carry no tags, found: {tags}"
|
||||
|
||||
current = _git(target, "rev-parse", "HEAD").stdout.strip()
|
||||
assert current == other_sha
|
||||
|
|
|
|||
|
|
@ -501,7 +501,10 @@ def run_proposer(
|
|||
auth_token=args.auth_token,
|
||||
base_url=args.base_url,
|
||||
),
|
||||
permission_mode="dontAsk",
|
||||
# No permission_mode: CLAUDE_CODE_SUBPROCESS_ENV_SCRUB
|
||||
# forces "default", so requesting dontAsk only warns. Tools
|
||||
# are pre-approved via settings permissions.allow
|
||||
# (proposer_sandbox.build_claude_settings).
|
||||
command_prefix=sandbox.command_prefix,
|
||||
require_pid_namespace=True,
|
||||
bare=True,
|
||||
|
|
|
|||
|
|
@ -26,6 +26,8 @@ SANDBOX_HOME = "/home/agent"
|
|||
SANDBOX_TMP = "/tmp"
|
||||
SANDBOX_CLAUDE = "/opt/claude/claude"
|
||||
SANDBOX_SHELL_PREFIX = "/opt/claude/shell-prefix"
|
||||
SANDBOX_PYTHON3 = "/opt/claude/python3"
|
||||
SANDBOX_NODE = "/opt/claude/node"
|
||||
SANDBOX_PATH = "/opt/claude:/usr/local/bin:/usr/bin:/bin"
|
||||
SANDBOX_GITNEXUS = "/opt/gitnexus"
|
||||
SANDBOX_GITNEXUS_SHARED = "/opt/gitnexus-shared"
|
||||
|
|
@ -329,7 +331,14 @@ def build_claude_settings() -> str:
|
|||
},
|
||||
},
|
||||
"permissions": {
|
||||
"defaultMode": "dontAsk",
|
||||
# CLAUDE_CODE_SUBPROCESS_ENV_SCRUB forces permission mode to
|
||||
# "default" (allowed_non_write_users hardening), so requesting a
|
||||
# non-default mode only emits a warning and never takes effect.
|
||||
# Under "default" a tool runs without a prompt only if it matches an
|
||||
# allow rule, so pre-approve the proposer's exact tool surface. Bash
|
||||
# is the only writable tool under --bare (it writes the candidate
|
||||
# overlay) and stays sandbox-confined by the sandbox.* policy above.
|
||||
"allow": ["Read", "Grep", "Glob", "Bash"],
|
||||
"disableBypassPermissionsMode": "disable",
|
||||
},
|
||||
"env": {
|
||||
|
|
@ -346,6 +355,20 @@ def _runtime_mount_args() -> list[str]:
|
|||
path = Path(raw)
|
||||
if path.exists():
|
||||
args += ["--ro-bind", raw, raw]
|
||||
# sanitized_graph.py and runner_sessions.py invoke the sandboxed graph
|
||||
# CLI via SANDBOX_NODE. Bind whatever `node` actually resolves to on PATH
|
||||
# there -- true node location varies by host (GitHub-hosted runner images
|
||||
# happen to have one under /usr/local/bin; a self-hosted runner's
|
||||
# actions/setup-node installs into its own tool-cache directory instead).
|
||||
# Target must be a fresh path like /opt/claude/... rather than anywhere
|
||||
# under /usr, /bin, /lib, or /lib64: those are already read-only bound
|
||||
# above, and bwrap can't create a new mount-point file inside an
|
||||
# already-read-only tree when the real path doesn't already exist there
|
||||
# (the exact case a self-hosted runner hits, and the reason this bind
|
||||
# exists at all).
|
||||
node_bin = shutil.which("node")
|
||||
if node_bin:
|
||||
args += ["--ro-bind", node_bin, SANDBOX_NODE]
|
||||
for raw in (
|
||||
"/etc/ssl",
|
||||
"/etc/hosts",
|
||||
|
|
@ -377,6 +400,24 @@ def _create_shell_prefix_wrapper(private_root: Path) -> Path:
|
|||
return wrapper
|
||||
|
||||
|
||||
def _create_python3_wrapper(private_root: Path) -> Path:
|
||||
"""A trusted, self-owned Python 3 launcher for evidence-provenance.mjs's atomic mover.
|
||||
|
||||
/usr/bin/python3 is a real system binary, but it's root-owned on the host.
|
||||
Inside this --unshare-user sandbox only the calling uid is mapped (root is
|
||||
not), so root-owned files surface as the kernel's overflow uid — which
|
||||
evidence-provenance.mjs's PATH-scan correctly refuses to trust. This
|
||||
wrapper is freshly created by the same host process that owns
|
||||
home/temp/shell-prefix, so it maps to the sandbox's own trusted uid
|
||||
instead, and simply execs the real interpreter through to do the work.
|
||||
"""
|
||||
|
||||
wrapper = private_root / "python3"
|
||||
wrapper.write_text('#!/bin/bash\nset -eu\nexec /usr/bin/python3 "$@"\n')
|
||||
wrapper.chmod(0o500)
|
||||
return wrapper
|
||||
|
||||
|
||||
def _resolve_executable(executable: Path | str | None, default: str) -> Path:
|
||||
raw = os.fspath(executable) if executable is not None else shutil.which(default)
|
||||
if not raw:
|
||||
|
|
@ -634,6 +675,7 @@ def prepare_sandbox(
|
|||
directory.mkdir(mode=0o700)
|
||||
directory.chmod(0o700)
|
||||
shell_prefix = _create_shell_prefix_wrapper(private_root)
|
||||
python3_wrapper = _create_python3_wrapper(private_root)
|
||||
# Claude may discover user-level skills below HOME. Keep the rest of HOME
|
||||
# writable for normal CLI state, but overlay an immutable empty skills root
|
||||
# so a model cannot shadow the evaluated repository/plugin skill by name.
|
||||
|
|
@ -644,6 +686,7 @@ def prepare_sandbox(
|
|||
*read_only_mounts,
|
||||
ReadOnlyMount(source=user_skills, target=SANDBOX_USER_SKILLS),
|
||||
ReadOnlyMount(source=shell_prefix, target=SANDBOX_SHELL_PREFIX),
|
||||
ReadOnlyMount(source=python3_wrapper, target=SANDBOX_PYTHON3),
|
||||
)
|
||||
primary: BaseException | None = None
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -78,6 +78,7 @@ from .proposer_sandbox import (
|
|||
SANDBOX_GITNEXUS as SANDBOX_GITNEXUS,
|
||||
SANDBOX_GITNEXUS_REGISTRY,
|
||||
SANDBOX_GITNEXUS_SHARED as SANDBOX_GITNEXUS_SHARED,
|
||||
SANDBOX_NODE as SANDBOX_NODE,
|
||||
SANDBOX_WORKSPACE,
|
||||
ReadOnlyMount,
|
||||
SandboxError,
|
||||
|
|
@ -402,6 +403,13 @@ def run_arm(
|
|||
auth_token=args.auth_token,
|
||||
base_url=args.base_url,
|
||||
)
|
||||
# --bare hard-disables the Skill tool and every mcp__* tool — by Claude
|
||||
# Code design, not a bug (--allowedTools can't restore what --bare
|
||||
# removes). Every arm except baseline_nomcp needs Skill and/or MCP tools,
|
||||
# so only baseline_nomcp can keep --bare's tighter isolation; the rest
|
||||
# rely on ANTHROPIC_API_KEY alone (the sandboxed HOME has no OAuth/
|
||||
# keychain state to conflict with it).
|
||||
bare = arm == "baseline_nomcp"
|
||||
common = {
|
||||
"claude_bin": sandbox.claude_bin,
|
||||
"timeout": args.timeout,
|
||||
|
|
@ -412,7 +420,7 @@ def run_arm(
|
|||
read_only_paths=_evaluated_skill_roots(worktree, arm),
|
||||
),
|
||||
"require_pid_namespace": True,
|
||||
"bare": True,
|
||||
"bare": bare,
|
||||
"settings_json": sandbox.settings_json,
|
||||
"strict_mcp_config": True,
|
||||
"mcp_config_json": sandbox_mcp_config(),
|
||||
|
|
@ -672,13 +680,21 @@ def aggregate(records: list[dict[str, Any]]) -> dict[str, Any]:
|
|||
# unmeasured run makes the whole median unavailable so the gate won't rank
|
||||
# a candidate on a cost that was never actually captured.
|
||||
valid_costs = [r.get("cost_usd") for r in valid]
|
||||
out["cost_usd"] = None if (not valid or any(cost is None for cost in valid_costs)) else statistics.median(valid_costs)
|
||||
out["cost_usd"] = (
|
||||
None if (not valid or any(cost is None for cost in valid_costs)) else statistics.median(valid_costs)
|
||||
)
|
||||
out["resolved"] = sum(1 for r in records if r["resolved"])
|
||||
out["runs"] = len(records)
|
||||
out["valid_runs"] = len(valid)
|
||||
out["excluded_runs"] = len(records) - len(valid)
|
||||
out["transcripts_missing"] = sum(1 for r in records if r.get("transcript_missing"))
|
||||
out["class"] = records[0].get("class", "")
|
||||
error_kinds: dict[str, int] = {}
|
||||
for r in records:
|
||||
kind = r.get("error_kind")
|
||||
if kind:
|
||||
error_kinds[kind] = error_kinds.get(kind, 0) + 1
|
||||
out["error_kinds"] = error_kinds
|
||||
return out
|
||||
|
||||
|
||||
|
|
@ -695,6 +711,33 @@ def savings(baseline: dict[str, Any], workflow: dict[str, Any]) -> dict[str, Any
|
|||
return out
|
||||
|
||||
|
||||
def broken_incumbent_arms(
|
||||
results: dict[str, dict[str, dict[str, Any]]],
|
||||
incumbent_arms: set[str],
|
||||
) -> list[str]:
|
||||
"""Incumbent arms that resolved nothing across every task they ran.
|
||||
|
||||
An incumbent arm is the currently-shipped, presumably-working skill: if it
|
||||
resolves NOTHING across every task it ran, that reads as an environment or
|
||||
harness failure (missing trusted interpreter, stale skill fingerprint,
|
||||
sandbox misconfiguration), not a skill regression. A candidate merely
|
||||
underperforming is a normal, expected outcome and must not trip this —
|
||||
only checking incumbents keeps that distinction.
|
||||
|
||||
Deliberately does NOT require valid_runs > 0 per task: an incumbent that
|
||||
fails every run with an excluded-but-non-systemic error_kind (e.g.
|
||||
"evidence-unverified", which the outage-streak breaker explicitly resets
|
||||
on rather than accumulates) would otherwise never accumulate a single
|
||||
valid run and sail through silently — the exact "quiet no-promotion"
|
||||
outcome this guard exists to catch, and arguably worse than the
|
||||
some-runs-resolved-zero case since here nothing completed at all.
|
||||
aggregate() never marks an excluded/unverifiable row resolved=True, so
|
||||
resolved == 0 alone already covers both cases.
|
||||
"""
|
||||
present = incumbent_arms & {arm for arms in results.values() for arm in arms}
|
||||
return sorted(arm for arm in present if all(arms[arm]["resolved"] == 0 for arms in results.values() if arm in arms))
|
||||
|
||||
|
||||
def _na(value: Any) -> Any:
|
||||
"""Render an unmeasured metric as ``n/a`` instead of a misleading number."""
|
||||
return "n/a" if value is None else value
|
||||
|
|
@ -719,8 +762,8 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str:
|
|||
"efficiency, sum usage from the session transcripts instead",
|
||||
"(dedup events sharing one message.id).",
|
||||
"",
|
||||
"| task | class | arm | resolved | input | cache_create | cache_read | output | cost $ | wall s | turns | churn |",
|
||||
"| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |",
|
||||
"| task | class | arm | resolved | input | cache_create | cache_read | output | cost $ | wall s | turns | churn | errors |",
|
||||
"| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |",
|
||||
]
|
||||
for task_id, arms in results.items():
|
||||
for arm, agg in arms.items():
|
||||
|
|
@ -728,12 +771,14 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str:
|
|||
resolved_cell = f"{agg['resolved']}/{agg.get('valid_runs', agg['runs'])}"
|
||||
if excluded:
|
||||
resolved_cell += f" ({excluded} excluded)"
|
||||
error_cell = ", ".join(f"{kind}×{count}" for kind, count in sorted(agg.get("error_kinds", {}).items()))
|
||||
lines.append(
|
||||
f"| {task_id} | {agg['class']} | {arm} | {resolved_cell} "
|
||||
f"| {agg['input_tokens']:.0f} | {agg['cache_creation_input_tokens']:.0f} "
|
||||
f"| {agg['cache_read_input_tokens']:.0f} | {agg['output_tokens']:.0f} "
|
||||
f"| {_cost_cell(agg['cost_usd'])} | {agg['duration_s']:.0f} | {agg['num_turns']:.0f} "
|
||||
f"| {agg['diff_files']:.0f}/+{agg['diff_insertions']:.0f}/−{agg['diff_deletions']:.0f} |"
|
||||
f"| {agg['diff_files']:.0f}/+{agg['diff_insertions']:.0f}/−{agg['diff_deletions']:.0f} "
|
||||
f"| {error_cell} |"
|
||||
)
|
||||
for arm in arms:
|
||||
if arm != "baseline" and "baseline" in arms:
|
||||
|
|
@ -742,7 +787,7 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str:
|
|||
f"| {task_id} | {arms[arm]['class']} | **{arm} savings %** | — "
|
||||
f"| {s['input_tokens']} | {s['cache_creation_input_tokens']} "
|
||||
f"| {s['cache_read_input_tokens']} | {s['output_tokens']} "
|
||||
f"| {_na(s['cost_usd'])} | {s['duration_s']} | — | — |"
|
||||
f"| {_na(s['cost_usd'])} | {s['duration_s']} | — | — | — |"
|
||||
)
|
||||
lines.append("")
|
||||
all_aggs = [agg for arms in results.values() for agg in arms.values()]
|
||||
|
|
@ -1326,6 +1371,17 @@ def main() -> None:
|
|||
}
|
||||
(out_dir / "promotion.json").write_text(json.dumps(promotion, indent=2) + "\n")
|
||||
print(f"\n{report}\n\nWritten to {out_dir}/")
|
||||
broken_incumbents = broken_incumbent_arms(results, set(CANDIDATE_ARMS.values()))
|
||||
if broken_incumbents:
|
||||
# Fail loudly rather than let a broken environment read as a quiet
|
||||
# "no promotion, incumbent stands."
|
||||
print(
|
||||
f"[harness-health] incumbent arm(s) {', '.join(broken_incumbents)} resolved zero "
|
||||
"tasks across every valid run — this looks like an environment/harness failure, "
|
||||
"not a normal candidate miss. See the errors column in report.md and error_detail "
|
||||
"in results.jsonl. Exiting non-zero rather than reporting a quiet no-promotion."
|
||||
)
|
||||
raise SystemExit(1)
|
||||
if outage_tripped:
|
||||
# Non-zero exit so a driver (evolve.py) treats the partial benchmark as a
|
||||
# failed run and halts instead of proposing from outage-truncated evidence.
|
||||
|
|
|
|||
|
|
@ -21,6 +21,40 @@ MAX_WORKSPACE_SNAPSHOT_ENTRIES = 100_000
|
|||
MAX_WORKSPACE_SNAPSHOT_PATH_BYTES = 16 * 1024 * 1024
|
||||
MAX_WORKSPACE_SNAPSHOT_FILE_BYTES = 1024 * 1024 * 1024
|
||||
|
||||
# Claude Code's own enableWeakerNestedSandbox bootstrap creates these paths on
|
||||
# EVERY session regardless of task or model output -- reproduced empirically
|
||||
# with a trivial "say OK" prompt: a synthetic package.json/lockfiles/
|
||||
# node_modules, a full set of .env variants, and .claude/agents,
|
||||
# .claude/commands, .claude/.cc-writes. None of this is something the model
|
||||
# decided to write, so it must not count as an "unauthorized" workspace
|
||||
# change during the planning-phase boundary check (the one thing this
|
||||
# snapshot is used for -- see workspace_snapshot's callers). Mirrors the
|
||||
# pre-existing .git exclusion below, which is the same kind of harness/tool
|
||||
# noise rather than substantive diff.
|
||||
WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE = frozenset(
|
||||
{
|
||||
".claude",
|
||||
".env",
|
||||
".env.development",
|
||||
".env.development.local",
|
||||
".env.local",
|
||||
".env.production",
|
||||
".env.production.local",
|
||||
".env.test",
|
||||
".env.test.local",
|
||||
".gitmodules",
|
||||
".npmrc",
|
||||
".yarnrc",
|
||||
".yarnrc.yml",
|
||||
"bunfig.toml",
|
||||
"node_modules",
|
||||
"package-lock.json",
|
||||
"package.json",
|
||||
"pnpm-lock.yaml",
|
||||
"yarn.lock",
|
||||
}
|
||||
)
|
||||
|
||||
IMPLEMENTATION_ARMS = frozenset(
|
||||
{
|
||||
"workflow",
|
||||
|
|
@ -53,7 +87,9 @@ class VerificationResult:
|
|||
|
||||
|
||||
def workspace_snapshot(worktree: Path) -> dict[str, str]:
|
||||
"""Hash the workspace without following links, excluding Git internals."""
|
||||
"""Hash the workspace without following links, excluding Git internals
|
||||
and Claude Code's own sandbox-bootstrap noise (see
|
||||
WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE)."""
|
||||
|
||||
root = worktree.expanduser().absolute()
|
||||
mode = root.lstat().st_mode
|
||||
|
|
@ -74,7 +110,7 @@ def workspace_snapshot(worktree: Path) -> dict[str, str]:
|
|||
raise ValueError(f"workspace snapshot directory is unreadable: {directory}: {exc}") from exc
|
||||
for entry in children:
|
||||
relative = relative_dir / entry.name
|
||||
if relative.parts[0] == ".git":
|
||||
if relative.parts[0] == ".git" or relative.parts[0] in WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE:
|
||||
continue
|
||||
entry_count += 1
|
||||
path_bytes += len(relative.as_posix().encode())
|
||||
|
|
@ -240,6 +276,7 @@ def make_worktree(repo: Path, ref: str, parent: Path) -> Path:
|
|||
"clone",
|
||||
"--no-local",
|
||||
"--no-hardlinks",
|
||||
"--no-tags",
|
||||
"--quiet",
|
||||
str(repo),
|
||||
str(target),
|
||||
|
|
|
|||
|
|
@ -18,6 +18,7 @@ from .proposer_sandbox import (
|
|||
SANDBOX_GITNEXUS,
|
||||
SANDBOX_GITNEXUS_REGISTRY,
|
||||
SANDBOX_HOME,
|
||||
SANDBOX_NODE,
|
||||
SANDBOX_TMP,
|
||||
SANDBOX_WORKSPACE,
|
||||
SandboxError,
|
||||
|
|
@ -50,6 +51,8 @@ def measured_cost(raw: Any) -> float | None:
|
|||
if not math.isfinite(raw) or raw < 0:
|
||||
return None
|
||||
return float(raw)
|
||||
|
||||
|
||||
SANDBOX_GITNEXUS_ENTRYPOINT = f"{SANDBOX_GITNEXUS}/dist/cli/index.js"
|
||||
SENSITIVE_EVENT_KEYS = frozenset(
|
||||
{
|
||||
|
|
@ -113,7 +116,7 @@ def sandbox_mcp_config() -> str:
|
|||
"PATH=/usr/local/bin:/usr/bin:/bin",
|
||||
"LANG=C.UTF-8",
|
||||
"GIT_TERMINAL_PROMPT=0",
|
||||
"/usr/local/bin/node",
|
||||
SANDBOX_NODE,
|
||||
SANDBOX_GITNEXUS_ENTRYPOINT,
|
||||
"mcp",
|
||||
],
|
||||
|
|
@ -377,6 +380,13 @@ def run_claude(
|
|||
if strict_mcp_config:
|
||||
cmd += ["--strict-mcp-config", "--mcp-config", mcp_config_json or '{"mcpServers":{}}']
|
||||
if allowed_tools:
|
||||
# --bare's own hard-coded Bash/Edit/Read ceiling already scopes bare
|
||||
# sessions; outside --bare the built-in toolset defaults to
|
||||
# everything (subagents, WebFetch, Task, ...), so --tools is needed
|
||||
# to actually restrict it — --allowedTools only pre-approves within
|
||||
# whatever set is available, it does not narrow that set.
|
||||
if not bare:
|
||||
cmd += ["--tools", *allowed_tools]
|
||||
cmd += ["--allowedTools", *allowed_tools]
|
||||
if disable_slash_commands:
|
||||
cmd.append("--disable-slash-commands")
|
||||
|
|
@ -434,6 +444,14 @@ def run_claude(
|
|||
"returncode": proc.returncode,
|
||||
"process_state": proc.state,
|
||||
"stderr_tail": proc.stderr_tail[-2000:],
|
||||
# A session can exit non-zero with an empty stderr (e.g. a
|
||||
# pre-flight sandbox failure before any model turn): the tail
|
||||
# of raw stdout is the only place the actual event stream
|
||||
# (permission_denials, tool_use/tool_result, is_error) shows
|
||||
# up, so surface it here rather than leaving the failure
|
||||
# opaque. Callers already redact this record before it is
|
||||
# written to disk or an uploaded artifact.
|
||||
"stdout_tail": proc.stdout_tail[-2000:],
|
||||
"process_detail": proc.detail,
|
||||
"event_stream_error": event_stream_error,
|
||||
}
|
||||
|
|
|
|||
|
|
@ -217,6 +217,12 @@ def trusted_gitnexus_runtime_mounts() -> tuple[ReadOnlyMount, ...]:
|
|||
f"{SANDBOX_GITNEXUS_SHARED}/package.json",
|
||||
directory=False,
|
||||
),
|
||||
_validated_runtime_component(
|
||||
runtime,
|
||||
"hooks/claude",
|
||||
f"{SANDBOX_GITNEXUS}/hooks/claude",
|
||||
directory=True,
|
||||
),
|
||||
)
|
||||
|
||||
entrypoint = mounts[0].source / "cli" / "index.js"
|
||||
|
|
|
|||
|
|
@ -16,6 +16,7 @@ from .process_control import ManagedProcessError, run_managed
|
|||
from .proposer_sandbox import (
|
||||
SANDBOX_GITNEXUS,
|
||||
SANDBOX_HOME,
|
||||
SANDBOX_NODE,
|
||||
SANDBOX_WORKSPACE,
|
||||
ReadOnlyMount,
|
||||
SandboxError,
|
||||
|
|
@ -248,7 +249,7 @@ def _run_graph_cli(
|
|||
) -> bytes | None:
|
||||
command = [
|
||||
*prefix,
|
||||
"/usr/local/bin/node",
|
||||
SANDBOX_NODE,
|
||||
SANDBOX_GITNEXUS_ENTRYPOINT,
|
||||
*arguments,
|
||||
]
|
||||
|
|
|
|||
|
|
@ -39,9 +39,14 @@ MAX_TASK_ASSET_ENTRIES = 100_000
|
|||
MAX_TASK_ASSET_PATH_BYTES = 4_096
|
||||
MAX_TASK_ASSET_BYTES = 2 * 1024 * 1024 * 1024
|
||||
|
||||
# A filesystem without reflink support may still run tiny fixtures. Large
|
||||
# assets fail closed instead of silently returning to one full copy per arm.
|
||||
MAX_BUFFERED_FALLBACK_BYTES = 16 * 1024 * 1024
|
||||
# The largest known real sandbox_copy asset in this harness is the shipped
|
||||
# index above (~428 MiB estimated, ~290 MiB measured); budget comfortably
|
||||
# above that so it can still materialize via buffered copy on a filesystem
|
||||
# that cannot reflink (ext4 CI runners, 9p-backed dev mounts), while staying
|
||||
# well below MAX_TASK_ASSET_BYTES so a genuinely oversized or malformed
|
||||
# declaration still fails closed instead of silently paying for a slow full
|
||||
# copy.
|
||||
MAX_BUFFERED_FALLBACK_BYTES = 512 * 1024 * 1024
|
||||
COPY_CHUNK_BYTES = 1024 * 1024
|
||||
|
||||
# linux/fs.h: #define FICLONE _IOW(0x94, 9, int)
|
||||
|
|
|
|||
6
gitnexus-web/package-lock.json
generated
6
gitnexus-web/package-lock.json
generated
|
|
@ -7894,9 +7894,9 @@
|
|||
}
|
||||
},
|
||||
"node_modules/tar": {
|
||||
"version": "7.5.16",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.16.tgz",
|
||||
"integrity": "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w==",
|
||||
"version": "7.5.20",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.20.tgz",
|
||||
"integrity": "sha512-9FcyK4PA6+WbzlTM9WhQm6vB5W7cP7dUiPsv1g7YDwEQnQ1CGpK3MGlKk/ITVWMk05kHZuBhmVhiv8LZoy/PFQ==",
|
||||
"dev": true,
|
||||
"license": "BlueOak-1.0.0",
|
||||
"dependencies": {
|
||||
|
|
|
|||
|
|
@ -284,6 +284,7 @@ Set these env vars to use a remote OpenAI-compatible `/v1/embeddings` endpoint i
|
|||
export GITNEXUS_EMBEDDING_URL=http://your-server:8080/v1
|
||||
export GITNEXUS_EMBEDDING_MODEL=BAAI/bge-large-en-v1.5
|
||||
export GITNEXUS_EMBEDDING_DIMS=1024 # optional, default 384
|
||||
export GITNEXUS_EMBEDDING_REQUEST_DIMS=omit # optional: omit "dimensions", or an integer to override it
|
||||
export GITNEXUS_EMBEDDING_API_KEY=your-key # optional, default: "unused"
|
||||
export GITNEXUS_EMBEDDING_MAX_ATTEMPTS=3 # optional, total attempts (1-20)
|
||||
export GITNEXUS_EMBEDDING_RETRY_CAP_MS=5000 # optional, maximum retry delay
|
||||
|
|
@ -291,6 +292,15 @@ export GITNEXUS_EMBEDDING_MIN_INTERVAL_MS=0 # optional, minimum request spacing
|
|||
gitnexus analyze . --embeddings
|
||||
```
|
||||
|
||||
`GITNEXUS_EMBEDDING_REQUEST_DIMS` controls only the `dimensions` field sent in
|
||||
the request body, independently of `GITNEXUS_EMBEDDING_DIMS` (which still
|
||||
validates the returned vector's length):
|
||||
|
||||
- `omit` (or `none`, `off`, `false`, `0`) — do not send `dimensions` at all, for
|
||||
strict backends that return the right vector size but reject the field.
|
||||
- a positive integer — send that value instead of `GITNEXUS_EMBEDDING_DIMS`.
|
||||
- unset — send `GITNEXUS_EMBEDDING_DIMS` (the previous behavior).
|
||||
|
||||
Works with Infinity, vLLM, TEI, llama.cpp, Ollama, LM Studio, or OpenAI. Retry and pacing settings are provider-neutral; provider-specific limits should be supplied through configuration. When unset, local embeddings are used unchanged.
|
||||
|
||||
## Multi-Repo Support
|
||||
|
|
@ -538,7 +548,7 @@ For repositories with very large source files, `GITNEXUS_WORKER_SUB_BATCH_MAX_BY
|
|||
|
||||
### Worker pool resilience tuning
|
||||
|
||||
Three env vars expose the pool's resilience layers (respawn budget, cumulative-timeout cap, circuit breaker). Defaults are tuned for typical repos; bump them when an analyze legitimately needs more retries, or lower them to fail-fast on a known-bad shape.
|
||||
Four env vars expose the pool's resilience layers (respawn budget, cumulative-timeout cap, circuit breaker, startup handshake). Defaults are tuned for typical repos; bump them when an analyze legitimately needs more retries, or lower them to fail-fast on a known-bad shape.
|
||||
|
||||
| Variable | Default | Effect |
|
||||
| ----------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
||||
|
|
@ -546,6 +556,7 @@ Three env vars expose the pool's resilience layers (respawn budget, cumulative-t
|
|||
| `GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS` | `5 × subBatchTimeoutMs` | Total retry wall-time budget per job before quarantining. Bounds exponentially-growing retry waits. |
|
||||
| `GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD` | `max(3, poolSize)` | Per-slot consecutive deaths before the pool's circuit breaker trips. After tripping, dispatches require a fresh pool. |
|
||||
| `GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS` | `30000` | Max wait at pool shutdown for a retired worker still inside native code — terminated at its next JS-safe point instead of mid-native-call, which would abort the process (`Napi::Error`, #2432). |
|
||||
| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. Raise it on a slow or heavily loaded host where a full pool cold-starting concurrently needs more than 5s. |
|
||||
| `GITNEXUS_CPP_CAPTURE_BUDGET_MS` | `20000` | Per-file wall-clock budget for C++ capture extraction; on breach the file keeps partial captures with a warning (#2432). `0` expires immediately. |
|
||||
|
||||
### Graph cleanup tuning
|
||||
|
|
|
|||
|
|
@ -46,8 +46,9 @@
|
|||
"_note": "#2046: F35 qualified-constructor captures now emit @reference.qualified-name + a simple-name @reference.name on `new Ns.Foo()`/`new A.B.Foo()`; namespace_declaration/file_scoped_namespace_declaration now emit @declaration.namespace name captures (feeding the non-destructive namespacePrefix sidecar for `new B.Foo()` same-tail disambiguation). + csharp-interface-only-base and csharp-namespace-qualified-ctor fixtures. Pure capture-additive + fixture-corpus drift; scaling stays linear (~1.11)."
|
||||
},
|
||||
"rust": {
|
||||
"fingerprint": "df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29",
|
||||
"fingerprint": "f7742f65f14d7d6590df7f16303fc3cc9dc0c233cd80bf90c98b084933cd3846",
|
||||
"scaling_budget": 1.5,
|
||||
"_rebaselined_dyn_trait_object_2604": "#2604: RUST_SCOPE_QUERY now captures function_signature_item (abstract trait methods, no body) as a scope + declaration, so a &dyn Trait receiver can dispatch a CALLS edge to the trait's own method. Additive capture shift across every bench fixture with a required trait method. Prior df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29 -> f7742f65f14d7d6590df7f16303fc3cc9dc0c233cd80bf90c98b084933cd3846; scaling 1.033 < 1.5.",
|
||||
"_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 65e5bca66bb1ca117949409e8fb5c80ee69d6f1b5318908eaaecf08da0482e5c -> df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29; scaling 1.065 < 1.5.",
|
||||
"_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Rust fn-value callable flow facts with invocation/constructor-result suppression. Prior ac610bbe97666bf285923479dd7b43a2fe4c5354aae8df1bcbafdc04fb220f82 -> 65e5bca66bb1ca117949409e8fb5c80ee69d6f1b5318908eaaecf08da0482e5c; scaling 1.024 < 1.5.",
|
||||
"_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) \u2014 legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.",
|
||||
|
|
@ -89,7 +90,7 @@
|
|||
"_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0."
|
||||
},
|
||||
"java": {
|
||||
"fingerprint": "975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca",
|
||||
"fingerprint": "d04298a91beec76d0fa7099b3d71265723be60c1df688969aa954f135dd49686",
|
||||
"scaling_budget": 1.5,
|
||||
"_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata; same-name lexical regions use an O(ancestor-depth) ID-set lookup. Prior d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a -> 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4; scaling 0.992 < 1.5.",
|
||||
"_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Java method-reference/SAM callable flow facts with invocation-result suppression. Prior 062d754764aaa8a6772fb90875c710502a63e3e7a300e633942381ed914faada -> d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a; scaling 1.074 < 1.5.",
|
||||
|
|
@ -97,7 +98,9 @@
|
|||
"_note": "#1928 / #2045: F35 adds qualified + qualified-generic constructor query captures (`new pkg.Foo()`, `new a.b.Foo()`, `new pkg.Box<T>()`); F38 synthesizes `@reference.call.constructor` on `super(...)`/`this(...)` explicit_constructor_invocation nodes; F41 generic-aware stripQualifier in interpret (type-binding normalization). + java-qualified-constructor and java-explicit-constructor fixtures. Pure capture-additive + fixture-corpus drift; scaling stays linear (~1.06).",
|
||||
"_rebaselined_2522_review_fixes": "PR #2522 review fixes: get/test dropped from callableProtocolMethods. Prior 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4 -> f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67; scaling ratio re-verified within budget.",
|
||||
"_rebaselined_2550_instance_model": "PR #2549 (#2550): anonymous class bodies emit synthesized @declaration.class/@declaration.name (Worker$N), an @reference.inherits to the constructed type, and receiver @type-binding.* captures; six new java-* fixtures joined the corpus. Prior f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67 -> d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90; scaling 1.058 < 1.5.",
|
||||
"_rebaselined_2555_enum_constant_bodies": "PR for #2555: enum constant bodies emit synthesized E$N classes + @reference.inherits to the host enum; anonymous naming follows JLS 13.1 immediately-enclosing-type chains INCLUDING anonymous enclosing types (NestHost$1$1, N$1$1); six new java-* fixtures joined the corpus. Prior d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90 -> 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca; scaling 1.05 < 1.5."
|
||||
"_rebaselined_2555_enum_constant_bodies": "PR for #2555: enum constant bodies emit synthesized E$N classes + @reference.inherits to the host enum; anonymous naming follows JLS 13.1 immediately-enclosing-type chains INCLUDING anonymous enclosing types (NestHost$1$1, N$1$1); six new java-* fixtures joined the corpus. Prior d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90 -> 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca; scaling 1.05 < 1.5.",
|
||||
"_rebaselined_2564_record_capture": "PR for #2564: JAVA_QUERIES gained a (record_declaration name: (identifier) @name) @definition.record capture, previously entirely missing (record_declaration had no structure-phase capture at all, unlike class/interface/enum) - a record's methods existed as ownerless Method nodes with no HAS_METHOD edge. Two new java-* fixtures (java-record-methods, java-new-expr-chain-call) joined the corpus. Prior 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca -> 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537; scaling 1.059 < 1.5.",
|
||||
"_rebaselined_2561_enum_constant_receiver": "PR for #2561: synthesizeJavaAnonymousClassDeclarations now emits a class-scope @type-binding.annotation/name/type per enum constant (constant simple name -> its E$N synthesized class when bodied, else the host enum) so E.CONST.method() resolves through the existing compound-receiver chain walk. Two drivers of the drift, both in the java-enum-constant-body fixture (this bench's corpus IS test/fixtures/lang-resolution): (1) one extra type-binding match per enum_constant from the capture change; (2) review follow-up added a body-less Plain.java enum + EnumConst.dispatchToConstant/dispatchInherited methods (bodied-override, inherited-via-MRO, and body-less dispatch call sites). The review's fail-safe hardening (bodied constant binds ONLY to E$N, never the host enum, when name synthesis fails on a malformed tree) is output-neutral on this well-formed corpus (verified: fingerprint identical with and without it). Prior 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537 -> d04298a91beec76d0fa7099b3d71265723be60c1df688969aa954f135dd49686; scaling < 1.5."
|
||||
},
|
||||
"typescript": {
|
||||
"fingerprint": "3280b13d3f9378ab23eee31c2edc779b5a9ae1e7bb510c23a24855b44406d2f4",
|
||||
|
|
|
|||
119
gitnexus/package-lock.json
generated
119
gitnexus/package-lock.json
generated
|
|
@ -59,8 +59,7 @@
|
|||
"@types/cors": "^2.8.17",
|
||||
"@types/express": "^5.0.6",
|
||||
"@types/js-yaml": "^4.0.9",
|
||||
"@types/node": "^25.6.0",
|
||||
"@types/uuid": "^11.0.0",
|
||||
"@types/node": "^26.0.0",
|
||||
"@vitest/coverage-v8": "^4.0.18",
|
||||
"gitnexus-shared": "file:../gitnexus-shared",
|
||||
"tsx": "^4.0.0",
|
||||
|
|
@ -68,7 +67,7 @@
|
|||
"vitest": "^4.0.18"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=22.0.0"
|
||||
"node": "^22.18.0 || >=24.11.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@huggingface/transformers": "^4.1.0",
|
||||
|
|
@ -1254,9 +1253,9 @@
|
|||
}
|
||||
},
|
||||
"node_modules/@ladybugdb/core": {
|
||||
"version": "0.18.1",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.18.1.tgz",
|
||||
"integrity": "sha512-0c1kXDpdv7z/GB0oyFYnLEjLsXFwPHz1YD4wxtrk9hav8zJX5T1PHQMr+XRfdDI1NQjx4iNdbPQGGT7Bx/X2aw==",
|
||||
"version": "0.18.2",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.18.2.tgz",
|
||||
"integrity": "sha512-222FjGciEO5Z+/MRQGU+b4IaGAjOgSQzj7fMpOuhMQN4F8nf654kuKRk1iybSiNy6XSw69hIJ0mKwUeBQ8y6Fg==",
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
|
|
@ -1265,17 +1264,17 @@
|
|||
"node-addon-api": "^6.0.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@ladybugdb/core-darwin-arm64": "0.18.1",
|
||||
"@ladybugdb/core-darwin-x64": "0.18.1",
|
||||
"@ladybugdb/core-linux-arm64": "0.18.1",
|
||||
"@ladybugdb/core-linux-x64": "0.18.1",
|
||||
"@ladybugdb/core-win32-x64": "0.18.1"
|
||||
"@ladybugdb/core-darwin-arm64": "0.18.2",
|
||||
"@ladybugdb/core-darwin-x64": "0.18.2",
|
||||
"@ladybugdb/core-linux-arm64": "0.18.2",
|
||||
"@ladybugdb/core-linux-x64": "0.18.2",
|
||||
"@ladybugdb/core-win32-x64": "0.18.2"
|
||||
}
|
||||
},
|
||||
"node_modules/@ladybugdb/core-darwin-arm64": {
|
||||
"version": "0.18.1",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.18.1.tgz",
|
||||
"integrity": "sha512-M5YZuAONRAv3awkr+cfaibn9Da+3pgDzRiek/JabWQuz48xgzW3Vh9yQH4s8Dq/bfQo6YTsaLIBRcUCCUzCtcg==",
|
||||
"version": "0.18.2",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.18.2.tgz",
|
||||
"integrity": "sha512-gAwxsdijBFTz4aZ9ITG6zdQw3lAki0eY33hNBLCfKXjKJLvW/8wCvgCVBglqfqBF5WyI7icFUt+wfy/Fbdfl5A==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
|
|
@ -1286,9 +1285,9 @@
|
|||
]
|
||||
},
|
||||
"node_modules/@ladybugdb/core-darwin-x64": {
|
||||
"version": "0.18.1",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.18.1.tgz",
|
||||
"integrity": "sha512-kq+pyTskfCx++Mrbk7QssE/f/CpSuU50T8lhRtv4PaOKhC2Jf8/wAUOA17UxI594wAru3ERpqVBFUBWGcPk2ag==",
|
||||
"version": "0.18.2",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.18.2.tgz",
|
||||
"integrity": "sha512-oUjYLc1fW3ntCrO9te55PoPfvhFo8AKeNa/sU66fiQyEAZB7qZJxeHnnLgl/bLueTF2os3RSawq46ZftoD/9Eg==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
|
|
@ -1299,9 +1298,9 @@
|
|||
]
|
||||
},
|
||||
"node_modules/@ladybugdb/core-linux-arm64": {
|
||||
"version": "0.18.1",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.18.1.tgz",
|
||||
"integrity": "sha512-fu7ke1haa5rPINcQn0+kxQijZ0A8ZDWP9e+X8xcDH94RagDbPWwG8yFC890cGSdc/j7mTV+xkA/y/kVHpmVI6w==",
|
||||
"version": "0.18.2",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.18.2.tgz",
|
||||
"integrity": "sha512-UppokeTaPl9pN0xOsdMa+hmM68zbN2eKReTZhZNYM16qX0d2OlgbS/NlXi09Wdot+w5qvlZ9Q0iCCPfr7qvPaw==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
|
|
@ -1312,9 +1311,9 @@
|
|||
]
|
||||
},
|
||||
"node_modules/@ladybugdb/core-linux-x64": {
|
||||
"version": "0.18.1",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.18.1.tgz",
|
||||
"integrity": "sha512-qp5HilHzDGuArfOyD+VyA7lVJ7IwQDKd81NZKKTmUwIAOJtdwqniYx6JZICPnlr36zFJBx/lGYoSsEzbC+TVdw==",
|
||||
"version": "0.18.2",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.18.2.tgz",
|
||||
"integrity": "sha512-GypOxCnP2ix/FWM8YhQ41aQYlS+ruoNMJp7pmaF5laJHhL/a+P/apywNTE+9N41Walsl+Emgg9xwxwTC93slow==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
|
|
@ -1325,9 +1324,9 @@
|
|||
]
|
||||
},
|
||||
"node_modules/@ladybugdb/core-win32-x64": {
|
||||
"version": "0.18.1",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.18.1.tgz",
|
||||
"integrity": "sha512-vHcXr7Df2X1dbb5ORK+SBmNstd/3tApGFImbAnaWiTuLDFlAdfY8lbiSBSp3OgFjc0BB7F3GYUUdvgDRJjK3zA==",
|
||||
"version": "0.18.2",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.18.2.tgz",
|
||||
"integrity": "sha512-hvFwjhTYdwG2sijapx963a27jP9mlLW0ZFv5Yfj19e0B3T/FqD9CaKULPy23mU2XlkISN2LUkY5qdgvCFztl/g==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
|
|
@ -1946,13 +1945,13 @@
|
|||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@types/node": {
|
||||
"version": "25.9.5",
|
||||
"resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.5.tgz",
|
||||
"integrity": "sha512-OScDchr2fwuUmWdf4kZ9h7PcJiYDVInhJizG/biAq3cAvqwYktuy/TYGGdZNMtNTFUP7rnb0NU4TUdm82kt4Rg==",
|
||||
"version": "26.0.0",
|
||||
"resolved": "https://registry.npmjs.org/@types/node/-/node-26.0.0.tgz",
|
||||
"integrity": "sha512-vf2YFi1iY9lHGwNJMs01biZFbKJkrZR1T6/MlzjhJLPdntOHLhTrDSnSVcdtvjihi4VQNlrFRIxLsDBlQpAipA==",
|
||||
"devOptional": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"undici-types": ">=7.24.0 <7.24.7"
|
||||
"undici-types": "~8.3.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@types/qs": {
|
||||
|
|
@ -1990,17 +1989,6 @@
|
|||
"@types/node": "*"
|
||||
}
|
||||
},
|
||||
"node_modules/@types/uuid": {
|
||||
"version": "11.0.0",
|
||||
"resolved": "https://registry.npmjs.org/@types/uuid/-/uuid-11.0.0.tgz",
|
||||
"integrity": "sha512-HVyk8nj2m+jcFRNazzqyVKiZezyhDKrGUA3jlEcg/nZ6Ms+qHwocba1Y/AaVaznJTAM9xpdFSh+ptbNrhOGvZA==",
|
||||
"deprecated": "This is a stub types definition. uuid provides its own type definitions, so you do not need this installed.",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"uuid": "*"
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/coverage-v8": {
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/coverage-v8/-/coverage-v8-4.1.10.tgz",
|
||||
|
|
@ -2316,20 +2304,20 @@
|
|||
}
|
||||
},
|
||||
"node_modules/body-parser": {
|
||||
"version": "2.2.2",
|
||||
"resolved": "https://registry.npmjs.org/body-parser/-/body-parser-2.2.2.tgz",
|
||||
"integrity": "sha512-oP5VkATKlNwcgvxi0vM0p/D3n2C3EReYVX+DNYs5TjZFn/oQt2j+4sVJtSMr18pdRr8wjTcBl6LoV+FUwzPmNA==",
|
||||
"version": "2.3.0",
|
||||
"resolved": "https://registry.npmjs.org/body-parser/-/body-parser-2.3.0.tgz",
|
||||
"integrity": "sha512-2cGmJupaNgg+QUwVLAucDuWuoMZ6EX9iHDRswZ5lsNYEmwPaRknMPCLZz07yTzVq/83p4o/wzbDZbBrTvGGTIw==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"bytes": "^3.1.2",
|
||||
"content-type": "^1.0.5",
|
||||
"content-type": "^2.0.0",
|
||||
"debug": "^4.4.3",
|
||||
"http-errors": "^2.0.0",
|
||||
"iconv-lite": "^0.7.0",
|
||||
"http-errors": "^2.0.1",
|
||||
"iconv-lite": "^0.7.2",
|
||||
"on-finished": "^2.4.1",
|
||||
"qs": "^6.14.1",
|
||||
"raw-body": "^3.0.1",
|
||||
"type-is": "^2.0.1"
|
||||
"qs": "^6.15.2",
|
||||
"raw-body": "^3.0.2",
|
||||
"type-is": "^2.1.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
|
|
@ -2339,10 +2327,23 @@
|
|||
"url": "https://opencollective.com/express"
|
||||
}
|
||||
},
|
||||
"node_modules/body-parser/node_modules/content-type": {
|
||||
"version": "2.0.0",
|
||||
"resolved": "https://registry.npmjs.org/content-type/-/content-type-2.0.0.tgz",
|
||||
"integrity": "sha512-j/O/d7GcZCyNl7/hwZAb606rzqkyvaDctLmckbxLzHvFBzTJHuGEdodATcP3yIRoDrLHkIATJuvzbFlp/ki2cQ==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
},
|
||||
"funding": {
|
||||
"type": "opencollective",
|
||||
"url": "https://opencollective.com/express"
|
||||
}
|
||||
},
|
||||
"node_modules/brace-expansion": {
|
||||
"version": "5.0.6",
|
||||
"resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.6.tgz",
|
||||
"integrity": "sha512-kLpxurY4Z4r9sgMsyG0Z9uzsBlgiU/EFKhj/h91/8yHu0edo7XuixOIH3VcJ8kkxs6/jPzoI6U9Vj3WqbMQ94g==",
|
||||
"version": "5.0.7",
|
||||
"resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.7.tgz",
|
||||
"integrity": "sha512-7oFy703dxfY3/NLxC1fh2SUCQ0H9rmAY+5EpDVfXjUTTs+HEwR2nYaqLv+GWcTsumwxPfiz6CzCNkwXwBUwqCA==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"balanced-match": "^4.0.2"
|
||||
|
|
@ -5135,9 +5136,9 @@
|
|||
}
|
||||
},
|
||||
"node_modules/tar": {
|
||||
"version": "7.5.16",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.16.tgz",
|
||||
"integrity": "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w==",
|
||||
"version": "7.5.20",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.20.tgz",
|
||||
"integrity": "sha512-9FcyK4PA6+WbzlTM9WhQm6vB5W7cP7dUiPsv1g7YDwEQnQ1CGpK3MGlKk/ITVWMk05kHZuBhmVhiv8LZoy/PFQ==",
|
||||
"license": "BlueOak-1.0.0",
|
||||
"dependencies": {
|
||||
"@isaacs/fs-minipass": "^4.0.0",
|
||||
|
|
@ -5510,9 +5511,9 @@
|
|||
}
|
||||
},
|
||||
"node_modules/undici-types": {
|
||||
"version": "7.24.6",
|
||||
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.24.6.tgz",
|
||||
"integrity": "sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg==",
|
||||
"version": "8.3.0",
|
||||
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-8.3.0.tgz",
|
||||
"integrity": "sha512-j375ScV60dom+YkPFIfTLcOiPxkN/buHz5GobjLhixFuANaNs3C9l4GmrWqejgXWJ7BbJcFYpTEUkS1Ge8bpZQ==",
|
||||
"devOptional": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
|
|
|
|||
|
|
@ -106,8 +106,7 @@
|
|||
"@types/cors": "^2.8.17",
|
||||
"@types/express": "^5.0.6",
|
||||
"@types/js-yaml": "^4.0.9",
|
||||
"@types/node": "^25.6.0",
|
||||
"@types/uuid": "^11.0.0",
|
||||
"@types/node": "^26.0.0",
|
||||
"@vitest/coverage-v8": "^4.0.18",
|
||||
"gitnexus-shared": "file:../gitnexus-shared",
|
||||
"tsx": "^4.0.0",
|
||||
|
|
@ -120,6 +119,6 @@
|
|||
}
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=22.0.0"
|
||||
"node": "^22.18.0 || >=24.11.0"
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -29,6 +29,7 @@ interface HttpConfig {
|
|||
maxAttempts: number;
|
||||
retryCapMs: number;
|
||||
minIntervalMs: number;
|
||||
requestDimensions?: number;
|
||||
}
|
||||
|
||||
export interface EmbeddingRequestOptions {
|
||||
|
|
@ -106,20 +107,26 @@ const paceHttpRequest = async (minIntervalMs: number, signal?: AbortSignal): Pro
|
|||
};
|
||||
|
||||
/**
|
||||
* Stable lead of the {@link readConfig} malformed-`GITNEXUS_EMBEDDING_DIMS`
|
||||
* error. `readConfig` throws a plain `Error` (not an {@link HttpEmbeddingError})
|
||||
* because this is a *config* mistake, not an endpoint failure — so the CLI
|
||||
* recognizes it by this lead ({@link isHttpEmbeddingDimsError}) and prints a
|
||||
* clean config message instead of a raw stack dump. See #2385.
|
||||
* Stable lead of a {@link readConfig} malformed dims-env error. `readConfig`
|
||||
* throws a plain `Error` (not an {@link HttpEmbeddingError}) for a malformed
|
||||
* `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS` because it's a
|
||||
* *config* mistake, not an endpoint failure — so the CLI recognizes it by this
|
||||
* lead ({@link isHttpEmbeddingDimsError}) and prints a clean config message
|
||||
* instead of a raw stack dump. Each var names itself so the message points the
|
||||
* operator at the variable they actually set, not a sibling. See #2385.
|
||||
*/
|
||||
const EMBEDDING_DIMS_ENV_ERROR_LEAD = 'GITNEXUS_EMBEDDING_DIMS must be a positive integer';
|
||||
const dimsEnvErrorLead = (name: string): string => `${name} must be a positive integer`;
|
||||
const EMBEDDING_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_DIMS');
|
||||
const EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_REQUEST_DIMS');
|
||||
|
||||
/**
|
||||
* @internal Exported for the CLI analyze error handler. True when `message` is
|
||||
* the {@link readConfig} malformed-DIMS config error (a plain `Error`).
|
||||
* @internal Exported for the CLI analyze error handler. True when `message` is a
|
||||
* {@link readConfig} malformed dims-env config error (a plain `Error`) — for
|
||||
* either `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS`.
|
||||
*/
|
||||
export const isHttpEmbeddingDimsError = (message: string): boolean =>
|
||||
message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD);
|
||||
message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD) ||
|
||||
message.includes(EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD);
|
||||
|
||||
/**
|
||||
* Build config from the current process.env snapshot.
|
||||
|
|
@ -147,6 +154,23 @@ const readConfig = (): HttpConfig | null => {
|
|||
dimensions = parsed;
|
||||
}
|
||||
|
||||
const rawRequestDims = process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS?.trim();
|
||||
let requestDimensions = dimensions;
|
||||
if (rawRequestDims) {
|
||||
if (/^(omit|none|off|false|0)$/i.test(rawRequestDims)) {
|
||||
requestDimensions = undefined;
|
||||
} else {
|
||||
if (!/^\d+$/.test(rawRequestDims)) {
|
||||
throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`);
|
||||
}
|
||||
const parsed = parseInt(rawRequestDims, 10);
|
||||
if (parsed <= 0) {
|
||||
throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`);
|
||||
}
|
||||
requestDimensions = parsed;
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
baseUrl: baseUrl.replace(/\/+$/, ''),
|
||||
model,
|
||||
|
|
@ -163,6 +187,7 @@ const readConfig = (): HttpConfig | null => {
|
|||
300_000,
|
||||
),
|
||||
minIntervalMs: parseNonNegativeIntegerEnv('GITNEXUS_EMBEDDING_MIN_INTERVAL_MS', 0, 300_000),
|
||||
requestDimensions,
|
||||
};
|
||||
};
|
||||
|
||||
|
|
@ -283,9 +308,9 @@ const isEmbeddingItem = (item: unknown): item is EmbeddingItem =>
|
|||
* the `dimensions` field in the request body. Endpoints that implement
|
||||
* Matryoshka truncation (OpenAI text-embedding-3-*, Cohere embed-v3,
|
||||
* Voyage) return a truncated vector at that size; endpoints that do not
|
||||
* recognise the field may ignore it or return 400. Leave
|
||||
* `GITNEXUS_EMBEDDING_DIMS` unset for strict backends that reject
|
||||
* unknown fields.
|
||||
* recognise the field may ignore it or return 400. Set
|
||||
* `GITNEXUS_EMBEDDING_REQUEST_DIMS=omit` for strict backends while keeping
|
||||
* `GITNEXUS_EMBEDDING_DIMS` set to the returned vector size.
|
||||
*/
|
||||
const httpEmbedBatch = async (
|
||||
url: string,
|
||||
|
|
@ -434,7 +459,7 @@ export const httpEmbed = async (
|
|||
config.model,
|
||||
config.apiKey,
|
||||
batchIndex,
|
||||
config.dimensions,
|
||||
config.requestDimensions,
|
||||
requestOptions,
|
||||
config.maxAttempts,
|
||||
config.retryCapMs,
|
||||
|
|
@ -491,7 +516,7 @@ export const httpEmbedQuery = async (
|
|||
config.model,
|
||||
config.apiKey,
|
||||
0,
|
||||
config.dimensions,
|
||||
config.requestDimensions,
|
||||
requestOptions,
|
||||
config.maxAttempts,
|
||||
config.retryCapMs,
|
||||
|
|
|
|||
|
|
@ -3,8 +3,10 @@
|
|||
*
|
||||
* `module.registerHooks` — the synchronous ESM/CJS resolution-hook API the
|
||||
* embedding-stack resolvers rely on — was added in Node 22.15.0 (and 23.5.0 on
|
||||
* the 23.x line). The gitnexus engines floor is `>=22.0.0`, which admits Node
|
||||
* 22.0–22.14 AND 23.0–23.4, where the export is absent.
|
||||
* the 23.x line). The gitnexus engines floor is `^22.18.0 || >=24.11.0`, so
|
||||
* every supported runtime exposes it — but `engines` is advisory (not
|
||||
* engine-strict), so a below-floor Node (22.0–22.14, or the unsupported
|
||||
* 23.0–23.4 line) can still run, where the export is absent.
|
||||
*
|
||||
* In this `"type": "module"` package, a *static named* import of a missing
|
||||
* builtin export (`import { registerHooks } from 'node:module'`) is a
|
||||
|
|
|
|||
|
|
@ -52,8 +52,8 @@
|
|||
* per-resolution cost is a single string comparison.
|
||||
*
|
||||
* `module.registerHooks` is marked `@experimental` and requires Node >= 22.15
|
||||
* (the gitnexus engines floor is >= 22.0.0). On older runtimes it is absent and
|
||||
* this is a graceful no-op: embeddings then resolve onnxruntime-common exactly
|
||||
* (below the gitnexus engines floor of `^22.18.0 || >=24.11.0`). On below-floor
|
||||
* runtimes it is absent and this is a graceful no-op: embeddings then resolve onnxruntime-common exactly
|
||||
* as before — fine on hoisted layouts. Any failure during installation is
|
||||
* swallowed.
|
||||
*/
|
||||
|
|
@ -100,9 +100,9 @@ export const ensureOnnxRuntimeCommonResolvable = (): void => {
|
|||
attempted = true;
|
||||
|
||||
try {
|
||||
// Node < 22.15 / < 23.5 (the gitnexus engines floor is >= 22.0.0): no
|
||||
// synchronous hooks API. Degrade gracefully — the import still works on
|
||||
// hoisted layouts.
|
||||
// Node < 22.15 / < 23.5 (below the gitnexus engines floor of
|
||||
// ^22.18.0 || >=24.11.0): no synchronous hooks API. Degrade gracefully —
|
||||
// the import still works on hoisted layouts.
|
||||
const registerHooks = getRegisterHooks();
|
||||
if (typeof registerHooks !== 'function') return;
|
||||
|
||||
|
|
|
|||
|
|
@ -36,8 +36,8 @@
|
|||
* So CUDA-12 hosts, Windows (DirectML), macOS, and CPU-only hosts are
|
||||
* untouched. Idempotent; any failure is swallowed and leaves the default
|
||||
* resolution exactly as before. `module.registerHooks` requires Node >= 22.15
|
||||
* (the gitnexus engines floor is >= 22.0.0); on older runtimes the redirect is
|
||||
* a no-op, but the default copy's CUDA major is still probed so an
|
||||
* (below the gitnexus engines floor of `^22.18.0 || >=24.11.0`); on below-floor
|
||||
* runtimes the redirect is a no-op, but the default copy's CUDA major is still probed so an
|
||||
* already-matching host (e.g. CUDA 12 + transformers' CUDA-12 build) keeps
|
||||
* auto-selecting the GPU.
|
||||
* `npm link` / symlinked local-dev checkouts are a known caveat: `resolveOurOrtNodeDir`/
|
||||
|
|
|
|||
|
|
@ -191,7 +191,7 @@ export const ensureEmbeddingStackResolvable = (): void => {
|
|||
hookAttempted = true;
|
||||
|
||||
try {
|
||||
// Node < 22.15 / < 23.5 (engines floor is >= 22.0.0): no synchronous hooks
|
||||
// Node < 22.15 / < 23.5 (below the engines floor of ^22.18.0 || >=24.11.0): no synchronous hooks
|
||||
// API. Degrade gracefully — normally-installed stacks still resolve; only
|
||||
// the runtime-prefix fallback is unavailable. Reachable now that the import
|
||||
// is a namespace access (see node-module-compat.ts) rather than a static
|
||||
|
|
|
|||
|
|
@ -59,6 +59,7 @@ export interface SpringBeanCandidateAdapter {
|
|||
}
|
||||
|
||||
type OwnedTypeNamesByOwner = ReadonlyMap<string, ReadonlySet<string>>;
|
||||
type RecognizedAnnotationNames = { readonly has: (value: string) => boolean };
|
||||
|
||||
function simpleNameOf(def: SymbolDefinition): string | undefined {
|
||||
const qualifiedName = def.qualifiedName;
|
||||
|
|
@ -152,7 +153,11 @@ function hasVisibleTypeBinding(
|
|||
return false;
|
||||
}
|
||||
|
||||
function wildcardImportTarget(parsed: ParsedFile, simpleName: string): string | undefined {
|
||||
function wildcardImportTarget(
|
||||
parsed: ParsedFile,
|
||||
simpleName: string,
|
||||
recognizedAnnotations: RecognizedAnnotationNames,
|
||||
): string | undefined {
|
||||
const wildcardPackages = new Set(
|
||||
parsed.parsedImports
|
||||
.filter((entry) => entry.kind === 'wildcard')
|
||||
|
|
@ -161,37 +166,40 @@ function wildcardImportTarget(parsed: ParsedFile, simpleName: string): string |
|
|||
if (wildcardPackages.size !== 1) return undefined;
|
||||
const [packageName] = wildcardPackages;
|
||||
const target = `${packageName}.${simpleName}`;
|
||||
return SPRING_BEAN_STEREOTYPES.has(target) ? target : undefined;
|
||||
return recognizedAnnotations.has(target) ? target : undefined;
|
||||
}
|
||||
|
||||
function resolveSpringAnnotation(
|
||||
rawName: string,
|
||||
parsed: ParsedFile,
|
||||
enclosingScope: ScopeId | null,
|
||||
indexes: ScopeResolutionIndexes,
|
||||
ownedTypeNamesByOwner: OwnedTypeNamesByOwner,
|
||||
isPackageVisibilityIncomplete: boolean,
|
||||
): string | undefined {
|
||||
if (rawName.includes('.')) {
|
||||
return SPRING_BEAN_STEREOTYPES.has(rawName) ? rawName : undefined;
|
||||
}
|
||||
/** Build a scope-aware Spring annotation resolver shared by framework hooks. */
|
||||
export function createSpringAnnotationNameResolver(indexes: ScopeResolutionIndexes) {
|
||||
const ownedTypeNamesByOwner = buildOwnedTypeNamesByOwner(indexes);
|
||||
return (
|
||||
rawName: string,
|
||||
parsed: ParsedFile,
|
||||
enclosingScope: ScopeId | null,
|
||||
recognizedAnnotations: RecognizedAnnotationNames,
|
||||
isPackageVisibilityIncomplete: boolean,
|
||||
): string | undefined => {
|
||||
if (rawName.includes('.')) {
|
||||
return recognizedAnnotations.has(rawName) ? rawName : undefined;
|
||||
}
|
||||
|
||||
if (hasLexicalTypeDeclaration(enclosingScope, rawName, indexes)) return undefined;
|
||||
if (hasInheritedTypeDeclaration(enclosingScope, rawName, indexes, ownedTypeNamesByOwner)) {
|
||||
return undefined;
|
||||
}
|
||||
if (hasLexicalTypeDeclaration(enclosingScope, rawName, indexes)) return undefined;
|
||||
if (hasInheritedTypeDeclaration(enclosingScope, rawName, indexes, ownedTypeNamesByOwner)) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const explicitImports = explicitImportTargets(parsed, rawName);
|
||||
if (explicitImports.size > 0) {
|
||||
if (explicitImports.size !== 1) return undefined;
|
||||
const [imported] = explicitImports;
|
||||
return SPRING_BEAN_STEREOTYPES.has(imported) ? imported : undefined;
|
||||
}
|
||||
const explicitImports = explicitImportTargets(parsed, rawName);
|
||||
if (explicitImports.size > 0) {
|
||||
if (explicitImports.size !== 1) return undefined;
|
||||
const [imported] = explicitImports;
|
||||
return recognizedAnnotations.has(imported) ? imported : undefined;
|
||||
}
|
||||
|
||||
const wildcardTarget = wildcardImportTarget(parsed, rawName);
|
||||
if (wildcardTarget === undefined || isPackageVisibilityIncomplete) return undefined;
|
||||
const wildcardTarget = wildcardImportTarget(parsed, rawName, recognizedAnnotations);
|
||||
if (wildcardTarget === undefined || isPackageVisibilityIncomplete) return undefined;
|
||||
|
||||
return hasVisibleTypeBinding(enclosingScope, rawName, indexes) ? undefined : wildcardTarget;
|
||||
return hasVisibleTypeBinding(enclosingScope, rawName, indexes) ? undefined : wildcardTarget;
|
||||
};
|
||||
}
|
||||
|
||||
/** Build a language hook that enriches Class nodes after scope resolution. */
|
||||
|
|
@ -202,7 +210,7 @@ export function createSpringBeanCandidateAttacher(adapter: SpringBeanCandidateAd
|
|||
nodeLookup: GraphNodeLookup,
|
||||
indexes: ScopeResolutionIndexes,
|
||||
): void => {
|
||||
const ownedTypeNamesByOwner = buildOwnedTypeNamesByOwner(indexes);
|
||||
const resolveSpringAnnotation = createSpringAnnotationNameResolver(indexes);
|
||||
for (const parsed of parsedFiles) {
|
||||
for (const fact of adapter.getClassAnnotationFacts(parsed.filePath)) {
|
||||
const classScope = indexes.scopeTree.getScope(fact.classScopeId);
|
||||
|
|
@ -221,8 +229,7 @@ export function createSpringBeanCandidateAttacher(adapter: SpringBeanCandidateAd
|
|||
rawName,
|
||||
parsed,
|
||||
classScope.parent,
|
||||
indexes,
|
||||
ownedTypeNamesByOwner,
|
||||
SPRING_BEAN_STEREOTYPES,
|
||||
adapter.isPackageVisibilityIncomplete(parsed.filePath),
|
||||
);
|
||||
if (annotation !== undefined) recognized.add(annotation);
|
||||
|
|
|
|||
166
gitnexus/src/core/ingestion/frameworks/spring/config-bindings.ts
Normal file
166
gitnexus/src/core/ingestion/frameworks/spring/config-bindings.ts
Normal file
|
|
@ -0,0 +1,166 @@
|
|||
import type { GraphNode } from 'gitnexus-shared';
|
||||
import type { KnowledgeGraph } from '../../../graph/types.js';
|
||||
import { generateId } from '../../../../lib/utils.js';
|
||||
|
||||
export const SPRING_CONFIG_DESCRIPTION = 'Spring configuration property';
|
||||
|
||||
export interface SpringValueConsumer {
|
||||
readonly kind: 'value';
|
||||
readonly fieldName: string;
|
||||
readonly line: number;
|
||||
readonly keys: readonly string[];
|
||||
}
|
||||
|
||||
export interface SpringConfigurationPropertiesConsumer {
|
||||
readonly kind: 'configuration-properties';
|
||||
readonly className: string;
|
||||
readonly line: number;
|
||||
readonly prefix: string;
|
||||
}
|
||||
|
||||
export type SpringConfigConsumer = SpringValueConsumer | SpringConfigurationPropertiesConsumer;
|
||||
|
||||
export interface SpringConfigConsumerBatch {
|
||||
readonly filePath: string;
|
||||
readonly consumers: readonly SpringConfigConsumer[];
|
||||
}
|
||||
|
||||
function closestNode(
|
||||
candidates: readonly GraphNode[],
|
||||
filePath: string,
|
||||
name: string,
|
||||
line: number,
|
||||
): GraphNode | undefined {
|
||||
return candidates
|
||||
.filter((node) => node.properties.filePath === filePath && node.properties.name === name)
|
||||
.sort(
|
||||
(left, right) =>
|
||||
Math.abs(Number(left.properties.startLine ?? 0) - line) -
|
||||
Math.abs(Number(right.properties.startLine ?? 0) - line),
|
||||
)[0];
|
||||
}
|
||||
|
||||
function markUnresolved(node: GraphNode, key: string): void {
|
||||
const marker = `Spring config unresolved: ${key}`;
|
||||
const existing =
|
||||
typeof node.properties.description === 'string' ? node.properties.description : '';
|
||||
if (existing.includes(marker)) return;
|
||||
node.properties.description = existing.length > 0 ? `${existing}; ${marker}` : marker;
|
||||
}
|
||||
|
||||
function relaxedName(value: string): string {
|
||||
return value.toLowerCase().replace(/[-_.]/g, '');
|
||||
}
|
||||
|
||||
function isSpringConfigNode(node: GraphNode): boolean {
|
||||
return (
|
||||
node.label === 'Property' &&
|
||||
typeof node.properties.description === 'string' &&
|
||||
node.properties.description.startsWith(SPRING_CONFIG_DESCRIPTION)
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Attach normalized, language-provider-produced Spring consumers to config
|
||||
* keys already present in the shared graph.
|
||||
*/
|
||||
export function bindSpringConfigConsumers(
|
||||
graph: KnowledgeGraph,
|
||||
batches: readonly SpringConfigConsumerBatch[],
|
||||
): void {
|
||||
if (batches.length === 0) return;
|
||||
|
||||
const configNodes: GraphNode[] = [];
|
||||
const propertyNodes: GraphNode[] = [];
|
||||
const classNodes: GraphNode[] = [];
|
||||
for (const node of graph.iterNodes()) {
|
||||
if (isSpringConfigNode(node)) configNodes.push(node);
|
||||
else if (node.label === 'Property') propertyNodes.push(node);
|
||||
else if (node.label === 'Class' || node.label === 'Record') classNodes.push(node);
|
||||
}
|
||||
|
||||
const keyNodes = new Map<string, GraphNode[]>();
|
||||
for (const node of configNodes) {
|
||||
const key = String(node.properties.name);
|
||||
const bucket = keyNodes.get(key) ?? [];
|
||||
bucket.push(node);
|
||||
keyNodes.set(key, bucket);
|
||||
}
|
||||
|
||||
const propertiesByOwner = new Map<string, GraphNode[]>();
|
||||
for (const rel of graph.iterRelationshipsByType('HAS_PROPERTY')) {
|
||||
const property = graph.getNode(rel.targetId);
|
||||
if (property?.label !== 'Property' || isSpringConfigNode(property)) continue;
|
||||
const members = propertiesByOwner.get(rel.sourceId) ?? [];
|
||||
members.push(property);
|
||||
propertiesByOwner.set(rel.sourceId, members);
|
||||
}
|
||||
|
||||
const addBinding = (
|
||||
source: GraphNode,
|
||||
target: GraphNode,
|
||||
reason: string,
|
||||
confidence: number,
|
||||
): void => {
|
||||
const edgeId = generateId('USES', `${source.id}->${target.id}:${reason}`);
|
||||
graph.addRelationship({
|
||||
id: edgeId,
|
||||
sourceId: source.id,
|
||||
targetId: target.id,
|
||||
type: 'USES',
|
||||
confidence,
|
||||
reason,
|
||||
});
|
||||
};
|
||||
|
||||
for (const { filePath, consumers } of batches) {
|
||||
for (const consumer of consumers) {
|
||||
if (consumer.kind === 'value') {
|
||||
const field = closestNode(propertyNodes, filePath, consumer.fieldName, consumer.line);
|
||||
if (field === undefined) continue;
|
||||
for (const key of consumer.keys) {
|
||||
const matches = keyNodes.get(key) ?? [];
|
||||
if (matches.length === 0) {
|
||||
markUnresolved(field, key);
|
||||
continue;
|
||||
}
|
||||
for (const match of matches) {
|
||||
addBinding(field, match, `spring-config:@Value ${key}`, 1);
|
||||
}
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
const owner = closestNode(classNodes, filePath, consumer.className, consumer.line);
|
||||
if (owner === undefined) continue;
|
||||
const prefix = `${consumer.prefix}.`;
|
||||
const matches = configNodes.filter((node) => {
|
||||
const key = String(node.properties.name);
|
||||
return key === consumer.prefix || key.startsWith(prefix);
|
||||
});
|
||||
if (matches.length === 0) {
|
||||
markUnresolved(owner, consumer.prefix);
|
||||
continue;
|
||||
}
|
||||
for (const match of matches) {
|
||||
addBinding(owner, match, `spring-config:@ConfigurationProperties ${consumer.prefix}`, 0.95);
|
||||
}
|
||||
|
||||
for (const field of propertiesByOwner.get(owner.id) ?? []) {
|
||||
const fieldName = relaxedName(String(field.properties.name));
|
||||
for (const match of matches) {
|
||||
const key = String(match.properties.name);
|
||||
const suffix = key === consumer.prefix ? '' : key.slice(prefix.length);
|
||||
const firstSegment = suffix.split(/[.\[]/, 1)[0];
|
||||
if (firstSegment.length === 0 || relaxedName(firstSegment) !== fieldName) continue;
|
||||
addBinding(
|
||||
field,
|
||||
match,
|
||||
`spring-config:@ConfigurationProperties field ${consumer.prefix}`,
|
||||
0.95,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -0,0 +1,19 @@
|
|||
import type { AnalysisFeatureDescriptor } from '../../../analysis-features.js';
|
||||
|
||||
function isSpringApplicationConfig(filePath: string): boolean {
|
||||
const base = filePath.replaceAll('\\', '/').split('/').pop() ?? '';
|
||||
return /^application(?:-[^.]+)?\.(?:properties|ya?ml)$/i.test(base);
|
||||
}
|
||||
|
||||
/** Durable completeness contract for Java Spring configuration bindings. */
|
||||
export const SPRING_CONFIG_BINDINGS_FEATURE: AnalysisFeatureDescriptor = {
|
||||
id: 'spring.config-bindings',
|
||||
version: 1,
|
||||
// Java sources need consumer extraction even without config files (missing
|
||||
// placeholders still get unresolved markers). Config-only repositories also
|
||||
// need a one-time rebuild to backfill language-agnostic Property nodes.
|
||||
appliesTo: (filePaths) =>
|
||||
filePaths.some(
|
||||
(filePath) => filePath.toLowerCase().endsWith('.java') || isSpringApplicationConfig(filePath),
|
||||
),
|
||||
};
|
||||
|
|
@ -9,6 +9,7 @@ import {
|
|||
type JvmPackageFact,
|
||||
} from '../jvm/package-facts.js';
|
||||
import { getJavaPackageFact, setJavaPackageFact } from './package-facts.js';
|
||||
import type { JavaSpringConfigConsumerFact } from './spring-config-bindings.js';
|
||||
|
||||
export type JavaClassAnnotationFact = ClassAnnotationFact;
|
||||
|
||||
|
|
@ -16,13 +17,16 @@ export interface JavaCaptureSideChannel {
|
|||
readonly kind: 'java';
|
||||
readonly packageFact: JvmPackageFact;
|
||||
readonly classAnnotations: readonly JavaClassAnnotationFact[];
|
||||
readonly springConfigConsumers?: readonly JavaSpringConfigConsumerFact[];
|
||||
}
|
||||
|
||||
const classAnnotations = createClassAnnotationFactStore();
|
||||
const springConfigConsumers = new Map<string, readonly JavaSpringConfigConsumerFact[]>();
|
||||
|
||||
/** Clear facts retained by a prior workspace pass in a long-lived process. */
|
||||
export function clearJavaClassAnnotationFacts(): void {
|
||||
classAnnotations.clear();
|
||||
springConfigConsumers.clear();
|
||||
}
|
||||
|
||||
/** Store the annotation syntax collected by Java's existing scope-query traversal. */
|
||||
|
|
@ -33,17 +37,35 @@ export function setJavaClassAnnotationFacts(
|
|||
classAnnotations.set(filePath, facts);
|
||||
}
|
||||
|
||||
export function setJavaSpringConfigConsumerFacts(
|
||||
filePath: string,
|
||||
facts: readonly JavaSpringConfigConsumerFact[],
|
||||
): void {
|
||||
if (facts.length === 0) springConfigConsumers.delete(filePath);
|
||||
else springConfigConsumers.set(filePath, facts);
|
||||
}
|
||||
|
||||
export function getJavaSpringConfigConsumerFacts(
|
||||
filePath: string,
|
||||
): readonly JavaSpringConfigConsumerFact[] {
|
||||
return springConfigConsumers.get(filePath) ?? [];
|
||||
}
|
||||
|
||||
/** Snapshot worker-local Java annotation facts for ParsedFile serialization. */
|
||||
export function collectJavaCaptureSideChannel(
|
||||
filePath: string,
|
||||
): JavaCaptureSideChannel | undefined {
|
||||
const facts = classAnnotations.get(filePath);
|
||||
const configConsumers = springConfigConsumers.get(filePath) ?? [];
|
||||
const packageFact = getJavaPackageFact(filePath);
|
||||
if (facts.length === 0 && packageFact === undefined) return undefined;
|
||||
if (facts.length === 0 && configConsumers.length === 0 && packageFact === undefined) {
|
||||
return undefined;
|
||||
}
|
||||
return {
|
||||
kind: 'java',
|
||||
packageFact: packageFact ?? UNKNOWN_JVM_PACKAGE_FACT,
|
||||
classAnnotations: facts,
|
||||
...(configConsumers.length > 0 ? { springConfigConsumers: configConsumers } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
|
|
@ -62,10 +84,15 @@ export function applyJavaCaptureSideChannel(parsed: ParsedFile): void {
|
|||
!Array.isArray(data.classAnnotations)
|
||||
) {
|
||||
setJavaClassAnnotationFacts(parsed.filePath, []);
|
||||
setJavaSpringConfigConsumerFacts(parsed.filePath, []);
|
||||
setJavaPackageFact(parsed.filePath, UNKNOWN_JVM_PACKAGE_FACT);
|
||||
return;
|
||||
}
|
||||
setJavaClassAnnotationFacts(parsed.filePath, data.classAnnotations);
|
||||
setJavaSpringConfigConsumerFacts(
|
||||
parsed.filePath,
|
||||
Array.isArray(data.springConfigConsumers) ? data.springConfigConsumers : [],
|
||||
);
|
||||
setJavaPackageFact(
|
||||
parsed.filePath,
|
||||
isJvmPackageFact(data.packageFact) ? data.packageFact : UNKNOWN_JVM_PACKAGE_FACT,
|
||||
|
|
|
|||
|
|
@ -32,9 +32,13 @@ import { getJavaParser, getJavaScopeQuery } from './query.js';
|
|||
import { recordCacheHit, recordCacheMiss } from './cache-stats.js';
|
||||
import { getTreeSitterBufferSize } from '../../constants.js';
|
||||
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
|
||||
import { setJavaClassAnnotationFacts } from './capture-side-channel.js';
|
||||
import {
|
||||
setJavaClassAnnotationFacts,
|
||||
setJavaSpringConfigConsumerFacts,
|
||||
} from './capture-side-channel.js';
|
||||
import { captureJavaPackageFact } from './package-facts.js';
|
||||
import { synthesizeCallableFlowCaptures } from '../../utils/callable-flow-captures.js';
|
||||
import { captureJavaSpringConfigConsumerFacts } from './spring-config-bindings.js';
|
||||
|
||||
/** Declaration anchors that carry function-like arity metadata. */
|
||||
const FUNCTION_DECL_TAGS = ['@declaration.method', '@declaration.constructor'] as const;
|
||||
|
|
@ -150,6 +154,29 @@ export function emitJavaScopeCaptures(
|
|||
continue;
|
||||
}
|
||||
|
||||
// Normalize a `new`-expression receiver to its constructed type's simple
|
||||
// name: `new Local().inner()` binds the WHOLE `object_creation_expression`
|
||||
// as `@reference.receiver`, so its raw text is `"new Local()"` — a string
|
||||
// that can never match a scope binding, so the compound-receiver resolver
|
||||
// silently falls through to name-only fallback resolution and picks the
|
||||
// wrong same-named method on a collision (#2564). Rewriting the text to
|
||||
// just `Local` lets Case 2 (class-name / static receiver) in
|
||||
// receiver-bound-calls.ts resolve it via its normal MRO walk. Mirrors the
|
||||
// established `normalizePhpReceiver` precedent (php/captures.ts) — a
|
||||
// language-local capture rewrite, no shared-pipeline change.
|
||||
if (grouped['@reference.receiver'] !== undefined) {
|
||||
const receiverNode = nodeIfType(nodeMap['@reference.receiver'], 'object_creation_expression');
|
||||
const typeNode = receiverNode?.childForFieldName('type');
|
||||
const simpleName = typeNode ? javaBaseSimpleNameOf(typeNode) : undefined;
|
||||
if (simpleName !== undefined) {
|
||||
grouped['@reference.receiver'] = syntheticCapture(
|
||||
'@reference.receiver',
|
||||
receiverNode!,
|
||||
simpleName,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Filter read.member when it's a child of method_invocation or assignment.
|
||||
// `@reference.read.member` is captured directly on the `field_access` node.
|
||||
if (grouped['@reference.read.member'] !== undefined) {
|
||||
|
|
@ -257,6 +284,10 @@ export function emitJavaScopeCaptures(
|
|||
}
|
||||
|
||||
setJavaClassAnnotationFacts(filePath, materializeClassAnnotationFacts(classAnnotations));
|
||||
setJavaSpringConfigConsumerFacts(
|
||||
filePath,
|
||||
captureJavaSpringConfigConsumerFacts(tree.rootNode, filePath),
|
||||
);
|
||||
|
||||
return [
|
||||
...resolveVarTypeBindings(out),
|
||||
|
|
@ -341,23 +372,49 @@ function synthesizeJavaAnonymousClassDeclarations(rootNode: SyntaxNode): Capture
|
|||
// constant's class extends its HOST ENUM (javac semantics), so the
|
||||
// inherits reference names the enum — giving `mroFor(E$N) ∋ E` and
|
||||
// keeping bare calls from the body to the enum's own helpers alive
|
||||
// through the ownership gate's MRO arm. No receiver typeBinding piece:
|
||||
// constants are not variable initializers; `E.A.hook()` dispatch rides
|
||||
// the existing enum receiver machinery.
|
||||
// through the ownership gate's MRO arm.
|
||||
for (const constant of rootNode.descendantsOfType('enum_constant')) {
|
||||
const name = synthesizeJavaAnonymousClassName(constant);
|
||||
if (name === undefined) continue;
|
||||
const body = constant.childForFieldName?.('body');
|
||||
if (body === null || body === undefined || body.type !== 'class_body') continue;
|
||||
out.push({
|
||||
'@declaration.class': nodeToCapture('@declaration.class', body),
|
||||
'@declaration.name': syntheticCapture('@declaration.name', body, name),
|
||||
});
|
||||
const hostEnum = javaEnclosingEnumNameOf(constant);
|
||||
if (hostEnum !== undefined) {
|
||||
const bodyNode = constant.childForFieldName?.('body');
|
||||
const isBodied = bodyNode !== null && bodyNode !== undefined && bodyNode.type === 'class_body';
|
||||
const bodiedName = synthesizeJavaAnonymousClassName(constant);
|
||||
if (bodiedName !== undefined && isBodied) {
|
||||
out.push({
|
||||
'@reference.inherits': nodeToCapture('@reference.inherits', body),
|
||||
'@reference.name': syntheticCapture('@reference.name', body, hostEnum),
|
||||
'@declaration.class': nodeToCapture('@declaration.class', bodyNode),
|
||||
'@declaration.name': syntheticCapture('@declaration.name', bodyNode, bodiedName),
|
||||
});
|
||||
if (hostEnum !== undefined) {
|
||||
out.push({
|
||||
'@reference.inherits': nodeToCapture('@reference.inherits', bodyNode),
|
||||
'@reference.name': syntheticCapture('@reference.name', bodyNode, hostEnum),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Receiver dispatch (#2561): `E.CONST.method()` resolves through the
|
||||
// generic compound-receiver chain walk, which looks up each dotted
|
||||
// segment via the owning class scope's `typeBindings` map — the same
|
||||
// mechanism a field declaration uses (`private User user;` binds
|
||||
// `user` on the class scope). Binding the constant's own simple name
|
||||
// there — to its synthesized `E$N` class when bodied (MRO includes E,
|
||||
// so members inherited from the enum still resolve), or to the host
|
||||
// enum itself when body-less — makes `E.CONST.method()` resolve with
|
||||
// no changes to the shared receiver-binding machinery.
|
||||
//
|
||||
// A bodied constant binds ONLY to its `E$N` class, never the host enum:
|
||||
// if name synthesis fails on a malformed/error-recovery tree (`bodiedName`
|
||||
// undefined despite a real body), emit nothing rather than silently
|
||||
// misattributing an OVERRIDING constant's receiver to the enum's own
|
||||
// (non-overridden) method — a wrong edge is worse than no edge. Mirrors
|
||||
// the `object_creation_expression` branch, which skips on synthesis
|
||||
// failure. `hostEnum` is used only for genuinely body-less constants.
|
||||
const constantNameNode = constant.childForFieldName?.('name');
|
||||
const constantType = isBodied ? bodiedName : hostEnum;
|
||||
if (constantNameNode !== null && constantNameNode !== undefined && constantType !== undefined) {
|
||||
out.push({
|
||||
'@type-binding.annotation': nodeToCapture('@type-binding.annotation', constant),
|
||||
'@type-binding.name': nodeToCapture('@type-binding.name', constantNameNode),
|
||||
'@type-binding.type': syntheticCapture('@type-binding.type', constant, constantType),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -30,6 +30,7 @@ import {
|
|||
} from './index.js';
|
||||
import { populateJavaPackageSiblings } from './package-siblings.js';
|
||||
import { attachSpringBeanCandidateMetadata } from './spring-bean-metadata.js';
|
||||
import { attachJavaSpringConfigBindings } from './spring-config-bindings.js';
|
||||
import {
|
||||
applyJavaCaptureSideChannel,
|
||||
clearJavaClassAnnotationFacts,
|
||||
|
|
@ -83,7 +84,10 @@ const javaScopeResolver: ScopeResolver = {
|
|||
|
||||
populateNamespaceSiblings: populateJavaPackageSiblings,
|
||||
populateRangeBindings: populateJavaCrossFileReturnTypes,
|
||||
emitPostResolutionEdges: attachSpringBeanCandidateMetadata,
|
||||
emitPostResolutionEdges: (graph, parsedFiles, nodeLookup, indexes, ctx) => {
|
||||
attachSpringBeanCandidateMetadata(graph, parsedFiles, nodeLookup, indexes);
|
||||
attachJavaSpringConfigBindings(graph, parsedFiles, nodeLookup, indexes, ctx);
|
||||
},
|
||||
};
|
||||
|
||||
export { javaScopeResolver };
|
||||
|
|
|
|||
|
|
@ -0,0 +1,267 @@
|
|||
import type { KnowledgeGraph } from '../../../graph/types.js';
|
||||
import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js';
|
||||
import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js';
|
||||
import { makeScopeId, type ParsedFile, type ScopeId } from 'gitnexus-shared';
|
||||
import {
|
||||
bindSpringConfigConsumers,
|
||||
type SpringConfigConsumer,
|
||||
} from '../../frameworks/spring/config-bindings.js';
|
||||
import { createSpringAnnotationNameResolver } from '../../frameworks/spring/bean-candidates.js';
|
||||
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
|
||||
import { nodeToCapture, type SyntaxNode } from '../../utils/ast-helpers.js';
|
||||
import { getJavaParser } from './query.js';
|
||||
import { getJavaSpringConfigConsumerFacts } from './capture-side-channel.js';
|
||||
import { isJavaPackageSiblingVisibilityIncomplete } from './package-siblings.js';
|
||||
|
||||
const VALUE_ANNOTATION = 'org.springframework.beans.factory.annotation.Value';
|
||||
const CONFIGURATION_PROPERTIES_ANNOTATION =
|
||||
'org.springframework.boot.context.properties.ConfigurationProperties';
|
||||
|
||||
interface JavaAnnotation {
|
||||
readonly name: string;
|
||||
readonly node: SyntaxNode;
|
||||
}
|
||||
|
||||
interface JavaImports {
|
||||
readonly exact: ReadonlySet<string>;
|
||||
readonly wildcard: ReadonlySet<string>;
|
||||
readonly localTypes: ReadonlySet<string>;
|
||||
}
|
||||
|
||||
export interface JavaSpringConfigConsumerFact {
|
||||
readonly consumer: SpringConfigConsumer;
|
||||
readonly annotationName: string;
|
||||
readonly classScopeId: ScopeId;
|
||||
}
|
||||
|
||||
function collectJavaImports(root: SyntaxNode): JavaImports {
|
||||
const exact = new Set<string>();
|
||||
const wildcard = new Set<string>();
|
||||
const localTypes = new Set<string>();
|
||||
|
||||
for (const node of root.descendantsOfType('import_declaration')) {
|
||||
const imported = node.text
|
||||
.replace(/^\s*import\s+(?:static\s+)?/, '')
|
||||
.replace(/;\s*$/, '')
|
||||
.trim();
|
||||
if (imported.endsWith('.*')) wildcard.add(imported.slice(0, -2));
|
||||
else exact.add(imported);
|
||||
}
|
||||
|
||||
for (const type of [
|
||||
'class_declaration',
|
||||
'interface_declaration',
|
||||
'enum_declaration',
|
||||
'record_declaration',
|
||||
'annotation_type_declaration',
|
||||
]) {
|
||||
for (const node of root.descendantsOfType(type)) {
|
||||
const name = node.childForFieldName('name')?.text;
|
||||
if (name) localTypes.add(name);
|
||||
}
|
||||
}
|
||||
return { exact, wildcard, localTypes };
|
||||
}
|
||||
|
||||
function annotationsOn(node: SyntaxNode): JavaAnnotation[] {
|
||||
const modifiers = node.namedChildren.find((child) => child.type === 'modifiers');
|
||||
if (modifiers === undefined) return [];
|
||||
const annotations: JavaAnnotation[] = [];
|
||||
for (const child of modifiers.namedChildren) {
|
||||
if (child.type !== 'annotation' && child.type !== 'marker_annotation') continue;
|
||||
const name = child.childForFieldName('name')?.text ?? child.firstNamedChild?.text;
|
||||
if (name) annotations.push({ name, node: child });
|
||||
}
|
||||
return annotations;
|
||||
}
|
||||
|
||||
function resolvesToAnnotation(
|
||||
rawName: string,
|
||||
canonicalName: string,
|
||||
imports: JavaImports,
|
||||
): boolean {
|
||||
if (rawName.includes('.')) return rawName === canonicalName;
|
||||
if (imports.localTypes.has(rawName)) return false;
|
||||
if (imports.exact.has(canonicalName)) return true;
|
||||
const packageName = canonicalName.slice(0, canonicalName.lastIndexOf('.'));
|
||||
return imports.wildcard.has(packageName);
|
||||
}
|
||||
|
||||
function decodeJavaStringLiteral(literal: string): string {
|
||||
const delimiterLength = literal.startsWith('"""') && literal.endsWith('"""') ? 3 : 1;
|
||||
return literal
|
||||
.slice(delimiterLength, -delimiterLength)
|
||||
.replace(/\\u([0-9a-fA-F]{4})/g, (_match, hex: string) =>
|
||||
String.fromCharCode(Number.parseInt(hex, 16)),
|
||||
)
|
||||
.replace(/\\(["'\\btnfr])/g, (_match, escaped: string) => {
|
||||
const controls: Record<string, string> = {
|
||||
b: '\b',
|
||||
t: '\t',
|
||||
n: '\n',
|
||||
f: '\f',
|
||||
r: '\r',
|
||||
};
|
||||
return controls[escaped] ?? escaped;
|
||||
});
|
||||
}
|
||||
|
||||
function javaStringLiterals(annotation: SyntaxNode): string[] {
|
||||
return annotation
|
||||
.descendantsOfType('string_literal')
|
||||
.map((literal) => decodeJavaStringLiteral(literal.text));
|
||||
}
|
||||
|
||||
/** Extract statically readable Spring placeholder keys from a Java annotation. */
|
||||
export function parseValuePlaceholderKeys(annotation: SyntaxNode): string[] {
|
||||
const keys = new Set<string>();
|
||||
for (const literal of javaStringLiterals(annotation)) {
|
||||
for (const match of literal.matchAll(/\$\{([^{}]+)\}/g)) {
|
||||
const key = match[1].split(':', 1)[0].trim();
|
||||
if (/^[A-Za-z0-9_.-]+$/.test(key)) keys.add(key);
|
||||
}
|
||||
}
|
||||
return [...keys];
|
||||
}
|
||||
|
||||
/** Extract `prefix`/`value` (or the positional value) from the annotation. */
|
||||
export function parseConfigurationPropertiesPrefix(annotation: SyntaxNode): string | null {
|
||||
const named = annotation.descendantsOfType('element_value_pair').find((pair) => {
|
||||
const key = pair.childForFieldName('key')?.text;
|
||||
return key === 'prefix' || key === 'value';
|
||||
});
|
||||
const namedValue = named?.childForFieldName('value');
|
||||
const argumentsNode = annotation.childForFieldName('arguments');
|
||||
const literalNode =
|
||||
(namedValue?.type === 'string_literal'
|
||||
? namedValue
|
||||
: namedValue?.descendantsOfType('string_literal')[0]) ??
|
||||
(named === undefined
|
||||
? argumentsNode?.namedChildren.find((child) => child.type === 'string_literal')
|
||||
: undefined);
|
||||
if (literalNode === undefined) return null;
|
||||
const prefix = decodeJavaStringLiteral(literalNode.text)
|
||||
.trim()
|
||||
.replace(/^\.+|\.+$/g, '');
|
||||
return /^[A-Za-z0-9_.-]+$/.test(prefix) ? prefix : null;
|
||||
}
|
||||
|
||||
function classScopeId(filePath: string, declaration: SyntaxNode): ScopeId {
|
||||
return makeScopeId({
|
||||
filePath,
|
||||
range: nodeToCapture('@scope.class', declaration).range,
|
||||
kind: 'Class',
|
||||
});
|
||||
}
|
||||
|
||||
function enclosingClass(node: SyntaxNode): SyntaxNode | undefined {
|
||||
let current = node.parent;
|
||||
while (current !== null) {
|
||||
if (current.type === 'class_declaration' || current.type === 'record_declaration') {
|
||||
return current;
|
||||
}
|
||||
current = current.parent;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/** Collect config facts from the Java parser's existing AST (no reparse). */
|
||||
export function captureJavaSpringConfigConsumerFacts(
|
||||
root: SyntaxNode,
|
||||
filePath: string,
|
||||
): JavaSpringConfigConsumerFact[] {
|
||||
const imports = collectJavaImports(root);
|
||||
const facts: JavaSpringConfigConsumerFact[] = [];
|
||||
|
||||
for (const field of root.descendantsOfType('field_declaration')) {
|
||||
const annotations = annotationsOn(field).filter((annotation) =>
|
||||
resolvesToAnnotation(annotation.name, VALUE_ANNOTATION, imports),
|
||||
);
|
||||
if (annotations.length === 0) continue;
|
||||
const owner = enclosingClass(field);
|
||||
if (owner === undefined) continue;
|
||||
for (const declarator of field.namedChildren.filter(
|
||||
(child) => child.type === 'variable_declarator',
|
||||
)) {
|
||||
const fieldName = declarator.childForFieldName('name')?.text;
|
||||
if (!fieldName) continue;
|
||||
for (const annotation of annotations) {
|
||||
const keys = parseValuePlaceholderKeys(annotation.node);
|
||||
if (keys.length > 0) {
|
||||
facts.push({
|
||||
consumer: { kind: 'value', fieldName, line: field.startPosition.row + 1, keys },
|
||||
annotationName: annotation.name,
|
||||
classScopeId: classScopeId(filePath, owner),
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (const type of ['class_declaration', 'record_declaration']) {
|
||||
for (const declaration of root.descendantsOfType(type)) {
|
||||
const className = declaration.childForFieldName('name')?.text;
|
||||
if (!className) continue;
|
||||
for (const annotation of annotationsOn(declaration)) {
|
||||
if (!resolvesToAnnotation(annotation.name, CONFIGURATION_PROPERTIES_ANNOTATION, imports)) {
|
||||
continue;
|
||||
}
|
||||
const prefix = parseConfigurationPropertiesPrefix(annotation.node);
|
||||
if (prefix !== null) {
|
||||
facts.push({
|
||||
consumer: {
|
||||
kind: 'configuration-properties',
|
||||
className,
|
||||
line: declaration.startPosition.row + 1,
|
||||
prefix,
|
||||
},
|
||||
annotationName: annotation.name,
|
||||
classScopeId: classScopeId(filePath, declaration),
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return facts;
|
||||
}
|
||||
|
||||
/** Parse Java consumers for focused unit tests; production reuses the worker AST. */
|
||||
export function extractJavaSpringConfigConsumers(source: string): SpringConfigConsumer[] {
|
||||
const tree = parseSourceSafe(getJavaParser(), source);
|
||||
return captureJavaSpringConfigConsumerFacts(tree.rootNode, '<memory>').map(
|
||||
(fact) => fact.consumer,
|
||||
);
|
||||
}
|
||||
|
||||
/** Java ScopeResolver post-resolution hook for Spring configuration consumers. */
|
||||
export function attachJavaSpringConfigBindings(
|
||||
graph: KnowledgeGraph,
|
||||
parsedFiles: readonly ParsedFile[],
|
||||
_nodeLookup: GraphNodeLookup,
|
||||
indexes: ScopeResolutionIndexes,
|
||||
_ctx: { readonly fileContents: ReadonlyMap<string, string> },
|
||||
): void {
|
||||
const resolveAnnotation = createSpringAnnotationNameResolver(indexes);
|
||||
const recognizedAnnotations = new Set([VALUE_ANNOTATION, CONFIGURATION_PROPERTIES_ANNOTATION]);
|
||||
const batches: Array<{ filePath: string; consumers: SpringConfigConsumer[] }> = [];
|
||||
for (const parsed of parsedFiles) {
|
||||
const consumers: SpringConfigConsumer[] = [];
|
||||
for (const fact of getJavaSpringConfigConsumerFacts(parsed.filePath)) {
|
||||
const classScope = indexes.scopeTree.getScope(fact.classScopeId);
|
||||
if (classScope === undefined || classScope.kind !== 'Class') continue;
|
||||
const expectedAnnotation =
|
||||
fact.consumer.kind === 'value' ? VALUE_ANNOTATION : CONFIGURATION_PROPERTIES_ANNOTATION;
|
||||
const enclosingScope = fact.consumer.kind === 'value' ? classScope.id : classScope.parent;
|
||||
const resolved = resolveAnnotation(
|
||||
fact.annotationName,
|
||||
parsed,
|
||||
enclosingScope,
|
||||
recognizedAnnotations,
|
||||
isJavaPackageSiblingVisibilityIncomplete(parsed.filePath),
|
||||
);
|
||||
if (resolved === expectedAnnotation) consumers.push(fact.consumer);
|
||||
}
|
||||
if (consumers.length > 0) batches.push({ filePath: parsed.filePath, consumers });
|
||||
}
|
||||
bindSpringConfigConsumers(graph, batches);
|
||||
}
|
||||
|
|
@ -2,8 +2,23 @@ import type { CaptureMatch, ParsedImport, ParsedTypeBinding, TypeRef } from 'git
|
|||
|
||||
const REF_PREFIX_RE = /^&\s*(mut\s+)?/;
|
||||
const PTR_PREFIX_RE = /^\*\s*(const|mut)?\s*/;
|
||||
const DYN_PREFIX_RE = /^dyn\s+/;
|
||||
const ENUM_VARIANT_NAMES = new Set(['Some', 'None', 'Ok', 'Err']);
|
||||
|
||||
// `dyn Trait`, `&dyn Trait`, `Box<dyn Trait>` all name a trait object whose
|
||||
// receiver-dispatch target is the trait itself (#2604) — strip the `dyn`
|
||||
// keyword and any auto-trait/lifetime bound list (`dyn Trait + Send`) down to
|
||||
// the principal trait name. Reference/pointer sigils are stripped by the
|
||||
// caller first; wrapper unwrapping (Box<T> etc.) runs before this so the
|
||||
// unwrapped inner text still gets the same treatment.
|
||||
function stripDynBound(t: string): string {
|
||||
if (!DYN_PREFIX_RE.test(t)) return t;
|
||||
t = t.replace(DYN_PREFIX_RE, '');
|
||||
const plus = t.indexOf('+');
|
||||
if (plus !== -1) t = t.slice(0, plus);
|
||||
return t.trim();
|
||||
}
|
||||
|
||||
// ─── interpretImport ──────────────────────────────────────────────────────
|
||||
|
||||
export function interpretRustImport(captures: CaptureMatch): ParsedImport | null {
|
||||
|
|
@ -98,6 +113,7 @@ export function normalizeRustTypeName(text: string): string {
|
|||
const inner = extractFirstGenericArg(t);
|
||||
if (inner !== null) t = inner;
|
||||
}
|
||||
t = stripDynBound(t);
|
||||
const bracket = t.indexOf('<');
|
||||
if (bracket !== -1) t = t.slice(0, bracket);
|
||||
// Take last segment of qualified paths (crate::foo::Bar → Bar)
|
||||
|
|
@ -158,6 +174,7 @@ function normalizeRustReturnType(text: string): string {
|
|||
}
|
||||
}
|
||||
}
|
||||
t = stripDynBound(t);
|
||||
const bracket = t.indexOf('<');
|
||||
if (bracket !== -1) t = t.slice(0, bracket);
|
||||
const lastColon = t.lastIndexOf('::');
|
||||
|
|
|
|||
|
|
@ -10,6 +10,7 @@ const RUST_SCOPE_QUERY = `
|
|||
(enum_item) @scope.class
|
||||
(union_item) @scope.class
|
||||
(function_item) @scope.function
|
||||
(function_signature_item) @scope.function
|
||||
(closure_expression) @scope.function
|
||||
(block) @scope.block
|
||||
(if_expression) @scope.block
|
||||
|
|
@ -55,6 +56,14 @@ const RUST_SCOPE_QUERY = `
|
|||
(function_item
|
||||
name: (identifier) @declaration.name) @declaration.function
|
||||
|
||||
;; Declarations — trait method signature (required method, no body,
|
||||
;; e.g. fn foo(self) -> T; inside a trait body). Without this, an abstract
|
||||
;; trait method is invisible to scope resolution — never owned by its
|
||||
;; trait's Class scope, so a dyn Trait receiver can never dispatch to
|
||||
;; it (#2604).
|
||||
(function_signature_item
|
||||
name: (identifier) @declaration.name) @declaration.function
|
||||
|
||||
;; Declarations — struct fields
|
||||
(field_declaration
|
||||
name: (field_identifier) @declaration.name
|
||||
|
|
|
|||
|
|
@ -20,6 +20,7 @@ export {
|
|||
scopeResolutionPhase,
|
||||
type ScopeResolutionOutput,
|
||||
} from '../scope-resolution/pipeline/phase.js';
|
||||
export { springConfigPhase, type SpringConfigOutput } from './spring-config.js';
|
||||
export { pruneLocalSymbolsPhase, type PruneLocalSymbolsOutput } from './prune-local-symbols.js';
|
||||
export { taintSummariesPhase, type TaintSummariesOutput } from './taint-summaries.js';
|
||||
export { callSummariesPhase, type CallSummariesOutput } from './call-summaries.js';
|
||||
|
|
|
|||
|
|
@ -25,6 +25,17 @@ export interface ProcessesOutput {
|
|||
processResult: ProcessDetectionResult;
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute the dynamic max-processes budget from the symbol count.
|
||||
*
|
||||
* Scales proportionally (symbolCount / 10) with a floor of 20.
|
||||
* Prior to #2198 this was capped at 300 via `Math.min(300, …)`,
|
||||
* silently truncating process detection on large repositories.
|
||||
*/
|
||||
export function computeDynamicMaxProcesses(symbolCount: number): number {
|
||||
return Math.max(20, Math.round(symbolCount / 10));
|
||||
}
|
||||
|
||||
export const processesPhase: PipelinePhase<ProcessesOutput> = {
|
||||
name: 'processes',
|
||||
// `structure` supplies `totalFiles` (progress counter) without the spurious
|
||||
|
|
@ -53,7 +64,7 @@ export const processesPhase: PipelinePhase<ProcessesOutput> = {
|
|||
ctx.graph.forEachNode((n) => {
|
||||
if (n.label !== 'File') symbolCount++;
|
||||
});
|
||||
const dynamicMaxProcesses = Math.max(20, Math.min(300, Math.round(symbolCount / 10)));
|
||||
const dynamicMaxProcesses = computeDynamicMaxProcesses(symbolCount);
|
||||
|
||||
const processResult = await processProcesses(
|
||||
ctx.graph,
|
||||
|
|
|
|||
426
gitnexus/src/core/ingestion/pipeline-phases/spring-config.ts
Normal file
426
gitnexus/src/core/ingestion/pipeline-phases/spring-config.ts
Normal file
|
|
@ -0,0 +1,426 @@
|
|||
/**
|
||||
* Phase: springConfig
|
||||
*
|
||||
* Adds key-only nodes for statically readable Spring
|
||||
* `application*.properties` / `application*.yml` / `application*.yaml` files.
|
||||
* Language-specific ScopeResolver hooks attach consumers later. Configuration
|
||||
* values are deliberately never copied into the graph because they may contain
|
||||
* credentials and key identity is sufficient for impact analysis.
|
||||
*
|
||||
* @deps structure
|
||||
* @reads Spring application configuration files
|
||||
* @writes Property nodes and DEFINES edges
|
||||
*/
|
||||
|
||||
import fs from 'node:fs/promises';
|
||||
import path from 'node:path';
|
||||
import { createRequire } from 'node:module';
|
||||
import type { EventType as YamlEventType, State as YamlState } from 'js-yaml';
|
||||
import { SPRING_CONFIG_DESCRIPTION } from '../frameworks/spring/config-bindings.js';
|
||||
import { generateId } from '../../../lib/utils.js';
|
||||
import type { PipelineContext, PipelinePhase, PhaseResult } from './types.js';
|
||||
import { getPhaseOutput } from './types.js';
|
||||
import type { StructureOutput } from './structure.js';
|
||||
|
||||
const require = createRequire(import.meta.url);
|
||||
const yaml = require('js-yaml') as typeof import('js-yaml');
|
||||
const MAX_CONFIG_FILE_BYTES = 2 * 1024 * 1024;
|
||||
const MAX_YAML_TRAVERSAL_DEPTH = 128;
|
||||
const MAX_YAML_TRAVERSAL_NODES = 100_000;
|
||||
|
||||
export interface SpringConfigKey {
|
||||
readonly key: string;
|
||||
readonly filePath: string;
|
||||
readonly line: number;
|
||||
readonly profile?: string;
|
||||
readonly format: 'properties' | 'yaml';
|
||||
}
|
||||
|
||||
interface SpringConfigFile {
|
||||
readonly filePath: string;
|
||||
readonly profile?: string;
|
||||
readonly format: SpringConfigKey['format'];
|
||||
}
|
||||
|
||||
export interface SpringConfigOutput {
|
||||
readonly configKeys: number;
|
||||
}
|
||||
|
||||
/** Match only Spring Boot's conventional application config file names. */
|
||||
export function classifySpringConfigFile(filePath: string): SpringConfigFile | null {
|
||||
const base = path.posix.basename(filePath.replaceAll('\\', '/'));
|
||||
const match = /^application(?:-([^.]+))?\.(properties|ya?ml)$/i.exec(base);
|
||||
if (match === null) return null;
|
||||
return {
|
||||
filePath,
|
||||
...(match[1] ? { profile: match[1] } : {}),
|
||||
format: match[2].toLowerCase() === 'properties' ? 'properties' : 'yaml',
|
||||
};
|
||||
}
|
||||
|
||||
function unescapePropertyKey(raw: string): string {
|
||||
return raw
|
||||
.replace(/\\u([0-9a-fA-F]{4})/g, (_match, hex: string) =>
|
||||
String.fromCharCode(Number.parseInt(hex, 16)),
|
||||
)
|
||||
.replace(/\\([:=#!\\ ])/g, '$1');
|
||||
}
|
||||
|
||||
function logicalPropertiesLines(content: string): Array<{ text: string; line: number }> {
|
||||
const physical = content.split(/\r?\n/);
|
||||
const logical: Array<{ text: string; line: number }> = [];
|
||||
let current = '';
|
||||
let startLine = 1;
|
||||
|
||||
for (let index = 0; index < physical.length; index++) {
|
||||
const line = physical[index];
|
||||
if (current.length === 0) startLine = index + 1;
|
||||
current += current.length === 0 ? line : line.trimStart();
|
||||
|
||||
let trailingBackslashes = 0;
|
||||
for (let cursor = current.length - 1; cursor >= 0 && current[cursor] === '\\'; cursor--) {
|
||||
trailingBackslashes++;
|
||||
}
|
||||
if (trailingBackslashes % 2 === 1) {
|
||||
current = current.slice(0, -1);
|
||||
continue;
|
||||
}
|
||||
logical.push({ text: current, line: startLine });
|
||||
current = '';
|
||||
}
|
||||
if (current.length > 0) logical.push({ text: current, line: startLine });
|
||||
return logical;
|
||||
}
|
||||
|
||||
/** Parse `.properties` keys without retaining their values. */
|
||||
export function parseSpringProperties(
|
||||
content: string,
|
||||
filePath: string,
|
||||
profile?: string,
|
||||
): SpringConfigKey[] {
|
||||
const keys: SpringConfigKey[] = [];
|
||||
const seen = new Set<string>();
|
||||
|
||||
for (const logical of logicalPropertiesLines(content)) {
|
||||
const trimmed = logical.text.trimStart();
|
||||
if (trimmed.length === 0 || trimmed.startsWith('#') || trimmed.startsWith('!')) continue;
|
||||
|
||||
let separator = -1;
|
||||
let escaped = false;
|
||||
for (let index = 0; index < trimmed.length; index++) {
|
||||
const char = trimmed[index];
|
||||
if (!escaped && (char === '=' || char === ':' || /\s/.test(char))) {
|
||||
separator = index;
|
||||
break;
|
||||
}
|
||||
escaped = !escaped && char === '\\';
|
||||
if (char !== '\\') escaped = false;
|
||||
}
|
||||
const rawKey = (separator === -1 ? trimmed : trimmed.slice(0, separator)).trim();
|
||||
const key = unescapePropertyKey(rawKey);
|
||||
if (key.length === 0 || seen.has(key)) continue;
|
||||
seen.add(key);
|
||||
keys.push({
|
||||
key,
|
||||
filePath,
|
||||
line: logical.line,
|
||||
...(profile ? { profile } : {}),
|
||||
format: 'properties',
|
||||
});
|
||||
}
|
||||
|
||||
return keys;
|
||||
}
|
||||
|
||||
interface YamlParseEvent {
|
||||
readonly startLine: number;
|
||||
kind: string | null;
|
||||
result: unknown;
|
||||
tag: string | null;
|
||||
readonly children: YamlParseEvent[];
|
||||
}
|
||||
|
||||
interface YamlMappingLocation {
|
||||
readonly valueEvent: YamlParseEvent;
|
||||
readonly line: number;
|
||||
}
|
||||
|
||||
interface YamlTraversalState {
|
||||
remainingNodes: number;
|
||||
readonly activeObjects: Set<object>;
|
||||
}
|
||||
|
||||
function consumeYamlTraversalBudget(state: YamlTraversalState, depth: number): void {
|
||||
if (depth > MAX_YAML_TRAVERSAL_DEPTH) {
|
||||
throw new Error(`Spring YAML traversal depth exceeds ${MAX_YAML_TRAVERSAL_DEPTH}`);
|
||||
}
|
||||
state.remainingNodes--;
|
||||
if (state.remainingNodes < 0) {
|
||||
throw new Error(`Spring YAML traversal exceeds ${MAX_YAML_TRAVERSAL_NODES} nodes`);
|
||||
}
|
||||
}
|
||||
|
||||
function isObjectValue(value: unknown): value is object {
|
||||
return value !== null && typeof value === 'object';
|
||||
}
|
||||
|
||||
function resolveYamlAliasEvent(
|
||||
event: YamlParseEvent | undefined,
|
||||
objectEvents: WeakMap<object, YamlParseEvent>,
|
||||
): YamlParseEvent | undefined {
|
||||
if (event?.kind !== null || !isObjectValue(event.result)) return event;
|
||||
return objectEvents.get(event.result) ?? event;
|
||||
}
|
||||
|
||||
function yamlMappingPairs(event: YamlParseEvent): Array<{
|
||||
key: string;
|
||||
keyEvent: YamlParseEvent;
|
||||
valueEvent: YamlParseEvent;
|
||||
}> {
|
||||
const pairs: Array<{ key: string; keyEvent: YamlParseEvent; valueEvent: YamlParseEvent }> = [];
|
||||
for (let index = 0; index + 1 < event.children.length; index += 2) {
|
||||
const keyEvent = event.children[index];
|
||||
const valueEvent = event.children[index + 1];
|
||||
if (keyEvent.kind !== 'scalar') continue;
|
||||
pairs.push({ key: String(keyEvent.result), keyEvent, valueEvent });
|
||||
}
|
||||
return pairs;
|
||||
}
|
||||
|
||||
function findYamlMappingLocation(
|
||||
event: YamlParseEvent | undefined,
|
||||
key: string,
|
||||
objectEvents: WeakMap<object, YamlParseEvent>,
|
||||
traversal: YamlTraversalState,
|
||||
visited = new Set<YamlParseEvent>(),
|
||||
depth = 0,
|
||||
): YamlMappingLocation | undefined {
|
||||
consumeYamlTraversalBudget(traversal, depth);
|
||||
const resolved = resolveYamlAliasEvent(event, objectEvents);
|
||||
if (resolved === undefined || visited.has(resolved)) return undefined;
|
||||
visited.add(resolved);
|
||||
|
||||
if (resolved.kind === 'sequence') {
|
||||
for (const child of resolved.children) {
|
||||
const found = findYamlMappingLocation(
|
||||
child,
|
||||
key,
|
||||
objectEvents,
|
||||
traversal,
|
||||
visited,
|
||||
depth + 1,
|
||||
);
|
||||
if (found !== undefined) return found;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
if (resolved.kind !== 'mapping') return undefined;
|
||||
|
||||
const pairs = yamlMappingPairs(resolved);
|
||||
const direct = pairs.find((pair) => pair.key === key);
|
||||
if (direct !== undefined) {
|
||||
return { valueEvent: direct.valueEvent, line: direct.keyEvent.startLine };
|
||||
}
|
||||
for (const merge of pairs.filter((pair) => pair.key === '<<')) {
|
||||
const found = findYamlMappingLocation(
|
||||
merge.valueEvent,
|
||||
key,
|
||||
objectEvents,
|
||||
traversal,
|
||||
visited,
|
||||
depth + 1,
|
||||
);
|
||||
if (found !== undefined) return found;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function flattenYamlValue(
|
||||
value: unknown,
|
||||
event: YamlParseEvent | undefined,
|
||||
prefix: string,
|
||||
out: Map<string, number>,
|
||||
objectEvents: WeakMap<object, YamlParseEvent>,
|
||||
traversal: YamlTraversalState,
|
||||
sourceLine = event?.startLine ?? 1,
|
||||
depth = 0,
|
||||
): void {
|
||||
consumeYamlTraversalBudget(traversal, depth);
|
||||
const resolvedEvent = resolveYamlAliasEvent(event, objectEvents);
|
||||
const trackedObject = isObjectValue(value) ? value : undefined;
|
||||
if (trackedObject !== undefined && traversal.activeObjects.has(trackedObject)) return;
|
||||
if (trackedObject !== undefined) traversal.activeObjects.add(trackedObject);
|
||||
try {
|
||||
if (Array.isArray(value)) {
|
||||
if (value.length === 0 && prefix.length > 0 && !out.has(prefix)) out.set(prefix, sourceLine);
|
||||
value.forEach((item, index) =>
|
||||
flattenYamlValue(
|
||||
item,
|
||||
resolvedEvent?.children[index],
|
||||
`${prefix}[${index}]`,
|
||||
out,
|
||||
objectEvents,
|
||||
traversal,
|
||||
sourceLine,
|
||||
depth + 1,
|
||||
),
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (
|
||||
value !== null &&
|
||||
typeof value === 'object' &&
|
||||
(resolvedEvent?.kind === 'mapping' || resolvedEvent === undefined)
|
||||
) {
|
||||
const entries = Object.entries(value as Record<string, unknown>);
|
||||
if (entries.length === 0 && prefix.length > 0 && !out.has(prefix))
|
||||
out.set(prefix, sourceLine);
|
||||
for (const [key, nested] of entries) {
|
||||
const next = prefix.length === 0 ? key : `${prefix}.${key}`;
|
||||
const location = findYamlMappingLocation(resolvedEvent, key, objectEvents, traversal);
|
||||
flattenYamlValue(
|
||||
nested,
|
||||
location?.valueEvent,
|
||||
next,
|
||||
out,
|
||||
objectEvents,
|
||||
traversal,
|
||||
location?.line ?? sourceLine,
|
||||
depth + 1,
|
||||
);
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (prefix.length > 0 && !out.has(prefix)) out.set(prefix, sourceLine);
|
||||
} finally {
|
||||
if (trackedObject !== undefined) traversal.activeObjects.delete(trackedObject);
|
||||
}
|
||||
}
|
||||
|
||||
/** Parse and flatten YAML leaves without retaining their values. */
|
||||
export function parseSpringYaml(
|
||||
content: string,
|
||||
filePath: string,
|
||||
profile?: string,
|
||||
): SpringConfigKey[] {
|
||||
const flattened = new Map<string, number>();
|
||||
const eventStack: YamlParseEvent[] = [];
|
||||
const documentEvents: YamlParseEvent[] = [];
|
||||
const objectEvents = new WeakMap<object, YamlParseEvent>();
|
||||
const documents: unknown[] = [];
|
||||
const traversal: YamlTraversalState = {
|
||||
remainingNodes: MAX_YAML_TRAVERSAL_NODES,
|
||||
activeObjects: new Set<object>(),
|
||||
};
|
||||
|
||||
yaml.loadAll(content, (document) => documents.push(document), {
|
||||
schema: yaml.DEFAULT_SCHEMA,
|
||||
json: true,
|
||||
listener: (eventType: YamlEventType, state: YamlState) => {
|
||||
if (eventType === 'open') {
|
||||
eventStack.push({
|
||||
startLine: state.line + 1,
|
||||
kind: null,
|
||||
result: undefined,
|
||||
tag: null,
|
||||
children: [],
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
const event = eventStack.pop();
|
||||
if (event === undefined) return;
|
||||
event.kind = state.kind ?? null;
|
||||
event.result = state.result;
|
||||
event.tag = (state as YamlState & { tag?: string | null }).tag ?? null;
|
||||
if (isObjectValue(event.result) && event.kind !== null) {
|
||||
objectEvents.set(event.result, event);
|
||||
}
|
||||
const parent = eventStack[eventStack.length - 1];
|
||||
if (parent === undefined) documentEvents.push(event);
|
||||
else parent.children.push(event);
|
||||
},
|
||||
});
|
||||
|
||||
documents.forEach((document, index) =>
|
||||
flattenYamlValue(document, documentEvents[index], '', flattened, objectEvents, traversal),
|
||||
);
|
||||
return [...flattened.entries()]
|
||||
.sort(([left], [right]) => left.localeCompare(right))
|
||||
.map(([key, line]) => ({
|
||||
key,
|
||||
filePath,
|
||||
line,
|
||||
...(profile ? { profile } : {}),
|
||||
format: 'yaml' as const,
|
||||
}));
|
||||
}
|
||||
|
||||
function configKeyNodeId(entry: SpringConfigKey): string {
|
||||
return generateId('Property', `spring-config:${entry.filePath}:${entry.key}`);
|
||||
}
|
||||
|
||||
async function readConfigKeys(
|
||||
repoPath: string,
|
||||
scannedFiles: StructureOutput['scannedFiles'],
|
||||
): Promise<SpringConfigKey[]> {
|
||||
const keys: SpringConfigKey[] = [];
|
||||
for (const scanned of scannedFiles) {
|
||||
const classified = classifySpringConfigFile(scanned.path);
|
||||
if (classified === null || scanned.size > MAX_CONFIG_FILE_BYTES) continue;
|
||||
try {
|
||||
const content = await fs.readFile(path.join(repoPath, scanned.path), 'utf8');
|
||||
keys.push(
|
||||
...(classified.format === 'properties'
|
||||
? parseSpringProperties(content, classified.filePath, classified.profile)
|
||||
: parseSpringYaml(content, classified.filePath, classified.profile)),
|
||||
);
|
||||
} catch {
|
||||
// Malformed configuration is not a reason to fail the entire code index.
|
||||
// Fail closed: no keys and therefore no misleading bindings for this file.
|
||||
}
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
|
||||
export const springConfigPhase: PipelinePhase<SpringConfigOutput> = {
|
||||
name: 'springConfig',
|
||||
deps: ['structure'],
|
||||
|
||||
async execute(
|
||||
ctx: PipelineContext,
|
||||
deps: ReadonlyMap<string, PhaseResult<unknown>>,
|
||||
): Promise<SpringConfigOutput> {
|
||||
const { scannedFiles } = getPhaseOutput<StructureOutput>(deps, 'structure');
|
||||
const configKeys = await readConfigKeys(ctx.repoPath, scannedFiles);
|
||||
for (const entry of configKeys) {
|
||||
const nodeId = configKeyNodeId(entry);
|
||||
ctx.graph.addNode({
|
||||
id: nodeId,
|
||||
label: 'Property',
|
||||
properties: {
|
||||
name: entry.key,
|
||||
filePath: entry.filePath,
|
||||
startLine: entry.line,
|
||||
endLine: entry.line,
|
||||
description: entry.profile
|
||||
? `${SPRING_CONFIG_DESCRIPTION} (profile: ${entry.profile})`
|
||||
: SPRING_CONFIG_DESCRIPTION,
|
||||
},
|
||||
});
|
||||
const fileId = generateId('File', entry.filePath);
|
||||
if (ctx.graph.getNode(fileId) !== undefined) {
|
||||
ctx.graph.addRelationship({
|
||||
id: generateId('DEFINES', `${fileId}->${nodeId}`),
|
||||
sourceId: fileId,
|
||||
targetId: nodeId,
|
||||
type: 'DEFINES',
|
||||
confidence: 1,
|
||||
reason: 'spring-config:key',
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return { configKeys: configKeys.length };
|
||||
},
|
||||
};
|
||||
|
|
@ -31,6 +31,7 @@ import {
|
|||
ormPhase,
|
||||
crossFilePhase,
|
||||
scopeResolutionPhase,
|
||||
springConfigPhase,
|
||||
pruneLocalSymbolsPhase,
|
||||
taintSummariesPhase,
|
||||
callSummariesPhase,
|
||||
|
|
@ -242,7 +243,7 @@ export interface PipelineOptions {
|
|||
*
|
||||
* Phase dependency graph:
|
||||
*
|
||||
* scan → structure → [markdown, cobol] → parse → [routes, tools, orm]
|
||||
* scan → structure → [springConfig, markdown, cobol] → parse → [routes, tools, orm]
|
||||
* → crossFile → scopeResolution → pruneLocalSymbols
|
||||
* → mro → di → communities → processes
|
||||
*
|
||||
|
|
@ -261,6 +262,7 @@ export function buildPhaseList(options?: PipelineOptions): PipelinePhase[] {
|
|||
new PhaseRegistry<PipelineOptions>()
|
||||
.register(scanPhase)
|
||||
.register(structurePhase)
|
||||
.register(springConfigPhase)
|
||||
.register(markdownPhase)
|
||||
.register(cobolPhase)
|
||||
.register(parsePhase)
|
||||
|
|
|
|||
|
|
@ -743,10 +743,11 @@ export const PYTHON_QUERIES = `
|
|||
|
||||
// Java queries - works with tree-sitter-java
|
||||
export const JAVA_QUERIES = `
|
||||
; Classes, Interfaces, Enums, Annotations
|
||||
; Classes, Interfaces, Enums, Records, Annotations
|
||||
(class_declaration name: (identifier) @name) @definition.class
|
||||
(interface_declaration name: (identifier) @name) @definition.interface
|
||||
(enum_declaration name: (identifier) @name) @definition.enum
|
||||
(record_declaration name: (identifier) @name) @definition.record
|
||||
(annotation_type_declaration name: (identifier) @name) @definition.annotation
|
||||
|
||||
; Anonymous class bodies: new Runnable() { ... } — no @name capture; the
|
||||
|
|
|
|||
|
|
@ -205,6 +205,19 @@ export interface WorkerPoolOptions {
|
|||
* created. Default `Math.max(3, poolSize)`.
|
||||
*/
|
||||
consecutiveFailureThreshold?: number;
|
||||
/**
|
||||
* Startup budget in milliseconds for a replacement worker to emit the
|
||||
* `{type:'ready'}` handshake before the pool treats it as a startup
|
||||
* crash (see {@link waitForWorkerReady}). Default 5000; also overridable
|
||||
* via `GITNEXUS_WORKER_READY_TIMEOUT_MS`, mirroring
|
||||
* `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS`. On a slow or heavily loaded
|
||||
* host, a full pool of workers cold-starting concurrently can
|
||||
* legitimately need more than 5s to load the native grammar bindings —
|
||||
* without the override every slot times out and the pool misclassifies
|
||||
* the slow start as a deterministic startup crash-loop, aborting the
|
||||
* whole analyze.
|
||||
*/
|
||||
workerReadyTimeoutMs?: number;
|
||||
/**
|
||||
* Test-only injection point for the Worker constructor. When provided,
|
||||
* the pool uses this factory instead of `new Worker(workerUrl)`. Production
|
||||
|
|
@ -406,17 +419,7 @@ const DEFAULT_TIMEOUT_BACKOFF_FACTOR = 2;
|
|||
const DEFAULT_MAX_RESPAWNS_PER_SLOT = 3;
|
||||
const DEFAULT_MAX_CUMULATIVE_TIMEOUT_FACTOR = 5;
|
||||
const DEFAULT_CONSECUTIVE_FAILURE_THRESHOLD_FLOOR = 3;
|
||||
/**
|
||||
* Bounded wait for a replacement worker to emit the `{type:'ready'}`
|
||||
* handshake from `parse-worker.ts`. Trusting Node's `online` event alone
|
||||
* lets a worker that crashes during top-of-script init slip past pool
|
||||
* startup — the pool only notices on the first dispatch's idle timeout
|
||||
* (default 30s). 5 seconds is a generous budget for parser + grammar
|
||||
* imports; if the worker hasn't reported ready by then, it's almost
|
||||
* certainly stuck or crashed and the pool should surface the failure
|
||||
* fast rather than wait out the dispatch idle timeout.
|
||||
*/
|
||||
const WORKER_READY_TIMEOUT_MS = 5_000;
|
||||
const DEFAULT_WORKER_READY_TIMEOUT_MS = 5_000;
|
||||
/**
|
||||
* Default upper bound on auto-resolved pool size. Past 16 workers the
|
||||
* dominant cost shifts from worker-side parsing to main-thread merge /
|
||||
|
|
@ -547,6 +550,7 @@ interface ResolvedWorkerPoolOptions {
|
|||
maxCumulativeTimeoutMs: number;
|
||||
consecutiveFailureThreshold: number;
|
||||
shutdownDrainMs: number;
|
||||
workerReadyTimeoutMs: number;
|
||||
}
|
||||
|
||||
export function resolveWorkerPoolOptions(
|
||||
|
|
@ -583,6 +587,10 @@ export function resolveWorkerPoolOptions(
|
|||
nonNegativeInteger(options.shutdownDrainMs) ??
|
||||
nonNegativeInteger(process.env.GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS) ??
|
||||
DEFAULT_SHUTDOWN_DRAIN_MS,
|
||||
workerReadyTimeoutMs:
|
||||
positiveInteger(options.workerReadyTimeoutMs) ??
|
||||
positiveInteger(process.env.GITNEXUS_WORKER_READY_TIMEOUT_MS) ??
|
||||
DEFAULT_WORKER_READY_TIMEOUT_MS,
|
||||
};
|
||||
}
|
||||
|
||||
|
|
@ -683,6 +691,27 @@ function captureWorkerStderr(worker: Worker): void {
|
|||
stream.on('error', () => undefined);
|
||||
}
|
||||
|
||||
/**
|
||||
* Forward a worker's piped stdout to the parent process's stdout, so worker
|
||||
* logs stay visible now that the production factory spawns with
|
||||
* `{ stdout: true }`. Workers with INHERITED stdout have been observed to
|
||||
* crash silently during top-of-script init (exit code 1, nothing on stderr,
|
||||
* roughly half of a concurrently spawned pool) on macOS 26.5 under both
|
||||
* Node 22 and 26; piping stdout eliminates the crash entirely. Piping also
|
||||
* matches the existing stderr handling, so worker output no longer races the
|
||||
* parent's raw fd. No-op when the worker has no `stdout` stream (test
|
||||
* factories).
|
||||
*/
|
||||
function forwardWorkerStdout(worker: Worker): void {
|
||||
const stream = worker.stdout;
|
||||
if (!stream) return;
|
||||
stream.on('data', (chunk: Buffer | string) => {
|
||||
process.stdout.write(chunk);
|
||||
});
|
||||
// A stdout stream error must never crash the pool.
|
||||
stream.on('error', () => undefined);
|
||||
}
|
||||
|
||||
/** Captured stderr tail for a worker, trimmed; '' when nothing was captured. */
|
||||
function workerStderrTail(worker: Worker): string {
|
||||
return workerStderrTails.get(worker)?.text.trim() ?? '';
|
||||
|
|
@ -722,13 +751,14 @@ function workerErrorReason(workerIndex: number, message: string, stack?: string)
|
|||
* (parser/grammar import failure, missing native binding) slip past
|
||||
* pool startup. The pool then only noticed the dead replacement on the
|
||||
* first dispatch's idle timeout (default 30s) — a long stall masking
|
||||
* an actual crash. This handshake bounds the wait at
|
||||
* {@link WORKER_READY_TIMEOUT_MS} and surfaces init failures as
|
||||
* `error` / `exit` / `messageerror` events directly. `messageerror` is
|
||||
* wired the same way: a V8 deserialization failure during startup is
|
||||
* treated as worker death and rejects the readiness promise.
|
||||
* an actual crash. This handshake bounds the wait at `readyTimeoutMs`
|
||||
* (see {@link WorkerPoolOptions.workerReadyTimeoutMs}) and surfaces init
|
||||
* failures as `error` / `exit` / `messageerror` events directly.
|
||||
* `messageerror` is wired the same way: a V8 deserialization failure
|
||||
* during startup is treated as worker death and rejects the readiness
|
||||
* promise.
|
||||
*/
|
||||
function waitForWorkerReady(worker: Worker): Promise<void> {
|
||||
function waitForWorkerReady(worker: Worker, readyTimeoutMs: number): Promise<void> {
|
||||
return new Promise<void>((resolve, reject) => {
|
||||
const cleanup = () => {
|
||||
clearTimeout(timer);
|
||||
|
|
@ -781,11 +811,11 @@ function waitForWorkerReady(worker: Worker): Promise<void> {
|
|||
new Error(
|
||||
withStderr(
|
||||
worker,
|
||||
`Replacement worker did not report ready within ${WORKER_READY_TIMEOUT_MS}ms — likely crashed during top-of-script init`,
|
||||
`Replacement worker did not report ready within ${readyTimeoutMs}ms — likely crashed during top-of-script init (slow host? raise GITNEXUS_WORKER_READY_TIMEOUT_MS)`,
|
||||
),
|
||||
),
|
||||
);
|
||||
}, WORKER_READY_TIMEOUT_MS);
|
||||
}, readyTimeoutMs);
|
||||
worker.on('message', onMessage);
|
||||
worker.once('error', onError);
|
||||
worker.once('exit', onExit);
|
||||
|
|
@ -931,6 +961,10 @@ export const createWorkerPool = (
|
|||
options?.workerFactory ??
|
||||
((url: URL) =>
|
||||
new Worker(url, {
|
||||
// Piped (not inherited) stdio: stderr for crash capture (#1741),
|
||||
// stdout because inherited stdout triggers silent startup crashes on
|
||||
// some hosts (see forwardWorkerStdout).
|
||||
stdout: true,
|
||||
stderr: true,
|
||||
workerData: workerStoreData,
|
||||
// The CFG visitors build per-function control-flow graphs by RECURSIVE
|
||||
|
|
@ -944,10 +978,11 @@ export const createWorkerPool = (
|
|||
// try/catch) and only that function's PDG is skipped, never a crash.
|
||||
resourceLimits: { stackSizeMb: 16 },
|
||||
}));
|
||||
/** Spawn + wire stderr capture in one step (used by all spawn sites). */
|
||||
/** Spawn + wire stdio capture/forwarding in one step (used by all spawn sites). */
|
||||
const spawnAndCapture = (url: URL): Worker => {
|
||||
const worker = spawnWorker(url);
|
||||
captureWorkerStderr(worker);
|
||||
forwardWorkerStdout(worker);
|
||||
return worker;
|
||||
};
|
||||
const workers: (Worker | undefined)[] = new Array(size);
|
||||
|
|
@ -1099,7 +1134,7 @@ export const createWorkerPool = (
|
|||
const worker = workers[i];
|
||||
if (!worker) return; // terminated mid-startup
|
||||
try {
|
||||
await waitForWorkerReady(worker);
|
||||
await waitForWorkerReady(worker, poolOptions.workerReadyTimeoutMs);
|
||||
anyWorkerReachedReady = true;
|
||||
return; // ready — slot stays in activeSlots
|
||||
} catch (err) {
|
||||
|
|
@ -1161,7 +1196,7 @@ export const createWorkerPool = (
|
|||
chunkHash?: string,
|
||||
): Promise<TResult[]> => {
|
||||
// Await the initial-spawn readiness gate (F13). On first dispatch
|
||||
// this blocks for up to WORKER_READY_TIMEOUT_MS while every initial
|
||||
// this blocks for up to poolOptions.workerReadyTimeoutMs while every initial
|
||||
// worker's `{type:'ready'}` handshake is checked; on subsequent
|
||||
// dispatches the promise is already settled and resolves
|
||||
// synchronously. Slots whose initial worker crashed in top-of-
|
||||
|
|
@ -1360,7 +1395,7 @@ export const createWorkerPool = (
|
|||
if (stopped) return false;
|
||||
const replacement = spawnAndCapture(workerUrl);
|
||||
try {
|
||||
await waitForWorkerReady(replacement);
|
||||
await waitForWorkerReady(replacement, poolOptions.workerReadyTimeoutMs);
|
||||
} catch (err) {
|
||||
await replacement.terminate().catch(() => undefined);
|
||||
logger.warn(
|
||||
|
|
|
|||
|
|
@ -2904,7 +2904,30 @@ export const queryFTS = async (
|
|||
};
|
||||
|
||||
/**
|
||||
* Drop an FTS index
|
||||
* True for the two benign "nothing to drop" `DROP_FTS_INDEX` failures —
|
||||
* both catalog/binder exceptions, LadybugDB's classes for "this name isn't
|
||||
* bound to anything right now" (probe-verified end-to-end through
|
||||
* `dropFTSIndex`'s real `conn.query()` path against @ladybugdb/core
|
||||
* 0.18.x): the named index was never created (`Binder exception: Table <T>
|
||||
* doesn't have an index with name <name>.`), or the FTS extension/function
|
||||
* isn't registered at all (`Catalog exception: function DROP_FTS_INDEX is
|
||||
* not defined...`). A real engine failure — e.g. the `Runtime exception:
|
||||
* FTS index '<name>' is inconsistent: ...` class from #2589 — is a
|
||||
* DIFFERENT exception class (an execution-time failure, not a catalog/bind
|
||||
* lookup miss), so this returns false for it. Anchored to the START of the
|
||||
* message (not a bare substring search): every probed LadybugDB error leads
|
||||
* with its exception class, and anchoring means a future message that merely
|
||||
* mentions "Binder exception" or "Catalog exception" further in in the body
|
||||
* of an otherwise-genuine failure can't be misclassified as benign. Pure
|
||||
* string logic so it is unit-testable without a native LadybugDB connection.
|
||||
*/
|
||||
export const isBenignDropFtsIndexError = (message: string): boolean =>
|
||||
message.startsWith('Binder exception:') || message.startsWith('Catalog exception:');
|
||||
|
||||
/**
|
||||
* Drop an FTS index. Tolerates only {@link isBenignDropFtsIndexError} —
|
||||
* anything else rethrows instead of being silently masked, which previously
|
||||
* let a corrupted index persist across analyze runs undetected.
|
||||
*/
|
||||
export const dropFTSIndex = async (tableName: string, indexName: string): Promise<void> => {
|
||||
if (!conn) {
|
||||
|
|
@ -2913,8 +2936,11 @@ export const dropFTSIndex = async (tableName: string, indexName: string): Promis
|
|||
|
||||
try {
|
||||
await queryAndDrain(conn, `CALL DROP_FTS_INDEX('${tableName}', '${indexName}')`);
|
||||
} catch {
|
||||
// Index may not exist
|
||||
} catch (e: unknown) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
if (!isBenignDropFtsIndexError(msg)) {
|
||||
throw e;
|
||||
}
|
||||
} finally {
|
||||
ensuredFTSIndexes.delete(ftsIndexKey(tableName, indexName));
|
||||
}
|
||||
|
|
|
|||
|
|
@ -321,6 +321,16 @@ const resolveCheckpointThreshold = (): number => {
|
|||
const DEFAULT_BUFFER_POOL_CAP = 2 * 1024 * 1024 * 1024;
|
||||
const BUFFER_POOL_FLOOR = 64 * 1024 * 1024;
|
||||
|
||||
// COPY-safety floor for the adaptive hint (below). LadybugDB's bulk COPY needs
|
||||
// working buffer-pool memory that scales with the repo: a 64 MiB pool fails
|
||||
// ("buffer pool is full and no memory could be freed") on any non-trivial repo,
|
||||
// and even the ~1800-file GitNexus checkout needs ≥256 MiB. So the adaptive
|
||||
// size never drops a repo below this — a distinct, higher floor than
|
||||
// BUFFER_POOL_FLOOR, which only guards defaultBufferPoolSize on tiny-RAM
|
||||
// machines. It is still clamped up to defaultBufferPoolSize, so a machine whose
|
||||
// default is below this floor keeps its default rather than over-committing.
|
||||
const ADAPTIVE_POOL_FLOOR = 256 * 1024 * 1024;
|
||||
|
||||
const parseBufferPoolSize = (raw: string | undefined): number | undefined => {
|
||||
if (raw === undefined) return undefined;
|
||||
const normalized = raw.trim();
|
||||
|
|
@ -333,16 +343,73 @@ const parseBufferPoolSize = (raw: string | undefined): number | undefined => {
|
|||
const defaultBufferPoolSize = (): number =>
|
||||
Math.min(DEFAULT_BUFFER_POOL_CAP, Math.max(BUFFER_POOL_FLOOR, Math.floor(os.totalmem() * 0.8)));
|
||||
|
||||
/**
|
||||
* Clamp an adaptive pool request to [ADAPTIVE_POOL_FLOOR, default]. The lower
|
||||
* bound keeps LadybugDB's COPY viable; the upper bound (defaultBufferPoolSize)
|
||||
* means the hint can only shrink the pool from today's default and can never
|
||||
* exceed the 2 GiB / 80%-RAM cap — and on a machine whose default is below the
|
||||
* COPY floor, the default wins, so the pool is never over-committed.
|
||||
*/
|
||||
const clampBufferPool = (bytes: number): number =>
|
||||
Math.min(defaultBufferPoolSize(), Math.max(ADAPTIVE_POOL_FLOOR, Math.floor(bytes)));
|
||||
|
||||
/**
|
||||
* Buffer-pool bytes to provision per graph element (node + relationship).
|
||||
*
|
||||
* The fixed 2 GiB default is far larger than most repos' working set, and
|
||||
* LadybugDB eagerly commits the pool at DB open — measured: a full
|
||||
* `analyze --force` of the GitNexus checkout takes ~51 s with the 2 GiB pool
|
||||
* vs ~35 s with the ~414 MiB this factor yields (31% faster; the oversized
|
||||
* pool's commit dominates). The pool is a page cache over the on-disk index,
|
||||
* which scales with node/edge count, so a per-element budget sizes it to the
|
||||
* repo. Kept generous so the whole index stays resident (no COPY thrash) and
|
||||
* always clamped to at least ADAPTIVE_POOL_FLOOR; tuned by timing a real
|
||||
* large-repo `analyze --force` at this factor vs a forced 2 GiB pool (the pool
|
||||
* is a native eager allocation, measured with a real analyze, not a build-free
|
||||
* bench — see the emit-path COPY timing note in bench/emit-persistence).
|
||||
*/
|
||||
const POOL_BYTES_PER_ELEMENT = 4 * 1024;
|
||||
|
||||
/**
|
||||
* Size the buffer pool to an estimated graph size (node + relationship count),
|
||||
* clamped to [ADAPTIVE_POOL_FLOOR, defaultBufferPoolSize()]. The estimate can
|
||||
* only *shrink* the pool from the default — never above the 2 GiB / 80%-RAM cap,
|
||||
* never below the COPY-safety floor — so no repo is under-sized or gets more
|
||||
* than the default it would have today.
|
||||
*/
|
||||
export const estimateBufferPool = (graphElementCount: number): number =>
|
||||
clampBufferPool(graphElementCount * POOL_BYTES_PER_ELEMENT);
|
||||
|
||||
/**
|
||||
* Optional per-run buffer-pool size hint (bytes). The analyze orchestrator sets
|
||||
* it once the graph size is known (after the pipeline, before the DB open) so
|
||||
* the pool is sized to the repo instead of the fixed 2 GiB default, and clears
|
||||
* it at run end. Non-analyze opens (MCP serve, `native-check` `:memory:`) never
|
||||
* set it and keep the default.
|
||||
*/
|
||||
let bufferPoolSizeHint: number | undefined;
|
||||
|
||||
/** Set (bytes) or clear (`undefined`) the per-run buffer-pool size hint. */
|
||||
export const setBufferPoolSizeHint = (bytes: number | undefined): void => {
|
||||
bufferPoolSizeHint = bytes;
|
||||
};
|
||||
|
||||
/**
|
||||
* Resolve the `bufferManagerSize` passed to every `new lbug.Database(...)`.
|
||||
* `GITNEXUS_LBUG_BUFFER_POOL_SIZE` (bytes) overrides the default; `0` is a
|
||||
* `GITNEXUS_LBUG_BUFFER_POOL_SIZE` (bytes) overrides everything; `0` is a
|
||||
* deliberate escape hatch that restores LadybugDB's native unbounded
|
||||
* 80%-of-RAM default. Resolved at call time (not module load) so tests can
|
||||
* stub the env var and `os.totalmem`.
|
||||
* 80%-of-RAM default. With no env override, a per-run `setBufferPoolSizeHint`
|
||||
* (clamped to [floor, default]) sizes the pool to the repo; otherwise the
|
||||
* default. Resolved at call time (not module load) so tests can stub the env
|
||||
* var, the hint, and `os.totalmem`.
|
||||
*/
|
||||
const resolveBufferManagerSize = (): number => {
|
||||
const raw = process.env.GITNEXUS_LBUG_BUFFER_POOL_SIZE;
|
||||
if (raw === undefined) return defaultBufferPoolSize();
|
||||
if (raw === undefined) {
|
||||
return bufferPoolSizeHint !== undefined
|
||||
? clampBufferPool(bufferPoolSizeHint)
|
||||
: defaultBufferPoolSize();
|
||||
}
|
||||
const parsed = parseBufferPoolSize(raw);
|
||||
if (parsed !== undefined) return parsed;
|
||||
// Non-empty but unparseable input: warn the operator and fall back —
|
||||
|
|
|
|||
|
|
@ -402,6 +402,7 @@ CREATE REL TABLE ${REL_TABLE_NAME} (
|
|||
FROM \`Static\` TO Community,
|
||||
FROM \`Variable\` TO Community,
|
||||
FROM \`Property\` TO Community,
|
||||
FROM \`Property\` TO \`Property\`,
|
||||
FROM \`Record\` TO Method,
|
||||
FROM \`Record\` TO \`Constructor\`,
|
||||
FROM \`Record\` TO \`Property\`,
|
||||
|
|
|
|||
|
|
@ -34,10 +34,12 @@ import {
|
|||
LbugWipeError,
|
||||
DELETE_FILES_CHUNK_SIZE,
|
||||
} from './lbug/lbug-adapter.js';
|
||||
import { estimateBufferPool, setBufferPoolSizeHint } from './lbug/lbug-config.js';
|
||||
import { escapeCypherString } from './lbug/cypher-escape.js';
|
||||
import {
|
||||
buildSearchIndexesOrDegrade,
|
||||
createSearchFTSIndexes,
|
||||
dropSearchFTSIndexes,
|
||||
initialiseSearchFTSStemmer,
|
||||
verifySearchFTSIndexes,
|
||||
} from './search/fts-indexes.js';
|
||||
|
|
@ -123,6 +125,7 @@ import { EMBEDDING_TABLE_NAME } from './lbug/schema.js';
|
|||
import { STALE_HASH_SENTINEL } from './lbug/schema.js';
|
||||
import { isSpringBeanCandidateSourceFile } from './ingestion/frameworks/spring/bean-catalog.js';
|
||||
import { SPRING_BEAN_INVENTORY_FEATURE } from './ingestion/frameworks/spring/analysis-features.js';
|
||||
import { SPRING_CONFIG_BINDINGS_FEATURE } from './ingestion/languages/java/analysis-features.js';
|
||||
import {
|
||||
CLASS_FRAMEWORK_ANNOTATIONS_FEATURE,
|
||||
findAnalysisFeatureMismatches,
|
||||
|
|
@ -137,6 +140,7 @@ import {
|
|||
const ANALYSIS_FEATURES = [
|
||||
CLASS_FRAMEWORK_ANNOTATIONS_FEATURE,
|
||||
SPRING_BEAN_INVENTORY_FEATURE,
|
||||
SPRING_CONFIG_BINDINGS_FEATURE,
|
||||
] as const;
|
||||
|
||||
interface PersistedFrameworkAnnotationRow {
|
||||
|
|
@ -645,6 +649,11 @@ export async function runFullAnalysis(
|
|||
// and are shared across branches (#2106 KTD7).
|
||||
const { storagePath } = getStoragePaths(repoPath);
|
||||
|
||||
// Start each analyze with a clean buffer-pool hint: any pre-pipeline DB open
|
||||
// (e.g. the embeddings-cache open) falls back to the default until the hint is
|
||||
// set from the built graph below, so a prior run's size can't leak in.
|
||||
setBufferPoolSizeHint(undefined);
|
||||
|
||||
// Clean up stale KuzuDB files from before the LadybugDB migration.
|
||||
const kuzuResult = await cleanupOldKuzuFiles(storagePath);
|
||||
if (kuzuResult.found && kuzuResult.needsReindex) {
|
||||
|
|
@ -1374,6 +1383,16 @@ export async function runFullAnalysis(
|
|||
await wipeLbugDbFiles(lbugPath);
|
||||
}
|
||||
|
||||
// Size the buffer pool to the graph just built by the pipeline (a page cache
|
||||
// over the on-disk index, which scales with node/edge count) instead of the
|
||||
// fixed 2 GiB default, whose eager commit dominates large-repo analyze. The
|
||||
// size is clamped to [COPY-safety floor, default], so it only ever shrinks
|
||||
// the pool; env override / no-hint paths are unchanged. See
|
||||
// resolveBufferManagerSize / estimateBufferPool.
|
||||
setBufferPoolSizeHint(
|
||||
estimateBufferPool(pipelineResult.graph.nodeCount + pipelineResult.graph.relationshipCount),
|
||||
);
|
||||
|
||||
await initLbug(lbugPath);
|
||||
|
||||
// Manual WAL checkpoint driver (#1741): periodically drain the WAL
|
||||
|
|
@ -1645,7 +1664,20 @@ export async function runFullAnalysis(
|
|||
progress('lbug', pct, msg);
|
||||
});
|
||||
} else {
|
||||
// 1a. Remove the write set's existing rows — batched (#2409): one
|
||||
// 1a. Drop every FTS index before touching a single row (#2589).
|
||||
// `deleteNodesForFiles` below DETACH DELETEs rows out of tables
|
||||
// that otherwise still carry the FTS index built at the end of
|
||||
// the PREVIOUS analyze run — Phase 3 doesn't drop+rebuild it
|
||||
// until well after this delete completes. LadybugDB's FTS
|
||||
// extension is not proven to survive DML against an indexed
|
||||
// table (its own docs never demonstrate it), and that ordering
|
||||
// is exactly what produced "FTS index 'file_fts' is
|
||||
// inconsistent: term is missing during delete". Dropping first
|
||||
// removes the hazard outright; Phase 3's createSearchFTSIndexes
|
||||
// rebuilds every index from the final row set regardless, so
|
||||
// this is a no-op on its own drop step there.
|
||||
await dropSearchFTSIndexes();
|
||||
// 1b. Remove the write set's existing rows — batched (#2409): one
|
||||
// DETACH DELETE per table per 200-file chunk. The former per-file
|
||||
// loop issued a count + delete per table per FILE — ~13k
|
||||
// single-row write transactions on a ~700-file write set — which
|
||||
|
|
|
|||
|
|
@ -121,6 +121,20 @@ export function getSearchFTSStemmer(): string {
|
|||
return resolvedStemmer ?? resolveFTSStemmer();
|
||||
}
|
||||
|
||||
/**
|
||||
* Drop every configured FTS index (no-op per index when absent or unloadable
|
||||
* — `dropFTSIndex` tolerates both). Callable ahead of any DML that mutates an
|
||||
* FTS-indexed table's rows: LadybugDB's FTS extension is not proven to
|
||||
* survive a DETACH DELETE against a table that still carries a live index
|
||||
* from a prior run (#2589) — dropping first removes that hazard entirely,
|
||||
* regardless of whether it also fixed a specific native inconsistency.
|
||||
*/
|
||||
export async function dropSearchFTSIndexes(): Promise<void> {
|
||||
for (const { table, indexName } of FTS_INDEXES) {
|
||||
await dropFTSIndex(table, indexName);
|
||||
}
|
||||
}
|
||||
|
||||
export async function createSearchFTSIndexes(
|
||||
options?: CreateSearchFTSIndexesOptions,
|
||||
): Promise<void> {
|
||||
|
|
|
|||
|
|
@ -4361,44 +4361,31 @@ export class LocalBackend {
|
|||
return { error: 'New name is the same as the current name.' };
|
||||
}
|
||||
|
||||
// Step 2: Collect edits from graph (high confidence)
|
||||
const changes = new Map<string, { file_path: string; edits: any[] }>();
|
||||
|
||||
const addEdit = (
|
||||
filePath: string,
|
||||
line: number,
|
||||
oldText: string,
|
||||
newText: string,
|
||||
confidence: string,
|
||||
) => {
|
||||
if (!changes.has(filePath)) {
|
||||
changes.set(filePath, { file_path: filePath, edits: [] });
|
||||
}
|
||||
changes.get(filePath)!.edits.push({ line, old_text: oldText, new_text: newText, confidence });
|
||||
// Steps 2+3: Determine the set of files the apply step will rewrite, then
|
||||
// enumerate every occurrence in each. The apply step (Step 4) does a
|
||||
// whole-file `\boldName\b` global replace on every file in `changes`, so the
|
||||
// reported edit list MUST enumerate every matching line in every such file —
|
||||
// otherwise the preview under-reports what lands, and the same partial list
|
||||
// comes back after apply (#2605). Building `changes` from one file set makes
|
||||
// the preview enumerate exactly the files the apply loop rewrites, using the
|
||||
// same word-boundary regex. (This is per-call consistency; the apply loop
|
||||
// still re-reads each file, so an external write landing between preview and
|
||||
// apply is a pre-existing gap this method does not lock against.)
|
||||
type RenameEdit = {
|
||||
line: number;
|
||||
old_text: string;
|
||||
new_text: string;
|
||||
confidence: 'graph' | 'text_search';
|
||||
};
|
||||
const escapedOldName = oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||||
|
||||
// The definition itself
|
||||
if (sym.filePath && sym.startLine) {
|
||||
try {
|
||||
const content = await fs.readFile(assertSafePath(sym.filePath), 'utf-8');
|
||||
const lines = content.split('\n');
|
||||
const lineIdx = sym.startLine - 1;
|
||||
if (lineIdx >= 0 && lineIdx < lines.length && lines[lineIdx].includes(oldName)) {
|
||||
const defRegex = new RegExp(
|
||||
`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`,
|
||||
'g',
|
||||
);
|
||||
addEdit(
|
||||
sym.filePath,
|
||||
sym.startLine,
|
||||
lines[lineIdx].trim(),
|
||||
lines[lineIdx].replace(defRegex, new_name).trim(),
|
||||
'graph',
|
||||
);
|
||||
}
|
||||
} catch (e) {
|
||||
logQueryError('rename:read-definition', e);
|
||||
}
|
||||
// Classify each file to rewrite by how it was discovered. Definition and
|
||||
// graph-ref files carry graph confidence; files found only by text search
|
||||
// carry text_search confidence. A graph-classified file is never downgraded.
|
||||
const fileConfidence = new Map<string, 'graph' | 'text_search'>();
|
||||
|
||||
if (sym.filePath) {
|
||||
fileConfidence.set(sym.filePath, 'graph');
|
||||
}
|
||||
|
||||
// All incoming refs from graph (callers, importers, etc.)
|
||||
|
|
@ -4408,44 +4395,13 @@ export class LocalBackend {
|
|||
...(lookupResult.incoming.extends || []),
|
||||
...(lookupResult.incoming.implements || []),
|
||||
];
|
||||
|
||||
let graphEdits = changes.size > 0 ? 1 : 0; // count definition edit
|
||||
|
||||
for (const ref of allIncoming) {
|
||||
if (!ref.filePath) continue;
|
||||
try {
|
||||
const content = await fs.readFile(assertSafePath(ref.filePath), 'utf-8');
|
||||
const lines = content.split('\n');
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
if (lines[i].includes(oldName)) {
|
||||
addEdit(
|
||||
ref.filePath,
|
||||
i + 1,
|
||||
lines[i].trim(),
|
||||
lines[i]
|
||||
.replace(
|
||||
new RegExp(`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'g'),
|
||||
new_name,
|
||||
)
|
||||
.trim(),
|
||||
'graph',
|
||||
);
|
||||
graphEdits++;
|
||||
break; // one edit per file from graph refs
|
||||
}
|
||||
}
|
||||
} catch (e) {
|
||||
logQueryError('rename:read-ref', e);
|
||||
if (ref.filePath) {
|
||||
fileConfidence.set(ref.filePath, 'graph');
|
||||
}
|
||||
}
|
||||
|
||||
// Step 3: Text search for refs the graph might have missed
|
||||
let astSearchEdits = 0;
|
||||
const graphFiles = new Set(
|
||||
[sym.filePath, ...allIncoming.map((r) => r.filePath)].filter(Boolean),
|
||||
);
|
||||
|
||||
// Simple text search across the repo for the old name (in files not already covered by graph)
|
||||
// Text search for files the graph might have missed entirely.
|
||||
try {
|
||||
const { execFileSync } = await import('child_process');
|
||||
const rgArgs = [
|
||||
|
|
@ -4472,67 +4428,98 @@ export class LocalBackend {
|
|||
|
||||
for (const file of files) {
|
||||
const normalizedFile = file.replace(/\\/g, '/').replace(/^\.\//, '');
|
||||
if (graphFiles.has(normalizedFile)) continue; // already covered by graph
|
||||
|
||||
try {
|
||||
const content = await fs.readFile(assertSafePath(normalizedFile), 'utf-8');
|
||||
const lines = content.split('\n');
|
||||
const regex = new RegExp(`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'g');
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
regex.lastIndex = 0;
|
||||
if (regex.test(lines[i])) {
|
||||
regex.lastIndex = 0;
|
||||
addEdit(
|
||||
normalizedFile,
|
||||
i + 1,
|
||||
lines[i].trim(),
|
||||
lines[i].replace(regex, new_name).trim(),
|
||||
'text_search',
|
||||
);
|
||||
astSearchEdits++;
|
||||
}
|
||||
}
|
||||
} catch (e) {
|
||||
logQueryError('rename:text-search-read', e);
|
||||
// Never downgrade a graph-classified file to text_search.
|
||||
if (!fileConfidence.has(normalizedFile)) {
|
||||
fileConfidence.set(normalizedFile, 'text_search');
|
||||
}
|
||||
}
|
||||
} catch (e) {
|
||||
logQueryError('rename:ripgrep', e);
|
||||
}
|
||||
|
||||
// Step 4: Apply or preview
|
||||
const allChanges = Array.from(changes.values());
|
||||
const totalEdits = allChanges.reduce((sum, c) => sum + c.edits.length, 0);
|
||||
// Enumerate every `\boldName\b` line in each file to rewrite, so the previewed
|
||||
// file set is exactly the set the apply loop below rewrites. A file with no
|
||||
// matching line is dropped (apply would write nothing to it). `wordTest`
|
||||
// (non-global) probes each line; `wordReplace` (global) rewrites it and is
|
||||
// reused by the apply loop — compiled once each rather than once per line,
|
||||
// and one escaping formula serves both passes.
|
||||
const wordTest = new RegExp(`\\b${escapedOldName}\\b`);
|
||||
const wordReplace = new RegExp(`\\b${escapedOldName}\\b`, 'g');
|
||||
const changes = new Map<string, { file_path: string; edits: RenameEdit[] }>();
|
||||
|
||||
for (const [filePath, confidence] of fileConfidence) {
|
||||
try {
|
||||
const content = await fs.readFile(assertSafePath(filePath), 'utf-8');
|
||||
const lines = content.split('\n');
|
||||
const edits: RenameEdit[] = [];
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
if (!wordTest.test(lines[i])) {
|
||||
continue;
|
||||
}
|
||||
edits.push({
|
||||
line: i + 1,
|
||||
old_text: lines[i].trim(),
|
||||
new_text: lines[i].replace(wordReplace, new_name).trim(),
|
||||
confidence,
|
||||
});
|
||||
}
|
||||
if (edits.length > 0) {
|
||||
changes.set(filePath, { file_path: filePath, edits });
|
||||
}
|
||||
} catch (e) {
|
||||
logQueryError('rename:enumerate', e);
|
||||
}
|
||||
}
|
||||
|
||||
// Step 4: Apply or preview.
|
||||
const failedFiles: string[] = [];
|
||||
if (!dry_run) {
|
||||
// Apply edits to files
|
||||
for (const change of allChanges) {
|
||||
for (const change of changes.values()) {
|
||||
try {
|
||||
const fullPath = assertSafePath(change.file_path);
|
||||
let content = await fs.readFile(fullPath, 'utf-8');
|
||||
const regex = new RegExp(`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'g');
|
||||
content = content.replace(regex, new_name);
|
||||
await fs.writeFile(fullPath, content, 'utf-8');
|
||||
const content = await fs.readFile(fullPath, 'utf-8');
|
||||
await fs.writeFile(fullPath, content.replace(wordReplace, new_name), 'utf-8');
|
||||
} catch (e) {
|
||||
// A swallowed write failure must not be reported as a full success
|
||||
// (#2283): record the file so the result can degrade to 'partial'
|
||||
// with the unwritten files listed, rather than masquerading as done.
|
||||
// A swallowed write failure must not be reported as success (#2283):
|
||||
// record the file so the result degrades to 'partial'.
|
||||
logQueryError('rename:apply-edit', e);
|
||||
failedFiles.push(change.file_path);
|
||||
}
|
||||
}
|
||||
// A file whose write threw did not land, so drop its edits from the
|
||||
// reported result — total_edits/changes must describe what actually
|
||||
// reached disk, not what was attempted (#2605: the report matches reality
|
||||
// even on partial failure). failed_files still names every dropped file.
|
||||
for (const f of failedFiles) {
|
||||
changes.delete(f);
|
||||
}
|
||||
}
|
||||
|
||||
// Counts derive from the reported set (dry-run: every enumerated file;
|
||||
// apply: only files that landed), so the graph/text_search split always
|
||||
// sums to total_edits and never overstates a partial apply.
|
||||
const reported = Array.from(changes.values());
|
||||
let graphEdits = 0;
|
||||
let astSearchEdits = 0;
|
||||
for (const change of reported) {
|
||||
for (const edit of change.edits) {
|
||||
if (edit.confidence === 'graph') {
|
||||
graphEdits++;
|
||||
} else {
|
||||
astSearchEdits++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
status: failedFiles.length > 0 ? 'partial' : 'success',
|
||||
old_name: oldName,
|
||||
new_name,
|
||||
files_affected: allChanges.length,
|
||||
total_edits: totalEdits,
|
||||
files_affected: reported.length,
|
||||
total_edits: graphEdits + astSearchEdits,
|
||||
graph_edits: graphEdits,
|
||||
text_search_edits: astSearchEdits,
|
||||
changes: allChanges,
|
||||
changes: reported,
|
||||
applied: !dry_run,
|
||||
...(failedFiles.length > 0 && { failed_files: failedFiles }),
|
||||
};
|
||||
|
|
|
|||
|
|
@ -429,8 +429,23 @@ export interface RepoMeta {
|
|||
* `E.hook` to `E$1.hook`, and nested-host anonymous names re-key
|
||||
* (`EnumWrap$1` → `EnumWrap$Mode$1`). Same contract as v8: identities move
|
||||
* on unchanged files; force a full re-analyze.
|
||||
* v10: Java `record_declaration` now emits a first-class `Record` graph node
|
||||
* (#2564): a record's container node was previously never created (JAVA_QUERIES
|
||||
* had no capture for it), so its methods existed as ownerless Method nodes
|
||||
* with no `HAS_METHOD` edge. The incremental write set only covers changed
|
||||
* files — a top-up against a pre-v10 index would keep silently omitting the
|
||||
* `Record` node and its `HAS_METHOD` edges for every unchanged record file
|
||||
* (same v7 contract: new nodes/edges the incremental path would otherwise
|
||||
* never backfill); force a full re-analyze instead.
|
||||
* v11: Rust abstract trait methods (`fn foo(&self) -> T;`, no body) now get a
|
||||
* scope + declaration capture (#2604): RUST_SCOPE_QUERY had no
|
||||
* `function_signature_item` pattern, so a `&dyn Trait` receiver could never
|
||||
* dispatch a CALLS edge to the trait's own method. Same v7/v10 contract: the
|
||||
* incremental write set only covers changed files, so a top-up against a
|
||||
* pre-v11 index would keep silently missing these CALLS edges for every
|
||||
* unchanged Rust trait file; force a full re-analyze instead.
|
||||
*/
|
||||
export const INCREMENTAL_SCHEMA_VERSION = 9;
|
||||
export const INCREMENTAL_SCHEMA_VERSION = 11;
|
||||
|
||||
export interface IndexedRepo {
|
||||
repoPath: string;
|
||||
|
|
|
|||
|
|
@ -21,4 +21,12 @@ class Unrelated {
|
|||
public void caller() {
|
||||
hook();
|
||||
}
|
||||
|
||||
public void dispatchToConstant() {
|
||||
EnumConst.A.hook();
|
||||
}
|
||||
|
||||
public void dispatchInherited() {
|
||||
EnumConst.A.log();
|
||||
}
|
||||
}
|
||||
|
|
|
|||
13
gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/Plain.java
vendored
Normal file
13
gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/Plain.java
vendored
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
public enum Plain {
|
||||
A;
|
||||
|
||||
public void m() {
|
||||
System.out.println("plain m");
|
||||
}
|
||||
}
|
||||
|
||||
class PlainCaller {
|
||||
public void callPlain() {
|
||||
Plain.A.m();
|
||||
}
|
||||
}
|
||||
12
gitnexus/test/fixtures/lang-resolution/java-new-expr-chain-call/LocalChain.java
vendored
Normal file
12
gitnexus/test/fixtures/lang-resolution/java-new-expr-chain-call/LocalChain.java
vendored
Normal file
|
|
@ -0,0 +1,12 @@
|
|||
package probe;
|
||||
|
||||
public class LocalChain {
|
||||
void m() {
|
||||
class Local {
|
||||
void inner() {
|
||||
System.out.println("right target");
|
||||
}
|
||||
}
|
||||
new Local().inner();
|
||||
}
|
||||
}
|
||||
7
gitnexus/test/fixtures/lang-resolution/java-new-expr-chain-call/Other.java
vendored
Normal file
7
gitnexus/test/fixtures/lang-resolution/java-new-expr-chain-call/Other.java
vendored
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
package probe;
|
||||
|
||||
class Other {
|
||||
void inner() {
|
||||
System.out.println("wrong target");
|
||||
}
|
||||
}
|
||||
11
gitnexus/test/fixtures/lang-resolution/java-record-methods/Point.java
vendored
Normal file
11
gitnexus/test/fixtures/lang-resolution/java-record-methods/Point.java
vendored
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
package probe;
|
||||
|
||||
public record Point(int x, int y) {
|
||||
public int sum() {
|
||||
return x + y;
|
||||
}
|
||||
|
||||
public int scaled(int factor) {
|
||||
return sum() * factor;
|
||||
}
|
||||
}
|
||||
15
gitnexus/test/fixtures/lang-resolution/rust-dyn-trait-object/src/lib.rs
vendored
Normal file
15
gitnexus/test/fixtures/lang-resolution/rust-dyn-trait-object/src/lib.rs
vendored
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
pub trait Behaviour {
|
||||
fn trait_target(&self) -> u32;
|
||||
}
|
||||
|
||||
pub struct Impl1;
|
||||
|
||||
impl Behaviour for Impl1 {
|
||||
fn trait_target(&self) -> u32 {
|
||||
7
|
||||
}
|
||||
}
|
||||
|
||||
pub fn calls_via_dyn(b: &dyn Behaviour) -> u32 {
|
||||
b.trait_target()
|
||||
}
|
||||
|
|
@ -1,7 +1,7 @@
|
|||
{
|
||||
"rust-abstract-dispatch/src/lib.rs": {
|
||||
"captureGroups": 30,
|
||||
"digest": "88309004d1ab00054f81bc55c1d058fc4ca25781162d6b94b7e7ce631a5d61b2"
|
||||
"captureGroups": 34,
|
||||
"digest": "973679363065ecd54c4e5128a9fab214ea27eca24f0f079c63a3c6f285e678b0"
|
||||
},
|
||||
"rust-abstract-dispatch/src/main.rs": {
|
||||
"captureGroups": 21,
|
||||
|
|
@ -148,8 +148,8 @@
|
|||
"digest": "e0120e3f215282e68d83b4f8f5d8918945e0b3e7ce4e0128c6afd2aa43caa1c0"
|
||||
},
|
||||
"rust-cross-module-collision/src/traits.rs": {
|
||||
"captureGroups": 3,
|
||||
"digest": "88eef9d92ea6e370bd8ef7fbf64c42ec622ec933fb53c56b32bc67db87fa8e03"
|
||||
"captureGroups": 5,
|
||||
"digest": "c7150a5052e0e2b5fd7fc21cc8ce361530fe5cc60f67ad0e2ef945bb97965b53"
|
||||
},
|
||||
"rust-deep-field-chain/models.rs": {
|
||||
"captureGroups": 24,
|
||||
|
|
@ -171,6 +171,10 @@
|
|||
"captureGroups": 22,
|
||||
"digest": "c53db401a81fde2ffd5665393acb9cd605a62ec51c015c3aafb3f41c0897471f"
|
||||
},
|
||||
"rust-dyn-trait-object/src/lib.rs": {
|
||||
"captureGroups": 23,
|
||||
"digest": "720618dff6a43ab8e5b59aa354c0c448b9057dd6f2f7b3b22b13b82d53745943"
|
||||
},
|
||||
"rust-err-unwrap/src/error.rs": {
|
||||
"captureGroups": 9,
|
||||
"digest": "798c8e01c6e54792ba69e845248efc8abf0cba38fa3d16fb8e0d1f6dd2ad2b7e"
|
||||
|
|
@ -316,8 +320,8 @@
|
|||
"digest": "cd836a2a9c15ab240961d2e15f192f7e33d65eb5ebf2e1a8af2f620a47fe66ae"
|
||||
},
|
||||
"rust-method-enrichment/src/lib.rs": {
|
||||
"captureGroups": 40,
|
||||
"digest": "71627a8218e32514b6945e4e310686eb631ce37451c90c9644e3f5336a37820b"
|
||||
"captureGroups": 42,
|
||||
"digest": "a4d9ca570fbb1ff1859a0b4f737aa3507b236f99700d37ded2c8c36518add567"
|
||||
},
|
||||
"rust-method-enrichment/src/main.rs": {
|
||||
"captureGroups": 18,
|
||||
|
|
@ -360,8 +364,8 @@
|
|||
"digest": "141388068614e16d96f27cfdf18ac9001b9e202ce832fe10f38dab990637b3ab"
|
||||
},
|
||||
"rust-parent-resolution/src/serializable.rs": {
|
||||
"captureGroups": 3,
|
||||
"digest": "f35d44f44d81e3a0be40f68ba9dbd4bde6f01659fa15b6db34a458ad460f904e"
|
||||
"captureGroups": 5,
|
||||
"digest": "f33bb881dd937cdd5af2eca6b0284ea79ca2d296c79217513f882dcba8a82fd8"
|
||||
},
|
||||
"rust-parent-resolution/src/user.rs": {
|
||||
"captureGroups": 13,
|
||||
|
|
@ -372,8 +376,8 @@
|
|||
"digest": "bc8946d31db81b85d780633608fdaa7565258cd788285fa00cd6dcb0de3dd16c"
|
||||
},
|
||||
"rust-qualified-trait/src/traits.rs": {
|
||||
"captureGroups": 5,
|
||||
"digest": "15be069f28f1400e4beb0b0860acb59979f78549960486f36a92f56578f05a06"
|
||||
"captureGroups": 9,
|
||||
"digest": "10f3bba4c2a16cdac77de0498ff506910ac09cc1a85e7daebfc5743c54e26015"
|
||||
},
|
||||
"rust-qualified-trait/src/widget.rs": {
|
||||
"captureGroups": 23,
|
||||
|
|
@ -492,12 +496,12 @@
|
|||
"digest": "f1b9f72d74467be55a8b7679215b49bcabb4d0fced6080f752672070b32ed93d"
|
||||
},
|
||||
"rust-traits/src/traits/clickable.rs": {
|
||||
"captureGroups": 3,
|
||||
"digest": "3ed5b27c172d48f83929715ba92d1030a282f9f1e29ec2fcdd3d7e9efbc54a84"
|
||||
"captureGroups": 7,
|
||||
"digest": "83c36832f24446fe03a07288dd394bc7494a54e5979c9b71c1dcb8b19ab54341"
|
||||
},
|
||||
"rust-traits/src/traits/drawable.rs": {
|
||||
"captureGroups": 5,
|
||||
"digest": "1dca39bbc7c1b1b66f1a34730b9a5b4dba04c54ee9d2688255e0fd4e6bc48499"
|
||||
"captureGroups": 9,
|
||||
"digest": "cee5091f041038722f1f012394a75ba4e16870d05b2dafa37371200e785198f0"
|
||||
},
|
||||
"rust-union/lib.rs": {
|
||||
"captureGroups": 10,
|
||||
|
|
|
|||
25
gitnexus/test/fixtures/spring-config-app/src/main/java/com/example/ConfigConsumers.java
vendored
Normal file
25
gitnexus/test/fixtures/spring-config-app/src/main/java/com/example/ConfigConsumers.java
vendored
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
package com.example;
|
||||
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.boot.context.properties.ConfigurationProperties;
|
||||
|
||||
class DirectValues {
|
||||
@Value("${payment.timeout:30}")
|
||||
private int timeout;
|
||||
|
||||
@Value("${payment.missing}")
|
||||
private String missing;
|
||||
}
|
||||
|
||||
@ConfigurationProperties(prefix = "service")
|
||||
class ServiceProperties {
|
||||
private String endpoint;
|
||||
private Retry retry;
|
||||
}
|
||||
|
||||
@ConfigurationProperties("service")
|
||||
class UnmatchedServiceProperties {
|
||||
private String unrelated;
|
||||
}
|
||||
|
||||
class Retry {}
|
||||
6
gitnexus/test/fixtures/spring-config-app/src/main/resources/application-dev.yml
vendored
Normal file
6
gitnexus/test/fixtures/spring-config-app/src/main/resources/application-dev.yml
vendored
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
defaults: &defaults
|
||||
retry:
|
||||
max-attempts: 3
|
||||
service:
|
||||
<<: *defaults
|
||||
endpoint: https://service.example.test
|
||||
2
gitnexus/test/fixtures/spring-config-app/src/main/resources/application.properties
vendored
Normal file
2
gitnexus/test/fixtures/spring-config-app/src/main/resources/application.properties
vendored
Normal file
|
|
@ -0,0 +1,2 @@
|
|||
payment.timeout=30
|
||||
service.endpoint=https://base.example.test
|
||||
8
gitnexus/test/fixtures/spring-config-shadow-app/src/main/java/com/example/Shadowed.java
vendored
Normal file
8
gitnexus/test/fixtures/spring-config-shadow-app/src/main/java/com/example/Shadowed.java
vendored
Normal file
|
|
@ -0,0 +1,8 @@
|
|||
package com.example;
|
||||
|
||||
import org.springframework.beans.factory.annotation.*;
|
||||
|
||||
class Shadowed {
|
||||
@Value("${fake.key}")
|
||||
private String fake;
|
||||
}
|
||||
5
gitnexus/test/fixtures/spring-config-shadow-app/src/main/java/com/example/Value.java
vendored
Normal file
5
gitnexus/test/fixtures/spring-config-shadow-app/src/main/java/com/example/Value.java
vendored
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
package com.example;
|
||||
|
||||
public @interface Value {
|
||||
String value();
|
||||
}
|
||||
1
gitnexus/test/fixtures/spring-config-shadow-app/src/main/resources/application.properties
vendored
Normal file
1
gitnexus/test/fixtures/spring-config-shadow-app/src/main/resources/application.properties
vendored
Normal file
|
|
@ -0,0 +1 @@
|
|||
fake.key=must-not-bind
|
||||
|
|
@ -1,3 +1,46 @@
|
|||
import { existsSync, readdirSync, statSync } from 'node:fs';
|
||||
import { join } from 'node:path';
|
||||
|
||||
/** A valid `libfts.lbug_extension` is ~2.2MB; anything smaller is truncated/corrupt. */
|
||||
const MIN_VALID_FTS_EXTENSION_BYTES = 1024 * 1024;
|
||||
|
||||
/**
|
||||
* Find the installed FTS extension file under a `.lbdb/extension` root,
|
||||
* discovering the version directory instead of assuming it equals the npm
|
||||
* `@ladybugdb/core` package version. LadybugDB's native INSTALL/LOAD resolves
|
||||
* its own extension-ABI version directory, which does not always track the
|
||||
* npm package version — e.g. #2587: bumping the package from 0.18.1 to 0.18.2
|
||||
* still installs into a `0.18.1` directory, because the underlying
|
||||
* extension-ABI build did not change with that patch release.
|
||||
*
|
||||
* Scans every version subdirectory for a `<platform>/fts/libfts.lbug_extension`
|
||||
* file and returns the most recently modified one (the one an install/load
|
||||
* actually just resolved), or null when nothing is installed.
|
||||
*/
|
||||
export const findInstalledFtsExtension = (extensionRoot: string): string | null => {
|
||||
// Fail closed on any FS error (permission quirks, AV file locks on Windows,
|
||||
// a directory vanishing mid-scan) — same contract as the callers this
|
||||
// replaces: "not found" is a valid outcome, a thrown exception is not.
|
||||
try {
|
||||
if (!existsSync(extensionRoot)) return null;
|
||||
let best: { path: string; mtimeMs: number } | null = null;
|
||||
for (const versionEntry of readdirSync(extensionRoot)) {
|
||||
const versionDir = join(extensionRoot, versionEntry);
|
||||
if (!statSync(versionDir).isDirectory()) continue;
|
||||
for (const platformEntry of readdirSync(versionDir)) {
|
||||
const candidate = join(versionDir, platformEntry, 'fts', 'libfts.lbug_extension');
|
||||
if (!existsSync(candidate)) continue;
|
||||
const stat = statSync(candidate);
|
||||
if (stat.size < MIN_VALID_FTS_EXTENSION_BYTES) continue;
|
||||
if (!best || stat.mtimeMs > best.mtimeMs) best = { path: candidate, mtimeMs: stat.mtimeMs };
|
||||
}
|
||||
}
|
||||
return best?.path ?? null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
};
|
||||
|
||||
export const FTS_UNAVAILABLE_NOTE =
|
||||
'FTS extension unavailable (load-only policy; LOAD failed on this machine)';
|
||||
|
||||
|
|
|
|||
|
|
@ -2,21 +2,21 @@ import {
|
|||
copyFileSync,
|
||||
existsSync,
|
||||
mkdtempSync,
|
||||
readdirSync,
|
||||
readFileSync,
|
||||
rmSync,
|
||||
statSync,
|
||||
writeFileSync,
|
||||
} from 'node:fs';
|
||||
import { homedir, tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import { afterAll, describe, expect, it } from 'vitest';
|
||||
import lbug from '@ladybugdb/core';
|
||||
import {
|
||||
diagnoseExtensionLoad,
|
||||
inspectExtensionBinary,
|
||||
} from '../../src/core/lbug/extension-load-error.js';
|
||||
import { requireFtsResourceOrSkip } from '../helpers/fts-availability.js';
|
||||
import {
|
||||
findInstalledFtsExtension,
|
||||
requireFtsResourceOrSkip,
|
||||
} from '../helpers/fts-availability.js';
|
||||
|
||||
/**
|
||||
* #2374: exercise the language-independent structural classifier against REAL
|
||||
|
|
@ -46,20 +46,14 @@ function resolveLbugNative(): string | null {
|
|||
return null;
|
||||
}
|
||||
|
||||
/** The actual installed FTS extension binary for the running lbug version. */
|
||||
/**
|
||||
* The actual installed FTS extension binary for the running lbug version.
|
||||
* `os.homedir()` already honors `$HOME` (POSIX) / `%USERPROFILE%` (Windows) —
|
||||
* the same resolution LadybugDB's native layer uses — so it stays correct
|
||||
* under the hermetic-home overrides other tests in this suite set via env vars.
|
||||
*/
|
||||
function resolveInstalledFtsExtension(): string | null {
|
||||
const home = process.env.USERPROFILE ?? process.env.HOME ?? homedir();
|
||||
const base = join(home, '.lbdb', 'extension', lbug.VERSION);
|
||||
try {
|
||||
const platformDir = readdirSync(base).find((entry) =>
|
||||
statSync(join(base, entry)).isDirectory(),
|
||||
);
|
||||
if (!platformDir) return null;
|
||||
const ext = join(base, platformDir, 'fts', 'libfts.lbug_extension');
|
||||
return existsSync(ext) ? ext : null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
return findInstalledFtsExtension(join(homedir(), '.lbdb', 'extension'));
|
||||
}
|
||||
|
||||
const lbugNative = resolveLbugNative();
|
||||
|
|
|
|||
|
|
@ -25,9 +25,9 @@ import path from 'path';
|
|||
import fs from 'fs';
|
||||
import os from 'os';
|
||||
|
||||
import lbug from '@ladybugdb/core';
|
||||
import { getExtensionInstallChildProcessArgs } from '../../src/core/lbug/extension-loader.js';
|
||||
import { cleanupTempDirSync } from '../helpers/test-db.js';
|
||||
import { findInstalledFtsExtension } from '../helpers/fts-availability.js';
|
||||
|
||||
/** `.lbdb/extension/<version>/<platform>/fts/libfts.lbug_extension`, discovered not hardcoded. */
|
||||
let extensionRelPath: string;
|
||||
|
|
@ -52,16 +52,12 @@ const makeTmpDir = (label: string): string => {
|
|||
* home — the production installer script, not a reimplementation.
|
||||
*/
|
||||
const resolveSeedExtension = (): void => {
|
||||
const relBase = path.join('.lbdb', 'extension', lbug.VERSION);
|
||||
const realVersionDir = path.join(os.homedir(), relBase);
|
||||
const platformDirs = fs.existsSync(realVersionDir) ? fs.readdirSync(realVersionDir) : [];
|
||||
for (const platform of platformDirs) {
|
||||
const candidate = path.join(realVersionDir, platform, 'fts', 'libfts.lbug_extension');
|
||||
if (fs.existsSync(candidate) && fs.statSync(candidate).size > 1024 * 1024) {
|
||||
extensionRelPath = path.join(relBase, platform, 'fts', 'libfts.lbug_extension');
|
||||
seedExtensionFile = candidate;
|
||||
return;
|
||||
}
|
||||
const realExtensionRoot = path.join(os.homedir(), '.lbdb', 'extension');
|
||||
const installed = findInstalledFtsExtension(realExtensionRoot);
|
||||
if (installed) {
|
||||
extensionRelPath = path.relative(os.homedir(), installed);
|
||||
seedExtensionFile = installed;
|
||||
return;
|
||||
}
|
||||
// No local copy — run the real installer against a hermetic probe home.
|
||||
const probeHome = makeTmpDir('seed-home');
|
||||
|
|
@ -70,16 +66,13 @@ const resolveSeedExtension = (): void => {
|
|||
timeout: 120_000,
|
||||
env: { ...process.env, HOME: probeHome, USERPROFILE: probeHome },
|
||||
});
|
||||
const probeVersionDir = path.join(probeHome, relBase);
|
||||
const probePlatforms = fs.existsSync(probeVersionDir) ? fs.readdirSync(probeVersionDir) : [];
|
||||
for (const platform of probePlatforms) {
|
||||
const candidate = path.join(probeVersionDir, platform, 'fts', 'libfts.lbug_extension');
|
||||
if (install.status === 0 && fs.existsSync(candidate)) {
|
||||
extensionRelPath = path.join(relBase, platform, 'fts', 'libfts.lbug_extension');
|
||||
seedExtensionFile = candidate;
|
||||
networkAvailable = true;
|
||||
return;
|
||||
}
|
||||
const probeExtensionRoot = path.join(probeHome, '.lbdb', 'extension');
|
||||
const probeInstalled = findInstalledFtsExtension(probeExtensionRoot);
|
||||
if (install.status === 0 && probeInstalled) {
|
||||
extensionRelPath = path.relative(probeHome, probeInstalled);
|
||||
seedExtensionFile = probeInstalled;
|
||||
networkAvailable = true;
|
||||
return;
|
||||
}
|
||||
};
|
||||
|
||||
|
|
|
|||
|
|
@ -1174,6 +1174,75 @@ describe('Java chained method call resolution', () => {
|
|||
});
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Chained call on a new-expression receiver: new Local().inner()
|
||||
// The receiver of inner() is an object_creation_expression, not a variable.
|
||||
// Regression test for #2564: without treating `new Local()` as a typed
|
||||
// receiver, the call falls back to name-only resolution and can pick an
|
||||
// unrelated same-named method (Other.inner) instead of Local.inner.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describe('Java chained call on a new-expression receiver (#2564)', () => {
|
||||
let result: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
result = await runPipelineFromRepo(path.join(FIXTURES, 'java-new-expr-chain-call'), () => {});
|
||||
}, 60000);
|
||||
|
||||
it('detects LocalChain and Other classes', () => {
|
||||
const classes = getNodesByLabel(result, 'Class');
|
||||
expect(classes).toContain('LocalChain');
|
||||
expect(classes).toContain('Other');
|
||||
});
|
||||
|
||||
it('resolves new Local().inner() to the local Local#inner, NOT Other#inner', () => {
|
||||
const calls = getRelationships(result, 'CALLS');
|
||||
const localInner = calls.find(
|
||||
(c) =>
|
||||
c.target === 'inner' && c.source === 'm' && c.targetFilePath.includes('LocalChain.java'),
|
||||
);
|
||||
const otherInner = calls.find(
|
||||
(c) => c.target === 'inner' && c.source === 'm' && c.targetFilePath.includes('Other.java'),
|
||||
);
|
||||
expect(localInner).toBeDefined();
|
||||
expect(otherInner).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Java record: container node + HAS_METHOD edges
|
||||
// Regression test for #2564: JAVA_QUERIES previously had no @definition.record
|
||||
// capture, so a record never got a Class/Record graph node — its methods
|
||||
// existed as ownerless orphans with no HAS_METHOD edge.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describe('Java record method resolution (#2564)', () => {
|
||||
let result: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
result = await runPipelineFromRepo(path.join(FIXTURES, 'java-record-methods'), () => {});
|
||||
}, 60000);
|
||||
|
||||
it('detects a Record node for Point', () => {
|
||||
const records = getNodesByLabel(result, 'Record');
|
||||
expect(records).toContain('Point');
|
||||
});
|
||||
|
||||
it('emits HAS_METHOD edges linking sum and scaled to Point', () => {
|
||||
const hasMethod = getRelationships(result, 'HAS_METHOD');
|
||||
const sumEdge = hasMethod.find((e) => e.source === 'Point' && e.target === 'sum');
|
||||
const scaledEdge = hasMethod.find((e) => e.source === 'Point' && e.target === 'scaled');
|
||||
expect(sumEdge).toBeDefined();
|
||||
expect(scaledEdge).toBeDefined();
|
||||
});
|
||||
|
||||
it('resolves scaled() calling sum() via a CALLS edge', () => {
|
||||
const calls = getRelationships(result, 'CALLS');
|
||||
const sumCall = calls.find((c) => c.target === 'sum' && c.source === 'scaled');
|
||||
expect(sumCall).toBeDefined();
|
||||
});
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Java 16+ instanceof pattern variable: `if (obj instanceof User user)`
|
||||
// Phase 5.2: extractPatternBinding on instanceof_expression binds user → User.
|
||||
|
|
@ -2986,3 +3055,43 @@ describe('Java enum constant bodies (#2555)', () => {
|
|||
expect(misattributed).toBeUndefined();
|
||||
}, 60000);
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// #2561: E.CONST.method() emits no CALLS edge — the receiver-side follow-up
|
||||
// to #2555. A bodied constant's receiver must resolve to its synthesized
|
||||
// E$N class; a body-less constant's receiver must resolve to the host enum
|
||||
// itself.
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describe('Java enum-constant receiver dispatch (#2561)', () => {
|
||||
let result: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
result = await runPipelineFromRepo(path.join(FIXTURES, 'java-enum-constant-body'), () => {});
|
||||
}, 60000);
|
||||
|
||||
it('resolves EnumConst.A.hook() to the bodied constant override (EnumConst$1.hook#0)', () => {
|
||||
const calls = getRelationships(result, 'CALLS');
|
||||
const dispatch = calls.find((c) => c.source === 'dispatchToConstant' && c.target === 'hook');
|
||||
expect(dispatch).toBeDefined();
|
||||
expect(dispatch!.rel.targetId).toBe('Method:src/EnumConst.java:EnumConst$1.hook#0');
|
||||
});
|
||||
|
||||
it("resolves EnumConst.A.log() to the host enum's inherited method via E$N's MRO (EnumConst.log#0)", () => {
|
||||
// A's body overrides hook() but NOT log(); log() lives only on the enum.
|
||||
// The bodied constant's receiver binds to EnumConst$1, whose MRO includes
|
||||
// EnumConst (via @reference.inherits), so the qualified call reaches the
|
||||
// host enum's own method — the inherited-dispatch capability the fix enables.
|
||||
const calls = getRelationships(result, 'CALLS');
|
||||
const dispatch = calls.find((c) => c.source === 'dispatchInherited' && c.target === 'log');
|
||||
expect(dispatch).toBeDefined();
|
||||
expect(dispatch!.rel.targetId).toBe('Method:src/EnumConst.java:EnumConst.log#0');
|
||||
});
|
||||
|
||||
it("resolves Plain.A.m() to the body-less constant's inherited enum method (Plain.m#0)", () => {
|
||||
const calls = getRelationships(result, 'CALLS');
|
||||
const dispatch = calls.find((c) => c.source === 'callPlain' && c.target === 'm');
|
||||
expect(dispatch).toBeDefined();
|
||||
expect(dispatch!.rel.targetId).toBe('Method:src/Plain.java:Plain.m#0');
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -1951,6 +1951,31 @@ describe('Rust abstract dispatch (Repository trait)', () => {
|
|||
});
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// #2604: trait-object (&dyn Trait) receiver dispatch
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describe('Rust dyn trait-object dispatch (#2604)', () => {
|
||||
let result: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
result = await runPipelineFromRepo(path.join(FIXTURES, 'rust-dyn-trait-object'), () => {});
|
||||
}, 60000);
|
||||
|
||||
it('detects Impl1 struct and Behaviour trait', () => {
|
||||
expect(getNodesByLabel(result, 'Struct')).toContain('Impl1');
|
||||
expect(getNodesByLabel(result, 'Trait')).toContain('Behaviour');
|
||||
});
|
||||
|
||||
it('emits exactly one CALLS edge from calls_via_dyn(b: &dyn Behaviour) to trait_target', () => {
|
||||
const calls = getRelationships(result, 'CALLS');
|
||||
const dynCalls = calls.filter(
|
||||
(c) => c.source === 'calls_via_dyn' && c.target === 'trait_target',
|
||||
);
|
||||
expect(dynCalls.length).toBe(1);
|
||||
});
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// SM-11: Rust Child extends Parent — qualified-syntax MRO
|
||||
//
|
||||
|
|
|
|||
|
|
@ -2371,7 +2371,13 @@ export function createEntry(level: string, msg: string) {
|
|||
});
|
||||
result1 = runSkillsCli(tmpDir);
|
||||
result2 = runSkillsCli(tmpDir);
|
||||
}, 90000);
|
||||
// 120s to match the other describe hooks in this file. This hook runs
|
||||
// runSkillsCli TWICE, each capped at 45s, so a 90s budget has no headroom
|
||||
// over two worst-case analyzes plus fixture setup and git init — it times
|
||||
// out the *hook* on slow Windows runners (the test below already tolerates
|
||||
// an individual analyze hitting its own 45s timeout via status === null,
|
||||
// but a hook timeout fails before that tolerance can apply).
|
||||
}, 120000);
|
||||
|
||||
afterAll(() => {
|
||||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
||||
|
|
|
|||
77
gitnexus/test/integration/spring-config-mcp.test.ts
Normal file
77
gitnexus/test/integration/spring-config-mcp.test.ts
Normal file
|
|
@ -0,0 +1,77 @@
|
|||
import { beforeAll, describe, expect, it, vi } from 'vitest';
|
||||
import { LocalBackend } from '../../src/mcp/local/local-backend.js';
|
||||
import { listRegisteredRepos } from '../../src/storage/repo-manager.js';
|
||||
import { withTestLbugDB } from '../helpers/test-indexed-db.js';
|
||||
|
||||
vi.mock('../../src/storage/repo-manager.js', () => ({
|
||||
listRegisteredRepos: vi.fn().mockResolvedValue([]),
|
||||
cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }),
|
||||
findSiblingClones: vi.fn().mockResolvedValue([]),
|
||||
}));
|
||||
|
||||
const CONSUMER_ID = 'Property:src/Config.java:DirectValues.timeout';
|
||||
const CONFIG_ID = 'Property:spring-config:application.properties:payment.timeout';
|
||||
const SEED = [
|
||||
`CREATE (p:\`Property\` {id:'${CONSUMER_ID}', name:'timeout', filePath:'src/Config.java', startLine:4, endLine:4, content:'', description:'', declaredType:'int'})`,
|
||||
`CREATE (p:\`Property\` {id:'${CONFIG_ID}', name:'payment.timeout', filePath:'application.properties', startLine:1, endLine:1, content:'', description:'Spring configuration property', declaredType:''})`,
|
||||
`MATCH (consumer:\`Property\` {id:'${CONSUMER_ID}'}), (config:\`Property\` {id:'${CONFIG_ID}'}) CREATE (consumer)-[:CodeRelation {type:'USES', confidence:1.0, reason:'spring-config:@Value payment.timeout'}]->(config)`,
|
||||
];
|
||||
|
||||
withTestLbugDB(
|
||||
'spring-config-mcp',
|
||||
(handle) => {
|
||||
let backend: LocalBackend;
|
||||
|
||||
beforeAll(() => {
|
||||
backend = (handle as typeof handle & { _backend: LocalBackend })._backend;
|
||||
});
|
||||
|
||||
describe('Spring configuration context and impact visibility', () => {
|
||||
it('shows configuration dependencies in context without special query flags', async () => {
|
||||
const context = await backend.callTool('context', { uid: CONSUMER_ID });
|
||||
expect(context.outgoing.uses).toEqual([
|
||||
expect.objectContaining({
|
||||
uid: CONFIG_ID,
|
||||
name: 'payment.timeout',
|
||||
filePath: 'application.properties',
|
||||
}),
|
||||
]);
|
||||
});
|
||||
|
||||
it('shows consumers in upstream impact from a configuration key', async () => {
|
||||
const impact = await backend.callTool('impact', {
|
||||
target_uid: CONFIG_ID,
|
||||
target: 'payment.timeout',
|
||||
direction: 'upstream',
|
||||
});
|
||||
expect(impact.risk).not.toBe('UNKNOWN');
|
||||
expect(impact.byDepth[1]).toEqual([
|
||||
expect.objectContaining({
|
||||
id: CONSUMER_ID,
|
||||
name: 'timeout',
|
||||
relationType: 'USES',
|
||||
}),
|
||||
]);
|
||||
});
|
||||
});
|
||||
},
|
||||
{
|
||||
seed: SEED,
|
||||
poolAdapter: true,
|
||||
afterSetup: async (handle) => {
|
||||
vi.mocked(listRegisteredRepos).mockResolvedValue([
|
||||
{
|
||||
name: 'test-repo',
|
||||
path: '/test/repo',
|
||||
storagePath: handle.tmpHandle.dbPath,
|
||||
indexedAt: new Date().toISOString(),
|
||||
lastCommit: 'abc123',
|
||||
stats: { files: 2, nodes: 2, communities: 0, processes: 0 },
|
||||
},
|
||||
]);
|
||||
const backend = new LocalBackend();
|
||||
await backend.init();
|
||||
(handle as typeof handle & { _backend?: LocalBackend })._backend = backend;
|
||||
},
|
||||
},
|
||||
);
|
||||
156
gitnexus/test/integration/spring-config-pipeline.test.ts
Normal file
156
gitnexus/test/integration/spring-config-pipeline.test.ts
Normal file
|
|
@ -0,0 +1,156 @@
|
|||
import path from 'node:path';
|
||||
import { mkdir, writeFile } from 'node:fs/promises';
|
||||
import { beforeAll, describe, expect, it, vi } from 'vitest';
|
||||
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
|
||||
import { runPipelineFromRepo } from '../../src/core/ingestion/pipeline.js';
|
||||
import { SPRING_CONFIG_DESCRIPTION } from '../../src/core/ingestion/frameworks/spring/config-bindings.js';
|
||||
import type { PipelineResult } from '../../types/pipeline.js';
|
||||
import { createTempDir } from '../helpers/test-db.js';
|
||||
|
||||
const FIXTURE = path.resolve(__dirname, '..', 'fixtures', 'spring-config-app');
|
||||
const SHADOW_FIXTURE = path.resolve(__dirname, '..', 'fixtures', 'spring-config-shadow-app');
|
||||
|
||||
describe('Spring configuration binding pipeline', () => {
|
||||
let result: PipelineResult;
|
||||
let nodes: GraphNode[];
|
||||
let uses: GraphRelationship[];
|
||||
|
||||
beforeAll(async () => {
|
||||
result = await runPipelineFromRepo(FIXTURE, () => {}, { skipGraphPhases: true });
|
||||
nodes = [...result.graph.iterNodes()];
|
||||
uses = [...result.graph.iterRelationshipsByType('USES')].filter((edge) =>
|
||||
edge.reason.startsWith('spring-config:'),
|
||||
);
|
||||
}, 60_000);
|
||||
|
||||
const nodeNamed = (name: string, fileSuffix?: string): GraphNode | undefined =>
|
||||
nodes.find(
|
||||
(node) =>
|
||||
node.properties.name === name &&
|
||||
(fileSuffix === undefined || String(node.properties.filePath).endsWith(fileSuffix)),
|
||||
);
|
||||
|
||||
const targetsFrom = (source: GraphNode): string[] =>
|
||||
uses
|
||||
.filter((edge) => edge.sourceId === source.id)
|
||||
.map((edge) => String(result.graph.getNode(edge.targetId)?.properties.name))
|
||||
.sort();
|
||||
|
||||
const targetFilesFrom = (source: GraphNode, targetName: string): string[] =>
|
||||
uses
|
||||
.filter((edge) => edge.sourceId === source.id)
|
||||
.map((edge) => result.graph.getNode(edge.targetId))
|
||||
.filter((node) => node?.properties.name === targetName)
|
||||
.map((node) => String(node?.properties.filePath))
|
||||
.sort();
|
||||
|
||||
it('creates key-only Property nodes for properties and profile YAML files', () => {
|
||||
const propertiesKey = nodeNamed('payment.timeout', 'application.properties');
|
||||
expect(propertiesKey).toBeDefined();
|
||||
expect(propertiesKey?.properties).not.toHaveProperty('language');
|
||||
expect(nodeNamed('service.endpoint', 'application-dev.yml')?.properties.description).toContain(
|
||||
'profile: dev',
|
||||
);
|
||||
expect(
|
||||
nodeNamed('service.retry.max-attempts', 'application-dev.yml')?.properties.startLine,
|
||||
).toBe(3);
|
||||
expect(
|
||||
nodes.some((node) => JSON.stringify(node.properties).includes('service.example.test')),
|
||||
).toBe(false);
|
||||
});
|
||||
|
||||
it('links Value fields to exact keys and leaves missing placeholders unresolved', () => {
|
||||
const timeout = nodeNamed('timeout', 'ConfigConsumers.java');
|
||||
const missing = nodeNamed('missing', 'ConfigConsumers.java');
|
||||
expect(timeout).toBeDefined();
|
||||
expect(missing).toBeDefined();
|
||||
if (timeout === undefined || missing === undefined) throw new Error('fixture fields missing');
|
||||
expect(targetsFrom(timeout)).toEqual(['payment.timeout']);
|
||||
expect(targetsFrom(missing)).toEqual([]);
|
||||
expect(missing?.properties.description).toContain('Spring config unresolved: payment.missing');
|
||||
});
|
||||
|
||||
it('links ConfigurationProperties classes and relaxed field names to their prefix', () => {
|
||||
const owner = nodeNamed('ServiceProperties', 'ConfigConsumers.java');
|
||||
const endpoint = nodeNamed('endpoint', 'ConfigConsumers.java');
|
||||
const retry = nodeNamed('retry', 'ConfigConsumers.java');
|
||||
expect(owner).toBeDefined();
|
||||
if (owner === undefined || endpoint === undefined || retry === undefined) {
|
||||
throw new Error('fixture ConfigurationProperties symbols missing');
|
||||
}
|
||||
expect(targetsFrom(owner)).toEqual([
|
||||
'service.endpoint',
|
||||
'service.endpoint',
|
||||
'service.retry.max-attempts',
|
||||
]);
|
||||
expect(targetsFrom(endpoint)).toEqual(['service.endpoint', 'service.endpoint']);
|
||||
expect(targetFilesFrom(endpoint, 'service.endpoint')).toEqual([
|
||||
'src/main/resources/application-dev.yml',
|
||||
'src/main/resources/application.properties',
|
||||
]);
|
||||
expect(targetsFrom(retry)).toEqual(['service.retry.max-attempts']);
|
||||
});
|
||||
|
||||
it('keeps the class-level binding when no field relaxed-name matches', () => {
|
||||
const owner = nodeNamed('UnmatchedServiceProperties', 'ConfigConsumers.java');
|
||||
const unrelated = nodeNamed('unrelated', 'ConfigConsumers.java');
|
||||
if (owner === undefined || unrelated === undefined) {
|
||||
throw new Error('unmatched ConfigurationProperties symbols missing');
|
||||
}
|
||||
expect(targetsFrom(owner)).toEqual([
|
||||
'service.endpoint',
|
||||
'service.endpoint',
|
||||
'service.retry.max-attempts',
|
||||
]);
|
||||
expect(targetsFrom(unrelated)).toEqual([]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Spring configuration annotation attribution', () => {
|
||||
it('fails closed when a same-package annotation shadows a Spring wildcard import', async () => {
|
||||
const result = await runPipelineFromRepo(SHADOW_FIXTURE, () => {}, {
|
||||
skipGraphPhases: true,
|
||||
});
|
||||
const fake = [...result.graph.iterNodes()].find(
|
||||
(node) =>
|
||||
node.properties.name === 'fake' &&
|
||||
String(node.properties.filePath).endsWith('Shadowed.java'),
|
||||
);
|
||||
expect(fake).toBeDefined();
|
||||
if (fake === undefined) throw new Error('shadow fixture field missing');
|
||||
expect(
|
||||
[...result.graph.iterRelationshipsByType('USES')].filter(
|
||||
(edge) => edge.sourceId === fake.id && edge.reason.startsWith('spring-config:'),
|
||||
),
|
||||
).toEqual([]);
|
||||
expect(String(fake.properties.description ?? '')).not.toContain('Spring config unresolved:');
|
||||
});
|
||||
});
|
||||
|
||||
describe('Spring configuration file safety bounds', () => {
|
||||
it('fails closed for malformed and oversized configuration files', async () => {
|
||||
const repo = await createTempDir();
|
||||
try {
|
||||
const resources = path.join(repo.dbPath, 'src', 'main', 'resources');
|
||||
await mkdir(resources, { recursive: true });
|
||||
await writeFile(path.join(resources, 'application-broken.yml'), 'broken: [\n', 'utf8');
|
||||
await writeFile(
|
||||
path.join(resources, 'application-oversized.properties'),
|
||||
`oversized.key=${'x'.repeat(2 * 1024 * 1024)}\n`,
|
||||
'utf8',
|
||||
);
|
||||
|
||||
// Let the scanner admit the file so this exercises springConfig's
|
||||
// stricter 2 MiB cap rather than the scanner's default 512 KiB cap.
|
||||
vi.stubEnv('GITNEXUS_MAX_FILE_SIZE', '4096');
|
||||
const result = await runPipelineFromRepo(repo.dbPath, () => {}, { skipGraphPhases: true });
|
||||
const configNodes = [...result.graph.iterNodes()].filter((node) =>
|
||||
String(node.properties.description ?? '').startsWith(SPRING_CONFIG_DESCRIPTION),
|
||||
);
|
||||
expect(configNodes).toEqual([]);
|
||||
} finally {
|
||||
vi.unstubAllEnvs();
|
||||
await repo.cleanup();
|
||||
}
|
||||
});
|
||||
});
|
||||
|
|
@ -6,8 +6,13 @@ import {
|
|||
type AnalysisFeatureDescriptor,
|
||||
} from '../../src/core/analysis-features.js';
|
||||
import { SPRING_BEAN_INVENTORY_FEATURE } from '../../src/core/ingestion/frameworks/spring/analysis-features.js';
|
||||
import { SPRING_CONFIG_BINDINGS_FEATURE } from '../../src/core/ingestion/languages/java/analysis-features.js';
|
||||
|
||||
const FEATURES = [CLASS_FRAMEWORK_ANNOTATIONS_FEATURE, SPRING_BEAN_INVENTORY_FEATURE] as const;
|
||||
const FEATURES = [
|
||||
CLASS_FRAMEWORK_ANNOTATIONS_FEATURE,
|
||||
SPRING_BEAN_INVENTORY_FEATURE,
|
||||
SPRING_CONFIG_BINDINGS_FEATURE,
|
||||
] as const;
|
||||
|
||||
describe('analysis feature versions', () => {
|
||||
it('separates the global Class schema capability from JVM-only Bean evidence', () => {
|
||||
|
|
@ -17,11 +22,21 @@ describe('analysis feature versions', () => {
|
|||
expect(resolveAnalysisFeatureVersions(FEATURES, ['src/App.java'])).toEqual({
|
||||
'graph.class-framework-annotations': 1,
|
||||
'spring.bean-inventory': 1,
|
||||
'spring.config-bindings': 1,
|
||||
});
|
||||
expect(resolveAnalysisFeatureVersions(FEATURES, ['BUILD.GRADLE.KTS'])).toEqual({
|
||||
'graph.class-framework-annotations': 1,
|
||||
'spring.bean-inventory': 1,
|
||||
});
|
||||
expect(
|
||||
resolveAnalysisFeatureVersions(FEATURES, [
|
||||
'src/main/resources/application-local.yml',
|
||||
'README.md',
|
||||
]),
|
||||
).toEqual({
|
||||
'graph.class-framework-annotations': 1,
|
||||
'spring.config-bindings': 1,
|
||||
});
|
||||
});
|
||||
|
||||
it('requires an exact, well-formed feature set', () => {
|
||||
|
|
|
|||
|
|
@ -73,8 +73,8 @@ describe('CALL_SUMMARY relation-type exclusion (U-C1)', () => {
|
|||
});
|
||||
|
||||
describe('CALL_SUMMARY incremental reuse gate (U-C5)', () => {
|
||||
it('INCREMENTAL_SCHEMA_VERSION is bumped to 9 (Java enum-constant-body + JLS-naming re-index window)', () => {
|
||||
expect(INCREMENTAL_SCHEMA_VERSION).toBe(9);
|
||||
it('INCREMENTAL_SCHEMA_VERSION is bumped to 11 (Rust dyn-trait-object dispatch re-index window, #2604)', () => {
|
||||
expect(INCREMENTAL_SCHEMA_VERSION).toBe(11);
|
||||
});
|
||||
|
||||
it('a pre-current stamp fails the `=== INCREMENTAL_SCHEMA_VERSION` reuse gate → forces full re-analyze', () => {
|
||||
|
|
@ -108,7 +108,15 @@ describe('CALL_SUMMARY incremental reuse gate (U-C5)', () => {
|
|||
// topmost-anchored `EnumWrap$1`-style ids would be stranded alongside
|
||||
// the re-keyed ones on unchanged files → must NOT reuse.
|
||||
expect(passesReuseGate(8)).toBe(false);
|
||||
// A pre-v10 (v9) index predates the Java record container-node fix
|
||||
// (#2564) — a record's methods would keep being ownerless Method nodes
|
||||
// with no HAS_METHOD edge on unchanged files → must NOT reuse.
|
||||
expect(passesReuseGate(9)).toBe(false);
|
||||
// A pre-v11 (v10) index predates the Rust dyn-trait-object dispatch fix
|
||||
// (#2604) — abstract trait methods would keep being uncaptured (no
|
||||
// ownerId/CALLS resolution) on unchanged Rust trait files → must NOT reuse.
|
||||
expect(passesReuseGate(10)).toBe(false);
|
||||
// A current-version stamp passes the gate (incremental top-up eligible).
|
||||
expect(passesReuseGate(9)).toBe(true);
|
||||
expect(passesReuseGate(11)).toBe(true);
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -0,0 +1,66 @@
|
|||
/**
|
||||
* #2589: `dropFTSIndex` must tolerate only benign "nothing to drop"
|
||||
* `DROP_FTS_INDEX` failures and rethrow everything else — previously it
|
||||
* swallowed every error unconditionally, which could mask a genuinely
|
||||
* corrupted FTS index across analyze runs.
|
||||
*
|
||||
* `isBenignDropFtsIndexError` is pure string logic (no native connection
|
||||
* needed), so the classification itself is unit-tested directly, including
|
||||
* against the exact reported #2589 error text — a native repro of that
|
||||
* specific engine failure was not achieved during investigation, but the
|
||||
* classifier's behavior for it is still provable from the message alone.
|
||||
*/
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { isBenignDropFtsIndexError, dropFTSIndex } from '../../src/core/lbug/lbug-adapter.js';
|
||||
import { withTestLbugDB } from '../helpers/test-indexed-db.js';
|
||||
|
||||
describe('isBenignDropFtsIndexError', () => {
|
||||
it('is true for the FTS-extension/function-not-registered catalog error (probe-verified text)', () => {
|
||||
expect(
|
||||
isBenignDropFtsIndexError(
|
||||
"Catalog exception: function DROP_FTS_INDEX is not defined. This function exists in the FTS extension. You can install and load the extension by running 'INSTALL FTS; LOAD EXTENSION FTS;'.",
|
||||
),
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
it('is true for the index-never-created binder error (probe-verified against the real dropFTSIndex path)', () => {
|
||||
expect(
|
||||
isBenignDropFtsIndexError(
|
||||
"Binder exception: Table File doesn't have an index with name file_fts.",
|
||||
),
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
it('is false for the #2589 runtime inconsistency error (must surface, not be swallowed)', () => {
|
||||
expect(
|
||||
isBenignDropFtsIndexError(
|
||||
"Runtime exception: FTS index 'file_fts' is inconsistent: term 'wiki' is missing during delete.",
|
||||
),
|
||||
).toBe(false);
|
||||
});
|
||||
|
||||
it('is false for an unrelated failure', () => {
|
||||
expect(isBenignDropFtsIndexError('Connection Exception: database is closed')).toBe(false);
|
||||
});
|
||||
|
||||
it('is false for a genuine failure that merely mentions "Binder exception" mid-message (anchored, not a bare substring match)', () => {
|
||||
expect(
|
||||
isBenignDropFtsIndexError(
|
||||
'Runtime exception: internal state corrupted while processing Binder exception: recovery failed.',
|
||||
),
|
||||
).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
withTestLbugDB('drop-fts-index-benign-cases', (handle) => {
|
||||
describe('dropFTSIndex end-to-end benign cases (#2589)', () => {
|
||||
it('resolves cleanly when the named index was never created', async () => {
|
||||
void handle;
|
||||
const { executeQuery } = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
await executeQuery(
|
||||
`CREATE NODE TABLE IF NOT EXISTS DropProbe (id STRING PRIMARY KEY, content STRING)`,
|
||||
);
|
||||
await expect(dropFTSIndex('DropProbe', 'drop_probe_never_created')).resolves.toBeUndefined();
|
||||
}, 120_000);
|
||||
});
|
||||
});
|
||||
|
|
@ -9,6 +9,7 @@ const ENV_KEYS = [
|
|||
'GITNEXUS_EMBEDDING_MAX_ATTEMPTS',
|
||||
'GITNEXUS_EMBEDDING_RETRY_CAP_MS',
|
||||
'GITNEXUS_EMBEDDING_MIN_INTERVAL_MS',
|
||||
'GITNEXUS_EMBEDDING_REQUEST_DIMS',
|
||||
] as const;
|
||||
|
||||
/** 384d mock vector matching the default schema dimensions. */
|
||||
|
|
@ -166,6 +167,30 @@ describe('HTTP embedding backend', () => {
|
|||
expect(result.length).toBe(1024);
|
||||
});
|
||||
|
||||
it('can validate custom dims without forwarding dimensions to strict backends', async () => {
|
||||
process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1';
|
||||
process.env.GITNEXUS_EMBEDDING_MODEL = 'bge-m3';
|
||||
process.env.GITNEXUS_EMBEDDING_DIMS = '1024';
|
||||
process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = 'omit';
|
||||
|
||||
const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024);
|
||||
vi.stubGlobal(
|
||||
'fetch',
|
||||
vi.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
json: async () => ({ data: [{ embedding: vec1024 }] }),
|
||||
}),
|
||||
);
|
||||
|
||||
const { embedText } = await import('../../src/core/embeddings/embedder.js');
|
||||
const result = await embedText('test text');
|
||||
|
||||
const body = JSON.parse((fetch as any).mock.calls[0][1].body);
|
||||
expect('dimensions' in body).toBe(false);
|
||||
expect(body.model).toBe('bge-m3');
|
||||
expect(result.length).toBe(1024);
|
||||
});
|
||||
|
||||
it('forwards dimensions on the single-query path', async () => {
|
||||
process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1';
|
||||
process.env.GITNEXUS_EMBEDDING_MODEL = 'text-embedding-3-large';
|
||||
|
|
@ -188,6 +213,93 @@ describe('HTTP embedding backend', () => {
|
|||
expect(result.length).toBe(512);
|
||||
});
|
||||
|
||||
it('can omit dimensions on the single-query path while validating custom dims', async () => {
|
||||
process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1';
|
||||
process.env.GITNEXUS_EMBEDDING_MODEL = 'bge-m3';
|
||||
process.env.GITNEXUS_EMBEDDING_DIMS = '1024';
|
||||
process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = 'omit';
|
||||
|
||||
const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024);
|
||||
vi.stubGlobal(
|
||||
'fetch',
|
||||
vi.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
json: async () => ({ data: [{ embedding: vec1024 }] }),
|
||||
}),
|
||||
);
|
||||
|
||||
const mod = await import('../../src/mcp/core/embedder.js');
|
||||
const result = await mod.embedQuery('query text');
|
||||
|
||||
const body = JSON.parse((fetch as any).mock.calls[0][1].body);
|
||||
expect('dimensions' in body).toBe(false);
|
||||
expect(result.length).toBe(1024);
|
||||
});
|
||||
|
||||
it.each(['none', 'off', 'false', '0'])(
|
||||
'treats GITNEXUS_EMBEDDING_REQUEST_DIMS=%s as omit and drops the request dimensions field',
|
||||
async (alias) => {
|
||||
process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1';
|
||||
process.env.GITNEXUS_EMBEDDING_MODEL = 'bge-m3';
|
||||
process.env.GITNEXUS_EMBEDDING_DIMS = '1024';
|
||||
process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = alias;
|
||||
|
||||
const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024);
|
||||
vi.stubGlobal(
|
||||
'fetch',
|
||||
vi.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
json: async () => ({ data: [{ embedding: vec1024 }] }),
|
||||
}),
|
||||
);
|
||||
|
||||
const { embedText } = await import('../../src/core/embeddings/embedder.js');
|
||||
const result = await embedText('test text');
|
||||
|
||||
const body = JSON.parse((fetch as any).mock.calls[0][1].body);
|
||||
expect('dimensions' in body).toBe(false);
|
||||
expect(result.length).toBe(1024);
|
||||
},
|
||||
);
|
||||
|
||||
it('sends REQUEST_DIMS as the request dimensions while DIMS validates the response', async () => {
|
||||
process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1';
|
||||
process.env.GITNEXUS_EMBEDDING_MODEL = 'text-embedding-3-large';
|
||||
process.env.GITNEXUS_EMBEDDING_DIMS = '1024';
|
||||
process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = '512';
|
||||
|
||||
// Response keeps the DIMS-validated length; only the outgoing request differs.
|
||||
const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024);
|
||||
vi.stubGlobal(
|
||||
'fetch',
|
||||
vi.fn().mockResolvedValue({
|
||||
ok: true,
|
||||
json: async () => ({ data: [{ embedding: vec1024 }] }),
|
||||
}),
|
||||
);
|
||||
|
||||
const { embedText } = await import('../../src/core/embeddings/embedder.js');
|
||||
const result = await embedText('test text');
|
||||
|
||||
const body = JSON.parse((fetch as any).mock.calls[0][1].body);
|
||||
expect(body.dimensions).toBe(512);
|
||||
expect(result.length).toBe(1024);
|
||||
});
|
||||
|
||||
it('rejects a malformed GITNEXUS_EMBEDDING_REQUEST_DIMS with an error naming that var', async () => {
|
||||
process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1';
|
||||
process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model';
|
||||
process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = 'garbage';
|
||||
|
||||
const { embedText } = await import('../../src/core/embeddings/embedder.js');
|
||||
const { isHttpEmbeddingDimsError } = await import('../../src/core/embeddings/http-client.js');
|
||||
const err = await embedText('test').catch((e: unknown) => e);
|
||||
// Recognizable as a config error so the CLI prints a clean message...
|
||||
expect(isHttpEmbeddingDimsError(String(err))).toBe(true);
|
||||
// ...and it points the operator at the var they set, not GITNEXUS_EMBEDDING_DIMS.
|
||||
expect(String(err)).toContain('GITNEXUS_EMBEDDING_REQUEST_DIMS must be a positive integer');
|
||||
});
|
||||
|
||||
it('retries on server error', async () => {
|
||||
process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1';
|
||||
process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model';
|
||||
|
|
|
|||
138
gitnexus/test/unit/incremental-fts-drop-ordering.test.ts
Normal file
138
gitnexus/test/unit/incremental-fts-drop-ordering.test.ts
Normal file
|
|
@ -0,0 +1,138 @@
|
|||
/**
|
||||
* #2589: the incremental writeback must drop every FTS index BEFORE
|
||||
* `deleteNodesForFiles` runs its batched DETACH DELETE — not only in
|
||||
* Phase 3, after the delete already ran against a table still carrying the
|
||||
* PREVIOUS run's index. This drives the real `runFullAnalysis` incremental
|
||||
* path (real git repo, real LadybugDB, real FTS extension) and asserts,
|
||||
* at the moment `deleteNodesForFiles` is invoked, that `SHOW_INDEXES()`
|
||||
* already reports every FTS index absent — proving the drop-before-delete
|
||||
* ordering end-to-end rather than only unit-testing the call sequence.
|
||||
*/
|
||||
import { readFile, writeFile } from 'fs/promises';
|
||||
import { execSync } from 'child_process';
|
||||
import path from 'path';
|
||||
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
import { setupMiniRepo } from '../helpers/mini-repo.js';
|
||||
import { getStoragePaths } from '../../src/storage/repo-manager.js';
|
||||
import { FTS_INDEXES } from '../../src/core/search/fts-schema.js';
|
||||
import { createTempDir } from '../helpers/test-db.js';
|
||||
import { resolveAnalyzeInstallPolicy } from '../../src/core/lbug/extension-loader.js';
|
||||
|
||||
const ftsMustBeAvailable = process.env.GITNEXUS_REQUIRE_FTS === '1';
|
||||
|
||||
describe('runFullAnalysis incremental writeback — FTS drop-before-delete ordering (#2589)', () => {
|
||||
let ftsAvailable = true;
|
||||
let skipWarned = false;
|
||||
|
||||
beforeAll(async () => {
|
||||
const lbugAdapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
// Cheap standalone probe — matches the withTestLbugDB/lbug-vector-extension
|
||||
// convention of checking availability once, up front, rather than deep
|
||||
// inside the (expensive) test body.
|
||||
const probe = await createTempDir('gitnexus-2589-fts-probe-');
|
||||
try {
|
||||
await lbugAdapter.initLbug(probe.dbPath);
|
||||
ftsAvailable = await lbugAdapter.loadFTSExtension(undefined, {
|
||||
policy: resolveAnalyzeInstallPolicy(),
|
||||
});
|
||||
} finally {
|
||||
await lbugAdapter.closeLbug();
|
||||
await probe.cleanup();
|
||||
}
|
||||
}, 120_000);
|
||||
|
||||
// Skip VISIBLY (ctx.skip() marks the test as skipped, not passed) when the
|
||||
// extension is unavailable — silently `return`ing from inside `it()` would
|
||||
// report a false pass and hide a regression in the drop-before-delete
|
||||
// ordering in exactly the environments least likely to have a human notice.
|
||||
beforeEach((ctx) => {
|
||||
if (!ftsAvailable) {
|
||||
if (ftsMustBeAvailable) {
|
||||
throw new Error(
|
||||
'GITNEXUS_REQUIRE_FTS=1 but the FTS extension is unavailable — cannot verify the #2589 ordering fix.',
|
||||
);
|
||||
}
|
||||
if (!skipWarned) {
|
||||
skipWarned = true;
|
||||
console.warn(
|
||||
'[incremental-fts-drop-ordering] Skipping — the LadybugDB FTS extension is unavailable.',
|
||||
);
|
||||
}
|
||||
ctx.skip();
|
||||
}
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.doUnmock('../../src/core/lbug/lbug-adapter.js');
|
||||
vi.resetModules();
|
||||
});
|
||||
|
||||
it('SHOW_INDEXES() reports every FTS index absent by the time deleteNodesForFiles runs', async () => {
|
||||
const lbugAdapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
const { runFullAnalysis } = await import('../../src/core/run-analyze.js');
|
||||
|
||||
const repo = await setupMiniRepo('gitnexus-2589-fts-order-');
|
||||
try {
|
||||
// First run: full rebuild, builds every FTS index for real.
|
||||
await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} });
|
||||
|
||||
// runFullAnalysis closes its own connection on return — open a fresh
|
||||
// one just to probe SHOW_INDEXES(), then close it before the second
|
||||
// run opens its own (LadybugDB is single-writer/single-connection).
|
||||
const { lbugPath } = getStoragePaths(repo.dbPath);
|
||||
await lbugAdapter.initLbug(lbugPath);
|
||||
const showIndexNames = async (): Promise<string[]> => {
|
||||
const rows = (await lbugAdapter.executeQuery('CALL SHOW_INDEXES() RETURN *')) as Array<
|
||||
Record<string, unknown>
|
||||
>;
|
||||
return rows.map((r) => r.index_name).filter((n): n is string => typeof n === 'string');
|
||||
};
|
||||
const beforeChange = await showIndexNames();
|
||||
await lbugAdapter.closeLbug();
|
||||
|
||||
// Hard assertion, not a soft skip: the beforeEach gate already proved
|
||||
// the extension loads, so every index failing to build here is a real
|
||||
// bug in the full-rebuild FTS phase, not an environment gap.
|
||||
for (const { indexName } of FTS_INDEXES) {
|
||||
expect(beforeChange).toContain(indexName);
|
||||
}
|
||||
|
||||
// Spy on the real deleteNodesForFiles, recording the FTS index list at
|
||||
// the exact moment it's invoked (before it does anything), then
|
||||
// delegating to the real implementation so the run completes normally.
|
||||
let indexNamesAtDeleteTime: string[] | undefined;
|
||||
const originalDeleteNodesForFiles = lbugAdapter.deleteNodesForFiles;
|
||||
vi.spyOn(lbugAdapter, 'deleteNodesForFiles').mockImplementation(async (filePaths, opts) => {
|
||||
indexNamesAtDeleteTime = await showIndexNames();
|
||||
return originalDeleteNodesForFiles(filePaths, opts);
|
||||
});
|
||||
|
||||
// Small change to a single file — stays well under the escalation
|
||||
// threshold (50 files) on this 7-file mini-repo, so it takes the
|
||||
// non-escalated (surgical) incremental branch this test targets.
|
||||
const handlerPath = path.join(repo.dbPath, 'src', 'handler.ts');
|
||||
await writeFile(
|
||||
handlerPath,
|
||||
(await readFile(handlerPath, 'utf-8')) + '\n// #2589 ordering-test touch\n',
|
||||
'utf-8',
|
||||
);
|
||||
execSync('git -c user.name=test -c user.email=t@t -c commit.gpgsign=false add -A', {
|
||||
cwd: repo.dbPath,
|
||||
stdio: 'pipe',
|
||||
});
|
||||
execSync(
|
||||
'git -c user.name=test -c user.email=t@t -c commit.gpgsign=false commit -q -m "#2589 ordering touch"',
|
||||
{ cwd: repo.dbPath, stdio: 'pipe' },
|
||||
);
|
||||
|
||||
await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} });
|
||||
|
||||
expect(indexNamesAtDeleteTime).toBeDefined();
|
||||
for (const { indexName } of FTS_INDEXES) {
|
||||
expect(indexNamesAtDeleteTime).not.toContain(indexName);
|
||||
}
|
||||
} finally {
|
||||
await repo.cleanup();
|
||||
}
|
||||
}, 300_000);
|
||||
});
|
||||
|
|
@ -44,6 +44,7 @@ import {
|
|||
} from '../helpers/embedding-seed.js';
|
||||
import { CLASS_FRAMEWORK_ANNOTATIONS_FEATURE } from '../../src/core/analysis-features.js';
|
||||
import { SPRING_BEAN_INVENTORY_FEATURE } from '../../src/core/ingestion/frameworks/spring/analysis-features.js';
|
||||
import { SPRING_CONFIG_BINDINGS_FEATURE } from '../../src/core/ingestion/languages/java/analysis-features.js';
|
||||
|
||||
const setupMiniRepo = () => setupSharedMiniRepo('gitnexus-incr-orch-');
|
||||
|
||||
|
|
@ -102,6 +103,16 @@ async function setupKotlinSpringBeanIncrementalRepo() {
|
|||
return repo;
|
||||
}
|
||||
|
||||
async function setupSpringConfigIncrementalRepo() {
|
||||
const repo = await createTempDir('gitnexus-incr-spring-config-');
|
||||
const resources = path.join(repo.dbPath, 'src', 'main', 'resources');
|
||||
await mkdir(resources, { recursive: true });
|
||||
await writeFile(path.join(resources, 'application.properties'), 'service.timeout=30\n', 'utf-8');
|
||||
execSync('git init', { cwd: repo.dbPath, stdio: 'pipe' });
|
||||
gitCommitAll(repo.dbPath, 'initial spring configuration');
|
||||
return repo;
|
||||
}
|
||||
|
||||
async function readWildcardServiceAnnotations(repoPath: string): Promise<string[]> {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
const { lbugPath } = getStoragePaths(repoPath);
|
||||
|
|
@ -120,6 +131,21 @@ async function readWildcardServiceAnnotations(repoPath: string): Promise<string[
|
|||
}
|
||||
}
|
||||
|
||||
async function readSpringConfigPropertyNames(repoPath: string): Promise<string[]> {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
const { lbugPath } = getStoragePaths(repoPath);
|
||||
await adapter.initLbug(lbugPath);
|
||||
try {
|
||||
const rows = (await adapter.executeQuery(
|
||||
"MATCH (p:Property) WHERE p.filePath = 'src/main/resources/application.properties' " +
|
||||
'RETURN p.name AS name ORDER BY p.name',
|
||||
)) as Array<{ name?: unknown }>;
|
||||
return rows.map((row) => String(row.name));
|
||||
} finally {
|
||||
await adapter.closeLbug();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Direct count over INJECTS CodeRelation rows — mirrors pdg-mode-flip's
|
||||
* countBasicBlocks: reopen the repo DB, count, close (runFullAnalysis closes
|
||||
|
|
@ -271,6 +297,38 @@ describe('runFullAnalysis — incremental orchestration', () => {
|
|||
}
|
||||
}, 300_000);
|
||||
|
||||
it('a config-only index missing Spring config evidence rebuilds and restores the scoped stamp', async () => {
|
||||
const repo = await setupSpringConfigIncrementalRepo();
|
||||
try {
|
||||
const { runFullAnalysis } = await import('../../src/core/run-analyze.js');
|
||||
await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} });
|
||||
const { storagePath } = getStoragePaths(repo.dbPath);
|
||||
const meta = await loadMeta(storagePath);
|
||||
expect(meta!.analysisFeatures).toEqual({
|
||||
[CLASS_FRAMEWORK_ANNOTATIONS_FEATURE.id]: CLASS_FRAMEWORK_ANNOTATIONS_FEATURE.version,
|
||||
[SPRING_CONFIG_BINDINGS_FEATURE.id]: SPRING_CONFIG_BINDINGS_FEATURE.version,
|
||||
});
|
||||
|
||||
await saveMeta(storagePath, withoutAnalysisFeature(meta!, SPRING_CONFIG_BINDINGS_FEATURE.id));
|
||||
const logs: string[] = [];
|
||||
const reanalyzed = await runFullAnalysis(
|
||||
repo.dbPath,
|
||||
{ skipAgentsMd: true },
|
||||
{ onProgress: () => {}, onLog: (message) => logs.push(message) },
|
||||
);
|
||||
|
||||
expect(reanalyzed.alreadyUpToDate).toBeUndefined();
|
||||
expect(logs.join('\n')).toContain(`missing:${SPRING_CONFIG_BINDINGS_FEATURE.id}`);
|
||||
expect(await readSpringConfigPropertyNames(repo.dbPath)).toEqual(['service.timeout']);
|
||||
expect((await loadMeta(storagePath))!.analysisFeatures).toEqual({
|
||||
[CLASS_FRAMEWORK_ANNOTATIONS_FEATURE.id]: CLASS_FRAMEWORK_ANNOTATIONS_FEATURE.version,
|
||||
[SPRING_CONFIG_BINDINGS_FEATURE.id]: SPRING_CONFIG_BINDINGS_FEATURE.version,
|
||||
});
|
||||
} finally {
|
||||
await repo.cleanup();
|
||||
}
|
||||
}, 300_000);
|
||||
|
||||
it('adding the first JVM file re-evaluates capabilities after the pipeline and avoids a top-up', async () => {
|
||||
const repo = await setupMiniRepo();
|
||||
try {
|
||||
|
|
|
|||
|
|
@ -66,6 +66,7 @@ describe('PhaseRegistry', () => {
|
|||
const FULL_ORDER = [
|
||||
'scan',
|
||||
'structure',
|
||||
'springConfig',
|
||||
'markdown',
|
||||
'cobol',
|
||||
'parse',
|
||||
|
|
|
|||
|
|
@ -1,9 +1,11 @@
|
|||
import os from 'os';
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest';
|
||||
import {
|
||||
createLbugDatabase,
|
||||
estimateBufferPool,
|
||||
isLbugCheckpointIoError,
|
||||
isWalCorruptionError,
|
||||
setBufferPoolSizeHint,
|
||||
} from '../../src/core/lbug/lbug-config.js';
|
||||
import { _captureLogger } from '../../src/core/logger.js';
|
||||
|
||||
|
|
@ -252,6 +254,74 @@ describe('createLbugDatabase buffer pool size (#2557)', () => {
|
|||
});
|
||||
});
|
||||
|
||||
describe('adaptive buffer pool hint', () => {
|
||||
const GiB = 1024 * 1024 * 1024;
|
||||
const MiB = 1024 * 1024;
|
||||
const bufferPoolArg = (Database: ReturnType<typeof vi.fn>): unknown => Database.mock.calls[0][1];
|
||||
|
||||
afterEach(() => setBufferPoolSizeHint(undefined));
|
||||
|
||||
describe('estimateBufferPool', () => {
|
||||
it.each([
|
||||
['tiny graph clamps up to the 256 MiB COPY-safety floor', 41, 256 * MiB],
|
||||
['a graph under the floor still clamps up to 256 MiB', 40_000, 256 * MiB],
|
||||
['mid graph scales linearly (100k elements * 4 KiB = 400 MiB)', 100_000, 100_000 * 4 * 1024],
|
||||
['huge graph caps at the 2 GiB / 80%-RAM default', 10_000_000, 2 * GiB],
|
||||
])('%s', (_label, elements, expected) => {
|
||||
const totalmemSpy = vi.spyOn(os, 'totalmem').mockReturnValue(32 * GiB);
|
||||
try {
|
||||
expect(estimateBufferPool(elements)).toBe(expected);
|
||||
} finally {
|
||||
totalmemSpy.mockRestore();
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
it.each([
|
||||
['a hint within range passes through', 512 * MiB, 512 * MiB],
|
||||
['a hint below the COPY-safety floor clamps up to 256 MiB', 100 * MiB, 256 * MiB],
|
||||
['a hint above the default clamps down to the 2 GiB cap', 8 * GiB, 2 * GiB],
|
||||
])(
|
||||
'createLbugDatabase uses the clamped hint when no env override is set: %s',
|
||||
(_label, hint, expected) => {
|
||||
const totalmemSpy = vi.spyOn(os, 'totalmem').mockReturnValue(32 * GiB);
|
||||
try {
|
||||
setBufferPoolSizeHint(hint);
|
||||
const Database = vi.fn(function (this: any) {});
|
||||
createLbugDatabase({ Database } as any, '/tmp/lbug-hint');
|
||||
expect(bufferPoolArg(Database)).toBe(expected);
|
||||
} finally {
|
||||
totalmemSpy.mockRestore();
|
||||
}
|
||||
},
|
||||
);
|
||||
|
||||
it('env override wins over the hint (including 0 = native default)', () => {
|
||||
try {
|
||||
setBufferPoolSizeHint(128 * MiB);
|
||||
vi.stubEnv('GITNEXUS_LBUG_BUFFER_POOL_SIZE', '0');
|
||||
const Database = vi.fn(function (this: any) {});
|
||||
createLbugDatabase({ Database } as any, '/tmp/lbug-hint-env');
|
||||
expect(bufferPoolArg(Database)).toBe(0);
|
||||
} finally {
|
||||
vi.unstubAllEnvs();
|
||||
}
|
||||
});
|
||||
|
||||
it('falls back to the default when the hint is cleared', () => {
|
||||
const totalmemSpy = vi.spyOn(os, 'totalmem').mockReturnValue(32 * GiB);
|
||||
try {
|
||||
setBufferPoolSizeHint(128 * MiB);
|
||||
setBufferPoolSizeHint(undefined);
|
||||
const Database = vi.fn(function (this: any) {});
|
||||
createLbugDatabase({ Database } as any, '/tmp/lbug-hint-cleared');
|
||||
expect(bufferPoolArg(Database)).toBe(2 * GiB);
|
||||
} finally {
|
||||
totalmemSpy.mockRestore();
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Finding 8: strict + permissive checkpoint IO matchers ─────────────────
|
||||
describe('isLbugCheckpointIoError', () => {
|
||||
it.each([
|
||||
|
|
|
|||
|
|
@ -2,8 +2,9 @@ import { describe, it, expect, vi, afterEach } from 'vitest';
|
|||
|
||||
/**
|
||||
* Tests for the #2372 `node:module` compat seam. `module.registerHooks` was
|
||||
* added in Node 22.15 / 23.5, but the engines floor is >=22.0.0, so on
|
||||
* 22.0–22.14 and 23.0–23.4 the export is absent. `getRegisterHooks()` must
|
||||
* added in Node 22.15 / 23.5. The engines floor is ^22.18.0 || >=24.11.0 (all
|
||||
* >=22.15), but engines is advisory, so a below-floor 22.0–22.14 / 23.0–23.4
|
||||
* runtime can still run, where the export is absent. `getRegisterHooks()` must
|
||||
* hand back the real function when present and `undefined` when not — the value
|
||||
* the resolver guards degrade on. `isPrefixRuntimeLoadable()` (exported from
|
||||
* runtime-install.ts so CLI code never imports the compat module) is the
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ import {
|
|||
processProcesses,
|
||||
type ProcessDetectionConfig,
|
||||
} from '../../src/core/ingestion/process-processor.js';
|
||||
import { computeDynamicMaxProcesses } from '../../src/core/ingestion/pipeline-phases/processes.js';
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import type { CommunityMembership } from '../../src/core/ingestion/community-processor.js';
|
||||
|
||||
|
|
@ -522,4 +523,40 @@ describe('processProcesses', () => {
|
|||
expect(result.processes.length).toBeLessThanOrEqual(3);
|
||||
expect(result.stats.totalProcesses).toBeLessThanOrEqual(3);
|
||||
});
|
||||
|
||||
// Regression for #2198: the processesPhase dynamic sizing used to cap at
|
||||
// Math.min(300, symbolCount/10). On large repos (>3000 symbols) that silently
|
||||
// truncated the process index. The cap was removed by extracting
|
||||
// computeDynamicMaxProcesses() — this test exercises the helper directly
|
||||
// so it fails if someone reintroduces the 300 ceiling.
|
||||
describe('computeDynamicMaxProcesses (#2198)', () => {
|
||||
it('returns at least the floor of 20 for tiny repos', () => {
|
||||
expect(computeDynamicMaxProcesses(0)).toBe(20);
|
||||
expect(computeDynamicMaxProcesses(50)).toBe(20); // 50/10 = 5, floored to 20
|
||||
expect(computeDynamicMaxProcesses(199)).toBe(20); // 199/10 ≈ 20
|
||||
});
|
||||
|
||||
it('scales linearly within the old 0–3000 range', () => {
|
||||
expect(computeDynamicMaxProcesses(500)).toBe(50);
|
||||
expect(computeDynamicMaxProcesses(1000)).toBe(100);
|
||||
expect(computeDynamicMaxProcesses(2999)).toBe(300);
|
||||
});
|
||||
|
||||
it('grows past 300 for large repos — the regression that #2198 fixes', () => {
|
||||
// 3001 symbols → 300 (just at the boundary)
|
||||
expect(computeDynamicMaxProcesses(3001)).toBe(300);
|
||||
// 3100 symbols → 310 — would have been capped to 300 before the fix
|
||||
expect(computeDynamicMaxProcesses(3100)).toBe(310);
|
||||
// 5000 symbols → 500
|
||||
expect(computeDynamicMaxProcesses(5000)).toBe(500);
|
||||
// 28000 symbols (real-world large repo) → 2800
|
||||
expect(computeDynamicMaxProcesses(28000)).toBe(2800);
|
||||
});
|
||||
|
||||
it('does NOT cap at 300 — fails if Math.min(300, ...) is reintroduced', () => {
|
||||
const largeRepo = computeDynamicMaxProcesses(10000);
|
||||
expect(largeRepo).toBe(1000);
|
||||
expect(largeRepo).toBeGreaterThan(300);
|
||||
});
|
||||
});
|
||||
});
|
||||
|
|
|
|||
239
gitnexus/test/unit/rename-edit-report.test.ts
Normal file
239
gitnexus/test/unit/rename-edit-report.test.ts
Normal file
|
|
@ -0,0 +1,239 @@
|
|||
/**
|
||||
* Regression test for issue #2605: `rename` must report every edit it applies.
|
||||
*
|
||||
* The apply step does a whole-file `\boldName\b` global replace on each touched
|
||||
* file, but the reported `changes`/`total_edits` were built from a partial
|
||||
* enumeration that (a) recorded only the definition line, (b) recorded one edit
|
||||
* per graph-ref file then broke, and (c) skipped text-search on any file already
|
||||
* covered by the graph. When a private symbol's definition and all its call
|
||||
* sites live in one file, only the definition line was reported (total_edits: 1)
|
||||
* while apply rewrote every occurrence. These tests drive the single-file repro,
|
||||
* a mixed graph/text_search multi-file rename, and a partial write failure, and
|
||||
* assert the report matches what apply actually writes in each case.
|
||||
*/
|
||||
import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest';
|
||||
import { promises as fs } from 'node:fs';
|
||||
import fsPromises from 'fs/promises';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
|
||||
// Prevent onnxruntime / native search adapters from loading at import time
|
||||
// (mirrors test/unit/calltool-dispatch.test.ts). We drive the private rename()
|
||||
// directly, so the graph/DB/embedding layers are never exercised.
|
||||
vi.mock('../../src/core/search/bm25-index.js', () => ({
|
||||
searchFTSFromLbug: vi.fn().mockResolvedValue({ results: [], ftsAvailable: true }),
|
||||
}));
|
||||
vi.mock('../../src/mcp/core/embedder.js', () => ({
|
||||
embedQuery: vi.fn().mockResolvedValue([]),
|
||||
getEmbeddingDims: vi.fn().mockReturnValue(384),
|
||||
}));
|
||||
|
||||
// rename() shells out to `rg -l` to discover text-search files. Stub it so the
|
||||
// ripgrep-discovery branch is deterministic and driveable (rg is not reliably
|
||||
// on PATH inside the vitest worker). Default: no hits.
|
||||
const { execFileSyncMock } = vi.hoisted(() => ({ execFileSyncMock: vi.fn(() => '') }));
|
||||
vi.mock('child_process', async (importActual) => {
|
||||
const actual = await importActual<typeof import('child_process')>();
|
||||
return { ...actual, execFileSync: execFileSyncMock };
|
||||
});
|
||||
|
||||
import { LocalBackend } from '../../src/mcp/local/local-backend.js';
|
||||
|
||||
type Incoming = {
|
||||
calls: { filePath: string }[];
|
||||
imports: { filePath: string }[];
|
||||
extends: { filePath: string }[];
|
||||
implements: { filePath: string }[];
|
||||
};
|
||||
const EMPTY_INCOMING: Incoming = { calls: [], imports: [], extends: [], implements: [] };
|
||||
|
||||
type RenameResult = {
|
||||
status: string;
|
||||
applied: boolean;
|
||||
files_affected: number;
|
||||
total_edits: number;
|
||||
graph_edits: number;
|
||||
text_search_edits: number;
|
||||
changes: { file_path: string; edits: { line: number; confidence: string }[] }[];
|
||||
failed_files?: string[];
|
||||
};
|
||||
|
||||
// The #2605 repro: a private free fn with exactly 4 textual occurrences of
|
||||
// `rename_target` — the definition, one production call, two test calls — all
|
||||
// in the same file.
|
||||
const RUST_SRC = `fn rename_target(x: u32) -> u32 {
|
||||
x + 1
|
||||
}
|
||||
|
||||
pub fn prod_call() -> u32 {
|
||||
rename_target(1)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn unit_one() {
|
||||
assert_eq!(rename_target(1), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unit_two() {
|
||||
assert_eq!(rename_target(2), 3);
|
||||
}
|
||||
}
|
||||
`;
|
||||
|
||||
/** 1-based occurrence lines of `rename_target` in `src` — the ground truth the
|
||||
* report must match. Computed (not hardcoded) so editing a fixture cannot
|
||||
* silently desync the expectation. */
|
||||
function occurrenceLines(src: string): number[] {
|
||||
return src
|
||||
.split('\n')
|
||||
.map((line, i) => (/\brename_target\b/.test(line) ? i + 1 : 0))
|
||||
.filter((n) => n > 0);
|
||||
}
|
||||
const OCCURRENCE_LINES = occurrenceLines(RUST_SRC);
|
||||
|
||||
/** Build a backend whose graph lookup returns the symbol (definition at
|
||||
* src/lib.rs) with the given incoming refs. */
|
||||
function stubbedBackend(incoming: Incoming = EMPTY_INCOMING): LocalBackend {
|
||||
const backend = new LocalBackend();
|
||||
vi.spyOn(
|
||||
backend as unknown as { ensureInitialized: () => Promise<void> },
|
||||
'ensureInitialized',
|
||||
).mockResolvedValue(undefined);
|
||||
vi.spyOn(backend as unknown as { context: () => Promise<unknown> }, 'context').mockResolvedValue({
|
||||
status: 'success',
|
||||
symbol: { name: 'rename_target', filePath: 'src/lib.rs', startLine: OCCURRENCE_LINES[0] },
|
||||
incoming,
|
||||
});
|
||||
return backend;
|
||||
}
|
||||
|
||||
function callRename(
|
||||
backend: LocalBackend,
|
||||
repoPath: string,
|
||||
params: Record<string, unknown>,
|
||||
): Promise<RenameResult> {
|
||||
return (
|
||||
backend as unknown as { rename: (r: unknown, p: unknown) => Promise<RenameResult> }
|
||||
).rename({ repoPath }, { symbol_name: 'rename_target', new_name: 'renamed_fn', ...params });
|
||||
}
|
||||
|
||||
const editsFor = (r: RenameResult, file: string) =>
|
||||
r.changes.find((c) => c.file_path === file)?.edits ?? [];
|
||||
|
||||
describe('rename edit report is faithful to apply (#2605)', () => {
|
||||
let tmpDir: string;
|
||||
|
||||
beforeEach(async () => {
|
||||
execFileSyncMock.mockReturnValue(''); // default: no ripgrep hits
|
||||
tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'gn-2605-'));
|
||||
await fs.mkdir(path.join(tmpDir, 'src'));
|
||||
await fs.writeFile(path.join(tmpDir, 'src', 'lib.rs'), RUST_SRC, 'utf-8');
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
vi.restoreAllMocks();
|
||||
await fs.rm(tmpDir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
it('previews every occurrence that apply will rewrite (dry_run)', async () => {
|
||||
const result = await callRename(stubbedBackend(), tmpDir, { dry_run: true });
|
||||
|
||||
expect(result.applied).toBe(false);
|
||||
expect(result.files_affected).toBe(1);
|
||||
expect(result.total_edits).toBe(OCCURRENCE_LINES.length); // 4, not 1
|
||||
// Concrete split, not just the sum: all occurrences are in the definition
|
||||
// file, so they are graph-confidence and text_search is zero.
|
||||
expect(result.graph_edits).toBe(OCCURRENCE_LINES.length);
|
||||
expect(result.text_search_edits).toBe(0);
|
||||
|
||||
const edits = editsFor(result, 'src/lib.rs');
|
||||
expect(edits.map((e) => e.line).sort((a, b) => a - b)).toEqual(OCCURRENCE_LINES);
|
||||
expect(edits.every((e) => e.confidence === 'graph')).toBe(true);
|
||||
|
||||
// A dry run leaves the file untouched.
|
||||
const onDisk = await fs.readFile(path.join(tmpDir, 'src', 'lib.rs'), 'utf-8');
|
||||
expect(onDisk).toContain('rename_target');
|
||||
});
|
||||
|
||||
it('reports exactly what it wrote (apply)', async () => {
|
||||
const result = await callRename(stubbedBackend(), tmpDir, { dry_run: false });
|
||||
|
||||
expect(result.applied).toBe(true);
|
||||
expect(result.total_edits).toBe(OCCURRENCE_LINES.length);
|
||||
|
||||
const onDisk = await fs.readFile(path.join(tmpDir, 'src', 'lib.rs'), 'utf-8');
|
||||
const renamedCount = (onDisk.match(/\brenamed_fn\b/g) || []).length;
|
||||
const stragglers = (onDisk.match(/\brename_target\b/g) || []).length;
|
||||
expect(renamedCount).toBe(OCCURRENCE_LINES.length); // all 4 rewritten
|
||||
expect(stragglers).toBe(0);
|
||||
|
||||
// The reported edit count equals the number of replacements that landed.
|
||||
const reportedEdits = result.changes.reduce((n, c) => n + c.edits.length, 0);
|
||||
expect(reportedEdits).toBe(renamedCount);
|
||||
});
|
||||
|
||||
it('enumerates all occurrences across graph-ref and text-search files, keeping confidence per file', async () => {
|
||||
// A graph-referencing file (not the definition) with MULTIPLE occurrences —
|
||||
// the exact "one edit per file then break" bug's other original trigger.
|
||||
const CALLER = 'use crate::rename_target;\nfn a() { rename_target(1); rename_target(2); }\n';
|
||||
// A file discovered only by ripgrep — the text_search branch.
|
||||
const NOTES = '// see rename_target for details\n';
|
||||
await fs.writeFile(path.join(tmpDir, 'src', 'caller.rs'), CALLER, 'utf-8');
|
||||
await fs.writeFile(path.join(tmpDir, 'src', 'notes.rs'), NOTES, 'utf-8');
|
||||
// rg reports the definition file (already graph — exercises never-downgrade)
|
||||
// and the text-only file.
|
||||
execFileSyncMock.mockReturnValue('src/lib.rs\nsrc/notes.rs\n');
|
||||
|
||||
const backend = stubbedBackend({
|
||||
...EMPTY_INCOMING,
|
||||
calls: [{ filePath: 'src/caller.rs' }],
|
||||
});
|
||||
const result = await callRename(backend, tmpDir, { dry_run: true });
|
||||
|
||||
const callerOcc = occurrenceLines(CALLER).length; // 3
|
||||
const notesOcc = occurrenceLines(NOTES).length; // 1
|
||||
|
||||
expect(result.files_affected).toBe(3);
|
||||
expect(result.total_edits).toBe(OCCURRENCE_LINES.length + callerOcc + notesOcc);
|
||||
// Split is concrete: definition + graph-ref file are graph; the rg-only file
|
||||
// is text_search. A file reached by both graph and rg keeps graph (never
|
||||
// downgraded).
|
||||
expect(result.graph_edits).toBe(OCCURRENCE_LINES.length + callerOcc);
|
||||
expect(result.text_search_edits).toBe(notesOcc);
|
||||
|
||||
expect(editsFor(result, 'src/lib.rs').every((e) => e.confidence === 'graph')).toBe(true);
|
||||
expect(editsFor(result, 'src/caller.rs').map((e) => e.confidence)).toEqual(['graph', 'graph']);
|
||||
expect(editsFor(result, 'src/notes.rs').map((e) => e.confidence)).toEqual(['text_search']);
|
||||
});
|
||||
|
||||
it('reports only files that landed when a write fails mid-apply (#2605 partial)', async () => {
|
||||
const CALLER = 'fn a() { rename_target(1); rename_target(2); }\n';
|
||||
await fs.writeFile(path.join(tmpDir, 'src', 'caller.rs'), CALLER, 'utf-8');
|
||||
const backend = stubbedBackend({ ...EMPTY_INCOMING, calls: [{ filePath: 'src/caller.rs' }] });
|
||||
|
||||
// caller.rs write throws; lib.rs succeeds.
|
||||
vi.spyOn(fsPromises, 'writeFile').mockImplementation(
|
||||
async (p: Parameters<typeof fsPromises.writeFile>[0]) => {
|
||||
if (String(p).endsWith(`${path.sep}caller.rs`)) {
|
||||
throw Object.assign(new Error('EACCES'), { code: 'EACCES' });
|
||||
}
|
||||
},
|
||||
);
|
||||
|
||||
const result = await callRename(backend, tmpDir, { dry_run: false });
|
||||
|
||||
expect(result.status).toBe('partial');
|
||||
expect(result.failed_files).toEqual(['src/caller.rs']);
|
||||
// The failed file's edits are NOT reported as applied: totals describe only
|
||||
// what reached disk (lib.rs), never the attempted caller.rs occurrences.
|
||||
expect(result.files_affected).toBe(1);
|
||||
expect(result.total_edits).toBe(OCCURRENCE_LINES.length);
|
||||
expect(result.graph_edits).toBe(OCCURRENCE_LINES.length);
|
||||
expect(result.changes.map((c) => c.file_path)).toEqual(['src/lib.rs']);
|
||||
});
|
||||
});
|
||||
|
|
@ -589,12 +589,12 @@ describe('gitnexus review-agent workflow security contract', () => {
|
|||
}
|
||||
|
||||
expect(runtimePackage.dependencies?.gitnexus).toBe('1.6.9');
|
||||
expect(runtimePackage.engines?.node).toBe('22.16.0');
|
||||
expect(runtimePackage.engines?.node).toBe('22.18.0');
|
||||
expect(runtimeLock.packages?.['node_modules/gitnexus']?.version).toBe('1.6.9');
|
||||
expect(runtimeLock.packages?.['node_modules/gitnexus']?.integrity).toMatch(/^sha512-/);
|
||||
expect(workflow).not.toMatch(/gitnexus@(latest|next|beta)/);
|
||||
expect(workflow).toContain("node-version: '22.16.0'");
|
||||
expect(workflow).toContain('test "$(node --version)" = \'v22.16.0\'');
|
||||
expect(workflow).toContain("node-version: '22.18.0'");
|
||||
expect(workflow).toContain('test "$(node --version)" = \'v22.18.0\'');
|
||||
expect(workflow).toContain('npm ci');
|
||||
expect(workflow).not.toContain('--package-lock=false');
|
||||
expect(workflow).toContain(
|
||||
|
|
@ -682,7 +682,7 @@ describe('gitnexus review-agent workflow security contract', () => {
|
|||
'${{ runner.temp }}/gitnexus-review-claude-runtime/node_modules/@anthropic-ai/claude-code/bin/claude.exe';
|
||||
|
||||
expect(claudeRuntimePackage.dependencies?.['@anthropic-ai/claude-code']).toBe('2.1.214');
|
||||
expect(claudeRuntimePackage.engines?.node).toBe('22.16.0');
|
||||
expect(claudeRuntimePackage.engines?.node).toBe('22.18.0');
|
||||
expect(claudeRuntimeLock.lockfileVersion).toBe(3);
|
||||
expect(claudeRuntimeLock.packages?.['node_modules/@anthropic-ai/claude-code']).toMatchObject({
|
||||
version: '2.1.214',
|
||||
|
|
@ -1119,7 +1119,7 @@ describe('gitnexus review-agent workflow security contract', () => {
|
|||
'CLAUDE_CONFIG_DIR: ${{ runner.temp }}/gitnexus-review-claude-config',
|
||||
);
|
||||
expect(analyze).toContain('CLAUDE_WORKING_DIR: ${{ runner.temp }}/gitnexus-review-control');
|
||||
expect(analyze).toContain("NODE_VERSION: '22.16.0'");
|
||||
expect(analyze).toContain("NODE_VERSION: '22.18.0'");
|
||||
expect(analyze).toContain('checkout-index --all --force');
|
||||
expect(analyze).toContain('find "${review_dir}" -type l -print0');
|
||||
expect(analyze).toContain('Escaping copied review symlink');
|
||||
|
|
|
|||
|
|
@ -0,0 +1,61 @@
|
|||
import { describe, expect, it } from 'vitest';
|
||||
import type { CaptureMatch } from 'gitnexus-shared';
|
||||
import {
|
||||
normalizeRustTypeName,
|
||||
interpretRustTypeBinding,
|
||||
} from '../../../../src/core/ingestion/languages/rust/interpret.js';
|
||||
|
||||
const RANGE = { startLine: 0, startCol: 0, endLine: 0, endCol: 0 };
|
||||
|
||||
/** Builds a minimal @type-binding.return CaptureMatch to exercise
|
||||
* normalizeRustReturnType (private, only reachable through this hook). */
|
||||
function returnTypeBinding(type: string): CaptureMatch {
|
||||
return {
|
||||
'@type-binding.name': { name: '@type-binding.name', range: RANGE, text: 'f' },
|
||||
'@type-binding.type': { name: '@type-binding.type', range: RANGE, text: type },
|
||||
'@type-binding.return': { name: '@type-binding.return', range: RANGE, text: '' },
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* #2604 coverage gap (GitNexus review-agent finding): stripDynBound's
|
||||
* documented Box<dyn Trait> and bound-list (dyn Trait + Send) shapes had no
|
||||
* test anywhere, even though the interpret.ts comment claims they're handled.
|
||||
* These exercise normalizeRustTypeName/normalizeRustReturnType directly —
|
||||
* stripDynBound itself is a private helper reached only through them.
|
||||
*/
|
||||
describe('Rust dyn-trait-object type-name normalization (#2604)', () => {
|
||||
it('strips a bare dyn Trait parameter type', () => {
|
||||
expect(normalizeRustTypeName('&dyn Behaviour')).toBe('Behaviour');
|
||||
expect(normalizeRustTypeName('dyn Behaviour')).toBe('Behaviour');
|
||||
});
|
||||
|
||||
it('strips dyn through Box/Rc/Arc wrappers', () => {
|
||||
expect(normalizeRustTypeName('Box<dyn Trait>')).toBe('Trait');
|
||||
expect(normalizeRustTypeName('Rc<dyn Trait>')).toBe('Trait');
|
||||
expect(normalizeRustTypeName('Arc<dyn Trait>')).toBe('Trait');
|
||||
});
|
||||
|
||||
it('drops an auto-trait/lifetime bound list after dyn', () => {
|
||||
expect(normalizeRustTypeName('dyn Trait + Send')).toBe('Trait');
|
||||
expect(normalizeRustTypeName("dyn Trait + Send + 'static")).toBe('Trait');
|
||||
expect(normalizeRustTypeName("Box<dyn Trait + 'static>")).toBe('Trait');
|
||||
});
|
||||
|
||||
it("truncates a dyn trait's own generic arguments after stripping dyn", () => {
|
||||
expect(normalizeRustTypeName('dyn Iterator<Item = u32>')).toBe('Iterator');
|
||||
});
|
||||
|
||||
it('strips dyn in return-type position, including through &', () => {
|
||||
expect(interpretRustTypeBinding(returnTypeBinding('&dyn Trait'))?.rawTypeName).toBe('Trait');
|
||||
expect(interpretRustTypeBinding(returnTypeBinding('dyn Trait + Send'))?.rawTypeName).toBe(
|
||||
'Trait',
|
||||
);
|
||||
});
|
||||
|
||||
it('leaves ordinary (non-dyn) type names untouched', () => {
|
||||
expect(normalizeRustTypeName('Behaviour')).toBe('Behaviour');
|
||||
expect(normalizeRustTypeName('&Behaviour')).toBe('Behaviour');
|
||||
expect(normalizeRustTypeName('Box<Behaviour>')).toBe('Behaviour');
|
||||
});
|
||||
});
|
||||
193
gitnexus/test/unit/spring-config-bindings.test.ts
Normal file
193
gitnexus/test/unit/spring-config-bindings.test.ts
Normal file
|
|
@ -0,0 +1,193 @@
|
|||
import { describe, expect, it, vi } from 'vitest';
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import { bindSpringConfigConsumers } from '../../src/core/ingestion/frameworks/spring/config-bindings.js';
|
||||
import {
|
||||
classifySpringConfigFile,
|
||||
parseSpringProperties,
|
||||
parseSpringYaml,
|
||||
} from '../../src/core/ingestion/pipeline-phases/spring-config.js';
|
||||
import { extractJavaSpringConfigConsumers } from '../../src/core/ingestion/languages/java/spring-config-bindings.js';
|
||||
|
||||
describe('Spring configuration parsing', () => {
|
||||
it('recognizes base and profile-specific application config files', () => {
|
||||
const base = classifySpringConfigFile('src/main/resources/application.properties');
|
||||
expect(base).toMatchObject({ format: 'properties' });
|
||||
expect(base).not.toHaveProperty('profile');
|
||||
expect(classifySpringConfigFile('src/main/resources/application-local.yml')).toMatchObject({
|
||||
format: 'yaml',
|
||||
profile: 'local',
|
||||
});
|
||||
expect(classifySpringConfigFile('src/main/resources/bootstrap.yml')).toBeNull();
|
||||
});
|
||||
|
||||
it('extracts properties keys, continuations, and escaped separators without values', () => {
|
||||
const keys = parseSpringProperties(
|
||||
'# comment\nserver.port=8080\nservice\\:name: demo\nlong.\\\n key = secret\n',
|
||||
'application.properties',
|
||||
);
|
||||
expect(keys.map((entry) => [entry.key, entry.line])).toEqual([
|
||||
['server.port', 2],
|
||||
['service:name', 3],
|
||||
['long.key', 4],
|
||||
]);
|
||||
expect(JSON.stringify(keys)).not.toContain('8080');
|
||||
expect(JSON.stringify(keys)).not.toContain('secret');
|
||||
});
|
||||
|
||||
it('flattens YAML maps and arrays while retaining profile identity', () => {
|
||||
const keys = parseSpringYaml(
|
||||
'service:\n endpoint: https://example.test\n retries:\n - delay: 10\n',
|
||||
'application-dev.yml',
|
||||
'dev',
|
||||
);
|
||||
expect(keys).toEqual([
|
||||
expect.objectContaining({
|
||||
key: 'service.endpoint',
|
||||
line: 2,
|
||||
profile: 'dev',
|
||||
format: 'yaml',
|
||||
}),
|
||||
expect.objectContaining({
|
||||
key: 'service.retries[0].delay',
|
||||
line: 4,
|
||||
profile: 'dev',
|
||||
format: 'yaml',
|
||||
}),
|
||||
]);
|
||||
expect(JSON.stringify(keys)).not.toContain('example.test');
|
||||
});
|
||||
|
||||
it('expands YAML merge keys and retains the declaration line for merged values', () => {
|
||||
const keys = parseSpringYaml(
|
||||
[
|
||||
'defaults: &defaults',
|
||||
' endpoint: https://base.example.test',
|
||||
' timeout: 30',
|
||||
'service:',
|
||||
' <<: *defaults',
|
||||
' endpoint: https://override.example.test',
|
||||
].join('\n'),
|
||||
'application.yml',
|
||||
);
|
||||
|
||||
expect(keys).toEqual([
|
||||
expect.objectContaining({ key: 'defaults.endpoint', line: 2 }),
|
||||
expect.objectContaining({ key: 'defaults.timeout', line: 3 }),
|
||||
expect.objectContaining({ key: 'service.endpoint', line: 6 }),
|
||||
expect.objectContaining({ key: 'service.timeout', line: 3 }),
|
||||
]);
|
||||
expect(keys.some((entry) => entry.key.includes('<<'))).toBe(false);
|
||||
});
|
||||
|
||||
it('terminates cyclic YAML aliases and bounds deeply nested expansion', () => {
|
||||
expect(
|
||||
parseSpringYaml('cycle: &cycle { self: *cycle }\nhealthy: true\n', 'application.yml'),
|
||||
).toEqual([expect.objectContaining({ key: 'healthy', line: 2 })]);
|
||||
|
||||
const aliasChain = ['level0: &level0 { leaf: true }'];
|
||||
for (let index = 1; index <= 130; index++) {
|
||||
aliasChain.push(`level${index}: &level${index} { next: *level${index - 1} }`);
|
||||
}
|
||||
expect(() => parseSpringYaml(aliasChain.join('\n'), 'application.yml')).toThrow(
|
||||
'Spring YAML traversal depth',
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Java Spring configuration consumers', () => {
|
||||
it('resolves official imports and ignores shadowed annotation names', () => {
|
||||
const consumers = extractJavaSpringConfigConsumers(`
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.boot.context.properties.ConfigurationProperties;
|
||||
|
||||
@ConfigurationProperties(prefix = "service")
|
||||
class ServiceProperties {
|
||||
@Value("\${service.timeout:30}") private int timeout;
|
||||
}
|
||||
`);
|
||||
expect(consumers).toEqual([
|
||||
expect.objectContaining({ kind: 'value', fieldName: 'timeout', keys: ['service.timeout'] }),
|
||||
expect.objectContaining({
|
||||
kind: 'configuration-properties',
|
||||
className: 'ServiceProperties',
|
||||
prefix: 'service',
|
||||
}),
|
||||
]);
|
||||
|
||||
expect(
|
||||
extractJavaSpringConfigConsumers(`
|
||||
@interface Value { String value(); }
|
||||
class Local { @Value("\${fake.key}") String field; }
|
||||
`),
|
||||
).toEqual([]);
|
||||
});
|
||||
|
||||
it('supports wildcard/FQN annotations and every declarator in a field', () => {
|
||||
const consumers = extractJavaSpringConfigConsumers(`
|
||||
import org.springframework.beans.factory.annotation.*;
|
||||
|
||||
class DirectValues {
|
||||
@Value("\${shared.key}") String first, second;
|
||||
}
|
||||
|
||||
@org.springframework.boot.context.properties.ConfigurationProperties("service")
|
||||
record ServiceProperties(String endpoint) {}
|
||||
`);
|
||||
|
||||
expect(consumers).toEqual([
|
||||
expect.objectContaining({ kind: 'value', fieldName: 'first', keys: ['shared.key'] }),
|
||||
expect.objectContaining({ kind: 'value', fieldName: 'second', keys: ['shared.key'] }),
|
||||
expect.objectContaining({
|
||||
kind: 'configuration-properties',
|
||||
className: 'ServiceProperties',
|
||||
prefix: 'service',
|
||||
}),
|
||||
]);
|
||||
});
|
||||
|
||||
it('reads only string-literal AST nodes and ignores placeholders inside comments', () => {
|
||||
const consumers = extractJavaSpringConfigConsumers(`
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.boot.context.properties.ConfigurationProperties;
|
||||
|
||||
@ConfigurationProperties(
|
||||
// legacy prefix: "old.unsafe"
|
||||
value = "service"
|
||||
)
|
||||
class ServiceProperties {
|
||||
@Value(
|
||||
/* legacy: "\${old.unsafe.key}" */
|
||||
"\${service.timeout:30}"
|
||||
)
|
||||
private int timeout;
|
||||
}
|
||||
`);
|
||||
|
||||
expect(consumers).toEqual([
|
||||
expect.objectContaining({ kind: 'value', keys: ['service.timeout'] }),
|
||||
expect.objectContaining({ kind: 'configuration-properties', prefix: 'service' }),
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Spring configuration graph binding', () => {
|
||||
it('indexes the graph once for all consumer files and skips empty work', () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
const iterNodes = vi.spyOn(graph, 'iterNodes');
|
||||
|
||||
bindSpringConfigConsumers(graph, []);
|
||||
expect(iterNodes).not.toHaveBeenCalled();
|
||||
|
||||
bindSpringConfigConsumers(graph, [
|
||||
{
|
||||
filePath: 'First.java',
|
||||
consumers: [{ kind: 'value', fieldName: 'first', line: 1, keys: ['first.key'] }],
|
||||
},
|
||||
{
|
||||
filePath: 'Second.java',
|
||||
consumers: [{ kind: 'value', fieldName: 'second', line: 1, keys: ['second.key'] }],
|
||||
},
|
||||
]);
|
||||
expect(iterNodes).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
});
|
||||
91
gitnexus/test/unit/worker-pool-ready-timeout-env.test.ts
Normal file
91
gitnexus/test/unit/worker-pool-ready-timeout-env.test.ts
Normal file
|
|
@ -0,0 +1,91 @@
|
|||
/**
|
||||
* `GITNEXUS_WORKER_READY_TIMEOUT_MS` overrides the worker ready budget.
|
||||
*
|
||||
* The 5s default is a startup budget for parser + grammar imports. On a slow
|
||||
* or heavily loaded host a full pool of workers cold-starting concurrently
|
||||
* can legitimately need more: without an override every slot misses the
|
||||
* handshake, the identical timeout messages reproduce across respawns, and
|
||||
* the pool misclassifies the slow start as a deterministic startup
|
||||
* crash-loop — aborting the whole analyze. The env var mirrors
|
||||
* `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS`.
|
||||
*
|
||||
* `resolveWorkerPoolOptions` reads the env var fresh on every
|
||||
* `createWorkerPool` call, so each test just sets the env var before
|
||||
* constructing the pool — no module reset needed.
|
||||
*/
|
||||
import { describe, expect, it, beforeEach, afterEach } from 'vitest';
|
||||
import { EventEmitter } from 'node:events';
|
||||
import path from 'node:path';
|
||||
import os from 'node:os';
|
||||
import fs from 'node:fs';
|
||||
import { pathToFileURL } from 'node:url';
|
||||
import {
|
||||
createWorkerPool,
|
||||
WorkerPoolInitializationError,
|
||||
} from '../../src/core/ingestion/workers/worker-pool.js';
|
||||
|
||||
/** Worker double that never reports ready and never exits: a slow starter. */
|
||||
class NeverReadyWorker extends EventEmitter {
|
||||
readonly stderr = new EventEmitter();
|
||||
postMessage(): void {}
|
||||
async terminate(): Promise<number> {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
let tempDir: string;
|
||||
let workerUrl: URL;
|
||||
const ENV_KEY = 'GITNEXUS_WORKER_READY_TIMEOUT_MS';
|
||||
let savedEnv: string | undefined;
|
||||
|
||||
beforeEach(() => {
|
||||
tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-worker-ready-timeout-'));
|
||||
const workerPath = path.join(tempDir, 'fake-worker.js');
|
||||
fs.writeFileSync(workerPath, '// fake');
|
||||
workerUrl = pathToFileURL(workerPath) as URL;
|
||||
savedEnv = process.env[ENV_KEY];
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
if (savedEnv === undefined) delete process.env[ENV_KEY];
|
||||
else process.env[ENV_KEY] = savedEnv;
|
||||
try {
|
||||
fs.rmSync(tempDir, { recursive: true, force: true });
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
});
|
||||
|
||||
describe('worker pool — GITNEXUS_WORKER_READY_TIMEOUT_MS override', () => {
|
||||
it('applies the override to the readiness deadline and its failure message', async () => {
|
||||
process.env[ENV_KEY] = '50';
|
||||
const pool = createWorkerPool(workerUrl, 1, {
|
||||
workerFactory: () => new NeverReadyWorker() as unknown as Worker,
|
||||
});
|
||||
|
||||
const err = await pool
|
||||
.dispatch([{ path: 'a.ts', content: 'x' }])
|
||||
.catch((e: unknown) => e as InstanceType<typeof WorkerPoolInitializationError>);
|
||||
|
||||
expect(err).toBeInstanceOf(WorkerPoolInitializationError);
|
||||
expect(err.readinessFailures.join('\n')).toContain('within 50ms');
|
||||
|
||||
await pool.terminate().catch(() => undefined);
|
||||
});
|
||||
|
||||
it('falls back to the 5s default when the value is not a positive integer', async () => {
|
||||
process.env[ENV_KEY] = 'not-a-number';
|
||||
const pool = createWorkerPool(workerUrl, 1, {
|
||||
workerFactory: () => new NeverReadyWorker() as unknown as Worker,
|
||||
});
|
||||
|
||||
const err = await pool
|
||||
.dispatch([{ path: 'a.ts', content: 'x' }])
|
||||
.catch((e: unknown) => e as InstanceType<typeof WorkerPoolInitializationError>);
|
||||
|
||||
expect(err).toBeInstanceOf(WorkerPoolInitializationError);
|
||||
expect(err.readinessFailures.join('\n')).toContain('within 5000ms');
|
||||
|
||||
await pool.terminate().catch(() => undefined);
|
||||
});
|
||||
});
|
||||
108
gitnexus/test/unit/worker-pool-stdout-forward.test.ts
Normal file
108
gitnexus/test/unit/worker-pool-stdout-forward.test.ts
Normal file
|
|
@ -0,0 +1,108 @@
|
|||
/**
|
||||
* Worker stdout is piped and forwarded, not inherited.
|
||||
*
|
||||
* The production factory now spawns workers with `{ stdout: true }`: workers
|
||||
* with INHERITED stdout have been observed to crash silently during
|
||||
* top-of-script init (exit code 1, nothing on stderr, roughly half of a
|
||||
* concurrently spawned pool) on macOS 26.5 under both Node 22 and 26.
|
||||
* Piping avoids the crash, and `forwardWorkerStdout` mirrors the piped
|
||||
* stream back to the parent's stdout so worker logs stay visible — the same
|
||||
* tee shape `captureWorkerStderr` uses for stderr (#1741).
|
||||
*
|
||||
* This test injects a fake worker that writes to its `stdout` stream and
|
||||
* asserts the pool forwards it to `process.stdout`; a stdout-less test
|
||||
* factory must remain a no-op.
|
||||
*/
|
||||
import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest';
|
||||
import { EventEmitter } from 'node:events';
|
||||
import path from 'node:path';
|
||||
import os from 'node:os';
|
||||
import fs from 'node:fs';
|
||||
import { pathToFileURL } from 'node:url';
|
||||
import { createWorkerPool } from '../../src/core/ingestion/workers/worker-pool.js';
|
||||
|
||||
const WORKER_LOG_LINE = '{"level":30,"name":"gitnexus","msg":"parse-worker log line"}\n';
|
||||
|
||||
/**
|
||||
* Worker double that starts cleanly and emits a log line on its piped
|
||||
* `stdout` stream, mirroring a production worker spawned with
|
||||
* `{ stdout: true }`.
|
||||
*/
|
||||
class ReadyWorkerWithStdout extends EventEmitter {
|
||||
readonly stdout = new EventEmitter();
|
||||
readonly stderr = new EventEmitter();
|
||||
constructor() {
|
||||
super();
|
||||
queueMicrotask(() => {
|
||||
this.stdout.emit('data', Buffer.from(WORKER_LOG_LINE));
|
||||
this.emit('message', { type: 'ready' });
|
||||
});
|
||||
}
|
||||
postMessage(): void {}
|
||||
async terminate(): Promise<number> {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
/** Worker double with no stdio streams at all (typical test factory shape). */
|
||||
class ReadyWorkerWithoutStdio extends EventEmitter {
|
||||
constructor() {
|
||||
super();
|
||||
queueMicrotask(() => this.emit('message', { type: 'ready' }));
|
||||
}
|
||||
postMessage(): void {}
|
||||
async terminate(): Promise<number> {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
let tempDir: string;
|
||||
let workerUrl: URL;
|
||||
let stdoutSpy: ReturnType<typeof vi.spyOn>;
|
||||
|
||||
beforeEach(() => {
|
||||
tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-worker-stdout-forward-'));
|
||||
const workerPath = path.join(tempDir, 'fake-worker.js');
|
||||
fs.writeFileSync(workerPath, '// fake');
|
||||
workerUrl = pathToFileURL(workerPath) as URL;
|
||||
// Capture the forwarded worker stdout without polluting test output.
|
||||
stdoutSpy = vi.spyOn(process.stdout, 'write').mockReturnValue(true);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
stdoutSpy.mockRestore();
|
||||
try {
|
||||
fs.rmSync(tempDir, { recursive: true, force: true });
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
});
|
||||
|
||||
describe('worker pool — stdout forwarding', () => {
|
||||
it("forwards a worker's piped stdout to the parent process stdout", async () => {
|
||||
const pool = createWorkerPool(workerUrl, 1, {
|
||||
workerFactory: () => new ReadyWorkerWithStdout() as unknown as Worker,
|
||||
});
|
||||
|
||||
// Empty dispatch settles the initial-ready gate; the fake worker's stdout
|
||||
// line is emitted in the same microtask turn as its ready handshake.
|
||||
await pool.dispatch([]);
|
||||
|
||||
const forwarded = stdoutSpy.mock.calls.map((c) => String(c[0])).join('');
|
||||
expect(forwarded).toContain('parse-worker log line');
|
||||
|
||||
await pool.terminate().catch(() => undefined);
|
||||
});
|
||||
|
||||
it('is a no-op for workers without a stdout stream (test factories)', async () => {
|
||||
const pool = createWorkerPool(workerUrl, 1, {
|
||||
workerFactory: () => new ReadyWorkerWithoutStdio() as unknown as Worker,
|
||||
});
|
||||
|
||||
// Must not throw while wiring stdio on a stream-less worker.
|
||||
await pool.dispatch([]);
|
||||
expect(pool.getStats().activeSlots).toBe(1);
|
||||
|
||||
await pool.terminate().catch(() => undefined);
|
||||
});
|
||||
});
|
||||
Loading…
Add table
Reference in a new issue