From f83223946201a307ecb57df1e60d36f7c9f523b9 Mon Sep 17 00:00:00 2001 From: Abhigyan Patwari <126312502+abhigyanpatwari@users.noreply.github.com> Date: Mon, 20 Jul 2026 15:55:49 +0530 Subject: [PATCH 01/61] fix(eval): pre-approve proposer tools under Claude Code 2.1.214 ENV_SCRUB hardening (#2576) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The first real skill-evolution run got past task binding, then the proposer session exited 1 with "Permission mode forced to default — CLAUDE_CODE_SUBPROCESS_ENV_SCRUB is set (allowed_non_write_users hardening)". On 2.1.214 the permission resolver unconditionally forces permission mode to "default" whenever CLAUDE_CODE_SUBPROCESS_ENV_SCRUB is set — `--permission-mode dontAsk`, settings `permissions.defaultMode`, and `autoAllowBashIfSandboxed` are all ignored for the mode decision. The proposer runs headless `-p --bare` where Bash is its only writable tool (it writes the candidate overlay); under forced "default" Bash was no longer auto-approved, so the session blocked. We cannot set ENV_SCRUB=0 (it scrubs the Anthropic auth token from the sandboxed proposer's Bash subprocesses). Instead, align with the forced mode: pre-approve the proposer's exact tool surface via settings `permissions.allow` (["Read","Grep","Glob","Bash"]) — under "default" a tool runs without a prompt iff it matches an allow rule — and stop requesting a non-default mode so no warning fires. ENV_SCRUB and the full sandbox filesystem/network lockdown are unchanged. The real-binary containment canary is updated to the new invocation (no --permission-mode) so the CI job is the authoritative empirical gate, and a fast unit assertion pins the new permissions.allow / absent defaultMode. Claude-Session: https://claude.ai/code/session_01Va5uu9Ar3e45QZ5xFsG4AZ Co-authored-by: Gergo Magyar Co-authored-by: Claude Opus 4.8 --- eval/tests/test_proposer_sandbox.py | 11 +++++++++-- eval/workflow_bench/evolve.py | 5 ++++- eval/workflow_bench/proposer_sandbox.py | 9 ++++++++- 3 files changed, 21 insertions(+), 4 deletions(-) diff --git a/eval/tests/test_proposer_sandbox.py b/eval/tests/test_proposer_sandbox.py index 338eead52..e5ceb7b2c 100644 --- a/eval/tests/test_proposer_sandbox.py +++ b/eval/tests/test_proposer_sandbox.py @@ -58,6 +58,11 @@ def test_environment_is_allowlisted_and_shell_children_are_credential_free(monke assert settings["sandbox"]["failIfUnavailable"] is True assert settings["sandbox"]["allowUnsandboxedCommands"] is False assert settings["sandbox"]["network"]["deniedDomains"] == ["*"] + # ENV_SCRUB forces "default" mode; the proposer's tools (Bash writes the + # overlay) run headless only because they are explicitly pre-approved. + # Requesting a non-default defaultMode would merely warn, so it must be gone. + assert settings["permissions"]["allow"] == ["Read", "Grep", "Glob", "Bash"] + assert "defaultMode" not in settings["permissions"] @pytest.mark.parametrize( @@ -716,8 +721,10 @@ for line in sys.stdin: "--strict-mcp-config", "--mcp-config", mcp_config, - "--permission-mode", - "dontAsk", + # No --permission-mode: mirrors production (run_proposer). + # ENV_SCRUB forces "default"; Bash runs only because + # settings permissions.allow pre-approves it. This is the + # authoritative empirical gate for that behavior. "--model", "claude-canary-20260718", "--allowedTools", diff --git a/eval/workflow_bench/evolve.py b/eval/workflow_bench/evolve.py index 8cf6be495..d917abe88 100644 --- a/eval/workflow_bench/evolve.py +++ b/eval/workflow_bench/evolve.py @@ -501,7 +501,10 @@ def run_proposer( auth_token=args.auth_token, base_url=args.base_url, ), - permission_mode="dontAsk", + # No permission_mode: CLAUDE_CODE_SUBPROCESS_ENV_SCRUB + # forces "default", so requesting dontAsk only warns. Tools + # are pre-approved via settings permissions.allow + # (proposer_sandbox.build_claude_settings). command_prefix=sandbox.command_prefix, require_pid_namespace=True, bare=True, diff --git a/eval/workflow_bench/proposer_sandbox.py b/eval/workflow_bench/proposer_sandbox.py index 3e492fee9..254906139 100644 --- a/eval/workflow_bench/proposer_sandbox.py +++ b/eval/workflow_bench/proposer_sandbox.py @@ -329,7 +329,14 @@ def build_claude_settings() -> str: }, }, "permissions": { - "defaultMode": "dontAsk", + # CLAUDE_CODE_SUBPROCESS_ENV_SCRUB forces permission mode to + # "default" (allowed_non_write_users hardening), so requesting a + # non-default mode only emits a warning and never takes effect. + # Under "default" a tool runs without a prompt only if it matches an + # allow rule, so pre-approve the proposer's exact tool surface. Bash + # is the only writable tool under --bare (it writes the candidate + # overlay) and stays sandbox-confined by the sandbox.* policy above. + "allow": ["Read", "Grep", "Glob", "Bash"], "disableBypassPermissionsMode": "disable", }, "env": { From 6b1c4d4540370a0f1c25c5f78f21e61f82c4d9e2 Mon Sep 17 00:00:00 2001 From: Abhigyan Patwari <126312502+abhigyanpatwari@users.noreply.github.com> Date: Mon, 20 Jul 2026 16:36:02 +0530 Subject: [PATCH 02/61] fix(eval): surface stdout tail on session failure, not just stderr (#2577) The proposer's second real-run failure (after the ENV_SCRUB permission fix landed) exited 1 with subtype "success" and an EMPTY stderr_tail -- opaque: the downloaded CI artifact showed num_turns:1, cost_usd:0, tokens:0, duration_s:0.1, meaning the session terminated before any real model turn completed (consistent with an early, pre-flight-style failure), but nothing in the persisted record said why. The actual JSON event stream (permission_denials, tool_use/tool_result, is_error) lives in stdout, which run_managed already captures as proc.stdout_tail -- it just never made it into the session's error_detail. Add it there, bounded and truncated the same way stderr_tail already is. It flows through evolve.py's existing whole-record redaction before being written to disk / the uploaded artifact, so this closes the diagnostic gap without a new blind CI dispatch: the next failure of this shape is readable directly from the artifact. Claude-Session: https://claude.ai/code/session_01Va5uu9Ar3e45QZ5xFsG4AZ Co-authored-by: Gergo Magyar Co-authored-by: Claude Opus 4.8 --- eval/tests/test_workflow_bench_sessions.py | 13 +++++++++++++ eval/workflow_bench/runner_sessions.py | 8 ++++++++ 2 files changed, 21 insertions(+) diff --git a/eval/tests/test_workflow_bench_sessions.py b/eval/tests/test_workflow_bench_sessions.py index 1ffabf760..194377ec3 100644 --- a/eval/tests/test_workflow_bench_sessions.py +++ b/eval/tests/test_workflow_bench_sessions.py @@ -192,11 +192,24 @@ def test_run_claude_keeps_raw_subtype_and_stderr_tail(monkeypatch, tmp_path): "returncode": 1, "process_state": "exited", "stderr_tail": "rate limit hit", + "stdout_tail": VALID_REPORT, "process_detail": None, "event_stream_error": None, } +def test_run_claude_surfaces_stdout_tail_on_empty_stderr(monkeypatch, tmp_path): + # A session can exit non-zero with an EMPTY stderr (e.g. a pre-flight + # sandbox failure before any model turn ever runs) -- stdout_tail is then + # the only place the actual event stream is visible, so it must not be + # dropped just because stderr had nothing to say. + proc = fake_cli_result(VALID_REPORT, returncode=1, stderr="") + monkeypatch.setattr(runner_sessions, "run_managed", lambda *a, **k: proc) + rec = runner.run_claude("task", tmp_path, claude_bin="claude", timeout=5) + assert rec["error_detail"]["stderr_tail"] == "" + assert rec["error_detail"]["stdout_tail"] == VALID_REPORT + + def test_run_arm_labels_completed_but_unverified_runs_verify_failed(monkeypatch, tmp_path): monkeypatch.setattr(runner, "run_claude", lambda *a, **k: session_record()) monkeypatch.setattr(runner, "run_verify", lambda *a, **k: (False, "failed")) diff --git a/eval/workflow_bench/runner_sessions.py b/eval/workflow_bench/runner_sessions.py index a6f16e4d4..60e92b6bc 100644 --- a/eval/workflow_bench/runner_sessions.py +++ b/eval/workflow_bench/runner_sessions.py @@ -434,6 +434,14 @@ def run_claude( "returncode": proc.returncode, "process_state": proc.state, "stderr_tail": proc.stderr_tail[-2000:], + # A session can exit non-zero with an empty stderr (e.g. a + # pre-flight sandbox failure before any model turn): the tail + # of raw stdout is the only place the actual event stream + # (permission_denials, tool_use/tool_result, is_error) shows + # up, so surface it here rather than leaving the failure + # opaque. Callers already redact this record before it is + # written to disk or an uploaded artifact. + "stdout_tail": proc.stdout_tail[-2000:], "process_detail": proc.detail, "event_stream_error": event_stream_error, } From 497f1170753a6df9ea18562a1ae2378496da78df Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 20 Jul 2026 13:25:16 +0100 Subject: [PATCH 03/61] fix(eval): drop tags from the benchmark's per-arm clone (#2579) Every benchmark-arm session failed with "sanitized graph snapshot preparation failed: clone has more than 1024 references; refusing incomplete sanitization" (confirmed via a real workflow_dispatch run, 29738099937, after the prior activation fixes let the proposer succeed end-to-end for the first time). make_worktree() creates each arm's throwaway clone with a plain `git clone`, which inherits every tag and branch from the source. This repo's history has grown to 1144 tags (a v1.6.9-rc.N release-candidate series) out of 1650 total refs, exceeding oracle_assets.MAX_CLONE_REFS=1024 -- a fail-closed guard in sanitize_clone_for_hidden_oracles() that refuses to proceed unless it can enumerate and delete every ref before handing a sanitized snapshot to a benchmark session (so an agent can never discover oracle answers via a ref the sanitization missed). `ref` at every call site (evolve.py, runner.py, sanitized_graph.py) is always a bare SHA or the literal "HEAD", never a branch name, so `--single-branch --branch ` isn't viable (git clone's --branch requires a name). Tags are never used by the checkout fallback or by sanitization's own delete-everything behavior, so dropping them via --no-tags removes the 1144-ref majority without touching branch-fetch behavior or the existing ref/origin-ref checkout fallback, and without weakening MAX_CLONE_REFS itself. Verified against the real repository (not just the test fixture): cloning /workspace (1650 refs, 1144 tags) via the fixed make_worktree() now produces a clone with 237 total refs and 0 tags. Claude-Session: https://claude.ai/code/session_01Va5uu9Ar3e45QZ5xFsG4AZ Co-authored-by: Claude Opus 4.8 --- eval/tests/test_workflow_bench_sessions.py | 52 ++++++++++++++++++++++ eval/workflow_bench/runner_artifacts.py | 1 + 2 files changed, 53 insertions(+) diff --git a/eval/tests/test_workflow_bench_sessions.py b/eval/tests/test_workflow_bench_sessions.py index 194377ec3..2b424854f 100644 --- a/eval/tests/test_workflow_bench_sessions.py +++ b/eval/tests/test_workflow_bench_sessions.py @@ -1025,3 +1025,55 @@ def test_review_phase_rejects_workspace_or_skill_mutation( assert rec["resolved"] is False assert rec["error_kind"] == "review-evidence-invalid" assert expected_detail in rec["error_detail"] + + +def _git(repo, *args, check=True): + return subprocess.run(["git", "-C", str(repo), *args], check=check, capture_output=True, text=True) + + +def _git_commit(repo, message): + _git( + repo, + "-c", + "user.name=test", + "-c", + "user.email=test@invalid", + "commit", + "--quiet", + "--allow-empty", + "-m", + message, + ) + return _git(repo, "rev-parse", "HEAD").stdout.strip() + + +def test_make_worktree_clone_has_no_tags_but_keeps_all_branches(tmp_path): + # oracle_assets.MAX_CLONE_REFS refuses to sanitize a clone with more than + # 1024 refs; this repo's own history has 1000+ release-candidate tags, so + # a plain `git clone` of it (inheriting every tag) trips that cap on every + # benchmark session. make_worktree must not carry tags into its throwaway + # clone, but callers pass a bare SHA or "HEAD" as `ref` (never a branch + # name -- see evolve.py:476, runner.py:1037, sanitized_graph.py:345), so + # branch-fetching itself must stay untouched: a commit reachable only from + # a non-default branch must still resolve via the existing + # checkout(ref) -> checkout(origin/{ref}) fallback. + repo = tmp_path / "repo" + repo.mkdir() + _git(repo, "init", "--quiet") + _git(repo, "checkout", "--quiet", "-b", "main") + _git_commit(repo, "base") + _git(repo, "tag", "v1.0.0-rc.1") + + _git(repo, "checkout", "--quiet", "-b", "other") + other_sha = _git_commit(repo, "only on other") + _git(repo, "checkout", "--quiet", "main") + + clones = tmp_path / "clones" + clones.mkdir() + target = runner.make_worktree(repo, other_sha, clones) + + tags = _git(target, "tag").stdout.split() + assert tags == [], f"clone must carry no tags, found: {tags}" + + current = _git(target, "rev-parse", "HEAD").stdout.strip() + assert current == other_sha diff --git a/eval/workflow_bench/runner_artifacts.py b/eval/workflow_bench/runner_artifacts.py index b337ea415..0b56a0c9a 100644 --- a/eval/workflow_bench/runner_artifacts.py +++ b/eval/workflow_bench/runner_artifacts.py @@ -240,6 +240,7 @@ def make_worktree(repo: Path, ref: str, parent: Path) -> Path: "clone", "--no-local", "--no-hardlinks", + "--no-tags", "--quiet", str(repo), str(target), From cd93b8da23f1d71d8f0b48a5beefc4109db164c7 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Mon, 20 Jul 2026 12:53:24 +0000 Subject: [PATCH 04/61] fix(eval): mount hooks/claude into the benchmark sandbox resolve-invocation.ts requires hooks/claude/resolve-analyze-cmd.cjs at module load time, reached whenever the analyze command loads. The sandbox's curated mount list never exposed hooks/, so every benchmark-arm session failed with MODULE_NOT_FOUND (CI run 29742191562). Appends the new mount after the existing six so the function's hardcoded mounts[0]/[1]/[2]/[5] validation reads stay correct. Extends the real-bwrap canary to require analyze.js directly, since --version alone never reaches the lazy import that broke. --- eval/tests/test_workflow_bench_sessions.py | 20 ++++++++++++++++++++ eval/workflow_bench/runtime_mounts.py | 6 ++++++ 2 files changed, 26 insertions(+) diff --git a/eval/tests/test_workflow_bench_sessions.py b/eval/tests/test_workflow_bench_sessions.py index 2b424854f..c48ea10a1 100644 --- a/eval/tests/test_workflow_bench_sessions.py +++ b/eval/tests/test_workflow_bench_sessions.py @@ -290,10 +290,12 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm runtime / "dist" / "cli", runtime / "node_modules", runtime / "vendor", + runtime / "hooks" / "claude", shared / "dist", ): directory.mkdir(parents=True) (runtime / "dist" / "cli" / "index.js").write_text("") + (runtime / "hooks" / "claude" / "resolve-analyze-cmd.cjs").write_text("") (runtime / "package.json").write_text(json.dumps({"version": runner.PINNED_GITNEXUS_VERSION})) (runtime / "node_modules" / "gitnexus-shared").symlink_to(shared, target_is_directory=True) (shared / "package.json").write_text(json.dumps({"name": "gitnexus-shared"})) @@ -316,6 +318,7 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm (runtime / "vendor", f"{runner.SANDBOX_GITNEXUS}/vendor"), (shared / "dist", f"{runner.SANDBOX_GITNEXUS_SHARED}/dist"), (shared / "package.json", f"{runner.SANDBOX_GITNEXUS_SHARED}/package.json"), + (runtime / "hooks" / "claude", f"{runner.SANDBOX_GITNEXUS}/hooks/claude"), ] package = json.loads((runtime / "package.json").read_text()) assert package["version"] == runner.PINNED_GITNEXUS_VERSION @@ -330,6 +333,12 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm assert shared / forbidden not in mounted_sources assert f"{runner.SANDBOX_GITNEXUS_SHARED}/{forbidden}" not in mounted_targets + # Only hooks/claude is exposed, not the whole hooks/ directory (which also + # has an unrelated hooks/antigravity/ tree) and not the runtime root itself. + assert runtime / "hooks" not in mounted_sources + assert runtime / "hooks" / "antigravity" not in mounted_sources + assert f"{runner.SANDBOX_GITNEXUS}/hooks" not in mounted_targets + @pytest.mark.skipif( os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1", @@ -347,6 +356,7 @@ def test_real_bubblewrap_runtime_mount_imports_cli_without_exposing_checkout(tmp f"{runner.SANDBOX_GITNEXUS}/vendor", f"{runner.SANDBOX_GITNEXUS_SHARED}/dist/index.js", f"{runner.SANDBOX_GITNEXUS_SHARED}/package.json", + f"{runner.SANDBOX_GITNEXUS}/hooks/claude/resolve-analyze-cmd.cjs", ] forbidden = [ f"{runner.SANDBOX_GITNEXUS}/{relative}" @@ -376,9 +386,19 @@ def test_real_bubblewrap_runtime_mount_imports_cli_without_exposing_checkout(tmp ["/usr/local/bin/node", runner.SANDBOX_GITNEXUS_ENTRYPOINT, "--version"], timeout=10, ) + # --version never reaches the `analyze` command, which is loaded via a + # lazy dynamic import and is the only path that pulls in + # resolve-invocation.ts's module-load-time require of hooks/claude/ + # resolve-analyze-cmd.cjs. Require the compiled analyze module + # directly so this canary actually exercises that chain. + analyze_imported = sandbox.run( + ["/usr/local/bin/node", "-e", f"require('{runner.SANDBOX_GITNEXUS}/dist/cli/analyze.js')"], + timeout=10, + ) assert visibility.ok, visibility.stderr_tail assert imported.ok, imported.stderr_tail + assert analyze_imported.ok, analyze_imported.stderr_tail assert imported.stdout_tail.strip() == runner.PINNED_GITNEXUS_VERSION diff --git a/eval/workflow_bench/runtime_mounts.py b/eval/workflow_bench/runtime_mounts.py index 4fec1d166..3f2e6fcf2 100644 --- a/eval/workflow_bench/runtime_mounts.py +++ b/eval/workflow_bench/runtime_mounts.py @@ -217,6 +217,12 @@ def trusted_gitnexus_runtime_mounts() -> tuple[ReadOnlyMount, ...]: f"{SANDBOX_GITNEXUS_SHARED}/package.json", directory=False, ), + _validated_runtime_component( + runtime, + "hooks/claude", + f"{SANDBOX_GITNEXUS}/hooks/claude", + directory=True, + ), ) entrypoint = mounts[0].source / "cli" / "index.js" From c723d420d58a5d1d1a4d3b8589c513f21a27828b Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Mon, 20 Jul 2026 14:42:50 +0000 Subject: [PATCH 05/61] fix(eval): size the buffered-fallback budget for the real graph index trusted_gitnexus_runtime_mounts and sanitized_graph both hand the ~290 MiB graph index through TaskAssetSnapshot.materialize(), which reflinks into each arm clone or falls back to a buffered copy under a 16 MiB budget. Neither ext4 (CI runners) nor 9p (this container) support FICLONE, so every real materialization fell to the buffered path and blew the budget instantly (CI run 29750271566). Raises MAX_BUFFERED_FALLBACK_BYTES to 512 MiB - well above the real index size, still a full 4x below MAX_TASK_ASSET_BYTES so a genuinely oversized declaration still fails closed. --- eval/tests/test_task_assets.py | 21 +++++++++++++++++++++ eval/workflow_bench/task_assets.py | 11 ++++++++--- 2 files changed, 29 insertions(+), 3 deletions(-) diff --git a/eval/tests/test_task_assets.py b/eval/tests/test_task_assets.py index 276280ff8..4479fce4c 100644 --- a/eval/tests/test_task_assets.py +++ b/eval/tests/test_task_assets.py @@ -113,6 +113,27 @@ def test_small_assets_use_a_bounded_buffered_fallback(monkeypatch, tmp_path: Pat assert (clone / "second").read_bytes() == b"def" +def test_default_buffered_fallback_budget_covers_a_realistic_large_asset( + monkeypatch, + tmp_path: Path, +) -> None: + # 20 MiB exceeds the old 16 MiB default but must fit comfortably under + # the current default, proving the real (non-monkeypatched) budget + # constant is sized for a realistic large sandbox_copy asset such as the + # harness's own pre-built graph index, not just tiny fixtures. + payload = os.urandom(20 * 1024 * 1024) + repo, task = _repo_and_task(tmp_path, {"large": payload}) + clone = tmp_path / "clone" + clone.mkdir() + monkeypatch.setattr(task_assets, "_try_reflink", lambda *_args: False) + + with TaskAssetCache(tmp_path / "cache") as cache: + snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA) + snapshot.materialize(clone) + + assert (clone / "large").read_bytes() == payload + + def test_large_asset_without_reflink_fails_before_publish_and_cleans_staging( monkeypatch, tmp_path: Path, diff --git a/eval/workflow_bench/task_assets.py b/eval/workflow_bench/task_assets.py index 0b7855e84..14815e0d6 100644 --- a/eval/workflow_bench/task_assets.py +++ b/eval/workflow_bench/task_assets.py @@ -39,9 +39,14 @@ MAX_TASK_ASSET_ENTRIES = 100_000 MAX_TASK_ASSET_PATH_BYTES = 4_096 MAX_TASK_ASSET_BYTES = 2 * 1024 * 1024 * 1024 -# A filesystem without reflink support may still run tiny fixtures. Large -# assets fail closed instead of silently returning to one full copy per arm. -MAX_BUFFERED_FALLBACK_BYTES = 16 * 1024 * 1024 +# The largest known real sandbox_copy asset in this harness is the shipped +# index above (~428 MiB estimated, ~290 MiB measured); budget comfortably +# above that so it can still materialize via buffered copy on a filesystem +# that cannot reflink (ext4 CI runners, 9p-backed dev mounts), while staying +# well below MAX_TASK_ASSET_BYTES so a genuinely oversized or malformed +# declaration still fails closed instead of silently paying for a slow full +# copy. +MAX_BUFFERED_FALLBACK_BYTES = 512 * 1024 * 1024 COPY_CHUNK_BYTES = 1024 * 1024 # linux/fs.h: #define FICLONE _IOW(0x94, 9, int) From ad5ff42804d9ec15a7b3d5cc9ab6648f013f435e Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 15:45:31 +0000 Subject: [PATCH 06/61] feat(lbug): add adaptive buffer-pool size hint Adds the sizing lever without changing behavior yet: a module-scoped buffer-pool size hint plus estimateBufferPool(graphElementCount), read by resolveBufferManagerSize with precedence env-override > clamp(hint, 64MiB, default) > default. The hint can only shrink the pool from the default (clamped to [floor, default]), so the 2GiB/80%-RAM cap and the GITNEXUS_LBUG_BUFFER_POOL_SIZE escape hatch (incl. 0) are preserved. With no hint set, resolveBufferManagerSize returns exactly what it did before. Motivation: LadybugDB eagerly commits the buffer pool at DB open, so the fixed min(2GiB,80%RAM) pool adds a measured ~2.8s to every analyze even on a 3-file repo (dominant on Windows). Sizing the pool to the graph lets small repos use the fast 64MiB floor while large repos keep the cap. Co-Authored-By: Claude Fable 5 --- gitnexus/src/core/lbug/lbug-config.ts | 54 ++++++++++++++-- gitnexus/test/unit/lbug-config-wal.test.ts | 71 +++++++++++++++++++++- 2 files changed, 120 insertions(+), 5 deletions(-) diff --git a/gitnexus/src/core/lbug/lbug-config.ts b/gitnexus/src/core/lbug/lbug-config.ts index 540fe530a..f32a12898 100644 --- a/gitnexus/src/core/lbug/lbug-config.ts +++ b/gitnexus/src/core/lbug/lbug-config.ts @@ -333,16 +333,62 @@ const parseBufferPoolSize = (raw: string | undefined): number | undefined => { const defaultBufferPoolSize = (): number => Math.min(DEFAULT_BUFFER_POOL_CAP, Math.max(BUFFER_POOL_FLOOR, Math.floor(os.totalmem() * 0.8))); +/** Clamp a requested pool size to [floor, default] — never above the 2 GiB / 80%-RAM default. */ +const clampBufferPool = (bytes: number): number => + Math.min(defaultBufferPoolSize(), Math.max(BUFFER_POOL_FLOOR, Math.floor(bytes))); + +/** + * Buffer-pool bytes to provision per graph element (node + relationship). + * + * LadybugDB eagerly commits the buffer pool at DB open, so an oversized pool + * costs a fixed penalty on every analyze regardless of repo size (measured: + * ~2.8 s extra for a pool ≥128 MiB vs the 64 MiB floor, on a 3-file repo). + * The pool is only a page cache over the on-disk index, and the index scales + * with node/edge count — so a per-element budget lets a small repo run with a + * small, fast-to-commit pool while a large repo still reaches the 2 GiB cap. + * Kept deliberately generous (holds the working set without thrash); tuned by + * `bench/buffer-pool/measure.mjs` against a large-repo analyze. + */ +const POOL_BYTES_PER_ELEMENT = 4 * 1024; + +/** + * Size the buffer pool to an estimated graph size (node + relationship count), + * clamped to [BUFFER_POOL_FLOOR, defaultBufferPoolSize()]. The estimate can + * only *shrink* the pool from the default — it never exceeds the 2 GiB / 80%-RAM + * cap — so a large repo is never under the default it gets today. + */ +export const estimateBufferPool = (graphElementCount: number): number => + clampBufferPool(graphElementCount * POOL_BYTES_PER_ELEMENT); + +/** + * Optional per-run buffer-pool size hint (bytes). The analyze orchestrator sets + * it once the graph size is known (after the pipeline, before the DB open) so a + * small repo opens with a small pool, and clears it at run end. Non-analyze + * opens (MCP serve, `native-check` `:memory:`) never set it and keep the default. + */ +let bufferPoolSizeHint: number | undefined; + +/** Set (bytes) or clear (`undefined`) the per-run buffer-pool size hint. */ +export const setBufferPoolSizeHint = (bytes: number | undefined): void => { + bufferPoolSizeHint = bytes; +}; + /** * Resolve the `bufferManagerSize` passed to every `new lbug.Database(...)`. - * `GITNEXUS_LBUG_BUFFER_POOL_SIZE` (bytes) overrides the default; `0` is a + * `GITNEXUS_LBUG_BUFFER_POOL_SIZE` (bytes) overrides everything; `0` is a * deliberate escape hatch that restores LadybugDB's native unbounded - * 80%-of-RAM default. Resolved at call time (not module load) so tests can - * stub the env var and `os.totalmem`. + * 80%-of-RAM default. With no env override, a per-run `setBufferPoolSizeHint` + * (clamped to [floor, default]) sizes the pool to the repo; otherwise the + * default. Resolved at call time (not module load) so tests can stub the env + * var, the hint, and `os.totalmem`. */ const resolveBufferManagerSize = (): number => { const raw = process.env.GITNEXUS_LBUG_BUFFER_POOL_SIZE; - if (raw === undefined) return defaultBufferPoolSize(); + if (raw === undefined) { + return bufferPoolSizeHint !== undefined + ? clampBufferPool(bufferPoolSizeHint) + : defaultBufferPoolSize(); + } const parsed = parseBufferPoolSize(raw); if (parsed !== undefined) return parsed; // Non-empty but unparseable input: warn the operator and fall back — diff --git a/gitnexus/test/unit/lbug-config-wal.test.ts b/gitnexus/test/unit/lbug-config-wal.test.ts index a9912b9ba..4b6d8fe61 100644 --- a/gitnexus/test/unit/lbug-config-wal.test.ts +++ b/gitnexus/test/unit/lbug-config-wal.test.ts @@ -1,9 +1,11 @@ import os from 'os'; -import { describe, expect, it, vi } from 'vitest'; +import { afterEach, describe, expect, it, vi } from 'vitest'; import { createLbugDatabase, + estimateBufferPool, isLbugCheckpointIoError, isWalCorruptionError, + setBufferPoolSizeHint, } from '../../src/core/lbug/lbug-config.js'; import { _captureLogger } from '../../src/core/logger.js'; @@ -252,6 +254,73 @@ describe('createLbugDatabase buffer pool size (#2557)', () => { }); }); +describe('adaptive buffer pool hint', () => { + const GiB = 1024 * 1024 * 1024; + const MiB = 1024 * 1024; + const bufferPoolArg = (Database: ReturnType): unknown => Database.mock.calls[0][1]; + + afterEach(() => setBufferPoolSizeHint(undefined)); + + describe('estimateBufferPool', () => { + it.each([ + ['tiny graph clamps up to the 64 MiB floor', 41, 64 * MiB], + ['mid graph scales linearly (100k elements * 4 KiB = 400 MiB)', 100_000, 100_000 * 4 * 1024], + ['huge graph caps at the 2 GiB / 80%-RAM default', 10_000_000, 2 * GiB], + ])('%s', (_label, elements, expected) => { + const totalmemSpy = vi.spyOn(os, 'totalmem').mockReturnValue(32 * GiB); + try { + expect(estimateBufferPool(elements)).toBe(expected); + } finally { + totalmemSpy.mockRestore(); + } + }); + }); + + it.each([ + ['a hint within range passes through', 128 * MiB, 128 * MiB], + ['a hint below the floor clamps up to 64 MiB', 1 * MiB, 64 * MiB], + ['a hint above the default clamps down to the 2 GiB cap', 8 * GiB, 2 * GiB], + ])( + 'createLbugDatabase uses the clamped hint when no env override is set: %s', + (_label, hint, expected) => { + const totalmemSpy = vi.spyOn(os, 'totalmem').mockReturnValue(32 * GiB); + try { + setBufferPoolSizeHint(hint); + const Database = vi.fn(function (this: any) {}); + createLbugDatabase({ Database } as any, '/tmp/lbug-hint'); + expect(bufferPoolArg(Database)).toBe(expected); + } finally { + totalmemSpy.mockRestore(); + } + }, + ); + + it('env override wins over the hint (including 0 = native default)', () => { + try { + setBufferPoolSizeHint(128 * MiB); + vi.stubEnv('GITNEXUS_LBUG_BUFFER_POOL_SIZE', '0'); + const Database = vi.fn(function (this: any) {}); + createLbugDatabase({ Database } as any, '/tmp/lbug-hint-env'); + expect(bufferPoolArg(Database)).toBe(0); + } finally { + vi.unstubAllEnvs(); + } + }); + + it('falls back to the default when the hint is cleared', () => { + const totalmemSpy = vi.spyOn(os, 'totalmem').mockReturnValue(32 * GiB); + try { + setBufferPoolSizeHint(128 * MiB); + setBufferPoolSizeHint(undefined); + const Database = vi.fn(function (this: any) {}); + createLbugDatabase({ Database } as any, '/tmp/lbug-hint-cleared'); + expect(bufferPoolArg(Database)).toBe(2 * GiB); + } finally { + totalmemSpy.mockRestore(); + } + }); +}); + // ─── Finding 8: strict + permissive checkpoint IO matchers ───────────────── describe('isLbugCheckpointIoError', () => { it.each([ From 087b44a8658f6332bbd113f90de0b94554511802 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 15:51:54 +0000 Subject: [PATCH 07/61] feat(analyze): size the buffer pool to the graph before the DB open runFullAnalysis now sets the buffer-pool size hint from the built graph's node+relationship count (after the pipeline, before initLbug), and clears it at the top of each run so a prior run's size can't leak into a pre-pipeline open. A small repo opens with the fast 64MiB floor instead of eagerly committing the full 2GiB pool. Measured on a 3-file repo (this box): analyze drops 4.78s -> 1.9s for both fresh and incremental, matching a forced 64MiB pool; the 2GiB path still reproduces the old 4.78s. This roughly halves the skills-e2e Idempotency hook (two analyze passes) that was timing out on Windows CI. Co-Authored-By: Claude Fable 5 --- gitnexus/src/core/run-analyze.ts | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index 635b255a9..812dc5757 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -34,6 +34,7 @@ import { LbugWipeError, DELETE_FILES_CHUNK_SIZE, } from './lbug/lbug-adapter.js'; +import { estimateBufferPool, setBufferPoolSizeHint } from './lbug/lbug-config.js'; import { escapeCypherString } from './lbug/cypher-escape.js'; import { buildSearchIndexesOrDegrade, @@ -645,6 +646,11 @@ export async function runFullAnalysis( // and are shared across branches (#2106 KTD7). const { storagePath } = getStoragePaths(repoPath); + // Start each analyze with a clean buffer-pool hint: any pre-pipeline DB open + // (e.g. the embeddings-cache open) falls back to the default until the hint is + // set from the built graph below, so a prior run's size can't leak in. + setBufferPoolSizeHint(undefined); + // Clean up stale KuzuDB files from before the LadybugDB migration. const kuzuResult = await cleanupOldKuzuFiles(storagePath); if (kuzuResult.found && kuzuResult.needsReindex) { @@ -1374,6 +1380,15 @@ export async function runFullAnalysis( await wipeLbugDbFiles(lbugPath); } + // Size the buffer pool to the graph just built by the pipeline (a page cache + // over the on-disk index, which scales with node/edge count). A small repo + // opens with the fast 64 MiB floor instead of eagerly committing the full + // 2 GiB default; a large repo still reaches the cap. env override / no-hint + // paths are unchanged. See resolveBufferManagerSize / estimateBufferPool. + setBufferPoolSizeHint( + estimateBufferPool(pipelineResult.graph.nodeCount + pipelineResult.graph.relationshipCount), + ); + await initLbug(lbugPath); // Manual WAL checkpoint driver (#1741): periodically drain the WAL From 7f03a4ed2db51cb19e6c45e700a79c1aa505b711 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 15:57:38 +0000 Subject: [PATCH 08/61] docs(lbug): correct the POOL_BYTES_PER_ELEMENT tuning note MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The factor is validated by timing a full `analyze --force` of a large repo, not a build-free bench (the pool is a native eager allocation). Benchmark (GitNexus self, 101k graph elements): adaptive 414MiB pool = 35.3s vs forced 2GiB = 50.8s vs forced 64MiB = 26.5s — the adaptive pool is 31% faster than the old 2GiB default even on a large repo (the eager commit dominates), with no under-sizing thrash. 4KiB/element kept. Co-Authored-By: Claude Fable 5 --- gitnexus/src/core/lbug/lbug-config.ts | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/gitnexus/src/core/lbug/lbug-config.ts b/gitnexus/src/core/lbug/lbug-config.ts index f32a12898..06ee6b9e0 100644 --- a/gitnexus/src/core/lbug/lbug-config.ts +++ b/gitnexus/src/core/lbug/lbug-config.ts @@ -346,8 +346,11 @@ const clampBufferPool = (bytes: number): number => * The pool is only a page cache over the on-disk index, and the index scales * with node/edge count — so a per-element budget lets a small repo run with a * small, fast-to-commit pool while a large repo still reaches the 2 GiB cap. - * Kept deliberately generous (holds the working set without thrash); tuned by - * `bench/buffer-pool/measure.mjs` against a large-repo analyze. + * Kept deliberately generous (holds the working set without thrash): tuned by + * timing a full `analyze --force` of a large repo at this factor vs a forced + * 2 GiB pool and confirming no wall-time regression (the pool is a native + * eager allocation, so it is measured with a real analyze, not a build-free + * bench — see the emit-path `COPY` timing note in bench/emit-persistence). */ const POOL_BYTES_PER_ELEMENT = 4 * 1024; From 05dadb79504d20b326c99a33f9f04b69b05e8d9d Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 16:13:26 +0000 Subject: [PATCH 09/61] fix(lbug): raise the adaptive pool floor to a COPY-safe 256 MiB MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 64 MiB floor was too small: LadybugDB's bulk COPY needs working buffer-pool memory that scales with the repo, so a 64 MiB pool fails with "buffer pool is full and no memory could be freed" on any non-trivial repo (empirically: the 6-file skills-e2e idempotency fixture needs >=128 MiB; the ~1800-file GitNexus checkout needs >=256 MiB). Introduce a distinct ADAPTIVE_POOL_FLOOR (256 MiB) for the hint clamp, kept separate from BUFFER_POOL_FLOOR (64 MiB), which still guards defaultBufferPoolSize on tiny-RAM machines; the hint is still clamped up to the machine default so it can never over-commit. This keeps the change as what it actually is — a large-repo optimization: GitNexus full analyze is 51s (2 GiB) -> 35s (adaptive ~414 MiB). Small repos now open COPY-safely at 256 MiB instead of the 2 GiB default (same wall time on Linux, where commit is lazy; the eager commit is far cheaper than 2 GiB on Windows). The reframed comments drop the earlier unrepresentative "3-file repo / 64 MiB fast" claim. Co-Authored-By: Claude Fable 5 --- gitnexus/src/core/lbug/lbug-config.ts | 56 ++++++++++++++-------- gitnexus/src/core/run-analyze.ts | 9 ++-- gitnexus/test/unit/lbug-config-wal.test.ts | 7 +-- 3 files changed, 46 insertions(+), 26 deletions(-) diff --git a/gitnexus/src/core/lbug/lbug-config.ts b/gitnexus/src/core/lbug/lbug-config.ts index 06ee6b9e0..7b4ba4c65 100644 --- a/gitnexus/src/core/lbug/lbug-config.ts +++ b/gitnexus/src/core/lbug/lbug-config.ts @@ -321,6 +321,16 @@ const resolveCheckpointThreshold = (): number => { const DEFAULT_BUFFER_POOL_CAP = 2 * 1024 * 1024 * 1024; const BUFFER_POOL_FLOOR = 64 * 1024 * 1024; +// COPY-safety floor for the adaptive hint (below). LadybugDB's bulk COPY needs +// working buffer-pool memory that scales with the repo: a 64 MiB pool fails +// ("buffer pool is full and no memory could be freed") on any non-trivial repo, +// and even the ~1800-file GitNexus checkout needs ≥256 MiB. So the adaptive +// size never drops a repo below this — a distinct, higher floor than +// BUFFER_POOL_FLOOR, which only guards defaultBufferPoolSize on tiny-RAM +// machines. It is still clamped up to defaultBufferPoolSize, so a machine whose +// default is below this floor keeps its default rather than over-committing. +const ADAPTIVE_POOL_FLOOR = 256 * 1024 * 1024; + const parseBufferPoolSize = (raw: string | undefined): number | undefined => { if (raw === undefined) return undefined; const normalized = raw.trim(); @@ -333,41 +343,49 @@ const parseBufferPoolSize = (raw: string | undefined): number | undefined => { const defaultBufferPoolSize = (): number => Math.min(DEFAULT_BUFFER_POOL_CAP, Math.max(BUFFER_POOL_FLOOR, Math.floor(os.totalmem() * 0.8))); -/** Clamp a requested pool size to [floor, default] — never above the 2 GiB / 80%-RAM default. */ +/** + * Clamp an adaptive pool request to [ADAPTIVE_POOL_FLOOR, default]. The lower + * bound keeps LadybugDB's COPY viable; the upper bound (defaultBufferPoolSize) + * means the hint can only shrink the pool from today's default and can never + * exceed the 2 GiB / 80%-RAM cap — and on a machine whose default is below the + * COPY floor, the default wins, so the pool is never over-committed. + */ const clampBufferPool = (bytes: number): number => - Math.min(defaultBufferPoolSize(), Math.max(BUFFER_POOL_FLOOR, Math.floor(bytes))); + Math.min(defaultBufferPoolSize(), Math.max(ADAPTIVE_POOL_FLOOR, Math.floor(bytes))); /** * Buffer-pool bytes to provision per graph element (node + relationship). * - * LadybugDB eagerly commits the buffer pool at DB open, so an oversized pool - * costs a fixed penalty on every analyze regardless of repo size (measured: - * ~2.8 s extra for a pool ≥128 MiB vs the 64 MiB floor, on a 3-file repo). - * The pool is only a page cache over the on-disk index, and the index scales - * with node/edge count — so a per-element budget lets a small repo run with a - * small, fast-to-commit pool while a large repo still reaches the 2 GiB cap. - * Kept deliberately generous (holds the working set without thrash): tuned by - * timing a full `analyze --force` of a large repo at this factor vs a forced - * 2 GiB pool and confirming no wall-time regression (the pool is a native - * eager allocation, so it is measured with a real analyze, not a build-free - * bench — see the emit-path `COPY` timing note in bench/emit-persistence). + * The fixed 2 GiB default is far larger than most repos' working set, and + * LadybugDB eagerly commits the pool at DB open — measured: a full + * `analyze --force` of the GitNexus checkout takes ~51 s with the 2 GiB pool + * vs ~35 s with the ~414 MiB this factor yields (31% faster; the oversized + * pool's commit dominates). The pool is a page cache over the on-disk index, + * which scales with node/edge count, so a per-element budget sizes it to the + * repo. Kept generous so the whole index stays resident (no COPY thrash) and + * always clamped to at least ADAPTIVE_POOL_FLOOR; tuned by timing a real + * large-repo `analyze --force` at this factor vs a forced 2 GiB pool (the pool + * is a native eager allocation, measured with a real analyze, not a build-free + * bench — see the emit-path COPY timing note in bench/emit-persistence). */ const POOL_BYTES_PER_ELEMENT = 4 * 1024; /** * Size the buffer pool to an estimated graph size (node + relationship count), - * clamped to [BUFFER_POOL_FLOOR, defaultBufferPoolSize()]. The estimate can - * only *shrink* the pool from the default — it never exceeds the 2 GiB / 80%-RAM - * cap — so a large repo is never under the default it gets today. + * clamped to [ADAPTIVE_POOL_FLOOR, defaultBufferPoolSize()]. The estimate can + * only *shrink* the pool from the default — never above the 2 GiB / 80%-RAM cap, + * never below the COPY-safety floor — so no repo is under-sized or gets more + * than the default it would have today. */ export const estimateBufferPool = (graphElementCount: number): number => clampBufferPool(graphElementCount * POOL_BYTES_PER_ELEMENT); /** * Optional per-run buffer-pool size hint (bytes). The analyze orchestrator sets - * it once the graph size is known (after the pipeline, before the DB open) so a - * small repo opens with a small pool, and clears it at run end. Non-analyze - * opens (MCP serve, `native-check` `:memory:`) never set it and keep the default. + * it once the graph size is known (after the pipeline, before the DB open) so + * the pool is sized to the repo instead of the fixed 2 GiB default, and clears + * it at run end. Non-analyze opens (MCP serve, `native-check` `:memory:`) never + * set it and keep the default. */ let bufferPoolSizeHint: number | undefined; diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index 812dc5757..ef77ba3d9 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -1381,10 +1381,11 @@ export async function runFullAnalysis( } // Size the buffer pool to the graph just built by the pipeline (a page cache - // over the on-disk index, which scales with node/edge count). A small repo - // opens with the fast 64 MiB floor instead of eagerly committing the full - // 2 GiB default; a large repo still reaches the cap. env override / no-hint - // paths are unchanged. See resolveBufferManagerSize / estimateBufferPool. + // over the on-disk index, which scales with node/edge count) instead of the + // fixed 2 GiB default, whose eager commit dominates large-repo analyze. The + // size is clamped to [COPY-safety floor, default], so it only ever shrinks + // the pool; env override / no-hint paths are unchanged. See + // resolveBufferManagerSize / estimateBufferPool. setBufferPoolSizeHint( estimateBufferPool(pipelineResult.graph.nodeCount + pipelineResult.graph.relationshipCount), ); diff --git a/gitnexus/test/unit/lbug-config-wal.test.ts b/gitnexus/test/unit/lbug-config-wal.test.ts index 4b6d8fe61..94395541a 100644 --- a/gitnexus/test/unit/lbug-config-wal.test.ts +++ b/gitnexus/test/unit/lbug-config-wal.test.ts @@ -263,7 +263,8 @@ describe('adaptive buffer pool hint', () => { describe('estimateBufferPool', () => { it.each([ - ['tiny graph clamps up to the 64 MiB floor', 41, 64 * MiB], + ['tiny graph clamps up to the 256 MiB COPY-safety floor', 41, 256 * MiB], + ['a graph under the floor still clamps up to 256 MiB', 40_000, 256 * MiB], ['mid graph scales linearly (100k elements * 4 KiB = 400 MiB)', 100_000, 100_000 * 4 * 1024], ['huge graph caps at the 2 GiB / 80%-RAM default', 10_000_000, 2 * GiB], ])('%s', (_label, elements, expected) => { @@ -277,8 +278,8 @@ describe('adaptive buffer pool hint', () => { }); it.each([ - ['a hint within range passes through', 128 * MiB, 128 * MiB], - ['a hint below the floor clamps up to 64 MiB', 1 * MiB, 64 * MiB], + ['a hint within range passes through', 512 * MiB, 512 * MiB], + ['a hint below the COPY-safety floor clamps up to 256 MiB', 100 * MiB, 256 * MiB], ['a hint above the default clamps down to the 2 GiB cap', 8 * GiB, 2 * GiB], ])( 'createLbugDatabase uses the clamped hint when no env override is set: %s', From 8fd1f8a8d839c55337f203f2d6f0ca6ed85aed4b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 20 Jul 2026 19:54:54 +0100 Subject: [PATCH 10/61] test(skills-e2e): give the Idempotency setup hook the 120s budget its siblings use (#2583) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Idempotency beforeAll runs runSkillsCli (analyze --skills) twice, each capped at 45s, under a 90s hook budget — exactly 2x the per-call timeout, with no headroom for fixture creation and git init. On slow Windows CI runners the two analyzes plus setup exceed 90s and the hook times out ('Hook timed out in 90000ms'), failing the shard before the test's own status===null timeout tolerance can apply. Every other describe hook in this file already uses 120s; align this one. Co-authored-by: Claude Co-authored-by: Claude Fable 5 --- gitnexus/test/integration/skills-e2e.test.ts | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/gitnexus/test/integration/skills-e2e.test.ts b/gitnexus/test/integration/skills-e2e.test.ts index 77ecc922d..581497880 100644 --- a/gitnexus/test/integration/skills-e2e.test.ts +++ b/gitnexus/test/integration/skills-e2e.test.ts @@ -2371,7 +2371,13 @@ export function createEntry(level: string, msg: string) { }); result1 = runSkillsCli(tmpDir); result2 = runSkillsCli(tmpDir); - }, 90000); + // 120s to match the other describe hooks in this file. This hook runs + // runSkillsCli TWICE, each capped at 45s, so a 90s budget has no headroom + // over two worst-case analyzes plus fixture setup and git init — it times + // out the *hook* on slow Windows runners (the test below already tolerates + // an individual analyze hitting its own 45s timeout via status === null, + // but a hook timeout fails before that tolerance can apply). + }, 120000); afterAll(() => { fs.rmSync(tmpDir, { recursive: true, force: true }); From 9c87030edf88f5a5bfbff58ade979f49e7c86b8b Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Mon, 20 Jul 2026 20:15:31 +0000 Subject: [PATCH 11/61] chore(deps)(deps): bump @ladybugdb/core in /gitnexus Bumps [@ladybugdb/core](https://github.com/LadybugDB/ladybug) from 0.18.1 to 0.18.2. - [Release notes](https://github.com/LadybugDB/ladybug/releases) - [Commits](https://github.com/LadybugDB/ladybug/compare/v0.18.1...v0.18.2) --- updated-dependencies: - dependency-name: "@ladybugdb/core" dependency-version: 0.18.2 dependency-type: direct:production update-type: version-update:semver-patch ... Signed-off-by: dependabot[bot] --- gitnexus/package-lock.json | 46 +++++++++++++++++++------------------- 1 file changed, 23 insertions(+), 23 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 9aa2ef8da..fe1643b96 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -1254,9 +1254,9 @@ } }, "node_modules/@ladybugdb/core": { - "version": "0.18.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.18.1.tgz", - "integrity": "sha512-0c1kXDpdv7z/GB0oyFYnLEjLsXFwPHz1YD4wxtrk9hav8zJX5T1PHQMr+XRfdDI1NQjx4iNdbPQGGT7Bx/X2aw==", + "version": "0.18.2", + "resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.18.2.tgz", + "integrity": "sha512-222FjGciEO5Z+/MRQGU+b4IaGAjOgSQzj7fMpOuhMQN4F8nf654kuKRk1iybSiNy6XSw69hIJ0mKwUeBQ8y6Fg==", "hasInstallScript": true, "license": "MIT", "dependencies": { @@ -1265,17 +1265,17 @@ "node-addon-api": "^6.0.0" }, "optionalDependencies": { - "@ladybugdb/core-darwin-arm64": "0.18.1", - "@ladybugdb/core-darwin-x64": "0.18.1", - "@ladybugdb/core-linux-arm64": "0.18.1", - "@ladybugdb/core-linux-x64": "0.18.1", - "@ladybugdb/core-win32-x64": "0.18.1" + "@ladybugdb/core-darwin-arm64": "0.18.2", + "@ladybugdb/core-darwin-x64": "0.18.2", + "@ladybugdb/core-linux-arm64": "0.18.2", + "@ladybugdb/core-linux-x64": "0.18.2", + "@ladybugdb/core-win32-x64": "0.18.2" } }, "node_modules/@ladybugdb/core-darwin-arm64": { - "version": "0.18.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.18.1.tgz", - "integrity": "sha512-M5YZuAONRAv3awkr+cfaibn9Da+3pgDzRiek/JabWQuz48xgzW3Vh9yQH4s8Dq/bfQo6YTsaLIBRcUCCUzCtcg==", + "version": "0.18.2", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.18.2.tgz", + "integrity": "sha512-gAwxsdijBFTz4aZ9ITG6zdQw3lAki0eY33hNBLCfKXjKJLvW/8wCvgCVBglqfqBF5WyI7icFUt+wfy/Fbdfl5A==", "cpu": [ "arm64" ], @@ -1286,9 +1286,9 @@ ] }, "node_modules/@ladybugdb/core-darwin-x64": { - "version": "0.18.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.18.1.tgz", - "integrity": "sha512-kq+pyTskfCx++Mrbk7QssE/f/CpSuU50T8lhRtv4PaOKhC2Jf8/wAUOA17UxI594wAru3ERpqVBFUBWGcPk2ag==", + "version": "0.18.2", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.18.2.tgz", + "integrity": "sha512-oUjYLc1fW3ntCrO9te55PoPfvhFo8AKeNa/sU66fiQyEAZB7qZJxeHnnLgl/bLueTF2os3RSawq46ZftoD/9Eg==", "cpu": [ "x64" ], @@ -1299,9 +1299,9 @@ ] }, "node_modules/@ladybugdb/core-linux-arm64": { - "version": "0.18.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.18.1.tgz", - "integrity": "sha512-fu7ke1haa5rPINcQn0+kxQijZ0A8ZDWP9e+X8xcDH94RagDbPWwG8yFC890cGSdc/j7mTV+xkA/y/kVHpmVI6w==", + "version": "0.18.2", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.18.2.tgz", + "integrity": "sha512-UppokeTaPl9pN0xOsdMa+hmM68zbN2eKReTZhZNYM16qX0d2OlgbS/NlXi09Wdot+w5qvlZ9Q0iCCPfr7qvPaw==", "cpu": [ "arm64" ], @@ -1312,9 +1312,9 @@ ] }, "node_modules/@ladybugdb/core-linux-x64": { - "version": "0.18.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.18.1.tgz", - "integrity": "sha512-qp5HilHzDGuArfOyD+VyA7lVJ7IwQDKd81NZKKTmUwIAOJtdwqniYx6JZICPnlr36zFJBx/lGYoSsEzbC+TVdw==", + "version": "0.18.2", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.18.2.tgz", + "integrity": "sha512-GypOxCnP2ix/FWM8YhQ41aQYlS+ruoNMJp7pmaF5laJHhL/a+P/apywNTE+9N41Walsl+Emgg9xwxwTC93slow==", "cpu": [ "x64" ], @@ -1325,9 +1325,9 @@ ] }, "node_modules/@ladybugdb/core-win32-x64": { - "version": "0.18.1", - "resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.18.1.tgz", - "integrity": "sha512-vHcXr7Df2X1dbb5ORK+SBmNstd/3tApGFImbAnaWiTuLDFlAdfY8lbiSBSp3OgFjc0BB7F3GYUUdvgDRJjK3zA==", + "version": "0.18.2", + "resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.18.2.tgz", + "integrity": "sha512-hvFwjhTYdwG2sijapx963a27jP9mlLW0ZFv5Yfj19e0B3T/FqD9CaKULPy23mU2XlkISN2LUkY5qdgvCFztl/g==", "cpu": [ "x64" ], From 52b06b2642e447666f4b94160a1010bd8c839a4b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 20 Jul 2026 21:26:11 +0100 Subject: [PATCH 12/61] fix(eval): stop using --bare for arms that need Skill or MCP tools (#2584) --bare hard-disables the Skill tool and every mcp__* tool by Claude Code design (confirmed against the pinned 2.1.214 binary; --allowedTools cannot restore what --bare removes). Every workflow_bench arm except baseline_nomcp needs Skill and/or GitNexus MCP tools, so every one of those sessions has been silently unable to invoke gitnexus-plan/work/ review or the CE comparator skills -- the last skill-evolution run (gen 0) scored 0/3 resolution on both arms across every task with error_kind "skill-not-invoked", not because the candidate was bad but because the harness could never invoke either arm's skill at all. Only baseline_nomcp keeps --bare (it explicitly wants zero Skill/MCP access anyway). The rest drop --bare and rely on ANTHROPIC_API_KEY alone; the sandboxed HOME has no OAuth/keychain state to conflict with it, and there's no committed .claude/settings.json in this repo for dropping --bare to newly pick up. Outside --bare the built-in toolset defaults to everything (WebFetch, Task, subagents, ...), and --allowedTools only pre-approves within whatever's available -- it doesn't narrow it. Added --tools for non-bare sessions so the intended tool scope is still enforced instead of silently widening. Claude-Session: https://claude.ai/code/session_01Va5uu9Ar3e45QZ5xFsG4AZ Co-authored-by: Claude Sonnet 5 --- eval/tests/test_workflow_bench_sessions.py | 59 ++++++++++++++++++++++ eval/workflow_bench/runner.py | 9 +++- eval/workflow_bench/runner_sessions.py | 7 +++ 3 files changed, 74 insertions(+), 1 deletion(-) diff --git a/eval/tests/test_workflow_bench_sessions.py b/eval/tests/test_workflow_bench_sessions.py index c48ea10a1..e03104ed3 100644 --- a/eval/tests/test_workflow_bench_sessions.py +++ b/eval/tests/test_workflow_bench_sessions.py @@ -167,6 +167,56 @@ def test_run_claude_forwards_the_named_model_to_every_session(monkeypatch, tmp_p assert captured[captured.index("--model") + 1] == "claude-sonnet-4-20250514" +def test_run_claude_restricts_tools_via_tools_flag_outside_bare(monkeypatch, tmp_path): + # Outside --bare, the built-in toolset defaults to everything (subagents, + # WebFetch, Task, ...) and --allowedTools only pre-approves within that — + # it does not narrow it. --tools is what actually restricts the set, so a + # non-bare arm session must pass it or it silently gets a far wider + # toolset than intended. + captured: list[str] = [] + + def fake_run(command, **kwargs): + captured.extend(command) + return fake_cli_result(VALID_REPORT) + + monkeypatch.setattr(runner_sessions, "run_managed", fake_run) + runner.run_claude( + "task", + tmp_path, + claude_bin="claude", + timeout=5, + bare=False, + allowed_tools=["Read", "Edit", "Bash", "Skill"], + ) + tools_idx = captured.index("--tools") + assert captured[tools_idx + 1 : tools_idx + 5] == ["Read", "Edit", "Bash", "Skill"] + allowed_idx = captured.index("--allowedTools") + assert captured[allowed_idx + 1 : allowed_idx + 5] == ["Read", "Edit", "Bash", "Skill"] + + +def test_run_claude_omits_tools_flag_under_bare(monkeypatch, tmp_path): + # --bare already hard-restricts to Bash/Edit/Read on its own (a Claude + # Code design choice, not something --tools/--allowedTools can widen or + # narrow further), so bare sessions must not also pass --tools. + captured: list[str] = [] + + def fake_run(command, **kwargs): + captured.extend(command) + return fake_cli_result(VALID_REPORT) + + monkeypatch.setattr(runner_sessions, "run_managed", fake_run) + runner.run_claude( + "task", + tmp_path, + claude_bin="claude", + timeout=5, + bare=True, + allowed_tools=["Read", "Edit", "Bash", "Skill"], + ) + assert "--tools" not in captured + assert "--allowedTools" in captured + + @pytest.mark.parametrize( ("proc", "expected_kind"), [ @@ -282,6 +332,15 @@ def test_agent_tool_grants_are_exact_and_nomcp_has_no_graph_tools(monkeypatch, t assert captured[3]["mcp_config_json"] == '{"mcpServers":{}}' assert captured[3]["disallowed_tools"] == ["Skill", "mcp__gitnexus"] + # --bare hard-disables the Skill tool and every mcp__* tool regardless of + # --allowedTools (a Claude Code design choice, not something the harness + # can override) -- every arm here except baseline_nomcp needs Skill + # and/or MCP tools, so only baseline_nomcp may still run under --bare. + assert captured[0]["bare"] is False # workflow: planning session + assert captured[1]["bare"] is False # review + assert captured[2]["bare"] is False # workflow_direct + assert captured[3]["bare"] is True # baseline_nomcp + def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tmp_path): runtime = tmp_path / "gitnexus" diff --git a/eval/workflow_bench/runner.py b/eval/workflow_bench/runner.py index cd760eb98..2a180f43f 100644 --- a/eval/workflow_bench/runner.py +++ b/eval/workflow_bench/runner.py @@ -402,6 +402,13 @@ def run_arm( auth_token=args.auth_token, base_url=args.base_url, ) + # --bare hard-disables the Skill tool and every mcp__* tool — by Claude + # Code design, not a bug (--allowedTools can't restore what --bare + # removes). Every arm except baseline_nomcp needs Skill and/or MCP tools, + # so only baseline_nomcp can keep --bare's tighter isolation; the rest + # rely on ANTHROPIC_API_KEY alone (the sandboxed HOME has no OAuth/ + # keychain state to conflict with it). + bare = arm == "baseline_nomcp" common = { "claude_bin": sandbox.claude_bin, "timeout": args.timeout, @@ -412,7 +419,7 @@ def run_arm( read_only_paths=_evaluated_skill_roots(worktree, arm), ), "require_pid_namespace": True, - "bare": True, + "bare": bare, "settings_json": sandbox.settings_json, "strict_mcp_config": True, "mcp_config_json": sandbox_mcp_config(), diff --git a/eval/workflow_bench/runner_sessions.py b/eval/workflow_bench/runner_sessions.py index 60e92b6bc..53e7335ee 100644 --- a/eval/workflow_bench/runner_sessions.py +++ b/eval/workflow_bench/runner_sessions.py @@ -377,6 +377,13 @@ def run_claude( if strict_mcp_config: cmd += ["--strict-mcp-config", "--mcp-config", mcp_config_json or '{"mcpServers":{}}'] if allowed_tools: + # --bare's own hard-coded Bash/Edit/Read ceiling already scopes bare + # sessions; outside --bare the built-in toolset defaults to + # everything (subagents, WebFetch, Task, ...), so --tools is needed + # to actually restrict it — --allowedTools only pre-approves within + # whatever set is available, it does not narrow that set. + if not bare: + cmd += ["--tools", *allowed_tools] cmd += ["--allowedTools", *allowed_tools] if disable_slash_commands: cmd.append("--disable-slash-commands") From 9096f6924ca12ba33af2150288fbac4892306857 Mon Sep 17 00:00:00 2001 From: Shining Date: Tue, 21 Jul 2026 10:07:20 +0800 Subject: [PATCH 13/61] feat(spring): bind configuration consumers --- .../frameworks/spring/bean-candidates.ts | 65 +-- .../frameworks/spring/config-bindings.ts | 166 ++++++++ .../languages/java/analysis-features.ts | 12 + .../languages/java/capture-side-channel.ts | 29 +- .../core/ingestion/languages/java/captures.ts | 10 +- .../languages/java/scope-resolver.ts | 6 +- .../languages/java/spring-config-bindings.ts | 253 ++++++++++++ .../core/ingestion/pipeline-phases/index.ts | 1 + .../pipeline-phases/spring-config.ts | 374 ++++++++++++++++++ gitnexus/src/core/ingestion/pipeline.ts | 4 +- gitnexus/src/core/lbug/schema.ts | 1 + gitnexus/src/core/run-analyze.ts | 2 + .../java/com/example/ConfigConsumers.java | 20 + .../src/main/resources/application-dev.yml | 6 + .../src/main/resources/application.properties | 1 + .../src/main/java/com/example/Shadowed.java | 8 + .../src/main/java/com/example/Value.java | 5 + .../src/main/resources/application.properties | 1 + .../integration/spring-config-mcp.test.ts | 77 ++++ .../spring-config-pipeline.test.ts | 93 +++++ gitnexus/test/unit/analysis-features.test.ts | 8 +- .../ingestion/pipeline-phase-registry.test.ts | 1 + .../test/unit/spring-config-bindings.test.ts | 169 ++++++++ 23 files changed, 1278 insertions(+), 34 deletions(-) create mode 100644 gitnexus/src/core/ingestion/frameworks/spring/config-bindings.ts create mode 100644 gitnexus/src/core/ingestion/languages/java/analysis-features.ts create mode 100644 gitnexus/src/core/ingestion/languages/java/spring-config-bindings.ts create mode 100644 gitnexus/src/core/ingestion/pipeline-phases/spring-config.ts create mode 100644 gitnexus/test/fixtures/spring-config-app/src/main/java/com/example/ConfigConsumers.java create mode 100644 gitnexus/test/fixtures/spring-config-app/src/main/resources/application-dev.yml create mode 100644 gitnexus/test/fixtures/spring-config-app/src/main/resources/application.properties create mode 100644 gitnexus/test/fixtures/spring-config-shadow-app/src/main/java/com/example/Shadowed.java create mode 100644 gitnexus/test/fixtures/spring-config-shadow-app/src/main/java/com/example/Value.java create mode 100644 gitnexus/test/fixtures/spring-config-shadow-app/src/main/resources/application.properties create mode 100644 gitnexus/test/integration/spring-config-mcp.test.ts create mode 100644 gitnexus/test/integration/spring-config-pipeline.test.ts create mode 100644 gitnexus/test/unit/spring-config-bindings.test.ts diff --git a/gitnexus/src/core/ingestion/frameworks/spring/bean-candidates.ts b/gitnexus/src/core/ingestion/frameworks/spring/bean-candidates.ts index f18424598..5981a9e5a 100644 --- a/gitnexus/src/core/ingestion/frameworks/spring/bean-candidates.ts +++ b/gitnexus/src/core/ingestion/frameworks/spring/bean-candidates.ts @@ -59,6 +59,7 @@ export interface SpringBeanCandidateAdapter { } type OwnedTypeNamesByOwner = ReadonlyMap>; +type RecognizedAnnotationNames = { readonly has: (value: string) => boolean }; function simpleNameOf(def: SymbolDefinition): string | undefined { const qualifiedName = def.qualifiedName; @@ -152,7 +153,11 @@ function hasVisibleTypeBinding( return false; } -function wildcardImportTarget(parsed: ParsedFile, simpleName: string): string | undefined { +function wildcardImportTarget( + parsed: ParsedFile, + simpleName: string, + recognizedAnnotations: RecognizedAnnotationNames, +): string | undefined { const wildcardPackages = new Set( parsed.parsedImports .filter((entry) => entry.kind === 'wildcard') @@ -161,37 +166,40 @@ function wildcardImportTarget(parsed: ParsedFile, simpleName: string): string | if (wildcardPackages.size !== 1) return undefined; const [packageName] = wildcardPackages; const target = `${packageName}.${simpleName}`; - return SPRING_BEAN_STEREOTYPES.has(target) ? target : undefined; + return recognizedAnnotations.has(target) ? target : undefined; } -function resolveSpringAnnotation( - rawName: string, - parsed: ParsedFile, - enclosingScope: ScopeId | null, - indexes: ScopeResolutionIndexes, - ownedTypeNamesByOwner: OwnedTypeNamesByOwner, - isPackageVisibilityIncomplete: boolean, -): string | undefined { - if (rawName.includes('.')) { - return SPRING_BEAN_STEREOTYPES.has(rawName) ? rawName : undefined; - } +/** Build a scope-aware Spring annotation resolver shared by framework hooks. */ +export function createSpringAnnotationNameResolver(indexes: ScopeResolutionIndexes) { + const ownedTypeNamesByOwner = buildOwnedTypeNamesByOwner(indexes); + return ( + rawName: string, + parsed: ParsedFile, + enclosingScope: ScopeId | null, + recognizedAnnotations: RecognizedAnnotationNames, + isPackageVisibilityIncomplete: boolean, + ): string | undefined => { + if (rawName.includes('.')) { + return recognizedAnnotations.has(rawName) ? rawName : undefined; + } - if (hasLexicalTypeDeclaration(enclosingScope, rawName, indexes)) return undefined; - if (hasInheritedTypeDeclaration(enclosingScope, rawName, indexes, ownedTypeNamesByOwner)) { - return undefined; - } + if (hasLexicalTypeDeclaration(enclosingScope, rawName, indexes)) return undefined; + if (hasInheritedTypeDeclaration(enclosingScope, rawName, indexes, ownedTypeNamesByOwner)) { + return undefined; + } - const explicitImports = explicitImportTargets(parsed, rawName); - if (explicitImports.size > 0) { - if (explicitImports.size !== 1) return undefined; - const [imported] = explicitImports; - return SPRING_BEAN_STEREOTYPES.has(imported) ? imported : undefined; - } + const explicitImports = explicitImportTargets(parsed, rawName); + if (explicitImports.size > 0) { + if (explicitImports.size !== 1) return undefined; + const [imported] = explicitImports; + return recognizedAnnotations.has(imported) ? imported : undefined; + } - const wildcardTarget = wildcardImportTarget(parsed, rawName); - if (wildcardTarget === undefined || isPackageVisibilityIncomplete) return undefined; + const wildcardTarget = wildcardImportTarget(parsed, rawName, recognizedAnnotations); + if (wildcardTarget === undefined || isPackageVisibilityIncomplete) return undefined; - return hasVisibleTypeBinding(enclosingScope, rawName, indexes) ? undefined : wildcardTarget; + return hasVisibleTypeBinding(enclosingScope, rawName, indexes) ? undefined : wildcardTarget; + }; } /** Build a language hook that enriches Class nodes after scope resolution. */ @@ -202,7 +210,7 @@ export function createSpringBeanCandidateAttacher(adapter: SpringBeanCandidateAd nodeLookup: GraphNodeLookup, indexes: ScopeResolutionIndexes, ): void => { - const ownedTypeNamesByOwner = buildOwnedTypeNamesByOwner(indexes); + const resolveSpringAnnotation = createSpringAnnotationNameResolver(indexes); for (const parsed of parsedFiles) { for (const fact of adapter.getClassAnnotationFacts(parsed.filePath)) { const classScope = indexes.scopeTree.getScope(fact.classScopeId); @@ -221,8 +229,7 @@ export function createSpringBeanCandidateAttacher(adapter: SpringBeanCandidateAd rawName, parsed, classScope.parent, - indexes, - ownedTypeNamesByOwner, + SPRING_BEAN_STEREOTYPES, adapter.isPackageVisibilityIncomplete(parsed.filePath), ); if (annotation !== undefined) recognized.add(annotation); diff --git a/gitnexus/src/core/ingestion/frameworks/spring/config-bindings.ts b/gitnexus/src/core/ingestion/frameworks/spring/config-bindings.ts new file mode 100644 index 000000000..0baba4e88 --- /dev/null +++ b/gitnexus/src/core/ingestion/frameworks/spring/config-bindings.ts @@ -0,0 +1,166 @@ +import type { GraphNode } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../../../graph/types.js'; +import { generateId } from '../../../../lib/utils.js'; + +export const SPRING_CONFIG_DESCRIPTION = 'Spring configuration property'; + +export interface SpringValueConsumer { + readonly kind: 'value'; + readonly fieldName: string; + readonly line: number; + readonly keys: readonly string[]; +} + +export interface SpringConfigurationPropertiesConsumer { + readonly kind: 'configuration-properties'; + readonly className: string; + readonly line: number; + readonly prefix: string; +} + +export type SpringConfigConsumer = SpringValueConsumer | SpringConfigurationPropertiesConsumer; + +export interface SpringConfigConsumerBatch { + readonly filePath: string; + readonly consumers: readonly SpringConfigConsumer[]; +} + +function closestNode( + candidates: readonly GraphNode[], + filePath: string, + name: string, + line: number, +): GraphNode | undefined { + return candidates + .filter((node) => node.properties.filePath === filePath && node.properties.name === name) + .sort( + (left, right) => + Math.abs(Number(left.properties.startLine ?? 0) - line) - + Math.abs(Number(right.properties.startLine ?? 0) - line), + )[0]; +} + +function markUnresolved(node: GraphNode, key: string): void { + const marker = `Spring config unresolved: ${key}`; + const existing = + typeof node.properties.description === 'string' ? node.properties.description : ''; + if (existing.includes(marker)) return; + node.properties.description = existing.length > 0 ? `${existing}; ${marker}` : marker; +} + +function relaxedName(value: string): string { + return value.toLowerCase().replace(/[-_.]/g, ''); +} + +function isSpringConfigNode(node: GraphNode): boolean { + return ( + node.label === 'Property' && + typeof node.properties.description === 'string' && + node.properties.description.startsWith(SPRING_CONFIG_DESCRIPTION) + ); +} + +/** + * Attach normalized, language-provider-produced Spring consumers to config + * keys already present in the shared graph. + */ +export function bindSpringConfigConsumers( + graph: KnowledgeGraph, + batches: readonly SpringConfigConsumerBatch[], +): void { + if (batches.length === 0) return; + + const configNodes: GraphNode[] = []; + const propertyNodes: GraphNode[] = []; + const classNodes: GraphNode[] = []; + for (const node of graph.iterNodes()) { + if (isSpringConfigNode(node)) configNodes.push(node); + else if (node.label === 'Property') propertyNodes.push(node); + else if (node.label === 'Class' || node.label === 'Record') classNodes.push(node); + } + + const keyNodes = new Map(); + for (const node of configNodes) { + const key = String(node.properties.name); + const bucket = keyNodes.get(key) ?? []; + bucket.push(node); + keyNodes.set(key, bucket); + } + + const propertiesByOwner = new Map(); + for (const rel of graph.iterRelationshipsByType('HAS_PROPERTY')) { + const property = graph.getNode(rel.targetId); + if (property?.label !== 'Property' || isSpringConfigNode(property)) continue; + const members = propertiesByOwner.get(rel.sourceId) ?? []; + members.push(property); + propertiesByOwner.set(rel.sourceId, members); + } + + const addBinding = ( + source: GraphNode, + target: GraphNode, + reason: string, + confidence: number, + ): void => { + const edgeId = generateId('USES', `${source.id}->${target.id}:${reason}`); + graph.addRelationship({ + id: edgeId, + sourceId: source.id, + targetId: target.id, + type: 'USES', + confidence, + reason, + }); + }; + + for (const { filePath, consumers } of batches) { + for (const consumer of consumers) { + if (consumer.kind === 'value') { + const field = closestNode(propertyNodes, filePath, consumer.fieldName, consumer.line); + if (field === undefined) continue; + for (const key of consumer.keys) { + const matches = keyNodes.get(key) ?? []; + if (matches.length === 0) { + markUnresolved(field, key); + continue; + } + for (const match of matches) { + addBinding(field, match, `spring-config:@Value ${key}`, 1); + } + } + continue; + } + + const owner = closestNode(classNodes, filePath, consumer.className, consumer.line); + if (owner === undefined) continue; + const prefix = `${consumer.prefix}.`; + const matches = configNodes.filter((node) => { + const key = String(node.properties.name); + return key === consumer.prefix || key.startsWith(prefix); + }); + if (matches.length === 0) { + markUnresolved(owner, consumer.prefix); + continue; + } + for (const match of matches) { + addBinding(owner, match, `spring-config:@ConfigurationProperties ${consumer.prefix}`, 0.95); + } + + for (const field of propertiesByOwner.get(owner.id) ?? []) { + const fieldName = relaxedName(String(field.properties.name)); + for (const match of matches) { + const key = String(match.properties.name); + const suffix = key === consumer.prefix ? '' : key.slice(prefix.length); + const firstSegment = suffix.split(/[.\[]/, 1)[0]; + if (firstSegment.length === 0 || relaxedName(firstSegment) !== fieldName) continue; + addBinding( + field, + match, + `spring-config:@ConfigurationProperties field ${consumer.prefix}`, + 0.95, + ); + } + } + } + } +} diff --git a/gitnexus/src/core/ingestion/languages/java/analysis-features.ts b/gitnexus/src/core/ingestion/languages/java/analysis-features.ts new file mode 100644 index 000000000..55dee7c4b --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/analysis-features.ts @@ -0,0 +1,12 @@ +import type { AnalysisFeatureDescriptor } from '../../../analysis-features.js'; + +/** Durable completeness contract for Java Spring configuration bindings. */ +export const SPRING_CONFIG_BINDINGS_FEATURE: AnalysisFeatureDescriptor = { + id: 'spring.config-bindings', + version: 1, + // Annotation presence is content-derived, so the conservative upgrade gate + // is every Java repository. This guarantees the first analyzer version that + // ships bindings performs a full rebuild even when config files are absent + // (missing @Value placeholders still need their unresolved marker). + appliesTo: (filePaths) => filePaths.some((filePath) => filePath.toLowerCase().endsWith('.java')), +}; diff --git a/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts index 8cc55a382..348c91653 100644 --- a/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts +++ b/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts @@ -9,6 +9,7 @@ import { type JvmPackageFact, } from '../jvm/package-facts.js'; import { getJavaPackageFact, setJavaPackageFact } from './package-facts.js'; +import type { JavaSpringConfigConsumerFact } from './spring-config-bindings.js'; export type JavaClassAnnotationFact = ClassAnnotationFact; @@ -16,13 +17,16 @@ export interface JavaCaptureSideChannel { readonly kind: 'java'; readonly packageFact: JvmPackageFact; readonly classAnnotations: readonly JavaClassAnnotationFact[]; + readonly springConfigConsumers?: readonly JavaSpringConfigConsumerFact[]; } const classAnnotations = createClassAnnotationFactStore(); +const springConfigConsumers = new Map(); /** Clear facts retained by a prior workspace pass in a long-lived process. */ export function clearJavaClassAnnotationFacts(): void { classAnnotations.clear(); + springConfigConsumers.clear(); } /** Store the annotation syntax collected by Java's existing scope-query traversal. */ @@ -33,17 +37,35 @@ export function setJavaClassAnnotationFacts( classAnnotations.set(filePath, facts); } +export function setJavaSpringConfigConsumerFacts( + filePath: string, + facts: readonly JavaSpringConfigConsumerFact[], +): void { + if (facts.length === 0) springConfigConsumers.delete(filePath); + else springConfigConsumers.set(filePath, facts); +} + +export function getJavaSpringConfigConsumerFacts( + filePath: string, +): readonly JavaSpringConfigConsumerFact[] { + return springConfigConsumers.get(filePath) ?? []; +} + /** Snapshot worker-local Java annotation facts for ParsedFile serialization. */ export function collectJavaCaptureSideChannel( filePath: string, ): JavaCaptureSideChannel | undefined { const facts = classAnnotations.get(filePath); + const configConsumers = springConfigConsumers.get(filePath) ?? []; const packageFact = getJavaPackageFact(filePath); - if (facts.length === 0 && packageFact === undefined) return undefined; + if (facts.length === 0 && configConsumers.length === 0 && packageFact === undefined) { + return undefined; + } return { kind: 'java', packageFact: packageFact ?? UNKNOWN_JVM_PACKAGE_FACT, classAnnotations: facts, + ...(configConsumers.length > 0 ? { springConfigConsumers: configConsumers } : {}), }; } @@ -62,10 +84,15 @@ export function applyJavaCaptureSideChannel(parsed: ParsedFile): void { !Array.isArray(data.classAnnotations) ) { setJavaClassAnnotationFacts(parsed.filePath, []); + setJavaSpringConfigConsumerFacts(parsed.filePath, []); setJavaPackageFact(parsed.filePath, UNKNOWN_JVM_PACKAGE_FACT); return; } setJavaClassAnnotationFacts(parsed.filePath, data.classAnnotations); + setJavaSpringConfigConsumerFacts( + parsed.filePath, + Array.isArray(data.springConfigConsumers) ? data.springConfigConsumers : [], + ); setJavaPackageFact( parsed.filePath, isJvmPackageFact(data.packageFact) ? data.packageFact : UNKNOWN_JVM_PACKAGE_FACT, diff --git a/gitnexus/src/core/ingestion/languages/java/captures.ts b/gitnexus/src/core/ingestion/languages/java/captures.ts index 542d09d38..c4da21212 100644 --- a/gitnexus/src/core/ingestion/languages/java/captures.ts +++ b/gitnexus/src/core/ingestion/languages/java/captures.ts @@ -32,9 +32,13 @@ import { getJavaParser, getJavaScopeQuery } from './query.js'; import { recordCacheHit, recordCacheMiss } from './cache-stats.js'; import { getTreeSitterBufferSize } from '../../constants.js'; import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js'; -import { setJavaClassAnnotationFacts } from './capture-side-channel.js'; +import { + setJavaClassAnnotationFacts, + setJavaSpringConfigConsumerFacts, +} from './capture-side-channel.js'; import { captureJavaPackageFact } from './package-facts.js'; import { synthesizeCallableFlowCaptures } from '../../utils/callable-flow-captures.js'; +import { captureJavaSpringConfigConsumerFacts } from './spring-config-bindings.js'; /** Declaration anchors that carry function-like arity metadata. */ const FUNCTION_DECL_TAGS = ['@declaration.method', '@declaration.constructor'] as const; @@ -257,6 +261,10 @@ export function emitJavaScopeCaptures( } setJavaClassAnnotationFacts(filePath, materializeClassAnnotationFacts(classAnnotations)); + setJavaSpringConfigConsumerFacts( + filePath, + captureJavaSpringConfigConsumerFacts(tree.rootNode, filePath), + ); return [ ...resolveVarTypeBindings(out), diff --git a/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts index fd47fb69e..52266456b 100644 --- a/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts @@ -30,6 +30,7 @@ import { } from './index.js'; import { populateJavaPackageSiblings } from './package-siblings.js'; import { attachSpringBeanCandidateMetadata } from './spring-bean-metadata.js'; +import { attachJavaSpringConfigBindings } from './spring-config-bindings.js'; import { applyJavaCaptureSideChannel, clearJavaClassAnnotationFacts, @@ -83,7 +84,10 @@ const javaScopeResolver: ScopeResolver = { populateNamespaceSiblings: populateJavaPackageSiblings, populateRangeBindings: populateJavaCrossFileReturnTypes, - emitPostResolutionEdges: attachSpringBeanCandidateMetadata, + emitPostResolutionEdges: (graph, parsedFiles, nodeLookup, indexes, ctx) => { + attachSpringBeanCandidateMetadata(graph, parsedFiles, nodeLookup, indexes); + attachJavaSpringConfigBindings(graph, parsedFiles, nodeLookup, indexes, ctx); + }, }; export { javaScopeResolver }; diff --git a/gitnexus/src/core/ingestion/languages/java/spring-config-bindings.ts b/gitnexus/src/core/ingestion/languages/java/spring-config-bindings.ts new file mode 100644 index 000000000..5f2eb07c3 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/spring-config-bindings.ts @@ -0,0 +1,253 @@ +import type { KnowledgeGraph } from '../../../graph/types.js'; +import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js'; +import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; +import { makeScopeId, type ParsedFile, type ScopeId } from 'gitnexus-shared'; +import { + bindSpringConfigConsumers, + type SpringConfigConsumer, +} from '../../frameworks/spring/config-bindings.js'; +import { createSpringAnnotationNameResolver } from '../../frameworks/spring/bean-candidates.js'; +import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js'; +import { nodeToCapture, type SyntaxNode } from '../../utils/ast-helpers.js'; +import { getJavaParser } from './query.js'; +import { getJavaSpringConfigConsumerFacts } from './capture-side-channel.js'; +import { isJavaPackageSiblingVisibilityIncomplete } from './package-siblings.js'; + +const VALUE_ANNOTATION = 'org.springframework.beans.factory.annotation.Value'; +const CONFIGURATION_PROPERTIES_ANNOTATION = + 'org.springframework.boot.context.properties.ConfigurationProperties'; + +interface JavaAnnotation { + readonly name: string; + readonly text: string; +} + +interface JavaImports { + readonly exact: ReadonlySet; + readonly wildcard: ReadonlySet; + readonly localTypes: ReadonlySet; +} + +export interface JavaSpringConfigConsumerFact { + readonly consumer: SpringConfigConsumer; + readonly annotationName: string; + readonly classScopeId: ScopeId; +} + +function collectJavaImports(root: SyntaxNode): JavaImports { + const exact = new Set(); + const wildcard = new Set(); + const localTypes = new Set(); + + for (const node of root.descendantsOfType('import_declaration')) { + const imported = node.text + .replace(/^\s*import\s+(?:static\s+)?/, '') + .replace(/;\s*$/, '') + .trim(); + if (imported.endsWith('.*')) wildcard.add(imported.slice(0, -2)); + else exact.add(imported); + } + + for (const type of [ + 'class_declaration', + 'interface_declaration', + 'enum_declaration', + 'record_declaration', + 'annotation_type_declaration', + ]) { + for (const node of root.descendantsOfType(type)) { + const name = node.childForFieldName('name')?.text; + if (name) localTypes.add(name); + } + } + return { exact, wildcard, localTypes }; +} + +function annotationsOn(node: SyntaxNode): JavaAnnotation[] { + const modifiers = node.namedChildren.find((child) => child.type === 'modifiers'); + if (modifiers === undefined) return []; + const annotations: JavaAnnotation[] = []; + for (const child of modifiers.namedChildren) { + if (child.type !== 'annotation' && child.type !== 'marker_annotation') continue; + const name = child.childForFieldName('name')?.text ?? child.firstNamedChild?.text; + if (name) annotations.push({ name, text: child.text }); + } + return annotations; +} + +function resolvesToAnnotation( + rawName: string, + canonicalName: string, + imports: JavaImports, +): boolean { + if (rawName.includes('.')) return rawName === canonicalName; + if (imports.localTypes.has(rawName)) return false; + if (imports.exact.has(canonicalName)) return true; + const packageName = canonicalName.slice(0, canonicalName.lastIndexOf('.')); + return imports.wildcard.has(packageName); +} + +function decodeJavaStringLiteral(literal: string): string { + return literal + .slice(1, -1) + .replace(/\\u([0-9a-fA-F]{4})/g, (_match, hex: string) => + String.fromCharCode(Number.parseInt(hex, 16)), + ) + .replace(/\\(["'\\btnfr])/g, (_match, escaped: string) => { + const controls: Record = { + b: '\b', + t: '\t', + n: '\n', + f: '\f', + r: '\r', + }; + return controls[escaped] ?? escaped; + }); +} + +function javaStringLiterals(text: string): string[] { + return [...text.matchAll(/"(?:\\.|[^"\\])*"/g)].map((match) => decodeJavaStringLiteral(match[0])); +} + +/** Extract statically readable Spring placeholder keys from a Java annotation. */ +export function parseValuePlaceholderKeys(annotationText: string): string[] { + const keys = new Set(); + for (const literal of javaStringLiterals(annotationText)) { + for (const match of literal.matchAll(/\$\{([^{}]+)\}/g)) { + const key = match[1].split(':', 1)[0].trim(); + if (/^[A-Za-z0-9_.-]+$/.test(key)) keys.add(key); + } + } + return [...keys]; +} + +/** Extract `prefix`/`value` (or the positional value) from the annotation. */ +export function parseConfigurationPropertiesPrefix(annotationText: string): string | null { + const named = /\b(?:prefix|value)\s*=\s*("(?:\\.|[^"\\])*")/.exec(annotationText); + const literal = named?.[1] ?? annotationText.match(/"(?:\\.|[^"\\])*"/)?.[0]; + if (literal === undefined) return null; + const prefix = decodeJavaStringLiteral(literal) + .trim() + .replace(/^\.+|\.+$/g, ''); + return /^[A-Za-z0-9_.-]+$/.test(prefix) ? prefix : null; +} + +function classScopeId(filePath: string, declaration: SyntaxNode): ScopeId { + return makeScopeId({ + filePath, + range: nodeToCapture('@scope.class', declaration).range, + kind: 'Class', + }); +} + +function enclosingClass(node: SyntaxNode): SyntaxNode | undefined { + let current = node.parent; + while (current !== null) { + if (current.type === 'class_declaration' || current.type === 'record_declaration') { + return current; + } + current = current.parent; + } + return undefined; +} + +/** Collect config facts from the Java parser's existing AST (no reparse). */ +export function captureJavaSpringConfigConsumerFacts( + root: SyntaxNode, + filePath: string, +): JavaSpringConfigConsumerFact[] { + const imports = collectJavaImports(root); + const facts: JavaSpringConfigConsumerFact[] = []; + + for (const field of root.descendantsOfType('field_declaration')) { + const annotations = annotationsOn(field).filter((annotation) => + resolvesToAnnotation(annotation.name, VALUE_ANNOTATION, imports), + ); + if (annotations.length === 0) continue; + const owner = enclosingClass(field); + if (owner === undefined) continue; + for (const declarator of field.namedChildren.filter( + (child) => child.type === 'variable_declarator', + )) { + const fieldName = declarator.childForFieldName('name')?.text; + if (!fieldName) continue; + for (const annotation of annotations) { + const keys = parseValuePlaceholderKeys(annotation.text); + if (keys.length > 0) { + facts.push({ + consumer: { kind: 'value', fieldName, line: field.startPosition.row + 1, keys }, + annotationName: annotation.name, + classScopeId: classScopeId(filePath, owner), + }); + } + } + } + } + + for (const type of ['class_declaration', 'record_declaration']) { + for (const declaration of root.descendantsOfType(type)) { + const className = declaration.childForFieldName('name')?.text; + if (!className) continue; + for (const annotation of annotationsOn(declaration)) { + if (!resolvesToAnnotation(annotation.name, CONFIGURATION_PROPERTIES_ANNOTATION, imports)) { + continue; + } + const prefix = parseConfigurationPropertiesPrefix(annotation.text); + if (prefix !== null) { + facts.push({ + consumer: { + kind: 'configuration-properties', + className, + line: declaration.startPosition.row + 1, + prefix, + }, + annotationName: annotation.name, + classScopeId: classScopeId(filePath, declaration), + }); + } + } + } + } + return facts; +} + +/** Parse Java consumers for focused unit tests; production reuses the worker AST. */ +export function extractJavaSpringConfigConsumers(source: string): SpringConfigConsumer[] { + const tree = parseSourceSafe(getJavaParser(), source); + return captureJavaSpringConfigConsumerFacts(tree.rootNode, '').map( + (fact) => fact.consumer, + ); +} + +/** Java ScopeResolver post-resolution hook for Spring configuration consumers. */ +export function attachJavaSpringConfigBindings( + graph: KnowledgeGraph, + parsedFiles: readonly ParsedFile[], + _nodeLookup: GraphNodeLookup, + indexes: ScopeResolutionIndexes, + _ctx: { readonly fileContents: ReadonlyMap }, +): void { + const resolveAnnotation = createSpringAnnotationNameResolver(indexes); + const recognizedAnnotations = new Set([VALUE_ANNOTATION, CONFIGURATION_PROPERTIES_ANNOTATION]); + const batches: Array<{ filePath: string; consumers: SpringConfigConsumer[] }> = []; + for (const parsed of parsedFiles) { + const consumers: SpringConfigConsumer[] = []; + for (const fact of getJavaSpringConfigConsumerFacts(parsed.filePath)) { + const classScope = indexes.scopeTree.getScope(fact.classScopeId); + if (classScope === undefined || classScope.kind !== 'Class') continue; + const expectedAnnotation = + fact.consumer.kind === 'value' ? VALUE_ANNOTATION : CONFIGURATION_PROPERTIES_ANNOTATION; + const enclosingScope = fact.consumer.kind === 'value' ? classScope.id : classScope.parent; + const resolved = resolveAnnotation( + fact.annotationName, + parsed, + enclosingScope, + recognizedAnnotations, + isJavaPackageSiblingVisibilityIncomplete(parsed.filePath), + ); + if (resolved === expectedAnnotation) consumers.push(fact.consumer); + } + if (consumers.length > 0) batches.push({ filePath: parsed.filePath, consumers }); + } + bindSpringConfigConsumers(graph, batches); +} diff --git a/gitnexus/src/core/ingestion/pipeline-phases/index.ts b/gitnexus/src/core/ingestion/pipeline-phases/index.ts index f4996a8ac..a3fb83aeb 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/index.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/index.ts @@ -20,6 +20,7 @@ export { scopeResolutionPhase, type ScopeResolutionOutput, } from '../scope-resolution/pipeline/phase.js'; +export { springConfigPhase, type SpringConfigOutput } from './spring-config.js'; export { pruneLocalSymbolsPhase, type PruneLocalSymbolsOutput } from './prune-local-symbols.js'; export { taintSummariesPhase, type TaintSummariesOutput } from './taint-summaries.js'; export { callSummariesPhase, type CallSummariesOutput } from './call-summaries.js'; diff --git a/gitnexus/src/core/ingestion/pipeline-phases/spring-config.ts b/gitnexus/src/core/ingestion/pipeline-phases/spring-config.ts new file mode 100644 index 000000000..4b222ff93 --- /dev/null +++ b/gitnexus/src/core/ingestion/pipeline-phases/spring-config.ts @@ -0,0 +1,374 @@ +/** + * Phase: springConfig + * + * Adds key-only nodes for statically readable Spring + * `application*.properties` / `application*.yml` / `application*.yaml` files. + * Language-specific ScopeResolver hooks attach consumers later. Configuration + * values are deliberately never copied into the graph because they may contain + * credentials and key identity is sufficient for impact analysis. + * + * @deps structure + * @reads Spring application configuration files + * @writes Property nodes and DEFINES edges + */ + +import fs from 'node:fs/promises'; +import path from 'node:path'; +import { createRequire } from 'node:module'; +import type { EventType as YamlEventType, State as YamlState } from 'js-yaml'; +import { SPRING_CONFIG_DESCRIPTION } from '../frameworks/spring/config-bindings.js'; +import { generateId } from '../../../lib/utils.js'; +import type { PipelineContext, PipelinePhase, PhaseResult } from './types.js'; +import { getPhaseOutput } from './types.js'; +import type { StructureOutput } from './structure.js'; + +const require = createRequire(import.meta.url); +const yaml = require('js-yaml') as typeof import('js-yaml'); +const MAX_CONFIG_FILE_BYTES = 2 * 1024 * 1024; + +export interface SpringConfigKey { + readonly key: string; + readonly filePath: string; + readonly line: number; + readonly profile?: string; + readonly format: 'properties' | 'yaml'; +} + +interface SpringConfigFile { + readonly filePath: string; + readonly profile?: string; + readonly format: SpringConfigKey['format']; +} + +export interface SpringConfigOutput { + readonly configKeys: number; +} + +/** Match only Spring Boot's conventional application config file names. */ +export function classifySpringConfigFile(filePath: string): SpringConfigFile | null { + const base = path.posix.basename(filePath.replaceAll('\\', '/')); + const match = /^application(?:-([^.]+))?\.(properties|ya?ml)$/i.exec(base); + if (match === null) return null; + return { + filePath, + ...(match[1] ? { profile: match[1] } : {}), + format: match[2].toLowerCase() === 'properties' ? 'properties' : 'yaml', + }; +} + +function unescapePropertyKey(raw: string): string { + return raw + .replace(/\\u([0-9a-fA-F]{4})/g, (_match, hex: string) => + String.fromCharCode(Number.parseInt(hex, 16)), + ) + .replace(/\\([:=#!\\ ])/g, '$1'); +} + +function logicalPropertiesLines(content: string): Array<{ text: string; line: number }> { + const physical = content.split(/\r?\n/); + const logical: Array<{ text: string; line: number }> = []; + let current = ''; + let startLine = 1; + + for (let index = 0; index < physical.length; index++) { + const line = physical[index]; + if (current.length === 0) startLine = index + 1; + current += current.length === 0 ? line : line.trimStart(); + + let trailingBackslashes = 0; + for (let cursor = current.length - 1; cursor >= 0 && current[cursor] === '\\'; cursor--) { + trailingBackslashes++; + } + if (trailingBackslashes % 2 === 1) { + current = current.slice(0, -1); + continue; + } + logical.push({ text: current, line: startLine }); + current = ''; + } + if (current.length > 0) logical.push({ text: current, line: startLine }); + return logical; +} + +/** Parse `.properties` keys without retaining their values. */ +export function parseSpringProperties( + content: string, + filePath: string, + profile?: string, +): SpringConfigKey[] { + const keys: SpringConfigKey[] = []; + const seen = new Set(); + + for (const logical of logicalPropertiesLines(content)) { + const trimmed = logical.text.trimStart(); + if (trimmed.length === 0 || trimmed.startsWith('#') || trimmed.startsWith('!')) continue; + + let separator = -1; + let escaped = false; + for (let index = 0; index < trimmed.length; index++) { + const char = trimmed[index]; + if (!escaped && (char === '=' || char === ':' || /\s/.test(char))) { + separator = index; + break; + } + escaped = !escaped && char === '\\'; + if (char !== '\\') escaped = false; + } + const rawKey = (separator === -1 ? trimmed : trimmed.slice(0, separator)).trim(); + const key = unescapePropertyKey(rawKey); + if (key.length === 0 || seen.has(key)) continue; + seen.add(key); + keys.push({ + key, + filePath, + line: logical.line, + ...(profile ? { profile } : {}), + format: 'properties', + }); + } + + return keys; +} + +interface YamlParseEvent { + readonly startLine: number; + kind: string | null; + result: unknown; + tag: string | null; + readonly children: YamlParseEvent[]; +} + +interface YamlMappingLocation { + readonly valueEvent: YamlParseEvent; + readonly line: number; +} + +function isObjectValue(value: unknown): value is object { + return value !== null && typeof value === 'object'; +} + +function resolveYamlAliasEvent( + event: YamlParseEvent | undefined, + objectEvents: WeakMap, +): YamlParseEvent | undefined { + if (event?.kind !== null || !isObjectValue(event.result)) return event; + return objectEvents.get(event.result) ?? event; +} + +function yamlMappingPairs(event: YamlParseEvent): Array<{ + key: string; + keyEvent: YamlParseEvent; + valueEvent: YamlParseEvent; +}> { + const pairs: Array<{ key: string; keyEvent: YamlParseEvent; valueEvent: YamlParseEvent }> = []; + for (let index = 0; index + 1 < event.children.length; index += 2) { + const keyEvent = event.children[index]; + const valueEvent = event.children[index + 1]; + if (keyEvent.kind !== 'scalar') continue; + pairs.push({ key: String(keyEvent.result), keyEvent, valueEvent }); + } + return pairs; +} + +function findYamlMappingLocation( + event: YamlParseEvent | undefined, + key: string, + objectEvents: WeakMap, + visited = new Set(), +): YamlMappingLocation | undefined { + const resolved = resolveYamlAliasEvent(event, objectEvents); + if (resolved === undefined || visited.has(resolved)) return undefined; + visited.add(resolved); + + if (resolved.kind === 'sequence') { + for (const child of resolved.children) { + const found = findYamlMappingLocation(child, key, objectEvents, visited); + if (found !== undefined) return found; + } + return undefined; + } + if (resolved.kind !== 'mapping') return undefined; + + const pairs = yamlMappingPairs(resolved); + const direct = pairs.find((pair) => pair.key === key); + if (direct !== undefined) { + return { valueEvent: direct.valueEvent, line: direct.keyEvent.startLine }; + } + for (const merge of pairs.filter((pair) => pair.key === '<<')) { + const found = findYamlMappingLocation(merge.valueEvent, key, objectEvents, visited); + if (found !== undefined) return found; + } + return undefined; +} + +function flattenYamlValue( + value: unknown, + event: YamlParseEvent | undefined, + prefix: string, + out: Map, + objectEvents: WeakMap, + sourceLine = event?.startLine ?? 1, +): void { + const resolvedEvent = resolveYamlAliasEvent(event, objectEvents); + if (Array.isArray(value)) { + if (value.length === 0 && prefix.length > 0 && !out.has(prefix)) out.set(prefix, sourceLine); + value.forEach((item, index) => + flattenYamlValue( + item, + resolvedEvent?.children[index], + `${prefix}[${index}]`, + out, + objectEvents, + sourceLine, + ), + ); + return; + } + if ( + value !== null && + typeof value === 'object' && + (resolvedEvent?.kind === 'mapping' || resolvedEvent === undefined) + ) { + const entries = Object.entries(value as Record); + if (entries.length === 0 && prefix.length > 0 && !out.has(prefix)) out.set(prefix, sourceLine); + for (const [key, nested] of entries) { + const next = prefix.length === 0 ? key : `${prefix}.${key}`; + const location = findYamlMappingLocation(resolvedEvent, key, objectEvents); + flattenYamlValue( + nested, + location?.valueEvent, + next, + out, + objectEvents, + location?.line ?? sourceLine, + ); + } + return; + } + if (prefix.length > 0 && !out.has(prefix)) out.set(prefix, sourceLine); +} + +/** Parse and flatten YAML leaves without retaining their values. */ +export function parseSpringYaml( + content: string, + filePath: string, + profile?: string, +): SpringConfigKey[] { + const flattened = new Map(); + const eventStack: YamlParseEvent[] = []; + const documentEvents: YamlParseEvent[] = []; + const objectEvents = new WeakMap(); + const documents: unknown[] = []; + + yaml.loadAll(content, (document) => documents.push(document), { + schema: yaml.DEFAULT_SCHEMA, + json: true, + listener: (eventType: YamlEventType, state: YamlState) => { + if (eventType === 'open') { + eventStack.push({ + startLine: state.line + 1, + kind: null, + result: undefined, + tag: null, + children: [], + }); + return; + } + + const event = eventStack.pop(); + if (event === undefined) return; + event.kind = state.kind ?? null; + event.result = state.result; + event.tag = (state as YamlState & { tag?: string | null }).tag ?? null; + if (isObjectValue(event.result) && event.kind !== null) { + objectEvents.set(event.result, event); + } + const parent = eventStack[eventStack.length - 1]; + if (parent === undefined) documentEvents.push(event); + else parent.children.push(event); + }, + }); + + documents.forEach((document, index) => + flattenYamlValue(document, documentEvents[index], '', flattened, objectEvents), + ); + return [...flattened.entries()] + .sort(([left], [right]) => left.localeCompare(right)) + .map(([key, line]) => ({ + key, + filePath, + line, + ...(profile ? { profile } : {}), + format: 'yaml' as const, + })); +} + +function configKeyNodeId(entry: SpringConfigKey): string { + return generateId('Property', `spring-config:${entry.filePath}:${entry.key}`); +} + +async function readConfigKeys( + repoPath: string, + scannedFiles: StructureOutput['scannedFiles'], +): Promise { + const keys: SpringConfigKey[] = []; + for (const scanned of scannedFiles) { + const classified = classifySpringConfigFile(scanned.path); + if (classified === null || scanned.size > MAX_CONFIG_FILE_BYTES) continue; + try { + const content = await fs.readFile(path.join(repoPath, scanned.path), 'utf8'); + keys.push( + ...(classified.format === 'properties' + ? parseSpringProperties(content, classified.filePath, classified.profile) + : parseSpringYaml(content, classified.filePath, classified.profile)), + ); + } catch { + // Malformed configuration is not a reason to fail the entire code index. + // Fail closed: no keys and therefore no misleading bindings for this file. + } + } + return keys; +} + +export const springConfigPhase: PipelinePhase = { + name: 'springConfig', + deps: ['structure'], + + async execute( + ctx: PipelineContext, + deps: ReadonlyMap>, + ): Promise { + const { scannedFiles } = getPhaseOutput(deps, 'structure'); + const configKeys = await readConfigKeys(ctx.repoPath, scannedFiles); + for (const entry of configKeys) { + const nodeId = configKeyNodeId(entry); + ctx.graph.addNode({ + id: nodeId, + label: 'Property', + properties: { + name: entry.key, + filePath: entry.filePath, + startLine: entry.line, + endLine: entry.line, + language: entry.format, + description: entry.profile + ? `${SPRING_CONFIG_DESCRIPTION} (profile: ${entry.profile})` + : SPRING_CONFIG_DESCRIPTION, + }, + }); + const fileId = generateId('File', entry.filePath); + if (ctx.graph.getNode(fileId) !== undefined) { + ctx.graph.addRelationship({ + id: generateId('DEFINES', `${fileId}->${nodeId}`), + sourceId: fileId, + targetId: nodeId, + type: 'DEFINES', + confidence: 1, + reason: 'spring-config:key', + }); + } + } + + return { configKeys: configKeys.length }; + }, +}; diff --git a/gitnexus/src/core/ingestion/pipeline.ts b/gitnexus/src/core/ingestion/pipeline.ts index 58ec391c0..ebf290112 100644 --- a/gitnexus/src/core/ingestion/pipeline.ts +++ b/gitnexus/src/core/ingestion/pipeline.ts @@ -31,6 +31,7 @@ import { ormPhase, crossFilePhase, scopeResolutionPhase, + springConfigPhase, pruneLocalSymbolsPhase, taintSummariesPhase, callSummariesPhase, @@ -242,7 +243,7 @@ export interface PipelineOptions { * * Phase dependency graph: * - * scan → structure → [markdown, cobol] → parse → [routes, tools, orm] + * scan → structure → [springConfig, markdown, cobol] → parse → [routes, tools, orm] * → crossFile → scopeResolution → pruneLocalSymbols * → mro → di → communities → processes * @@ -261,6 +262,7 @@ export function buildPhaseList(options?: PipelineOptions): PipelinePhase[] { new PhaseRegistry() .register(scanPhase) .register(structurePhase) + .register(springConfigPhase) .register(markdownPhase) .register(cobolPhase) .register(parsePhase) diff --git a/gitnexus/src/core/lbug/schema.ts b/gitnexus/src/core/lbug/schema.ts index 22aed098d..c8943d719 100644 --- a/gitnexus/src/core/lbug/schema.ts +++ b/gitnexus/src/core/lbug/schema.ts @@ -402,6 +402,7 @@ CREATE REL TABLE ${REL_TABLE_NAME} ( FROM \`Static\` TO Community, FROM \`Variable\` TO Community, FROM \`Property\` TO Community, + FROM \`Property\` TO \`Property\`, FROM \`Record\` TO Method, FROM \`Record\` TO \`Constructor\`, FROM \`Record\` TO \`Property\`, diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index 635b255a9..e0bbfa116 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -123,6 +123,7 @@ import { EMBEDDING_TABLE_NAME } from './lbug/schema.js'; import { STALE_HASH_SENTINEL } from './lbug/schema.js'; import { isSpringBeanCandidateSourceFile } from './ingestion/frameworks/spring/bean-catalog.js'; import { SPRING_BEAN_INVENTORY_FEATURE } from './ingestion/frameworks/spring/analysis-features.js'; +import { SPRING_CONFIG_BINDINGS_FEATURE } from './ingestion/languages/java/analysis-features.js'; import { CLASS_FRAMEWORK_ANNOTATIONS_FEATURE, findAnalysisFeatureMismatches, @@ -137,6 +138,7 @@ import { const ANALYSIS_FEATURES = [ CLASS_FRAMEWORK_ANNOTATIONS_FEATURE, SPRING_BEAN_INVENTORY_FEATURE, + SPRING_CONFIG_BINDINGS_FEATURE, ] as const; interface PersistedFrameworkAnnotationRow { diff --git a/gitnexus/test/fixtures/spring-config-app/src/main/java/com/example/ConfigConsumers.java b/gitnexus/test/fixtures/spring-config-app/src/main/java/com/example/ConfigConsumers.java new file mode 100644 index 000000000..49e6847dd --- /dev/null +++ b/gitnexus/test/fixtures/spring-config-app/src/main/java/com/example/ConfigConsumers.java @@ -0,0 +1,20 @@ +package com.example; + +import org.springframework.beans.factory.annotation.Value; +import org.springframework.boot.context.properties.ConfigurationProperties; + +class DirectValues { + @Value("${payment.timeout:30}") + private int timeout; + + @Value("${payment.missing}") + private String missing; +} + +@ConfigurationProperties(prefix = "service") +class ServiceProperties { + private String endpoint; + private Retry retry; +} + +class Retry {} diff --git a/gitnexus/test/fixtures/spring-config-app/src/main/resources/application-dev.yml b/gitnexus/test/fixtures/spring-config-app/src/main/resources/application-dev.yml new file mode 100644 index 000000000..28cf5ab6a --- /dev/null +++ b/gitnexus/test/fixtures/spring-config-app/src/main/resources/application-dev.yml @@ -0,0 +1,6 @@ +defaults: &defaults + retry: + max-attempts: 3 +service: + <<: *defaults + endpoint: https://service.example.test diff --git a/gitnexus/test/fixtures/spring-config-app/src/main/resources/application.properties b/gitnexus/test/fixtures/spring-config-app/src/main/resources/application.properties new file mode 100644 index 000000000..b763cc27c --- /dev/null +++ b/gitnexus/test/fixtures/spring-config-app/src/main/resources/application.properties @@ -0,0 +1 @@ +payment.timeout=30 diff --git a/gitnexus/test/fixtures/spring-config-shadow-app/src/main/java/com/example/Shadowed.java b/gitnexus/test/fixtures/spring-config-shadow-app/src/main/java/com/example/Shadowed.java new file mode 100644 index 000000000..833184fc8 --- /dev/null +++ b/gitnexus/test/fixtures/spring-config-shadow-app/src/main/java/com/example/Shadowed.java @@ -0,0 +1,8 @@ +package com.example; + +import org.springframework.beans.factory.annotation.*; + +class Shadowed { + @Value("${fake.key}") + private String fake; +} diff --git a/gitnexus/test/fixtures/spring-config-shadow-app/src/main/java/com/example/Value.java b/gitnexus/test/fixtures/spring-config-shadow-app/src/main/java/com/example/Value.java new file mode 100644 index 000000000..85156a4cf --- /dev/null +++ b/gitnexus/test/fixtures/spring-config-shadow-app/src/main/java/com/example/Value.java @@ -0,0 +1,5 @@ +package com.example; + +public @interface Value { + String value(); +} diff --git a/gitnexus/test/fixtures/spring-config-shadow-app/src/main/resources/application.properties b/gitnexus/test/fixtures/spring-config-shadow-app/src/main/resources/application.properties new file mode 100644 index 000000000..4fded8755 --- /dev/null +++ b/gitnexus/test/fixtures/spring-config-shadow-app/src/main/resources/application.properties @@ -0,0 +1 @@ +fake.key=must-not-bind diff --git a/gitnexus/test/integration/spring-config-mcp.test.ts b/gitnexus/test/integration/spring-config-mcp.test.ts new file mode 100644 index 000000000..570c15b6b --- /dev/null +++ b/gitnexus/test/integration/spring-config-mcp.test.ts @@ -0,0 +1,77 @@ +import { beforeAll, describe, expect, it, vi } from 'vitest'; +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +import { listRegisteredRepos } from '../../src/storage/repo-manager.js'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; + +vi.mock('../../src/storage/repo-manager.js', () => ({ + listRegisteredRepos: vi.fn().mockResolvedValue([]), + cleanupOldKuzuFiles: vi.fn().mockResolvedValue({ found: false, needsReindex: false }), + findSiblingClones: vi.fn().mockResolvedValue([]), +})); + +const CONSUMER_ID = 'Property:src/Config.java:DirectValues.timeout'; +const CONFIG_ID = 'Property:spring-config:application.properties:payment.timeout'; +const SEED = [ + `CREATE (p:\`Property\` {id:'${CONSUMER_ID}', name:'timeout', filePath:'src/Config.java', startLine:4, endLine:4, content:'', description:'', declaredType:'int'})`, + `CREATE (p:\`Property\` {id:'${CONFIG_ID}', name:'payment.timeout', filePath:'application.properties', startLine:1, endLine:1, content:'', description:'Spring configuration property', declaredType:''})`, + `MATCH (consumer:\`Property\` {id:'${CONSUMER_ID}'}), (config:\`Property\` {id:'${CONFIG_ID}'}) CREATE (consumer)-[:CodeRelation {type:'USES', confidence:1.0, reason:'spring-config:@Value payment.timeout'}]->(config)`, +]; + +withTestLbugDB( + 'spring-config-mcp', + (handle) => { + let backend: LocalBackend; + + beforeAll(() => { + backend = (handle as typeof handle & { _backend: LocalBackend })._backend; + }); + + describe('Spring configuration context and impact visibility', () => { + it('shows configuration dependencies in context without special query flags', async () => { + const context = await backend.callTool('context', { uid: CONSUMER_ID }); + expect(context.outgoing.uses).toEqual([ + expect.objectContaining({ + uid: CONFIG_ID, + name: 'payment.timeout', + filePath: 'application.properties', + }), + ]); + }); + + it('shows consumers in upstream impact from a configuration key', async () => { + const impact = await backend.callTool('impact', { + target_uid: CONFIG_ID, + target: 'payment.timeout', + direction: 'upstream', + }); + expect(impact.risk).not.toBe('UNKNOWN'); + expect(impact.byDepth[1]).toEqual([ + expect.objectContaining({ + id: CONSUMER_ID, + name: 'timeout', + relationType: 'USES', + }), + ]); + }); + }); + }, + { + seed: SEED, + poolAdapter: true, + afterSetup: async (handle) => { + vi.mocked(listRegisteredRepos).mockResolvedValue([ + { + name: 'test-repo', + path: '/test/repo', + storagePath: handle.tmpHandle.dbPath, + indexedAt: new Date().toISOString(), + lastCommit: 'abc123', + stats: { files: 2, nodes: 2, communities: 0, processes: 0 }, + }, + ]); + const backend = new LocalBackend(); + await backend.init(); + (handle as typeof handle & { _backend?: LocalBackend })._backend = backend; + }, + }, +); diff --git a/gitnexus/test/integration/spring-config-pipeline.test.ts b/gitnexus/test/integration/spring-config-pipeline.test.ts new file mode 100644 index 000000000..797262816 --- /dev/null +++ b/gitnexus/test/integration/spring-config-pipeline.test.ts @@ -0,0 +1,93 @@ +import path from 'node:path'; +import { beforeAll, describe, expect, it } from 'vitest'; +import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; +import { runPipelineFromRepo } from '../../src/core/ingestion/pipeline.js'; +import type { PipelineResult } from '../../types/pipeline.js'; + +const FIXTURE = path.resolve(__dirname, '..', 'fixtures', 'spring-config-app'); +const SHADOW_FIXTURE = path.resolve(__dirname, '..', 'fixtures', 'spring-config-shadow-app'); + +describe('Spring configuration binding pipeline', () => { + let result: PipelineResult; + let nodes: GraphNode[]; + let uses: GraphRelationship[]; + + beforeAll(async () => { + result = await runPipelineFromRepo(FIXTURE, () => {}, { skipGraphPhases: true }); + nodes = [...result.graph.iterNodes()]; + uses = [...result.graph.iterRelationshipsByType('USES')].filter((edge) => + edge.reason.startsWith('spring-config:'), + ); + }, 60_000); + + const nodeNamed = (name: string, fileSuffix?: string): GraphNode | undefined => + nodes.find( + (node) => + node.properties.name === name && + (fileSuffix === undefined || String(node.properties.filePath).endsWith(fileSuffix)), + ); + + const targetsFrom = (source: GraphNode): string[] => + uses + .filter((edge) => edge.sourceId === source.id) + .map((edge) => String(result.graph.getNode(edge.targetId)?.properties.name)) + .sort(); + + it('creates key-only Property nodes for properties and profile YAML files', () => { + expect(nodeNamed('payment.timeout', 'application.properties')).toBeDefined(); + expect(nodeNamed('service.endpoint', 'application-dev.yml')?.properties.description).toContain( + 'profile: dev', + ); + expect( + nodeNamed('service.retry.max-attempts', 'application-dev.yml')?.properties.startLine, + ).toBe(3); + expect( + nodes.some((node) => JSON.stringify(node.properties).includes('service.example.test')), + ).toBe(false); + }); + + it('links Value fields to exact keys and leaves missing placeholders unresolved', () => { + const timeout = nodeNamed('timeout', 'ConfigConsumers.java'); + const missing = nodeNamed('missing', 'ConfigConsumers.java'); + expect(timeout).toBeDefined(); + expect(missing).toBeDefined(); + if (timeout === undefined || missing === undefined) throw new Error('fixture fields missing'); + expect(targetsFrom(timeout)).toEqual(['payment.timeout']); + expect(targetsFrom(missing)).toEqual([]); + expect(missing?.properties.description).toContain('Spring config unresolved: payment.missing'); + }); + + it('links ConfigurationProperties classes and relaxed field names to their prefix', () => { + const owner = nodeNamed('ServiceProperties', 'ConfigConsumers.java'); + const endpoint = nodeNamed('endpoint', 'ConfigConsumers.java'); + const retry = nodeNamed('retry', 'ConfigConsumers.java'); + expect(owner).toBeDefined(); + if (owner === undefined || endpoint === undefined || retry === undefined) { + throw new Error('fixture ConfigurationProperties symbols missing'); + } + expect(targetsFrom(owner)).toEqual(['service.endpoint', 'service.retry.max-attempts']); + expect(targetsFrom(endpoint)).toEqual(['service.endpoint']); + expect(targetsFrom(retry)).toEqual(['service.retry.max-attempts']); + }); +}); + +describe('Spring configuration annotation attribution', () => { + it('fails closed when a same-package annotation shadows a Spring wildcard import', async () => { + const result = await runPipelineFromRepo(SHADOW_FIXTURE, () => {}, { + skipGraphPhases: true, + }); + const fake = [...result.graph.iterNodes()].find( + (node) => + node.properties.name === 'fake' && + String(node.properties.filePath).endsWith('Shadowed.java'), + ); + expect(fake).toBeDefined(); + if (fake === undefined) throw new Error('shadow fixture field missing'); + expect( + [...result.graph.iterRelationshipsByType('USES')].filter( + (edge) => edge.sourceId === fake.id && edge.reason.startsWith('spring-config:'), + ), + ).toEqual([]); + expect(String(fake.properties.description ?? '')).not.toContain('Spring config unresolved:'); + }); +}); diff --git a/gitnexus/test/unit/analysis-features.test.ts b/gitnexus/test/unit/analysis-features.test.ts index 6965c99d5..c2592b494 100644 --- a/gitnexus/test/unit/analysis-features.test.ts +++ b/gitnexus/test/unit/analysis-features.test.ts @@ -6,8 +6,13 @@ import { type AnalysisFeatureDescriptor, } from '../../src/core/analysis-features.js'; import { SPRING_BEAN_INVENTORY_FEATURE } from '../../src/core/ingestion/frameworks/spring/analysis-features.js'; +import { SPRING_CONFIG_BINDINGS_FEATURE } from '../../src/core/ingestion/languages/java/analysis-features.js'; -const FEATURES = [CLASS_FRAMEWORK_ANNOTATIONS_FEATURE, SPRING_BEAN_INVENTORY_FEATURE] as const; +const FEATURES = [ + CLASS_FRAMEWORK_ANNOTATIONS_FEATURE, + SPRING_BEAN_INVENTORY_FEATURE, + SPRING_CONFIG_BINDINGS_FEATURE, +] as const; describe('analysis feature versions', () => { it('separates the global Class schema capability from JVM-only Bean evidence', () => { @@ -17,6 +22,7 @@ describe('analysis feature versions', () => { expect(resolveAnalysisFeatureVersions(FEATURES, ['src/App.java'])).toEqual({ 'graph.class-framework-annotations': 1, 'spring.bean-inventory': 1, + 'spring.config-bindings': 1, }); expect(resolveAnalysisFeatureVersions(FEATURES, ['BUILD.GRADLE.KTS'])).toEqual({ 'graph.class-framework-annotations': 1, diff --git a/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts b/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts index 51efd6d1f..7aeb8b431 100644 --- a/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts +++ b/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts @@ -66,6 +66,7 @@ describe('PhaseRegistry', () => { const FULL_ORDER = [ 'scan', 'structure', + 'springConfig', 'markdown', 'cobol', 'parse', diff --git a/gitnexus/test/unit/spring-config-bindings.test.ts b/gitnexus/test/unit/spring-config-bindings.test.ts new file mode 100644 index 000000000..321d3647a --- /dev/null +++ b/gitnexus/test/unit/spring-config-bindings.test.ts @@ -0,0 +1,169 @@ +import { describe, expect, it, vi } from 'vitest'; +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import { bindSpringConfigConsumers } from '../../src/core/ingestion/frameworks/spring/config-bindings.js'; +import { + classifySpringConfigFile, + parseSpringProperties, + parseSpringYaml, +} from '../../src/core/ingestion/pipeline-phases/spring-config.js'; +import { + extractJavaSpringConfigConsumers, + parseConfigurationPropertiesPrefix, + parseValuePlaceholderKeys, +} from '../../src/core/ingestion/languages/java/spring-config-bindings.js'; + +describe('Spring configuration parsing', () => { + it('recognizes base and profile-specific application config files', () => { + const base = classifySpringConfigFile('src/main/resources/application.properties'); + expect(base).toMatchObject({ format: 'properties' }); + expect(base).not.toHaveProperty('profile'); + expect(classifySpringConfigFile('src/main/resources/application-local.yml')).toMatchObject({ + format: 'yaml', + profile: 'local', + }); + expect(classifySpringConfigFile('src/main/resources/bootstrap.yml')).toBeNull(); + }); + + it('extracts properties keys, continuations, and escaped separators without values', () => { + const keys = parseSpringProperties( + '# comment\nserver.port=8080\nservice\\:name: demo\nlong.\\\n key = secret\n', + 'application.properties', + ); + expect(keys.map((entry) => [entry.key, entry.line])).toEqual([ + ['server.port', 2], + ['service:name', 3], + ['long.key', 4], + ]); + expect(JSON.stringify(keys)).not.toContain('8080'); + expect(JSON.stringify(keys)).not.toContain('secret'); + }); + + it('flattens YAML maps and arrays while retaining profile identity', () => { + const keys = parseSpringYaml( + 'service:\n endpoint: https://example.test\n retries:\n - delay: 10\n', + 'application-dev.yml', + 'dev', + ); + expect(keys).toEqual([ + expect.objectContaining({ + key: 'service.endpoint', + line: 2, + profile: 'dev', + format: 'yaml', + }), + expect.objectContaining({ + key: 'service.retries[0].delay', + line: 4, + profile: 'dev', + format: 'yaml', + }), + ]); + expect(JSON.stringify(keys)).not.toContain('example.test'); + }); + + it('expands YAML merge keys and retains the declaration line for merged values', () => { + const keys = parseSpringYaml( + [ + 'defaults: &defaults', + ' endpoint: https://base.example.test', + ' timeout: 30', + 'service:', + ' <<: *defaults', + ' endpoint: https://override.example.test', + ].join('\n'), + 'application.yml', + ); + + expect(keys).toEqual([ + expect.objectContaining({ key: 'defaults.endpoint', line: 2 }), + expect.objectContaining({ key: 'defaults.timeout', line: 3 }), + expect.objectContaining({ key: 'service.endpoint', line: 6 }), + expect.objectContaining({ key: 'service.timeout', line: 3 }), + ]); + expect(keys.some((entry) => entry.key.includes('<<'))).toBe(false); + }); + + it('parses Value defaults and ConfigurationProperties aliases', () => { + expect(parseValuePlaceholderKeys('@Value("${server.port:8080}")')).toEqual(['server.port']); + expect( + parseConfigurationPropertiesPrefix('@ConfigurationProperties(prefix = "acme.api")'), + ).toBe('acme.api'); + expect(parseConfigurationPropertiesPrefix('@ConfigurationProperties("acme.api")')).toBe( + 'acme.api', + ); + }); +}); + +describe('Java Spring configuration consumers', () => { + it('resolves official imports and ignores shadowed annotation names', () => { + const consumers = extractJavaSpringConfigConsumers(` + import org.springframework.beans.factory.annotation.Value; + import org.springframework.boot.context.properties.ConfigurationProperties; + + @ConfigurationProperties(prefix = "service") + class ServiceProperties { + @Value("\${service.timeout:30}") private int timeout; + } + `); + expect(consumers).toEqual([ + expect.objectContaining({ kind: 'value', fieldName: 'timeout', keys: ['service.timeout'] }), + expect.objectContaining({ + kind: 'configuration-properties', + className: 'ServiceProperties', + prefix: 'service', + }), + ]); + + expect( + extractJavaSpringConfigConsumers(` + @interface Value { String value(); } + class Local { @Value("\${fake.key}") String field; } + `), + ).toEqual([]); + }); + + it('supports wildcard/FQN annotations and every declarator in a field', () => { + const consumers = extractJavaSpringConfigConsumers(` + import org.springframework.beans.factory.annotation.*; + + class DirectValues { + @Value("\${shared.key}") String first, second; + } + + @org.springframework.boot.context.properties.ConfigurationProperties("service") + record ServiceProperties(String endpoint) {} + `); + + expect(consumers).toEqual([ + expect.objectContaining({ kind: 'value', fieldName: 'first', keys: ['shared.key'] }), + expect.objectContaining({ kind: 'value', fieldName: 'second', keys: ['shared.key'] }), + expect.objectContaining({ + kind: 'configuration-properties', + className: 'ServiceProperties', + prefix: 'service', + }), + ]); + }); +}); + +describe('Spring configuration graph binding', () => { + it('indexes the graph once for all consumer files and skips empty work', () => { + const graph = createKnowledgeGraph(); + const iterNodes = vi.spyOn(graph, 'iterNodes'); + + bindSpringConfigConsumers(graph, []); + expect(iterNodes).not.toHaveBeenCalled(); + + bindSpringConfigConsumers(graph, [ + { + filePath: 'First.java', + consumers: [{ kind: 'value', fieldName: 'first', line: 1, keys: ['first.key'] }], + }, + { + filePath: 'Second.java', + consumers: [{ kind: 'value', fieldName: 'second', line: 1, keys: ['second.key'] }], + }, + ]); + expect(iterNodes).toHaveBeenCalledTimes(1); + }); +}); From 3f93bf22d6765fe5bc61268ea20fd4f0ca91de57 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 21 Jul 2026 05:44:25 +0100 Subject: [PATCH 14/61] chore(deps)(deps): bump js-yaml from 4.3.0 to 5.0.0 in /gitnexus (#2586) Bumps [js-yaml](https://github.com/nodeca/js-yaml) from 4.3.0 to 5.0.0. - [Changelog](https://github.com/nodeca/js-yaml/blob/master/CHANGELOG.md) - [Commits](https://github.com/nodeca/js-yaml/compare/4.3.0...5.0.0) --- updated-dependencies: - dependency-name: js-yaml dependency-version: 5.0.0 dependency-type: direct:production update-type: version-update:semver-major ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- gitnexus/package-lock.json | 10 +++++----- gitnexus/package.json | 2 +- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 9aa2ef8da..174b31e55 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -24,7 +24,7 @@ "graphology-indices": "^0.17.0", "graphology-utils": "^2.3.0", "ignore": "^7.0.5", - "js-yaml": "^4.1.1", + "js-yaml": "^5.0.0", "jsonc-parser": "^3.3.1", "mnemonist": "^0.40.3", "node-addon-api": "^8.0.0", @@ -3589,9 +3589,9 @@ "license": "MIT" }, "node_modules/js-yaml": { - "version": "4.3.0", - "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.0.tgz", - "integrity": "sha512-1td788aAnnZ5qs7V2QIRl1owjtYpbKt749Y3xauqQgwIIGF/xXWz1wMTEBx5O3LK3lXLVuqXPdPxj2BoFHaW9Q==", + "version": "5.0.0", + "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-5.0.0.tgz", + "integrity": "sha512-GSvaPUbk1U+FMZ7rJzF+F8e5YVtu7KnD40et/5rBXXRBv2jCO9L3qCewvIDDdudC0QycTFlf6EAA+h3kxBsuUw==", "funding": [ { "type": "github", @@ -3607,7 +3607,7 @@ "argparse": "^2.0.1" }, "bin": { - "js-yaml": "bin/js-yaml.js" + "js-yaml": "bin/js-yaml.mjs" } }, "node_modules/jsesc": { diff --git a/gitnexus/package.json b/gitnexus/package.json index fb60f0dbe..696c74dad 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -70,7 +70,7 @@ "graphology-indices": "^0.17.0", "graphology-utils": "^2.3.0", "ignore": "^7.0.5", - "js-yaml": "^4.1.1", + "js-yaml": "^5.0.0", "jsonc-parser": "^3.3.1", "mnemonist": "^0.40.3", "node-addon-api": "^8.0.0", From 59359210794efdf36b4589254a949fb20814c752 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 21 Jul 2026 05:44:37 +0100 Subject: [PATCH 15/61] chore(deps)(deps-dev): bump @types/node in /gitnexus (#2588) Bumps [@types/node](https://github.com/DefinitelyTyped/DefinitelyTyped/tree/HEAD/types/node) from 25.9.5 to 26.0.0. - [Release notes](https://github.com/DefinitelyTyped/DefinitelyTyped/releases) - [Commits](https://github.com/DefinitelyTyped/DefinitelyTyped/commits/HEAD/types/node) --- updated-dependencies: - dependency-name: "@types/node" dependency-version: 26.0.0 dependency-type: direct:development update-type: version-update:semver-major ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- gitnexus/package-lock.json | 16 ++++++++-------- gitnexus/package.json | 2 +- 2 files changed, 9 insertions(+), 9 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 174b31e55..882a99731 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -59,7 +59,7 @@ "@types/cors": "^2.8.17", "@types/express": "^5.0.6", "@types/js-yaml": "^4.0.9", - "@types/node": "^25.6.0", + "@types/node": "^26.0.0", "@types/uuid": "^11.0.0", "@vitest/coverage-v8": "^4.0.18", "gitnexus-shared": "file:../gitnexus-shared", @@ -1946,13 +1946,13 @@ "license": "MIT" }, "node_modules/@types/node": { - "version": "25.9.5", - "resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.5.tgz", - "integrity": "sha512-OScDchr2fwuUmWdf4kZ9h7PcJiYDVInhJizG/biAq3cAvqwYktuy/TYGGdZNMtNTFUP7rnb0NU4TUdm82kt4Rg==", + "version": "26.0.0", + "resolved": "https://registry.npmjs.org/@types/node/-/node-26.0.0.tgz", + "integrity": "sha512-vf2YFi1iY9lHGwNJMs01biZFbKJkrZR1T6/MlzjhJLPdntOHLhTrDSnSVcdtvjihi4VQNlrFRIxLsDBlQpAipA==", "devOptional": true, "license": "MIT", "dependencies": { - "undici-types": ">=7.24.0 <7.24.7" + "undici-types": "~8.3.0" } }, "node_modules/@types/qs": { @@ -5510,9 +5510,9 @@ } }, "node_modules/undici-types": { - "version": "7.24.6", - "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.24.6.tgz", - "integrity": "sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg==", + "version": "8.3.0", + "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-8.3.0.tgz", + "integrity": "sha512-j375ScV60dom+YkPFIfTLcOiPxkN/buHz5GobjLhixFuANaNs3C9l4GmrWqejgXWJ7BbJcFYpTEUkS1Ge8bpZQ==", "devOptional": true, "license": "MIT" }, diff --git a/gitnexus/package.json b/gitnexus/package.json index 696c74dad..fc7e69e2a 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -106,7 +106,7 @@ "@types/cors": "^2.8.17", "@types/express": "^5.0.6", "@types/js-yaml": "^4.0.9", - "@types/node": "^25.6.0", + "@types/node": "^26.0.0", "@types/uuid": "^11.0.0", "@vitest/coverage-v8": "^4.0.18", "gitnexus-shared": "file:../gitnexus-shared", From 7534f53c27829fb35c3f91573fe4b323032d628e Mon Sep 17 00:00:00 2001 From: ChamHerry <51915924+ChamHerry@users.noreply.github.com> Date: Tue, 21 Jul 2026 12:46:20 +0800 Subject: [PATCH 16/61] feat(embeddings): control request-body dimensions via GITNEXUS_EMBEDDING_REQUEST_DIMS (#2574) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(embeddings): support GITNEXUS_EMBEDDING_REQUEST_DIMS=omit What: Honor GITNEXUS_EMBEDDING_REQUEST_DIMS=omit by suppressing the request-body `dimensions` field sent to HTTP embedding backends. Why: Strict OpenAI-compatible backends return vectors in the model's native size but reject an unfamiliar `dimensions` field, breaking `analyze --embeddings` against them. The var was parsed but never propagated, so `omit` was a no-op. How: Add `requestDimensions` to HttpConfig, return it from readConfig, and forward `config.requestDimensions` (not the validation-only `config.dimensions`) to `httpEmbedBatch`. Local dimension checks still use `config.dimensions`. Details: Coexists with the retry/pacing fields introduced upstream; both feature sets are preserved. Default behavior unchanged when REQUEST_DIMS is unset. Impact: gitnexus/src/core/embeddings/http-client.ts; README; unit tests. * fix(embeddings): name GITNEXUS_EMBEDDING_REQUEST_DIMS in its own config error Address the review findings on #2574. What: - A malformed GITNEXUS_EMBEDDING_REQUEST_DIMS now throws an error naming GITNEXUS_EMBEDDING_REQUEST_DIMS, not the sibling GITNEXUS_EMBEDDING_DIMS. - isHttpEmbeddingDimsError recognizes both leads, so the CLI still classifies the REQUEST_DIMS config mistake as a clean config error, not a stack dump. - Tests: numeric-override decoupling (DIMS=1024 validates the response while REQUEST_DIMS=512 is sent in the body), the omit aliases (none/off/false/0), and the malformed-value error path (which also pins the naming fix). - README documents the full accepted values: omit-aliases and integer override. Why: readConfig reused the DIMS error lead for the REQUEST_DIMS branch, so REQUEST_DIMS=garbage misdirected the operator to edit the wrong variable. The feature's actual decoupling and its non-omit inputs had no test coverage. Co-Authored-By: Claude Opus 4.8 (1M context) --------- Co-authored-by: wangxc Co-authored-by: Gergő Magyar Co-authored-by: Claude Opus 4.8 (1M context) --- gitnexus/README.md | 10 ++ gitnexus/src/core/embeddings/http-client.ts | 53 ++++++--- gitnexus/test/unit/http-embedder.test.ts | 112 ++++++++++++++++++++ 3 files changed, 161 insertions(+), 14 deletions(-) diff --git a/gitnexus/README.md b/gitnexus/README.md index abc193a40..dc8ef17e5 100644 --- a/gitnexus/README.md +++ b/gitnexus/README.md @@ -284,6 +284,7 @@ Set these env vars to use a remote OpenAI-compatible `/v1/embeddings` endpoint i export GITNEXUS_EMBEDDING_URL=http://your-server:8080/v1 export GITNEXUS_EMBEDDING_MODEL=BAAI/bge-large-en-v1.5 export GITNEXUS_EMBEDDING_DIMS=1024 # optional, default 384 +export GITNEXUS_EMBEDDING_REQUEST_DIMS=omit # optional: omit "dimensions", or an integer to override it export GITNEXUS_EMBEDDING_API_KEY=your-key # optional, default: "unused" export GITNEXUS_EMBEDDING_MAX_ATTEMPTS=3 # optional, total attempts (1-20) export GITNEXUS_EMBEDDING_RETRY_CAP_MS=5000 # optional, maximum retry delay @@ -291,6 +292,15 @@ export GITNEXUS_EMBEDDING_MIN_INTERVAL_MS=0 # optional, minimum request spacing gitnexus analyze . --embeddings ``` +`GITNEXUS_EMBEDDING_REQUEST_DIMS` controls only the `dimensions` field sent in +the request body, independently of `GITNEXUS_EMBEDDING_DIMS` (which still +validates the returned vector's length): + +- `omit` (or `none`, `off`, `false`, `0`) — do not send `dimensions` at all, for + strict backends that return the right vector size but reject the field. +- a positive integer — send that value instead of `GITNEXUS_EMBEDDING_DIMS`. +- unset — send `GITNEXUS_EMBEDDING_DIMS` (the previous behavior). + Works with Infinity, vLLM, TEI, llama.cpp, Ollama, LM Studio, or OpenAI. Retry and pacing settings are provider-neutral; provider-specific limits should be supplied through configuration. When unset, local embeddings are used unchanged. ## Multi-Repo Support diff --git a/gitnexus/src/core/embeddings/http-client.ts b/gitnexus/src/core/embeddings/http-client.ts index c85d9ff02..f3eb3bb5a 100644 --- a/gitnexus/src/core/embeddings/http-client.ts +++ b/gitnexus/src/core/embeddings/http-client.ts @@ -29,6 +29,7 @@ interface HttpConfig { maxAttempts: number; retryCapMs: number; minIntervalMs: number; + requestDimensions?: number; } export interface EmbeddingRequestOptions { @@ -106,20 +107,26 @@ const paceHttpRequest = async (minIntervalMs: number, signal?: AbortSignal): Pro }; /** - * Stable lead of the {@link readConfig} malformed-`GITNEXUS_EMBEDDING_DIMS` - * error. `readConfig` throws a plain `Error` (not an {@link HttpEmbeddingError}) - * because this is a *config* mistake, not an endpoint failure — so the CLI - * recognizes it by this lead ({@link isHttpEmbeddingDimsError}) and prints a - * clean config message instead of a raw stack dump. See #2385. + * Stable lead of a {@link readConfig} malformed dims-env error. `readConfig` + * throws a plain `Error` (not an {@link HttpEmbeddingError}) for a malformed + * `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS` because it's a + * *config* mistake, not an endpoint failure — so the CLI recognizes it by this + * lead ({@link isHttpEmbeddingDimsError}) and prints a clean config message + * instead of a raw stack dump. Each var names itself so the message points the + * operator at the variable they actually set, not a sibling. See #2385. */ -const EMBEDDING_DIMS_ENV_ERROR_LEAD = 'GITNEXUS_EMBEDDING_DIMS must be a positive integer'; +const dimsEnvErrorLead = (name: string): string => `${name} must be a positive integer`; +const EMBEDDING_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_DIMS'); +const EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_REQUEST_DIMS'); /** - * @internal Exported for the CLI analyze error handler. True when `message` is - * the {@link readConfig} malformed-DIMS config error (a plain `Error`). + * @internal Exported for the CLI analyze error handler. True when `message` is a + * {@link readConfig} malformed dims-env config error (a plain `Error`) — for + * either `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS`. */ export const isHttpEmbeddingDimsError = (message: string): boolean => - message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD); + message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD) || + message.includes(EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD); /** * Build config from the current process.env snapshot. @@ -147,6 +154,23 @@ const readConfig = (): HttpConfig | null => { dimensions = parsed; } + const rawRequestDims = process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS?.trim(); + let requestDimensions = dimensions; + if (rawRequestDims) { + if (/^(omit|none|off|false|0)$/i.test(rawRequestDims)) { + requestDimensions = undefined; + } else { + if (!/^\d+$/.test(rawRequestDims)) { + throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`); + } + const parsed = parseInt(rawRequestDims, 10); + if (parsed <= 0) { + throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`); + } + requestDimensions = parsed; + } + } + return { baseUrl: baseUrl.replace(/\/+$/, ''), model, @@ -163,6 +187,7 @@ const readConfig = (): HttpConfig | null => { 300_000, ), minIntervalMs: parseNonNegativeIntegerEnv('GITNEXUS_EMBEDDING_MIN_INTERVAL_MS', 0, 300_000), + requestDimensions, }; }; @@ -283,9 +308,9 @@ const isEmbeddingItem = (item: unknown): item is EmbeddingItem => * the `dimensions` field in the request body. Endpoints that implement * Matryoshka truncation (OpenAI text-embedding-3-*, Cohere embed-v3, * Voyage) return a truncated vector at that size; endpoints that do not - * recognise the field may ignore it or return 400. Leave - * `GITNEXUS_EMBEDDING_DIMS` unset for strict backends that reject - * unknown fields. + * recognise the field may ignore it or return 400. Set + * `GITNEXUS_EMBEDDING_REQUEST_DIMS=omit` for strict backends while keeping + * `GITNEXUS_EMBEDDING_DIMS` set to the returned vector size. */ const httpEmbedBatch = async ( url: string, @@ -434,7 +459,7 @@ export const httpEmbed = async ( config.model, config.apiKey, batchIndex, - config.dimensions, + config.requestDimensions, requestOptions, config.maxAttempts, config.retryCapMs, @@ -491,7 +516,7 @@ export const httpEmbedQuery = async ( config.model, config.apiKey, 0, - config.dimensions, + config.requestDimensions, requestOptions, config.maxAttempts, config.retryCapMs, diff --git a/gitnexus/test/unit/http-embedder.test.ts b/gitnexus/test/unit/http-embedder.test.ts index 63cd715f9..fb4dde0af 100644 --- a/gitnexus/test/unit/http-embedder.test.ts +++ b/gitnexus/test/unit/http-embedder.test.ts @@ -9,6 +9,7 @@ const ENV_KEYS = [ 'GITNEXUS_EMBEDDING_MAX_ATTEMPTS', 'GITNEXUS_EMBEDDING_RETRY_CAP_MS', 'GITNEXUS_EMBEDDING_MIN_INTERVAL_MS', + 'GITNEXUS_EMBEDDING_REQUEST_DIMS', ] as const; /** 384d mock vector matching the default schema dimensions. */ @@ -166,6 +167,30 @@ describe('HTTP embedding backend', () => { expect(result.length).toBe(1024); }); + it('can validate custom dims without forwarding dimensions to strict backends', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'bge-m3'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = 'omit'; + + const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec1024 }] }), + }), + ); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const result = await embedText('test text'); + + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect('dimensions' in body).toBe(false); + expect(body.model).toBe('bge-m3'); + expect(result.length).toBe(1024); + }); + it('forwards dimensions on the single-query path', async () => { process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; process.env.GITNEXUS_EMBEDDING_MODEL = 'text-embedding-3-large'; @@ -188,6 +213,93 @@ describe('HTTP embedding backend', () => { expect(result.length).toBe(512); }); + it('can omit dimensions on the single-query path while validating custom dims', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'bge-m3'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = 'omit'; + + const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec1024 }] }), + }), + ); + + const mod = await import('../../src/mcp/core/embedder.js'); + const result = await mod.embedQuery('query text'); + + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect('dimensions' in body).toBe(false); + expect(result.length).toBe(1024); + }); + + it.each(['none', 'off', 'false', '0'])( + 'treats GITNEXUS_EMBEDDING_REQUEST_DIMS=%s as omit and drops the request dimensions field', + async (alias) => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'bge-m3'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = alias; + + const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec1024 }] }), + }), + ); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const result = await embedText('test text'); + + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect('dimensions' in body).toBe(false); + expect(result.length).toBe(1024); + }, + ); + + it('sends REQUEST_DIMS as the request dimensions while DIMS validates the response', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'text-embedding-3-large'; + process.env.GITNEXUS_EMBEDDING_DIMS = '1024'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = '512'; + + // Response keeps the DIMS-validated length; only the outgoing request differs. + const vec1024 = Array.from({ length: 1024 }, (_, i) => i / 1024); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue({ + ok: true, + json: async () => ({ data: [{ embedding: vec1024 }] }), + }), + ); + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const result = await embedText('test text'); + + const body = JSON.parse((fetch as any).mock.calls[0][1].body); + expect(body.dimensions).toBe(512); + expect(result.length).toBe(1024); + }); + + it('rejects a malformed GITNEXUS_EMBEDDING_REQUEST_DIMS with an error naming that var', async () => { + process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; + process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; + process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS = 'garbage'; + + const { embedText } = await import('../../src/core/embeddings/embedder.js'); + const { isHttpEmbeddingDimsError } = await import('../../src/core/embeddings/http-client.js'); + const err = await embedText('test').catch((e: unknown) => e); + // Recognizable as a config error so the CLI prints a clean message... + expect(isHttpEmbeddingDimsError(String(err))).toBe(true); + // ...and it points the operator at the var they set, not GITNEXUS_EMBEDDING_DIMS. + expect(String(err)).toContain('GITNEXUS_EMBEDDING_REQUEST_DIMS must be a positive integer'); + }); + it('retries on server error', async () => { process.env.GITNEXUS_EMBEDDING_URL = 'http://test:8080/v1'; process.env.GITNEXUS_EMBEDDING_MODEL = 'test-model'; From 84b4402cd18a4f551a6eb05d0020fb24a8bdef16 Mon Sep 17 00:00:00 2001 From: ChunxueLi <54129170+ChunxueLi@users.noreply.github.com> Date: Tue, 21 Jul 2026 12:48:55 +0800 Subject: [PATCH 17/61] fix: remove hardcoded 300-flows cap for large repositories (#2198) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix: remove hardcoded 300-flows cap for large repositories The dynamicMaxProcesses was capped at 300 via Math.min(300, ...), causing large repositories (280K+ nodes) to lose execution flows. Change: Remove the Math.min(300, ...) cap, keep dynamic calculation. Effect: 280K-node repo: 300 → 1617 flows. * test: add regression for dynamic maxProcesses sizing (#2198) Verify that processProcesses honours maxProcesses > 300 without truncation. Addresses the optional follow-up suggested by @koriyoshi2041. * test: exercise computeDynamicMaxProcesses at the phase layer (#2198) Extract from the inline expression in so the regression test can exercise the function that actually contained the removed cap. The previous test called directly with , which passes regardless of whether the phase-level cap is present — never had the cap. The new test suite covers: - floor (20) for tiny repos - linear scaling in the 0–3000 range - growth past 300 for large repos (the actual regression) - explicit assertion that reintroducing Math.min(300, …) would fail Addresses review feedback from @azizur100389. --------- Co-authored-by: Ubuntu Co-authored-by: Gergő Magyar --- .../ingestion/pipeline-phases/processes.ts | 13 ++++++- gitnexus/test/unit/process-processor.test.ts | 37 +++++++++++++++++++ 2 files changed, 49 insertions(+), 1 deletion(-) diff --git a/gitnexus/src/core/ingestion/pipeline-phases/processes.ts b/gitnexus/src/core/ingestion/pipeline-phases/processes.ts index fd2f25cc8..462d1523f 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/processes.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/processes.ts @@ -25,6 +25,17 @@ export interface ProcessesOutput { processResult: ProcessDetectionResult; } +/** + * Compute the dynamic max-processes budget from the symbol count. + * + * Scales proportionally (symbolCount / 10) with a floor of 20. + * Prior to #2198 this was capped at 300 via `Math.min(300, …)`, + * silently truncating process detection on large repositories. + */ +export function computeDynamicMaxProcesses(symbolCount: number): number { + return Math.max(20, Math.round(symbolCount / 10)); +} + export const processesPhase: PipelinePhase = { name: 'processes', // `structure` supplies `totalFiles` (progress counter) without the spurious @@ -53,7 +64,7 @@ export const processesPhase: PipelinePhase = { ctx.graph.forEachNode((n) => { if (n.label !== 'File') symbolCount++; }); - const dynamicMaxProcesses = Math.max(20, Math.min(300, Math.round(symbolCount / 10))); + const dynamicMaxProcesses = computeDynamicMaxProcesses(symbolCount); const processResult = await processProcesses( ctx.graph, diff --git a/gitnexus/test/unit/process-processor.test.ts b/gitnexus/test/unit/process-processor.test.ts index 18d5c7be3..09600c6de 100644 --- a/gitnexus/test/unit/process-processor.test.ts +++ b/gitnexus/test/unit/process-processor.test.ts @@ -3,6 +3,7 @@ import { processProcesses, type ProcessDetectionConfig, } from '../../src/core/ingestion/process-processor.js'; +import { computeDynamicMaxProcesses } from '../../src/core/ingestion/pipeline-phases/processes.js'; import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; import type { CommunityMembership } from '../../src/core/ingestion/community-processor.js'; @@ -522,4 +523,40 @@ describe('processProcesses', () => { expect(result.processes.length).toBeLessThanOrEqual(3); expect(result.stats.totalProcesses).toBeLessThanOrEqual(3); }); + + // Regression for #2198: the processesPhase dynamic sizing used to cap at + // Math.min(300, symbolCount/10). On large repos (>3000 symbols) that silently + // truncated the process index. The cap was removed by extracting + // computeDynamicMaxProcesses() — this test exercises the helper directly + // so it fails if someone reintroduces the 300 ceiling. + describe('computeDynamicMaxProcesses (#2198)', () => { + it('returns at least the floor of 20 for tiny repos', () => { + expect(computeDynamicMaxProcesses(0)).toBe(20); + expect(computeDynamicMaxProcesses(50)).toBe(20); // 50/10 = 5, floored to 20 + expect(computeDynamicMaxProcesses(199)).toBe(20); // 199/10 ≈ 20 + }); + + it('scales linearly within the old 0–3000 range', () => { + expect(computeDynamicMaxProcesses(500)).toBe(50); + expect(computeDynamicMaxProcesses(1000)).toBe(100); + expect(computeDynamicMaxProcesses(2999)).toBe(300); + }); + + it('grows past 300 for large repos — the regression that #2198 fixes', () => { + // 3001 symbols → 300 (just at the boundary) + expect(computeDynamicMaxProcesses(3001)).toBe(300); + // 3100 symbols → 310 — would have been capped to 300 before the fix + expect(computeDynamicMaxProcesses(3100)).toBe(310); + // 5000 symbols → 500 + expect(computeDynamicMaxProcesses(5000)).toBe(500); + // 28000 symbols (real-world large repo) → 2800 + expect(computeDynamicMaxProcesses(28000)).toBe(2800); + }); + + it('does NOT cap at 300 — fails if Math.min(300, ...) is reintroduced', () => { + const largeRepo = computeDynamicMaxProcesses(10000); + expect(largeRepo).toBe(1000); + expect(largeRepo).toBeGreaterThan(300); + }); + }); }); From 2dabdd391e506bb6504f96e7f5b3e13376281287 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 21 Jul 2026 06:44:37 +0100 Subject: [PATCH 18/61] chore(deps)(deps-dev): bump tar (#2591) Bumps the npm_and_yarn group with 1 update in the /gitnexus-web directory: [tar](https://github.com/isaacs/node-tar). Updates `tar` from 7.5.16 to 7.5.20 - [Release notes](https://github.com/isaacs/node-tar/releases) - [Changelog](https://github.com/isaacs/node-tar/blob/main/CHANGELOG.md) - [Commits](https://github.com/isaacs/node-tar/compare/v7.5.16...v7.5.20) --- updated-dependencies: - dependency-name: tar dependency-version: 7.5.20 dependency-type: indirect dependency-group: npm_and_yarn ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- gitnexus-web/package-lock.json | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/gitnexus-web/package-lock.json b/gitnexus-web/package-lock.json index 638123bf4..fd00e725a 100644 --- a/gitnexus-web/package-lock.json +++ b/gitnexus-web/package-lock.json @@ -7894,9 +7894,9 @@ } }, "node_modules/tar": { - "version": "7.5.16", - "resolved": "https://registry.npmjs.org/tar/-/tar-7.5.16.tgz", - "integrity": "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w==", + "version": "7.5.20", + "resolved": "https://registry.npmjs.org/tar/-/tar-7.5.20.tgz", + "integrity": "sha512-9FcyK4PA6+WbzlTM9WhQm6vB5W7cP7dUiPsv1g7YDwEQnQ1CGpK3MGlKk/ITVWMk05kHZuBhmVhiv8LZoy/PFQ==", "dev": true, "license": "BlueOak-1.0.0", "dependencies": { From 41e590fed730dcbb22ad31f9316e5ed6d36d0d02 Mon Sep 17 00:00:00 2001 From: Shining Date: Tue, 21 Jul 2026 13:44:43 +0800 Subject: [PATCH 19/61] fix(spring): harden configuration bindings --- .../languages/java/analysis-features.ts | 17 ++- .../languages/java/spring-config-bindings.ts | 42 ++++-- .../pipeline-phases/spring-config.ts | 126 +++++++++++++----- .../java/com/example/ConfigConsumers.java | 5 + .../src/main/resources/application.properties | 1 + .../spring-config-pipeline.test.ts | 71 +++++++++- gitnexus/test/unit/analysis-features.test.ts | 9 ++ .../unit/incremental-orchestration.test.ts | 58 ++++++++ .../test/unit/spring-config-bindings.test.ts | 46 +++++-- 9 files changed, 304 insertions(+), 71 deletions(-) diff --git a/gitnexus/src/core/ingestion/languages/java/analysis-features.ts b/gitnexus/src/core/ingestion/languages/java/analysis-features.ts index 55dee7c4b..b85616602 100644 --- a/gitnexus/src/core/ingestion/languages/java/analysis-features.ts +++ b/gitnexus/src/core/ingestion/languages/java/analysis-features.ts @@ -1,12 +1,19 @@ import type { AnalysisFeatureDescriptor } from '../../../analysis-features.js'; +function isSpringApplicationConfig(filePath: string): boolean { + const base = filePath.replaceAll('\\', '/').split('/').pop() ?? ''; + return /^application(?:-[^.]+)?\.(?:properties|ya?ml)$/i.test(base); +} + /** Durable completeness contract for Java Spring configuration bindings. */ export const SPRING_CONFIG_BINDINGS_FEATURE: AnalysisFeatureDescriptor = { id: 'spring.config-bindings', version: 1, - // Annotation presence is content-derived, so the conservative upgrade gate - // is every Java repository. This guarantees the first analyzer version that - // ships bindings performs a full rebuild even when config files are absent - // (missing @Value placeholders still need their unresolved marker). - appliesTo: (filePaths) => filePaths.some((filePath) => filePath.toLowerCase().endsWith('.java')), + // Java sources need consumer extraction even without config files (missing + // placeholders still get unresolved markers). Config-only repositories also + // need a one-time rebuild to backfill language-agnostic Property nodes. + appliesTo: (filePaths) => + filePaths.some( + (filePath) => filePath.toLowerCase().endsWith('.java') || isSpringApplicationConfig(filePath), + ), }; diff --git a/gitnexus/src/core/ingestion/languages/java/spring-config-bindings.ts b/gitnexus/src/core/ingestion/languages/java/spring-config-bindings.ts index 5f2eb07c3..59e98fdd0 100644 --- a/gitnexus/src/core/ingestion/languages/java/spring-config-bindings.ts +++ b/gitnexus/src/core/ingestion/languages/java/spring-config-bindings.ts @@ -19,7 +19,7 @@ const CONFIGURATION_PROPERTIES_ANNOTATION = interface JavaAnnotation { readonly name: string; - readonly text: string; + readonly node: SyntaxNode; } interface JavaImports { @@ -70,7 +70,7 @@ function annotationsOn(node: SyntaxNode): JavaAnnotation[] { for (const child of modifiers.namedChildren) { if (child.type !== 'annotation' && child.type !== 'marker_annotation') continue; const name = child.childForFieldName('name')?.text ?? child.firstNamedChild?.text; - if (name) annotations.push({ name, text: child.text }); + if (name) annotations.push({ name, node: child }); } return annotations; } @@ -88,8 +88,9 @@ function resolvesToAnnotation( } function decodeJavaStringLiteral(literal: string): string { + const delimiterLength = literal.startsWith('"""') && literal.endsWith('"""') ? 3 : 1; return literal - .slice(1, -1) + .slice(delimiterLength, -delimiterLength) .replace(/\\u([0-9a-fA-F]{4})/g, (_match, hex: string) => String.fromCharCode(Number.parseInt(hex, 16)), ) @@ -105,14 +106,16 @@ function decodeJavaStringLiteral(literal: string): string { }); } -function javaStringLiterals(text: string): string[] { - return [...text.matchAll(/"(?:\\.|[^"\\])*"/g)].map((match) => decodeJavaStringLiteral(match[0])); +function javaStringLiterals(annotation: SyntaxNode): string[] { + return annotation + .descendantsOfType('string_literal') + .map((literal) => decodeJavaStringLiteral(literal.text)); } /** Extract statically readable Spring placeholder keys from a Java annotation. */ -export function parseValuePlaceholderKeys(annotationText: string): string[] { +export function parseValuePlaceholderKeys(annotation: SyntaxNode): string[] { const keys = new Set(); - for (const literal of javaStringLiterals(annotationText)) { + for (const literal of javaStringLiterals(annotation)) { for (const match of literal.matchAll(/\$\{([^{}]+)\}/g)) { const key = match[1].split(':', 1)[0].trim(); if (/^[A-Za-z0-9_.-]+$/.test(key)) keys.add(key); @@ -122,11 +125,22 @@ export function parseValuePlaceholderKeys(annotationText: string): string[] { } /** Extract `prefix`/`value` (or the positional value) from the annotation. */ -export function parseConfigurationPropertiesPrefix(annotationText: string): string | null { - const named = /\b(?:prefix|value)\s*=\s*("(?:\\.|[^"\\])*")/.exec(annotationText); - const literal = named?.[1] ?? annotationText.match(/"(?:\\.|[^"\\])*"/)?.[0]; - if (literal === undefined) return null; - const prefix = decodeJavaStringLiteral(literal) +export function parseConfigurationPropertiesPrefix(annotation: SyntaxNode): string | null { + const named = annotation.descendantsOfType('element_value_pair').find((pair) => { + const key = pair.childForFieldName('key')?.text; + return key === 'prefix' || key === 'value'; + }); + const namedValue = named?.childForFieldName('value'); + const argumentsNode = annotation.childForFieldName('arguments'); + const literalNode = + (namedValue?.type === 'string_literal' + ? namedValue + : namedValue?.descendantsOfType('string_literal')[0]) ?? + (named === undefined + ? argumentsNode?.namedChildren.find((child) => child.type === 'string_literal') + : undefined); + if (literalNode === undefined) return null; + const prefix = decodeJavaStringLiteral(literalNode.text) .trim() .replace(/^\.+|\.+$/g, ''); return /^[A-Za-z0-9_.-]+$/.test(prefix) ? prefix : null; @@ -172,7 +186,7 @@ export function captureJavaSpringConfigConsumerFacts( const fieldName = declarator.childForFieldName('name')?.text; if (!fieldName) continue; for (const annotation of annotations) { - const keys = parseValuePlaceholderKeys(annotation.text); + const keys = parseValuePlaceholderKeys(annotation.node); if (keys.length > 0) { facts.push({ consumer: { kind: 'value', fieldName, line: field.startPosition.row + 1, keys }, @@ -192,7 +206,7 @@ export function captureJavaSpringConfigConsumerFacts( if (!resolvesToAnnotation(annotation.name, CONFIGURATION_PROPERTIES_ANNOTATION, imports)) { continue; } - const prefix = parseConfigurationPropertiesPrefix(annotation.text); + const prefix = parseConfigurationPropertiesPrefix(annotation.node); if (prefix !== null) { facts.push({ consumer: { diff --git a/gitnexus/src/core/ingestion/pipeline-phases/spring-config.ts b/gitnexus/src/core/ingestion/pipeline-phases/spring-config.ts index 4b222ff93..90fa040c6 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/spring-config.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/spring-config.ts @@ -25,6 +25,8 @@ import type { StructureOutput } from './structure.js'; const require = createRequire(import.meta.url); const yaml = require('js-yaml') as typeof import('js-yaml'); const MAX_CONFIG_FILE_BYTES = 2 * 1024 * 1024; +const MAX_YAML_TRAVERSAL_DEPTH = 128; +const MAX_YAML_TRAVERSAL_NODES = 100_000; export interface SpringConfigKey { readonly key: string; @@ -143,6 +145,21 @@ interface YamlMappingLocation { readonly line: number; } +interface YamlTraversalState { + remainingNodes: number; + readonly activeObjects: Set; +} + +function consumeYamlTraversalBudget(state: YamlTraversalState, depth: number): void { + if (depth > MAX_YAML_TRAVERSAL_DEPTH) { + throw new Error(`Spring YAML traversal depth exceeds ${MAX_YAML_TRAVERSAL_DEPTH}`); + } + state.remainingNodes--; + if (state.remainingNodes < 0) { + throw new Error(`Spring YAML traversal exceeds ${MAX_YAML_TRAVERSAL_NODES} nodes`); + } +} + function isObjectValue(value: unknown): value is object { return value !== null && typeof value === 'object'; } @@ -174,15 +191,25 @@ function findYamlMappingLocation( event: YamlParseEvent | undefined, key: string, objectEvents: WeakMap, + traversal: YamlTraversalState, visited = new Set(), + depth = 0, ): YamlMappingLocation | undefined { + consumeYamlTraversalBudget(traversal, depth); const resolved = resolveYamlAliasEvent(event, objectEvents); if (resolved === undefined || visited.has(resolved)) return undefined; visited.add(resolved); if (resolved.kind === 'sequence') { for (const child of resolved.children) { - const found = findYamlMappingLocation(child, key, objectEvents, visited); + const found = findYamlMappingLocation( + child, + key, + objectEvents, + traversal, + visited, + depth + 1, + ); if (found !== undefined) return found; } return undefined; @@ -195,7 +222,14 @@ function findYamlMappingLocation( return { valueEvent: direct.valueEvent, line: direct.keyEvent.startLine }; } for (const merge of pairs.filter((pair) => pair.key === '<<')) { - const found = findYamlMappingLocation(merge.valueEvent, key, objectEvents, visited); + const found = findYamlMappingLocation( + merge.valueEvent, + key, + objectEvents, + traversal, + visited, + depth + 1, + ); if (found !== undefined) return found; } return undefined; @@ -207,45 +241,60 @@ function flattenYamlValue( prefix: string, out: Map, objectEvents: WeakMap, + traversal: YamlTraversalState, sourceLine = event?.startLine ?? 1, + depth = 0, ): void { + consumeYamlTraversalBudget(traversal, depth); const resolvedEvent = resolveYamlAliasEvent(event, objectEvents); - if (Array.isArray(value)) { - if (value.length === 0 && prefix.length > 0 && !out.has(prefix)) out.set(prefix, sourceLine); - value.forEach((item, index) => - flattenYamlValue( - item, - resolvedEvent?.children[index], - `${prefix}[${index}]`, - out, - objectEvents, - sourceLine, - ), - ); - return; - } - if ( - value !== null && - typeof value === 'object' && - (resolvedEvent?.kind === 'mapping' || resolvedEvent === undefined) - ) { - const entries = Object.entries(value as Record); - if (entries.length === 0 && prefix.length > 0 && !out.has(prefix)) out.set(prefix, sourceLine); - for (const [key, nested] of entries) { - const next = prefix.length === 0 ? key : `${prefix}.${key}`; - const location = findYamlMappingLocation(resolvedEvent, key, objectEvents); - flattenYamlValue( - nested, - location?.valueEvent, - next, - out, - objectEvents, - location?.line ?? sourceLine, + const trackedObject = isObjectValue(value) ? value : undefined; + if (trackedObject !== undefined && traversal.activeObjects.has(trackedObject)) return; + if (trackedObject !== undefined) traversal.activeObjects.add(trackedObject); + try { + if (Array.isArray(value)) { + if (value.length === 0 && prefix.length > 0 && !out.has(prefix)) out.set(prefix, sourceLine); + value.forEach((item, index) => + flattenYamlValue( + item, + resolvedEvent?.children[index], + `${prefix}[${index}]`, + out, + objectEvents, + traversal, + sourceLine, + depth + 1, + ), ); + return; } - return; + if ( + value !== null && + typeof value === 'object' && + (resolvedEvent?.kind === 'mapping' || resolvedEvent === undefined) + ) { + const entries = Object.entries(value as Record); + if (entries.length === 0 && prefix.length > 0 && !out.has(prefix)) + out.set(prefix, sourceLine); + for (const [key, nested] of entries) { + const next = prefix.length === 0 ? key : `${prefix}.${key}`; + const location = findYamlMappingLocation(resolvedEvent, key, objectEvents, traversal); + flattenYamlValue( + nested, + location?.valueEvent, + next, + out, + objectEvents, + traversal, + location?.line ?? sourceLine, + depth + 1, + ); + } + return; + } + if (prefix.length > 0 && !out.has(prefix)) out.set(prefix, sourceLine); + } finally { + if (trackedObject !== undefined) traversal.activeObjects.delete(trackedObject); } - if (prefix.length > 0 && !out.has(prefix)) out.set(prefix, sourceLine); } /** Parse and flatten YAML leaves without retaining their values. */ @@ -259,6 +308,10 @@ export function parseSpringYaml( const documentEvents: YamlParseEvent[] = []; const objectEvents = new WeakMap(); const documents: unknown[] = []; + const traversal: YamlTraversalState = { + remainingNodes: MAX_YAML_TRAVERSAL_NODES, + activeObjects: new Set(), + }; yaml.loadAll(content, (document) => documents.push(document), { schema: yaml.DEFAULT_SCHEMA, @@ -290,7 +343,7 @@ export function parseSpringYaml( }); documents.forEach((document, index) => - flattenYamlValue(document, documentEvents[index], '', flattened, objectEvents), + flattenYamlValue(document, documentEvents[index], '', flattened, objectEvents, traversal), ); return [...flattened.entries()] .sort(([left], [right]) => left.localeCompare(right)) @@ -350,7 +403,6 @@ export const springConfigPhase: PipelinePhase = { filePath: entry.filePath, startLine: entry.line, endLine: entry.line, - language: entry.format, description: entry.profile ? `${SPRING_CONFIG_DESCRIPTION} (profile: ${entry.profile})` : SPRING_CONFIG_DESCRIPTION, diff --git a/gitnexus/test/fixtures/spring-config-app/src/main/java/com/example/ConfigConsumers.java b/gitnexus/test/fixtures/spring-config-app/src/main/java/com/example/ConfigConsumers.java index 49e6847dd..0d9803d06 100644 --- a/gitnexus/test/fixtures/spring-config-app/src/main/java/com/example/ConfigConsumers.java +++ b/gitnexus/test/fixtures/spring-config-app/src/main/java/com/example/ConfigConsumers.java @@ -17,4 +17,9 @@ class ServiceProperties { private Retry retry; } +@ConfigurationProperties("service") +class UnmatchedServiceProperties { + private String unrelated; +} + class Retry {} diff --git a/gitnexus/test/fixtures/spring-config-app/src/main/resources/application.properties b/gitnexus/test/fixtures/spring-config-app/src/main/resources/application.properties index b763cc27c..fff1b7e68 100644 --- a/gitnexus/test/fixtures/spring-config-app/src/main/resources/application.properties +++ b/gitnexus/test/fixtures/spring-config-app/src/main/resources/application.properties @@ -1 +1,2 @@ payment.timeout=30 +service.endpoint=https://base.example.test diff --git a/gitnexus/test/integration/spring-config-pipeline.test.ts b/gitnexus/test/integration/spring-config-pipeline.test.ts index 797262816..002014f28 100644 --- a/gitnexus/test/integration/spring-config-pipeline.test.ts +++ b/gitnexus/test/integration/spring-config-pipeline.test.ts @@ -1,8 +1,11 @@ import path from 'node:path'; -import { beforeAll, describe, expect, it } from 'vitest'; +import { mkdir, writeFile } from 'node:fs/promises'; +import { beforeAll, describe, expect, it, vi } from 'vitest'; import type { GraphNode, GraphRelationship } from 'gitnexus-shared'; import { runPipelineFromRepo } from '../../src/core/ingestion/pipeline.js'; +import { SPRING_CONFIG_DESCRIPTION } from '../../src/core/ingestion/frameworks/spring/config-bindings.js'; import type { PipelineResult } from '../../types/pipeline.js'; +import { createTempDir } from '../helpers/test-db.js'; const FIXTURE = path.resolve(__dirname, '..', 'fixtures', 'spring-config-app'); const SHADOW_FIXTURE = path.resolve(__dirname, '..', 'fixtures', 'spring-config-shadow-app'); @@ -33,8 +36,18 @@ describe('Spring configuration binding pipeline', () => { .map((edge) => String(result.graph.getNode(edge.targetId)?.properties.name)) .sort(); + const targetFilesFrom = (source: GraphNode, targetName: string): string[] => + uses + .filter((edge) => edge.sourceId === source.id) + .map((edge) => result.graph.getNode(edge.targetId)) + .filter((node) => node?.properties.name === targetName) + .map((node) => String(node?.properties.filePath)) + .sort(); + it('creates key-only Property nodes for properties and profile YAML files', () => { - expect(nodeNamed('payment.timeout', 'application.properties')).toBeDefined(); + const propertiesKey = nodeNamed('payment.timeout', 'application.properties'); + expect(propertiesKey).toBeDefined(); + expect(propertiesKey?.properties).not.toHaveProperty('language'); expect(nodeNamed('service.endpoint', 'application-dev.yml')?.properties.description).toContain( 'profile: dev', ); @@ -65,10 +78,32 @@ describe('Spring configuration binding pipeline', () => { if (owner === undefined || endpoint === undefined || retry === undefined) { throw new Error('fixture ConfigurationProperties symbols missing'); } - expect(targetsFrom(owner)).toEqual(['service.endpoint', 'service.retry.max-attempts']); - expect(targetsFrom(endpoint)).toEqual(['service.endpoint']); + expect(targetsFrom(owner)).toEqual([ + 'service.endpoint', + 'service.endpoint', + 'service.retry.max-attempts', + ]); + expect(targetsFrom(endpoint)).toEqual(['service.endpoint', 'service.endpoint']); + expect(targetFilesFrom(endpoint, 'service.endpoint')).toEqual([ + 'src/main/resources/application-dev.yml', + 'src/main/resources/application.properties', + ]); expect(targetsFrom(retry)).toEqual(['service.retry.max-attempts']); }); + + it('keeps the class-level binding when no field relaxed-name matches', () => { + const owner = nodeNamed('UnmatchedServiceProperties', 'ConfigConsumers.java'); + const unrelated = nodeNamed('unrelated', 'ConfigConsumers.java'); + if (owner === undefined || unrelated === undefined) { + throw new Error('unmatched ConfigurationProperties symbols missing'); + } + expect(targetsFrom(owner)).toEqual([ + 'service.endpoint', + 'service.endpoint', + 'service.retry.max-attempts', + ]); + expect(targetsFrom(unrelated)).toEqual([]); + }); }); describe('Spring configuration annotation attribution', () => { @@ -91,3 +126,31 @@ describe('Spring configuration annotation attribution', () => { expect(String(fake.properties.description ?? '')).not.toContain('Spring config unresolved:'); }); }); + +describe('Spring configuration file safety bounds', () => { + it('fails closed for malformed and oversized configuration files', async () => { + const repo = await createTempDir(); + try { + const resources = path.join(repo.dbPath, 'src', 'main', 'resources'); + await mkdir(resources, { recursive: true }); + await writeFile(path.join(resources, 'application-broken.yml'), 'broken: [\n', 'utf8'); + await writeFile( + path.join(resources, 'application-oversized.properties'), + `oversized.key=${'x'.repeat(2 * 1024 * 1024)}\n`, + 'utf8', + ); + + // Let the scanner admit the file so this exercises springConfig's + // stricter 2 MiB cap rather than the scanner's default 512 KiB cap. + vi.stubEnv('GITNEXUS_MAX_FILE_SIZE', '4096'); + const result = await runPipelineFromRepo(repo.dbPath, () => {}, { skipGraphPhases: true }); + const configNodes = [...result.graph.iterNodes()].filter((node) => + String(node.properties.description ?? '').startsWith(SPRING_CONFIG_DESCRIPTION), + ); + expect(configNodes).toEqual([]); + } finally { + vi.unstubAllEnvs(); + await repo.cleanup(); + } + }); +}); diff --git a/gitnexus/test/unit/analysis-features.test.ts b/gitnexus/test/unit/analysis-features.test.ts index c2592b494..146965798 100644 --- a/gitnexus/test/unit/analysis-features.test.ts +++ b/gitnexus/test/unit/analysis-features.test.ts @@ -28,6 +28,15 @@ describe('analysis feature versions', () => { 'graph.class-framework-annotations': 1, 'spring.bean-inventory': 1, }); + expect( + resolveAnalysisFeatureVersions(FEATURES, [ + 'src/main/resources/application-local.yml', + 'README.md', + ]), + ).toEqual({ + 'graph.class-framework-annotations': 1, + 'spring.config-bindings': 1, + }); }); it('requires an exact, well-formed feature set', () => { diff --git a/gitnexus/test/unit/incremental-orchestration.test.ts b/gitnexus/test/unit/incremental-orchestration.test.ts index 12726a607..b74efcb47 100644 --- a/gitnexus/test/unit/incremental-orchestration.test.ts +++ b/gitnexus/test/unit/incremental-orchestration.test.ts @@ -44,6 +44,7 @@ import { } from '../helpers/embedding-seed.js'; import { CLASS_FRAMEWORK_ANNOTATIONS_FEATURE } from '../../src/core/analysis-features.js'; import { SPRING_BEAN_INVENTORY_FEATURE } from '../../src/core/ingestion/frameworks/spring/analysis-features.js'; +import { SPRING_CONFIG_BINDINGS_FEATURE } from '../../src/core/ingestion/languages/java/analysis-features.js'; const setupMiniRepo = () => setupSharedMiniRepo('gitnexus-incr-orch-'); @@ -102,6 +103,16 @@ async function setupKotlinSpringBeanIncrementalRepo() { return repo; } +async function setupSpringConfigIncrementalRepo() { + const repo = await createTempDir('gitnexus-incr-spring-config-'); + const resources = path.join(repo.dbPath, 'src', 'main', 'resources'); + await mkdir(resources, { recursive: true }); + await writeFile(path.join(resources, 'application.properties'), 'service.timeout=30\n', 'utf-8'); + execSync('git init', { cwd: repo.dbPath, stdio: 'pipe' }); + gitCommitAll(repo.dbPath, 'initial spring configuration'); + return repo; +} + async function readWildcardServiceAnnotations(repoPath: string): Promise { const adapter = await import('../../src/core/lbug/lbug-adapter.js'); const { lbugPath } = getStoragePaths(repoPath); @@ -120,6 +131,21 @@ async function readWildcardServiceAnnotations(repoPath: string): Promise { + const adapter = await import('../../src/core/lbug/lbug-adapter.js'); + const { lbugPath } = getStoragePaths(repoPath); + await adapter.initLbug(lbugPath); + try { + const rows = (await adapter.executeQuery( + "MATCH (p:Property) WHERE p.filePath = 'src/main/resources/application.properties' " + + 'RETURN p.name AS name ORDER BY p.name', + )) as Array<{ name?: unknown }>; + return rows.map((row) => String(row.name)); + } finally { + await adapter.closeLbug(); + } +} + /** * Direct count over INJECTS CodeRelation rows — mirrors pdg-mode-flip's * countBasicBlocks: reopen the repo DB, count, close (runFullAnalysis closes @@ -271,6 +297,38 @@ describe('runFullAnalysis — incremental orchestration', () => { } }, 300_000); + it('a config-only index missing Spring config evidence rebuilds and restores the scoped stamp', async () => { + const repo = await setupSpringConfigIncrementalRepo(); + try { + const { runFullAnalysis } = await import('../../src/core/run-analyze.js'); + await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} }); + const { storagePath } = getStoragePaths(repo.dbPath); + const meta = await loadMeta(storagePath); + expect(meta!.analysisFeatures).toEqual({ + [CLASS_FRAMEWORK_ANNOTATIONS_FEATURE.id]: CLASS_FRAMEWORK_ANNOTATIONS_FEATURE.version, + [SPRING_CONFIG_BINDINGS_FEATURE.id]: SPRING_CONFIG_BINDINGS_FEATURE.version, + }); + + await saveMeta(storagePath, withoutAnalysisFeature(meta!, SPRING_CONFIG_BINDINGS_FEATURE.id)); + const logs: string[] = []; + const reanalyzed = await runFullAnalysis( + repo.dbPath, + { skipAgentsMd: true }, + { onProgress: () => {}, onLog: (message) => logs.push(message) }, + ); + + expect(reanalyzed.alreadyUpToDate).toBeUndefined(); + expect(logs.join('\n')).toContain(`missing:${SPRING_CONFIG_BINDINGS_FEATURE.id}`); + expect(await readSpringConfigPropertyNames(repo.dbPath)).toEqual(['service.timeout']); + expect((await loadMeta(storagePath))!.analysisFeatures).toEqual({ + [CLASS_FRAMEWORK_ANNOTATIONS_FEATURE.id]: CLASS_FRAMEWORK_ANNOTATIONS_FEATURE.version, + [SPRING_CONFIG_BINDINGS_FEATURE.id]: SPRING_CONFIG_BINDINGS_FEATURE.version, + }); + } finally { + await repo.cleanup(); + } + }, 300_000); + it('adding the first JVM file re-evaluates capabilities after the pipeline and avoids a top-up', async () => { const repo = await setupMiniRepo(); try { diff --git a/gitnexus/test/unit/spring-config-bindings.test.ts b/gitnexus/test/unit/spring-config-bindings.test.ts index 321d3647a..a935039fc 100644 --- a/gitnexus/test/unit/spring-config-bindings.test.ts +++ b/gitnexus/test/unit/spring-config-bindings.test.ts @@ -6,11 +6,7 @@ import { parseSpringProperties, parseSpringYaml, } from '../../src/core/ingestion/pipeline-phases/spring-config.js'; -import { - extractJavaSpringConfigConsumers, - parseConfigurationPropertiesPrefix, - parseValuePlaceholderKeys, -} from '../../src/core/ingestion/languages/java/spring-config-bindings.js'; +import { extractJavaSpringConfigConsumers } from '../../src/core/ingestion/languages/java/spring-config-bindings.js'; describe('Spring configuration parsing', () => { it('recognizes base and profile-specific application config files', () => { @@ -83,13 +79,17 @@ describe('Spring configuration parsing', () => { expect(keys.some((entry) => entry.key.includes('<<'))).toBe(false); }); - it('parses Value defaults and ConfigurationProperties aliases', () => { - expect(parseValuePlaceholderKeys('@Value("${server.port:8080}")')).toEqual(['server.port']); + it('terminates cyclic YAML aliases and bounds deeply nested expansion', () => { expect( - parseConfigurationPropertiesPrefix('@ConfigurationProperties(prefix = "acme.api")'), - ).toBe('acme.api'); - expect(parseConfigurationPropertiesPrefix('@ConfigurationProperties("acme.api")')).toBe( - 'acme.api', + parseSpringYaml('cycle: &cycle { self: *cycle }\nhealthy: true\n', 'application.yml'), + ).toEqual([expect.objectContaining({ key: 'healthy', line: 2 })]); + + const aliasChain = ['level0: &level0 { leaf: true }']; + for (let index = 1; index <= 130; index++) { + aliasChain.push(`level${index}: &level${index} { next: *level${index - 1} }`); + } + expect(() => parseSpringYaml(aliasChain.join('\n'), 'application.yml')).toThrow( + 'Spring YAML traversal depth', ); }); }); @@ -144,6 +144,30 @@ describe('Java Spring configuration consumers', () => { }), ]); }); + + it('reads only string-literal AST nodes and ignores placeholders inside comments', () => { + const consumers = extractJavaSpringConfigConsumers(` + import org.springframework.beans.factory.annotation.Value; + import org.springframework.boot.context.properties.ConfigurationProperties; + + @ConfigurationProperties( + // legacy prefix: "old.unsafe" + value = "service" + ) + class ServiceProperties { + @Value( + /* legacy: "\${old.unsafe.key}" */ + "\${service.timeout:30}" + ) + private int timeout; + } + `); + + expect(consumers).toEqual([ + expect.objectContaining({ kind: 'value', keys: ['service.timeout'] }), + expect.objectContaining({ kind: 'configuration-properties', prefix: 'service' }), + ]); + }); }); describe('Spring configuration graph binding', () => { From b751418985169170f5d1c41f751502545989326a Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 06:06:42 +0000 Subject: [PATCH 20/61] fix(test): discover the installed FTS extension version dir instead of assuming it equals lbug.VERSION resolveInstalledFtsExtension (extension-binary-real.test.ts) and resolveSeedExtension (fts-extension-e2e.test.ts) both hardcoded the on-disk FTS extension path as .lbdb/extension//..., but LadybugDB's native INSTALL/LOAD resolves its own extension-ABI version directory, which does not always track the npm package version. Bumping @ladybugdb/core from 0.18.1 to 0.18.2 in this PR still installs into a 0.18.1 directory, so both hardcoded lookups came up empty and failed hard under GITNEXUS_REQUIRE_FTS=1 in CI (all platforms, shard 3). Add findInstalledFtsExtension() to discover the real installed file by scanning every version subdirectory, and use it from both test files. Co-Authored-By: Claude Sonnet 5 --- gitnexus/test/helpers/fts-availability.ts | 36 +++++++++++++++++++ .../integration/extension-binary-real.test.ts | 20 +++-------- .../integration/fts-extension-e2e.test.ts | 35 ++++++++---------- 3 files changed, 55 insertions(+), 36 deletions(-) diff --git a/gitnexus/test/helpers/fts-availability.ts b/gitnexus/test/helpers/fts-availability.ts index 28d8eb7dd..bc3620662 100644 --- a/gitnexus/test/helpers/fts-availability.ts +++ b/gitnexus/test/helpers/fts-availability.ts @@ -1,3 +1,39 @@ +import { existsSync, readdirSync, statSync } from 'node:fs'; +import { join } from 'node:path'; + +/** A valid `libfts.lbug_extension` is ~2.2MB; anything smaller is truncated/corrupt. */ +const MIN_VALID_FTS_EXTENSION_BYTES = 1024 * 1024; + +/** + * Find the installed FTS extension file under a `.lbdb/extension` root, + * discovering the version directory instead of assuming it equals the npm + * `@ladybugdb/core` package version. LadybugDB's native INSTALL/LOAD resolves + * its own extension-ABI version directory, which does not always track the + * npm package version — e.g. #2587: bumping the package from 0.18.1 to 0.18.2 + * still installs into a `0.18.1` directory, because the underlying + * extension-ABI build did not change with that patch release. + * + * Scans every version subdirectory for a `/fts/libfts.lbug_extension` + * file and returns the most recently modified one (the one an install/load + * actually just resolved), or null when nothing is installed. + */ +export const findInstalledFtsExtension = (extensionRoot: string): string | null => { + if (!existsSync(extensionRoot)) return null; + let best: { path: string; mtimeMs: number } | null = null; + for (const versionEntry of readdirSync(extensionRoot)) { + const versionDir = join(extensionRoot, versionEntry); + if (!statSync(versionDir).isDirectory()) continue; + for (const platformEntry of readdirSync(versionDir)) { + const candidate = join(versionDir, platformEntry, 'fts', 'libfts.lbug_extension'); + if (!existsSync(candidate)) continue; + const stat = statSync(candidate); + if (stat.size < MIN_VALID_FTS_EXTENSION_BYTES) continue; + if (!best || stat.mtimeMs > best.mtimeMs) best = { path: candidate, mtimeMs: stat.mtimeMs }; + } + } + return best?.path ?? null; +}; + export const FTS_UNAVAILABLE_NOTE = 'FTS extension unavailable (load-only policy; LOAD failed on this machine)'; diff --git a/gitnexus/test/integration/extension-binary-real.test.ts b/gitnexus/test/integration/extension-binary-real.test.ts index 27293c1a5..a2734d40f 100644 --- a/gitnexus/test/integration/extension-binary-real.test.ts +++ b/gitnexus/test/integration/extension-binary-real.test.ts @@ -2,21 +2,21 @@ import { copyFileSync, existsSync, mkdtempSync, - readdirSync, readFileSync, rmSync, - statSync, writeFileSync, } from 'node:fs'; import { homedir, tmpdir } from 'node:os'; import { join } from 'node:path'; import { afterAll, describe, expect, it } from 'vitest'; -import lbug from '@ladybugdb/core'; import { diagnoseExtensionLoad, inspectExtensionBinary, } from '../../src/core/lbug/extension-load-error.js'; -import { requireFtsResourceOrSkip } from '../helpers/fts-availability.js'; +import { + findInstalledFtsExtension, + requireFtsResourceOrSkip, +} from '../helpers/fts-availability.js'; /** * #2374: exercise the language-independent structural classifier against REAL @@ -49,17 +49,7 @@ function resolveLbugNative(): string | null { /** The actual installed FTS extension binary for the running lbug version. */ function resolveInstalledFtsExtension(): string | null { const home = process.env.USERPROFILE ?? process.env.HOME ?? homedir(); - const base = join(home, '.lbdb', 'extension', lbug.VERSION); - try { - const platformDir = readdirSync(base).find((entry) => - statSync(join(base, entry)).isDirectory(), - ); - if (!platformDir) return null; - const ext = join(base, platformDir, 'fts', 'libfts.lbug_extension'); - return existsSync(ext) ? ext : null; - } catch { - return null; - } + return findInstalledFtsExtension(join(home, '.lbdb', 'extension')); } const lbugNative = resolveLbugNative(); diff --git a/gitnexus/test/integration/fts-extension-e2e.test.ts b/gitnexus/test/integration/fts-extension-e2e.test.ts index 2c072a2fe..94d56cd1b 100644 --- a/gitnexus/test/integration/fts-extension-e2e.test.ts +++ b/gitnexus/test/integration/fts-extension-e2e.test.ts @@ -25,9 +25,9 @@ import path from 'path'; import fs from 'fs'; import os from 'os'; -import lbug from '@ladybugdb/core'; import { getExtensionInstallChildProcessArgs } from '../../src/core/lbug/extension-loader.js'; import { cleanupTempDirSync } from '../helpers/test-db.js'; +import { findInstalledFtsExtension } from '../helpers/fts-availability.js'; /** `.lbdb/extension///fts/libfts.lbug_extension`, discovered not hardcoded. */ let extensionRelPath: string; @@ -52,16 +52,12 @@ const makeTmpDir = (label: string): string => { * home — the production installer script, not a reimplementation. */ const resolveSeedExtension = (): void => { - const relBase = path.join('.lbdb', 'extension', lbug.VERSION); - const realVersionDir = path.join(os.homedir(), relBase); - const platformDirs = fs.existsSync(realVersionDir) ? fs.readdirSync(realVersionDir) : []; - for (const platform of platformDirs) { - const candidate = path.join(realVersionDir, platform, 'fts', 'libfts.lbug_extension'); - if (fs.existsSync(candidate) && fs.statSync(candidate).size > 1024 * 1024) { - extensionRelPath = path.join(relBase, platform, 'fts', 'libfts.lbug_extension'); - seedExtensionFile = candidate; - return; - } + const realExtensionRoot = path.join(os.homedir(), '.lbdb', 'extension'); + const installed = findInstalledFtsExtension(realExtensionRoot); + if (installed) { + extensionRelPath = path.relative(os.homedir(), installed); + seedExtensionFile = installed; + return; } // No local copy — run the real installer against a hermetic probe home. const probeHome = makeTmpDir('seed-home'); @@ -70,16 +66,13 @@ const resolveSeedExtension = (): void => { timeout: 120_000, env: { ...process.env, HOME: probeHome, USERPROFILE: probeHome }, }); - const probeVersionDir = path.join(probeHome, relBase); - const probePlatforms = fs.existsSync(probeVersionDir) ? fs.readdirSync(probeVersionDir) : []; - for (const platform of probePlatforms) { - const candidate = path.join(probeVersionDir, platform, 'fts', 'libfts.lbug_extension'); - if (install.status === 0 && fs.existsSync(candidate)) { - extensionRelPath = path.join(relBase, platform, 'fts', 'libfts.lbug_extension'); - seedExtensionFile = candidate; - networkAvailable = true; - return; - } + const probeExtensionRoot = path.join(probeHome, '.lbdb', 'extension'); + const probeInstalled = findInstalledFtsExtension(probeExtensionRoot); + if (install.status === 0 && probeInstalled) { + extensionRelPath = path.relative(probeHome, probeInstalled); + seedExtensionFile = probeInstalled; + networkAvailable = true; + return; } }; From 281664eb0a59570338bc573d4e2afc5e3d95eb6a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 21 Jul 2026 07:08:37 +0100 Subject: [PATCH 21/61] Revert "chore(deps)(deps): bump js-yaml from 4.3.0 to 5.0.0 in /gitnexus (#2586)" (#2596) This reverts commit 3f93bf22d6765fe5bc61268ea20fd4f0ca91de57. --- gitnexus/package-lock.json | 10 +++++----- gitnexus/package.json | 2 +- 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index 882a99731..d74fff71c 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -24,7 +24,7 @@ "graphology-indices": "^0.17.0", "graphology-utils": "^2.3.0", "ignore": "^7.0.5", - "js-yaml": "^5.0.0", + "js-yaml": "^4.1.1", "jsonc-parser": "^3.3.1", "mnemonist": "^0.40.3", "node-addon-api": "^8.0.0", @@ -3589,9 +3589,9 @@ "license": "MIT" }, "node_modules/js-yaml": { - "version": "5.0.0", - "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-5.0.0.tgz", - "integrity": "sha512-GSvaPUbk1U+FMZ7rJzF+F8e5YVtu7KnD40et/5rBXXRBv2jCO9L3qCewvIDDdudC0QycTFlf6EAA+h3kxBsuUw==", + "version": "4.3.0", + "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.0.tgz", + "integrity": "sha512-1td788aAnnZ5qs7V2QIRl1owjtYpbKt749Y3xauqQgwIIGF/xXWz1wMTEBx5O3LK3lXLVuqXPdPxj2BoFHaW9Q==", "funding": [ { "type": "github", @@ -3607,7 +3607,7 @@ "argparse": "^2.0.1" }, "bin": { - "js-yaml": "bin/js-yaml.mjs" + "js-yaml": "bin/js-yaml.js" } }, "node_modules/jsesc": { diff --git a/gitnexus/package.json b/gitnexus/package.json index fc7e69e2a..fbbafb3ec 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -70,7 +70,7 @@ "graphology-indices": "^0.17.0", "graphology-utils": "^2.3.0", "ignore": "^7.0.5", - "js-yaml": "^5.0.0", + "js-yaml": "^4.1.1", "jsonc-parser": "^3.3.1", "mnemonist": "^0.40.3", "node-addon-api": "^8.0.0", From 7c22975048ea268f018ba67b9af904be576512c7 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 21 Jul 2026 06:13:22 +0000 Subject: [PATCH 22/61] chore(deps)(deps): bump tar from 7.5.16 to 7.5.20 in /gitnexus Bumps [tar](https://github.com/isaacs/node-tar) from 7.5.16 to 7.5.20. - [Release notes](https://github.com/isaacs/node-tar/releases) - [Changelog](https://github.com/isaacs/node-tar/blob/main/CHANGELOG.md) - [Commits](https://github.com/isaacs/node-tar/compare/v7.5.16...v7.5.20) --- updated-dependencies: - dependency-name: tar dependency-version: 7.5.20 dependency-type: indirect ... Signed-off-by: dependabot[bot] --- gitnexus/package-lock.json | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index d74fff71c..b9567fb33 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -5135,9 +5135,9 @@ } }, "node_modules/tar": { - "version": "7.5.16", - "resolved": "https://registry.npmjs.org/tar/-/tar-7.5.16.tgz", - "integrity": "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w==", + "version": "7.5.20", + "resolved": "https://registry.npmjs.org/tar/-/tar-7.5.20.tgz", + "integrity": "sha512-9FcyK4PA6+WbzlTM9WhQm6vB5W7cP7dUiPsv1g7YDwEQnQ1CGpK3MGlKk/ITVWMk05kHZuBhmVhiv8LZoy/PFQ==", "license": "BlueOak-1.0.0", "dependencies": { "@isaacs/fs-minipass": "^4.0.0", From 30ec68fa7a03ac446c0ea6c898744e418e0f57a1 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 06:14:58 +0000 Subject: [PATCH 23/61] fix(test): harden findInstalledFtsExtension for cross-OS filesystem quirks MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Wrap the version-directory scan in try/catch so a transient FS error (permission denial, an AV file lock on Windows, a directory vanishing mid-scan) fails closed to null instead of throwing — matching the original callers' contract, and safer across the Windows/macOS/Linux CI matrix where these error modes differ. Also drop the redundant USERPROFILE/HOME manual chain in extension-binary-real.test.ts in favor of the repo's established os.homedir() convention (already used ~15 other places here), which Node resolves correctly per-OS and already honors env overrides. Co-Authored-By: Claude Sonnet 5 --- gitnexus/test/helpers/fts-availability.ts | 31 ++++++++++++------- .../integration/extension-binary-real.test.ts | 10 ++++-- 2 files changed, 26 insertions(+), 15 deletions(-) diff --git a/gitnexus/test/helpers/fts-availability.ts b/gitnexus/test/helpers/fts-availability.ts index bc3620662..30ba05b8a 100644 --- a/gitnexus/test/helpers/fts-availability.ts +++ b/gitnexus/test/helpers/fts-availability.ts @@ -18,20 +18,27 @@ const MIN_VALID_FTS_EXTENSION_BYTES = 1024 * 1024; * actually just resolved), or null when nothing is installed. */ export const findInstalledFtsExtension = (extensionRoot: string): string | null => { - if (!existsSync(extensionRoot)) return null; - let best: { path: string; mtimeMs: number } | null = null; - for (const versionEntry of readdirSync(extensionRoot)) { - const versionDir = join(extensionRoot, versionEntry); - if (!statSync(versionDir).isDirectory()) continue; - for (const platformEntry of readdirSync(versionDir)) { - const candidate = join(versionDir, platformEntry, 'fts', 'libfts.lbug_extension'); - if (!existsSync(candidate)) continue; - const stat = statSync(candidate); - if (stat.size < MIN_VALID_FTS_EXTENSION_BYTES) continue; - if (!best || stat.mtimeMs > best.mtimeMs) best = { path: candidate, mtimeMs: stat.mtimeMs }; + // Fail closed on any FS error (permission quirks, AV file locks on Windows, + // a directory vanishing mid-scan) — same contract as the callers this + // replaces: "not found" is a valid outcome, a thrown exception is not. + try { + if (!existsSync(extensionRoot)) return null; + let best: { path: string; mtimeMs: number } | null = null; + for (const versionEntry of readdirSync(extensionRoot)) { + const versionDir = join(extensionRoot, versionEntry); + if (!statSync(versionDir).isDirectory()) continue; + for (const platformEntry of readdirSync(versionDir)) { + const candidate = join(versionDir, platformEntry, 'fts', 'libfts.lbug_extension'); + if (!existsSync(candidate)) continue; + const stat = statSync(candidate); + if (stat.size < MIN_VALID_FTS_EXTENSION_BYTES) continue; + if (!best || stat.mtimeMs > best.mtimeMs) best = { path: candidate, mtimeMs: stat.mtimeMs }; + } } + return best?.path ?? null; + } catch { + return null; } - return best?.path ?? null; }; export const FTS_UNAVAILABLE_NOTE = diff --git a/gitnexus/test/integration/extension-binary-real.test.ts b/gitnexus/test/integration/extension-binary-real.test.ts index a2734d40f..f46958a2b 100644 --- a/gitnexus/test/integration/extension-binary-real.test.ts +++ b/gitnexus/test/integration/extension-binary-real.test.ts @@ -46,10 +46,14 @@ function resolveLbugNative(): string | null { return null; } -/** The actual installed FTS extension binary for the running lbug version. */ +/** + * The actual installed FTS extension binary for the running lbug version. + * `os.homedir()` already honors `$HOME` (POSIX) / `%USERPROFILE%` (Windows) — + * the same resolution LadybugDB's native layer uses — so it stays correct + * under the hermetic-home overrides other tests in this suite set via env vars. + */ function resolveInstalledFtsExtension(): string | null { - const home = process.env.USERPROFILE ?? process.env.HOME ?? homedir(); - return findInstalledFtsExtension(join(home, '.lbdb', 'extension')); + return findInstalledFtsExtension(join(homedir(), '.lbdb', 'extension')); } const lbugNative = resolveLbugNative(); From 1e190e6fdd79df868bb619691c46f71f96a46815 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 06:16:32 +0000 Subject: [PATCH 24/61] fix(java): emit a graph node for record_declaration (#2564) JAVA_QUERIES had no @definition.record capture, unlike its class_declaration/interface_declaration/enum_declaration siblings and unlike CSHARP_QUERIES' own record_declaration pattern. A Java record's container node was never created, so its HAS_METHOD edges were dropped at persistence even though ownership resolution computed a valid ownerId for its methods. Downstream label mapping, the class-extractor config, the dispatch table, and ownership reconciliation already treated 'Record' correctly - this was purely a missing structure-phase capture. --- .../src/core/ingestion/tree-sitter-queries.ts | 3 +- .../java-record-methods/Point.java | 11 ++++++ .../test/integration/resolvers/java.test.ts | 34 +++++++++++++++++++ 3 files changed, 47 insertions(+), 1 deletion(-) create mode 100644 gitnexus/test/fixtures/lang-resolution/java-record-methods/Point.java diff --git a/gitnexus/src/core/ingestion/tree-sitter-queries.ts b/gitnexus/src/core/ingestion/tree-sitter-queries.ts index d1b0359aa..3261913ba 100644 --- a/gitnexus/src/core/ingestion/tree-sitter-queries.ts +++ b/gitnexus/src/core/ingestion/tree-sitter-queries.ts @@ -743,10 +743,11 @@ export const PYTHON_QUERIES = ` // Java queries - works with tree-sitter-java export const JAVA_QUERIES = ` -; Classes, Interfaces, Enums, Annotations +; Classes, Interfaces, Enums, Records, Annotations (class_declaration name: (identifier) @name) @definition.class (interface_declaration name: (identifier) @name) @definition.interface (enum_declaration name: (identifier) @name) @definition.enum +(record_declaration name: (identifier) @name) @definition.record (annotation_type_declaration name: (identifier) @name) @definition.annotation ; Anonymous class bodies: new Runnable() { ... } — no @name capture; the diff --git a/gitnexus/test/fixtures/lang-resolution/java-record-methods/Point.java b/gitnexus/test/fixtures/lang-resolution/java-record-methods/Point.java new file mode 100644 index 000000000..f4471a1e1 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/java-record-methods/Point.java @@ -0,0 +1,11 @@ +package probe; + +public record Point(int x, int y) { + public int sum() { + return x + y; + } + + public int scaled(int factor) { + return sum() * factor; + } +} diff --git a/gitnexus/test/integration/resolvers/java.test.ts b/gitnexus/test/integration/resolvers/java.test.ts index 66bc820dd..0d53ce0eb 100644 --- a/gitnexus/test/integration/resolvers/java.test.ts +++ b/gitnexus/test/integration/resolvers/java.test.ts @@ -1174,6 +1174,40 @@ describe('Java chained method call resolution', () => { }); }); +// --------------------------------------------------------------------------- +// Java record: container node + HAS_METHOD edges +// Regression test for #2564: JAVA_QUERIES previously had no @definition.record +// capture, so a record never got a Class/Record graph node — its methods +// existed as ownerless orphans with no HAS_METHOD edge. +// --------------------------------------------------------------------------- + +describe('Java record method resolution (#2564)', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'java-record-methods'), () => {}); + }, 60000); + + it('detects a Record node for Point', () => { + const records = getNodesByLabel(result, 'Record'); + expect(records).toContain('Point'); + }); + + it('emits HAS_METHOD edges linking sum and scaled to Point', () => { + const hasMethod = getRelationships(result, 'HAS_METHOD'); + const sumEdge = hasMethod.find((e) => e.source === 'Point' && e.target === 'sum'); + const scaledEdge = hasMethod.find((e) => e.source === 'Point' && e.target === 'scaled'); + expect(sumEdge).toBeDefined(); + expect(scaledEdge).toBeDefined(); + }); + + it('resolves scaled() calling sum() via a CALLS edge', () => { + const calls = getRelationships(result, 'CALLS'); + const sumCall = calls.find((c) => c.target === 'sum' && c.source === 'scaled'); + expect(sumCall).toBeDefined(); + }); +}); + // --------------------------------------------------------------------------- // Java 16+ instanceof pattern variable: `if (obj instanceof User user)` // Phase 5.2: extractPatternBinding on instanceof_expression binds user → User. From 0b933aa43f8d354a49b150a73e0228cb2cdbd2f3 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 06:21:45 +0000 Subject: [PATCH 25/61] fix(java): treat a new-expression as a typed receiver for its chained call (#2564) new Local().inner() bound the whole object_creation_expression as @reference.receiver, so its raw source text ("new Local()") became the receiver name. That text can never match a scope binding, so the call silently fell through to name-only fallback resolution and could resolve to an unrelated same-named method on a collision. Normalize the receiver to the constructed type's simple name (reusing javaBaseSimpleNameOf, already used for the anonymous-class inheritance edge) so Case 2 (class-name / static receiver) in receiver-bound-calls.ts resolves it via its normal MRO walk. Mirrors the existing normalizePhpReceiver precedent in php/captures.ts - a language-local capture rewrite, no shared-pipeline change. --- .../core/ingestion/languages/java/captures.ts | 23 ++++++++++++ .../java-new-expr-chain-call/LocalChain.java | 12 +++++++ .../java-new-expr-chain-call/Other.java | 7 ++++ .../test/integration/resolvers/java.test.ts | 35 +++++++++++++++++++ 4 files changed, 77 insertions(+) create mode 100644 gitnexus/test/fixtures/lang-resolution/java-new-expr-chain-call/LocalChain.java create mode 100644 gitnexus/test/fixtures/lang-resolution/java-new-expr-chain-call/Other.java diff --git a/gitnexus/src/core/ingestion/languages/java/captures.ts b/gitnexus/src/core/ingestion/languages/java/captures.ts index 542d09d38..9864d0763 100644 --- a/gitnexus/src/core/ingestion/languages/java/captures.ts +++ b/gitnexus/src/core/ingestion/languages/java/captures.ts @@ -150,6 +150,29 @@ export function emitJavaScopeCaptures( continue; } + // Normalize a `new`-expression receiver to its constructed type's simple + // name: `new Local().inner()` binds the WHOLE `object_creation_expression` + // as `@reference.receiver`, so its raw text is `"new Local()"` — a string + // that can never match a scope binding, so the compound-receiver resolver + // silently falls through to name-only fallback resolution and picks the + // wrong same-named method on a collision (#2564). Rewriting the text to + // just `Local` lets Case 2 (class-name / static receiver) in + // receiver-bound-calls.ts resolve it via its normal MRO walk. Mirrors the + // established `normalizePhpReceiver` precedent (php/captures.ts) — a + // language-local capture rewrite, no shared-pipeline change. + if (grouped['@reference.receiver'] !== undefined) { + const receiverNode = nodeIfType(nodeMap['@reference.receiver'], 'object_creation_expression'); + const typeNode = receiverNode?.childForFieldName('type'); + const simpleName = typeNode ? javaBaseSimpleNameOf(typeNode) : undefined; + if (simpleName !== undefined) { + grouped['@reference.receiver'] = syntheticCapture( + '@reference.receiver', + receiverNode!, + simpleName, + ); + } + } + // Filter read.member when it's a child of method_invocation or assignment. // `@reference.read.member` is captured directly on the `field_access` node. if (grouped['@reference.read.member'] !== undefined) { diff --git a/gitnexus/test/fixtures/lang-resolution/java-new-expr-chain-call/LocalChain.java b/gitnexus/test/fixtures/lang-resolution/java-new-expr-chain-call/LocalChain.java new file mode 100644 index 000000000..599f8d5ed --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/java-new-expr-chain-call/LocalChain.java @@ -0,0 +1,12 @@ +package probe; + +public class LocalChain { + void m() { + class Local { + void inner() { + System.out.println("right target"); + } + } + new Local().inner(); + } +} diff --git a/gitnexus/test/fixtures/lang-resolution/java-new-expr-chain-call/Other.java b/gitnexus/test/fixtures/lang-resolution/java-new-expr-chain-call/Other.java new file mode 100644 index 000000000..2b722fc3d --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/java-new-expr-chain-call/Other.java @@ -0,0 +1,7 @@ +package probe; + +class Other { + void inner() { + System.out.println("wrong target"); + } +} diff --git a/gitnexus/test/integration/resolvers/java.test.ts b/gitnexus/test/integration/resolvers/java.test.ts index 0d53ce0eb..c7e0545c0 100644 --- a/gitnexus/test/integration/resolvers/java.test.ts +++ b/gitnexus/test/integration/resolvers/java.test.ts @@ -1174,6 +1174,41 @@ describe('Java chained method call resolution', () => { }); }); +// --------------------------------------------------------------------------- +// Chained call on a new-expression receiver: new Local().inner() +// The receiver of inner() is an object_creation_expression, not a variable. +// Regression test for #2564: without treating `new Local()` as a typed +// receiver, the call falls back to name-only resolution and can pick an +// unrelated same-named method (Other.inner) instead of Local.inner. +// --------------------------------------------------------------------------- + +describe('Java chained call on a new-expression receiver (#2564)', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'java-new-expr-chain-call'), () => {}); + }, 60000); + + it('detects LocalChain and Other classes', () => { + const classes = getNodesByLabel(result, 'Class'); + expect(classes).toContain('LocalChain'); + expect(classes).toContain('Other'); + }); + + it('resolves new Local().inner() to the local Local#inner, NOT Other#inner', () => { + const calls = getRelationships(result, 'CALLS'); + const localInner = calls.find( + (c) => + c.target === 'inner' && c.source === 'm' && c.targetFilePath.includes('LocalChain.java'), + ); + const otherInner = calls.find( + (c) => c.target === 'inner' && c.source === 'm' && c.targetFilePath.includes('Other.java'), + ); + expect(localInner).toBeDefined(); + expect(otherInner).toBeUndefined(); + }); +}); + // --------------------------------------------------------------------------- // Java record: container node + HAS_METHOD edges // Regression test for #2564: JAVA_QUERIES previously had no @definition.record From 1595a90a13681da02ebd82540d012c67a58eacbd Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 06:54:10 +0000 Subject: [PATCH 26/61] fix(storage): bump INCREMENTAL_SCHEMA_VERSION for the Java record fix (#2564) Review finding: the record_declaration container-node fix (894110bf) makes previously-uncaptured Record nodes and HAS_METHOD edges appear for the first time, but the incremental write set only covers changed files. Without this bump, an existing index would silently keep omitting the Record node and its HAS_METHOD edges for unchanged record files after an ordinary incremental analyze. Same contract as v7 (#2437/#2522) and the two closest precedents, v8 (#2550) and v9 (#2555), which bumped this constant for the identical "model X as first-class node" class of change. --- gitnexus/src/storage/repo-manager.ts | 10 +++++++++- gitnexus/test/unit/call-summary-schema-version.test.ts | 10 +++++++--- 2 files changed, 16 insertions(+), 4 deletions(-) diff --git a/gitnexus/src/storage/repo-manager.ts b/gitnexus/src/storage/repo-manager.ts index 568d288bd..e34432a7a 100644 --- a/gitnexus/src/storage/repo-manager.ts +++ b/gitnexus/src/storage/repo-manager.ts @@ -429,8 +429,16 @@ export interface RepoMeta { * `E.hook` to `E$1.hook`, and nested-host anonymous names re-key * (`EnumWrap$1` → `EnumWrap$Mode$1`). Same contract as v8: identities move * on unchanged files; force a full re-analyze. + * v10: Java `record_declaration` now emits a first-class `Record` graph node + * (#2564): a record's container node was previously never created (JAVA_QUERIES + * had no capture for it), so its methods existed as ownerless Method nodes + * with no `HAS_METHOD` edge. The incremental write set only covers changed + * files — a top-up against a pre-v10 index would keep silently omitting the + * `Record` node and its `HAS_METHOD` edges for every unchanged record file + * (same v7 contract: new nodes/edges the incremental path would otherwise + * never backfill); force a full re-analyze instead. */ -export const INCREMENTAL_SCHEMA_VERSION = 9; +export const INCREMENTAL_SCHEMA_VERSION = 10; export interface IndexedRepo { repoPath: string; diff --git a/gitnexus/test/unit/call-summary-schema-version.test.ts b/gitnexus/test/unit/call-summary-schema-version.test.ts index 8e7e3b5ce..a8240fa0f 100644 --- a/gitnexus/test/unit/call-summary-schema-version.test.ts +++ b/gitnexus/test/unit/call-summary-schema-version.test.ts @@ -73,8 +73,8 @@ describe('CALL_SUMMARY relation-type exclusion (U-C1)', () => { }); describe('CALL_SUMMARY incremental reuse gate (U-C5)', () => { - it('INCREMENTAL_SCHEMA_VERSION is bumped to 9 (Java enum-constant-body + JLS-naming re-index window)', () => { - expect(INCREMENTAL_SCHEMA_VERSION).toBe(9); + it('INCREMENTAL_SCHEMA_VERSION is bumped to 10 (Java record container-node re-index window)', () => { + expect(INCREMENTAL_SCHEMA_VERSION).toBe(10); }); it('a pre-current stamp fails the `=== INCREMENTAL_SCHEMA_VERSION` reuse gate → forces full re-analyze', () => { @@ -108,7 +108,11 @@ describe('CALL_SUMMARY incremental reuse gate (U-C5)', () => { // topmost-anchored `EnumWrap$1`-style ids would be stranded alongside // the re-keyed ones on unchanged files → must NOT reuse. expect(passesReuseGate(8)).toBe(false); + // A pre-v10 (v9) index predates the Java record container-node fix + // (#2564) — a record's methods would keep being ownerless Method nodes + // with no HAS_METHOD edge on unchanged files → must NOT reuse. + expect(passesReuseGate(9)).toBe(false); // A current-version stamp passes the gate (incremental top-up eligible). - expect(passesReuseGate(9)).toBe(true); + expect(passesReuseGate(10)).toBe(true); }); }); From 1fd1f14cee3c30b3ac3a8a55d9d98f62e4213e47 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 05:47:36 +0000 Subject: [PATCH 27/61] refactor(search): extract dropSearchFTSIndexes from createSearchFTSIndexes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Pulls the existing per-index dropFTSIndex loop out into its own exported function so the incremental writeback can drop FTS indexes up front, before deleteNodesForFiles runs (#2589). No behavior change here — createSearchFTSIndexes calls the new function and still rebuilds every index afterward. --- gitnexus/src/core/search/fts-indexes.ts | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/gitnexus/src/core/search/fts-indexes.ts b/gitnexus/src/core/search/fts-indexes.ts index dfc4f2eeb..1207861b9 100644 --- a/gitnexus/src/core/search/fts-indexes.ts +++ b/gitnexus/src/core/search/fts-indexes.ts @@ -121,6 +121,20 @@ export function getSearchFTSStemmer(): string { return resolvedStemmer ?? resolveFTSStemmer(); } +/** + * Drop every configured FTS index (no-op per index when absent or unloadable + * — `dropFTSIndex` tolerates both). Callable ahead of any DML that mutates an + * FTS-indexed table's rows: LadybugDB's FTS extension is not proven to + * survive a DETACH DELETE against a table that still carries a live index + * from a prior run (#2589) — dropping first removes that hazard entirely, + * regardless of whether it also fixed a specific native inconsistency. + */ +export async function dropSearchFTSIndexes(): Promise { + for (const { table, indexName } of FTS_INDEXES) { + await dropFTSIndex(table, indexName); + } +} + export async function createSearchFTSIndexes( options?: CreateSearchFTSIndexesOptions, ): Promise { From c548652eca213973eedeed59861a2737ac3a43ff Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 05:56:04 +0000 Subject: [PATCH 28/61] fix(lbug): stop dropFTSIndex from swallowing genuine engine failures dropFTSIndex previously caught and discarded every DROP_FTS_INDEX error unconditionally. Extract isBenignDropFtsIndexError, a pure classifier for the two legitimate "nothing to drop" cases (Binder/Catalog exceptions: index never created, or the FTS function isn't registered) verified end-to-end against @ladybugdb/core 0.18.x's real conn.query() error text. Anything else -- e.g. the Runtime exception "FTS index is inconsistent" class from #2589 -- now rethrows instead of being masked, so a corrupted index can no longer persist across analyze runs undetected. --- gitnexus/src/core/lbug/lbug-adapter.ts | 28 ++++++++- ...rop-fts-index-error-classification.test.ts | 58 +++++++++++++++++++ 2 files changed, 83 insertions(+), 3 deletions(-) create mode 100644 gitnexus/test/unit/drop-fts-index-error-classification.test.ts diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index 51af63b03..37cbd1bc6 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -2904,7 +2904,26 @@ export const queryFTS = async ( }; /** - * Drop an FTS index + * True for the two benign "nothing to drop" `DROP_FTS_INDEX` failures — + * both catalog/binder exceptions, LadybugDB's classes for "this name isn't + * bound to anything right now" (probe-verified end-to-end through + * `dropFTSIndex`'s real `conn.query()` path against @ladybugdb/core + * 0.18.x): the named index was never created (`Binder exception: Table + * doesn't have an index with name .`), or the FTS extension/function + * isn't registered at all (`Catalog exception: function DROP_FTS_INDEX is + * not defined...`). A real engine failure — e.g. the `Runtime exception: + * FTS index '' is inconsistent: ...` class from #2589 — is a + * DIFFERENT exception class (an execution-time failure, not a catalog/bind + * lookup miss), so this returns false for it. Pure string logic so it is + * unit-testable without a native LadybugDB connection. + */ +export const isBenignDropFtsIndexError = (message: string): boolean => + message.includes('Binder exception:') || message.includes('Catalog exception:'); + +/** + * Drop an FTS index. Tolerates only {@link isBenignDropFtsIndexError} — + * anything else rethrows instead of being silently masked, which previously + * let a corrupted index persist across analyze runs undetected. */ export const dropFTSIndex = async (tableName: string, indexName: string): Promise => { if (!conn) { @@ -2913,8 +2932,11 @@ export const dropFTSIndex = async (tableName: string, indexName: string): Promis try { await queryAndDrain(conn, `CALL DROP_FTS_INDEX('${tableName}', '${indexName}')`); - } catch { - // Index may not exist + } catch (e: unknown) { + const msg = e instanceof Error ? e.message : String(e); + if (!isBenignDropFtsIndexError(msg)) { + throw e; + } } finally { ensuredFTSIndexes.delete(ftsIndexKey(tableName, indexName)); } diff --git a/gitnexus/test/unit/drop-fts-index-error-classification.test.ts b/gitnexus/test/unit/drop-fts-index-error-classification.test.ts new file mode 100644 index 000000000..54d6d8597 --- /dev/null +++ b/gitnexus/test/unit/drop-fts-index-error-classification.test.ts @@ -0,0 +1,58 @@ +/** + * #2589: `dropFTSIndex` must tolerate only benign "nothing to drop" + * `DROP_FTS_INDEX` failures and rethrow everything else — previously it + * swallowed every error unconditionally, which could mask a genuinely + * corrupted FTS index across analyze runs. + * + * `isBenignDropFtsIndexError` is pure string logic (no native connection + * needed), so the classification itself is unit-tested directly, including + * against the exact reported #2589 error text — a native repro of that + * specific engine failure was not achieved during investigation, but the + * classifier's behavior for it is still provable from the message alone. + */ +import { describe, expect, it } from 'vitest'; +import { isBenignDropFtsIndexError, dropFTSIndex } from '../../src/core/lbug/lbug-adapter.js'; +import { withTestLbugDB } from '../helpers/test-indexed-db.js'; + +describe('isBenignDropFtsIndexError', () => { + it('is true for the FTS-extension/function-not-registered catalog error (probe-verified text)', () => { + expect( + isBenignDropFtsIndexError( + "Catalog exception: function DROP_FTS_INDEX is not defined. This function exists in the FTS extension. You can install and load the extension by running 'INSTALL FTS; LOAD EXTENSION FTS;'.", + ), + ).toBe(true); + }); + + it('is true for the index-never-created binder error (probe-verified against the real dropFTSIndex path)', () => { + expect( + isBenignDropFtsIndexError( + "Binder exception: Table File doesn't have an index with name file_fts.", + ), + ).toBe(true); + }); + + it('is false for the #2589 runtime inconsistency error (must surface, not be swallowed)', () => { + expect( + isBenignDropFtsIndexError( + "Runtime exception: FTS index 'file_fts' is inconsistent: term 'wiki' is missing during delete.", + ), + ).toBe(false); + }); + + it('is false for an unrelated failure', () => { + expect(isBenignDropFtsIndexError('Connection Exception: database is closed')).toBe(false); + }); +}); + +withTestLbugDB('drop-fts-index-benign-cases', (handle) => { + describe('dropFTSIndex end-to-end benign cases (#2589)', () => { + it('resolves cleanly when the named index was never created', async () => { + void handle; + const { executeQuery } = await import('../../src/core/lbug/lbug-adapter.js'); + await executeQuery( + `CREATE NODE TABLE IF NOT EXISTS DropProbe (id STRING PRIMARY KEY, content STRING)`, + ); + await expect(dropFTSIndex('DropProbe', 'drop_probe_never_created')).resolves.toBeUndefined(); + }, 120_000); + }); +}); From d7a4f4758064c7d5e7f3506d3edb89ad6c155079 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 06:09:57 +0000 Subject: [PATCH 29/61] fix(analyze): drop FTS indexes before the incremental DETACH DELETE Fixes #2589: incremental analyze intermittently crashed with "FTS index 'file_fts' is inconsistent: term is missing during delete" after markdown-only commits, and --repair-fts also failed in that state. deleteNodesForFiles' batched DETACH DELETE ran against tables that still carried the FTS index built at the end of the PREVIOUS analyze run -- createSearchFTSIndexes only drops+rebuilds every index in Phase 3, well after that delete already ran. LadybugDB's FTS extension is not proven to survive DML against an indexed table (its own docs never demonstrate the sequence). Call the new dropSearchFTSIndexes() up front in the non-escalated incremental branch, before deleteNodesForFiles -- Phase 3 still rebuilds every index from the final row set regardless. New end-to-end test drives a real runFullAnalysis full+incremental cycle and confirms it fails without this change (file_fts and 50 sibling indexes still present at delete time) and passes with it. --- gitnexus/src/core/run-analyze.ts | 16 ++- .../incremental-fts-drop-ordering.test.ts | 97 +++++++++++++++++++ 2 files changed, 112 insertions(+), 1 deletion(-) create mode 100644 gitnexus/test/unit/incremental-fts-drop-ordering.test.ts diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index ef77ba3d9..721d0ce17 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -39,6 +39,7 @@ import { escapeCypherString } from './lbug/cypher-escape.js'; import { buildSearchIndexesOrDegrade, createSearchFTSIndexes, + dropSearchFTSIndexes, initialiseSearchFTSStemmer, verifySearchFTSIndexes, } from './search/fts-indexes.js'; @@ -1661,7 +1662,20 @@ export async function runFullAnalysis( progress('lbug', pct, msg); }); } else { - // 1a. Remove the write set's existing rows — batched (#2409): one + // 1a. Drop every FTS index before touching a single row (#2589). + // `deleteNodesForFiles` below DETACH DELETEs rows out of tables + // that otherwise still carry the FTS index built at the end of + // the PREVIOUS analyze run — Phase 3 doesn't drop+rebuild it + // until well after this delete completes. LadybugDB's FTS + // extension is not proven to survive DML against an indexed + // table (its own docs never demonstrate it), and that ordering + // is exactly what produced "FTS index 'file_fts' is + // inconsistent: term is missing during delete". Dropping first + // removes the hazard outright; Phase 3's createSearchFTSIndexes + // rebuilds every index from the final row set regardless, so + // this is a no-op on its own drop step there. + await dropSearchFTSIndexes(); + // 1b. Remove the write set's existing rows — batched (#2409): one // DETACH DELETE per table per 200-file chunk. The former per-file // loop issued a count + delete per table per FILE — ~13k // single-row write transactions on a ~700-file write set — which diff --git a/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts b/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts new file mode 100644 index 000000000..3c034a616 --- /dev/null +++ b/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts @@ -0,0 +1,97 @@ +/** + * #2589: the incremental writeback must drop every FTS index BEFORE + * `deleteNodesForFiles` runs its batched DETACH DELETE — not only in + * Phase 3, after the delete already ran against a table still carrying the + * PREVIOUS run's index. This drives the real `runFullAnalysis` incremental + * path (real git repo, real LadybugDB, real FTS extension) and asserts, + * at the moment `deleteNodesForFiles` is invoked, that `SHOW_INDEXES()` + * already reports every FTS index absent — proving the drop-before-delete + * ordering end-to-end rather than only unit-testing the call sequence. + */ +import { readFile, writeFile } from 'fs/promises'; +import { execSync } from 'child_process'; +import path from 'path'; +import { afterEach, describe, expect, it, vi } from 'vitest'; +import { setupMiniRepo } from '../helpers/mini-repo.js'; +import { getStoragePaths } from '../../src/storage/repo-manager.js'; +import { FTS_INDEXES } from '../../src/core/search/fts-schema.js'; + +const ftsMustBeAvailable = process.env.GITNEXUS_REQUIRE_FTS === '1'; + +describe('runFullAnalysis incremental writeback — FTS drop-before-delete ordering (#2589)', () => { + afterEach(() => { + vi.doUnmock('../../src/core/lbug/lbug-adapter.js'); + vi.resetModules(); + }); + + it('SHOW_INDEXES() reports every FTS index absent by the time deleteNodesForFiles runs', async () => { + const lbugAdapter = await import('../../src/core/lbug/lbug-adapter.js'); + const { runFullAnalysis } = await import('../../src/core/run-analyze.js'); + + const repo = await setupMiniRepo('gitnexus-2589-fts-order-'); + try { + // First run: full rebuild, builds every FTS index for real (skipped + // gracefully below if the extension isn't available on this machine). + await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} }); + + // runFullAnalysis closes its own connection on return — open a fresh + // one just to probe SHOW_INDEXES(), then close it before the second + // run opens its own (LadybugDB is single-writer/single-connection). + const { lbugPath } = getStoragePaths(repo.dbPath); + await lbugAdapter.initLbug(lbugPath); + const showIndexNames = async (): Promise => { + const rows = (await lbugAdapter.executeQuery('CALL SHOW_INDEXES() RETURN *')) as Array< + Record + >; + return rows.map((r) => r.index_name).filter((n): n is string => typeof n === 'string'); + }; + const beforeChange = await showIndexNames(); + await lbugAdapter.closeLbug(); + + if (!FTS_INDEXES.every(({ indexName }) => beforeChange.includes(indexName))) { + if (ftsMustBeAvailable) { + throw new Error( + 'GITNEXUS_REQUIRE_FTS=1 but the first run did not build every FTS index — cannot verify the #2589 ordering fix.', + ); + } + console.warn( + '[incremental-fts-drop-ordering] Skipping — FTS extension unavailable or indexes not built.', + ); + return; + } + + // Spy on the real deleteNodesForFiles, recording the FTS index list at + // the exact moment it's invoked (before it does anything), then + // delegating to the real implementation so the run completes normally. + let indexNamesAtDeleteTime: string[] | undefined; + const originalDeleteNodesForFiles = lbugAdapter.deleteNodesForFiles; + vi.spyOn(lbugAdapter, 'deleteNodesForFiles').mockImplementation(async (filePaths, opts) => { + indexNamesAtDeleteTime = await showIndexNames(); + return originalDeleteNodesForFiles(filePaths, opts); + }); + + // Small change to a single file — stays well under the escalation + // threshold (50 files) on this 7-file mini-repo, so it takes the + // non-escalated (surgical) incremental branch this test targets. + const handlerPath = path.join(repo.dbPath, 'src', 'handler.ts'); + await writeFile(handlerPath, (await readFile(handlerPath, 'utf-8')) + '\n// #2589 ordering-test touch\n', 'utf-8'); + execSync('git -c user.name=test -c user.email=t@t -c commit.gpgsign=false add -A', { + cwd: repo.dbPath, + stdio: 'pipe', + }); + execSync( + 'git -c user.name=test -c user.email=t@t -c commit.gpgsign=false commit -q -m "#2589 ordering touch"', + { cwd: repo.dbPath, stdio: 'pipe' }, + ); + + await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} }); + + expect(indexNamesAtDeleteTime).toBeDefined(); + for (const { indexName } of FTS_INDEXES) { + expect(indexNamesAtDeleteTime).not.toContain(indexName); + } + } finally { + await repo.cleanup(); + } + }, 300_000); +}); From cdd0ce9b8e6fa9e3362c784ff0354137fa813c8f Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 21 Jul 2026 08:05:33 +0100 Subject: [PATCH 30/61] chore(deps)(deps): bump brace-expansion from 5.0.6 to 5.0.7 in /gitnexus (#2593) Bumps [brace-expansion](https://github.com/juliangruber/brace-expansion) from 5.0.6 to 5.0.7. - [Release notes](https://github.com/juliangruber/brace-expansion/releases) - [Commits](https://github.com/juliangruber/brace-expansion/compare/v5.0.6...v5.0.7) --- updated-dependencies: - dependency-name: brace-expansion dependency-version: 5.0.7 dependency-type: indirect ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- gitnexus/package-lock.json | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index b9567fb33..c761fb088 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -2340,9 +2340,9 @@ } }, "node_modules/brace-expansion": { - "version": "5.0.6", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.6.tgz", - "integrity": "sha512-kLpxurY4Z4r9sgMsyG0Z9uzsBlgiU/EFKhj/h91/8yHu0edo7XuixOIH3VcJ8kkxs6/jPzoI6U9Vj3WqbMQ94g==", + "version": "5.0.7", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.7.tgz", + "integrity": "sha512-7oFy703dxfY3/NLxC1fh2SUCQ0H9rmAY+5EpDVfXjUTTs+HEwR2nYaqLv+GWcTsumwxPfiz6CzCNkwXwBUwqCA==", "license": "MIT", "dependencies": { "balanced-match": "^4.0.2" From b028cc0212af49f73e56997215a9e883927add82 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Tue, 21 Jul 2026 08:05:49 +0100 Subject: [PATCH 31/61] chore(deps)(deps): bump body-parser from 2.2.2 to 2.3.0 in /gitnexus (#2594) Bumps [body-parser](https://github.com/expressjs/body-parser) from 2.2.2 to 2.3.0. - [Release notes](https://github.com/expressjs/body-parser/releases) - [Changelog](https://github.com/expressjs/body-parser/blob/master/HISTORY.md) - [Commits](https://github.com/expressjs/body-parser/compare/v2.2.2...v2.3.0) --- updated-dependencies: - dependency-name: body-parser dependency-version: 2.3.0 dependency-type: indirect ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- gitnexus/package-lock.json | 31 ++++++++++++++++++++++--------- 1 file changed, 22 insertions(+), 9 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index c761fb088..e10f45d6f 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -2316,20 +2316,20 @@ } }, "node_modules/body-parser": { - "version": "2.2.2", - "resolved": "https://registry.npmjs.org/body-parser/-/body-parser-2.2.2.tgz", - "integrity": "sha512-oP5VkATKlNwcgvxi0vM0p/D3n2C3EReYVX+DNYs5TjZFn/oQt2j+4sVJtSMr18pdRr8wjTcBl6LoV+FUwzPmNA==", + "version": "2.3.0", + "resolved": "https://registry.npmjs.org/body-parser/-/body-parser-2.3.0.tgz", + "integrity": "sha512-2cGmJupaNgg+QUwVLAucDuWuoMZ6EX9iHDRswZ5lsNYEmwPaRknMPCLZz07yTzVq/83p4o/wzbDZbBrTvGGTIw==", "license": "MIT", "dependencies": { "bytes": "^3.1.2", - "content-type": "^1.0.5", + "content-type": "^2.0.0", "debug": "^4.4.3", - "http-errors": "^2.0.0", - "iconv-lite": "^0.7.0", + "http-errors": "^2.0.1", + "iconv-lite": "^0.7.2", "on-finished": "^2.4.1", - "qs": "^6.14.1", - "raw-body": "^3.0.1", - "type-is": "^2.0.1" + "qs": "^6.15.2", + "raw-body": "^3.0.2", + "type-is": "^2.1.0" }, "engines": { "node": ">=18" @@ -2339,6 +2339,19 @@ "url": "https://opencollective.com/express" } }, + "node_modules/body-parser/node_modules/content-type": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/content-type/-/content-type-2.0.0.tgz", + "integrity": "sha512-j/O/d7GcZCyNl7/hwZAb606rzqkyvaDctLmckbxLzHvFBzTJHuGEdodATcP3yIRoDrLHkIATJuvzbFlp/ki2cQ==", + "license": "MIT", + "engines": { + "node": ">=18" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/express" + } + }, "node_modules/brace-expansion": { "version": "5.0.7", "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.7.tgz", From 5099e8ff1e983a2da420f0695440a1d9da7b2d79 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 07:18:38 +0000 Subject: [PATCH 32/61] test(bench): rebaseline java scope-capture fingerprint for record support (#2564) CI caught this: adding the record_declaration capture legitimately changes the pinned java capture fingerprint, same as every prior capture-behavior change to this language (#2550, #2555). Rebaselined following the established _rebaselined_* precedent; scaling ratio 1.059 stays well within the 1.5 budget. --- gitnexus/bench/scope-capture/baselines.json | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index d4e86debb..d1413db56 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -89,7 +89,7 @@ "_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0." }, "java": { - "fingerprint": "975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca", + "fingerprint": "85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537", "scaling_budget": 1.5, "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata; same-name lexical regions use an O(ancestor-depth) ID-set lookup. Prior d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a -> 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4; scaling 0.992 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Java method-reference/SAM callable flow facts with invocation-result suppression. Prior 062d754764aaa8a6772fb90875c710502a63e3e7a300e633942381ed914faada -> d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a; scaling 1.074 < 1.5.", @@ -97,7 +97,8 @@ "_note": "#1928 / #2045: F35 adds qualified + qualified-generic constructor query captures (`new pkg.Foo()`, `new a.b.Foo()`, `new pkg.Box()`); F38 synthesizes `@reference.call.constructor` on `super(...)`/`this(...)` explicit_constructor_invocation nodes; F41 generic-aware stripQualifier in interpret (type-binding normalization). + java-qualified-constructor and java-explicit-constructor fixtures. Pure capture-additive + fixture-corpus drift; scaling stays linear (~1.06).", "_rebaselined_2522_review_fixes": "PR #2522 review fixes: get/test dropped from callableProtocolMethods. Prior 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4 -> f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67; scaling ratio re-verified within budget.", "_rebaselined_2550_instance_model": "PR #2549 (#2550): anonymous class bodies emit synthesized @declaration.class/@declaration.name (Worker$N), an @reference.inherits to the constructed type, and receiver @type-binding.* captures; six new java-* fixtures joined the corpus. Prior f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67 -> d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90; scaling 1.058 < 1.5.", - "_rebaselined_2555_enum_constant_bodies": "PR for #2555: enum constant bodies emit synthesized E$N classes + @reference.inherits to the host enum; anonymous naming follows JLS 13.1 immediately-enclosing-type chains INCLUDING anonymous enclosing types (NestHost$1$1, N$1$1); six new java-* fixtures joined the corpus. Prior d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90 -> 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca; scaling 1.05 < 1.5." + "_rebaselined_2555_enum_constant_bodies": "PR for #2555: enum constant bodies emit synthesized E$N classes + @reference.inherits to the host enum; anonymous naming follows JLS 13.1 immediately-enclosing-type chains INCLUDING anonymous enclosing types (NestHost$1$1, N$1$1); six new java-* fixtures joined the corpus. Prior d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90 -> 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca; scaling 1.05 < 1.5.", + "_rebaselined_2564_record_capture": "PR for #2564: JAVA_QUERIES gained a (record_declaration name: (identifier) @name) @definition.record capture, previously entirely missing (record_declaration had no structure-phase capture at all, unlike class/interface/enum) - a record's methods existed as ownerless Method nodes with no HAS_METHOD edge. Two new java-* fixtures (java-record-methods, java-new-expr-chain-call) joined the corpus. Prior 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca -> 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537; scaling 1.059 < 1.5." }, "typescript": { "fingerprint": "3280b13d3f9378ab23eee31c2edc779b5a9ae1e7bb510c23a24855b44406d2f4", From f46015747030ecafabf00489b8123fc3d488dd70 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 07:25:15 +0000 Subject: [PATCH 33/61] style(test): fix prettier formatting in incremental-fts-drop-ordering.test.ts CI's prettier --check flagged one over-long line; no behavior change. --- gitnexus/test/unit/incremental-fts-drop-ordering.test.ts | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts b/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts index 3c034a616..da3339463 100644 --- a/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts +++ b/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts @@ -74,7 +74,11 @@ describe('runFullAnalysis incremental writeback — FTS drop-before-delete order // threshold (50 files) on this 7-file mini-repo, so it takes the // non-escalated (surgical) incremental branch this test targets. const handlerPath = path.join(repo.dbPath, 'src', 'handler.ts'); - await writeFile(handlerPath, (await readFile(handlerPath, 'utf-8')) + '\n// #2589 ordering-test touch\n', 'utf-8'); + await writeFile( + handlerPath, + (await readFile(handlerPath, 'utf-8')) + '\n// #2589 ordering-test touch\n', + 'utf-8', + ); execSync('git -c user.name=test -c user.email=t@t -c commit.gpgsign=false add -A', { cwd: repo.dbPath, stdio: 'pipe', From dac2b770a0728b8a5835c53049b2f7a6f9635f5e Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 07:57:28 +0000 Subject: [PATCH 34/61] fix(test): address gitnexus-review-agent findings on PR #2598 - Anchor isBenignDropFtsIndexError to the START of the message (startsWith, not includes) so a future genuine failure that merely mentions "Binder exception" or "Catalog exception" mid-message can't be misclassified as benign. New test proves the old substring match would have swallowed such a message. - incremental-fts-drop-ordering.test.ts: probe FTS availability once in beforeAll and skip VISIBLY via ctx.skip() in beforeEach (matching the withTestLbugDB/lbug-vector-extension convention) instead of a silent console.warn+return inside the test body, which reported a false pass with zero coverage of the ordering invariant when FTS was unavailable. The post-first-run FTS-index-built check is now a hard assertion instead of a second soft skip, since the beforeEach gate already proved the extension loads. --- gitnexus/src/core/lbug/lbug-adapter.ts | 10 ++- ...rop-fts-index-error-classification.test.ts | 8 +++ .../incremental-fts-drop-ordering.test.ts | 63 +++++++++++++++---- 3 files changed, 65 insertions(+), 16 deletions(-) diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index 37cbd1bc6..0be12b947 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -2914,11 +2914,15 @@ export const queryFTS = async ( * not defined...`). A real engine failure — e.g. the `Runtime exception: * FTS index '' is inconsistent: ...` class from #2589 — is a * DIFFERENT exception class (an execution-time failure, not a catalog/bind - * lookup miss), so this returns false for it. Pure string logic so it is - * unit-testable without a native LadybugDB connection. + * lookup miss), so this returns false for it. Anchored to the START of the + * message (not a bare substring search): every probed LadybugDB error leads + * with its exception class, and anchoring means a future message that merely + * mentions "Binder exception" or "Catalog exception" further in in the body + * of an otherwise-genuine failure can't be misclassified as benign. Pure + * string logic so it is unit-testable without a native LadybugDB connection. */ export const isBenignDropFtsIndexError = (message: string): boolean => - message.includes('Binder exception:') || message.includes('Catalog exception:'); + message.startsWith('Binder exception:') || message.startsWith('Catalog exception:'); /** * Drop an FTS index. Tolerates only {@link isBenignDropFtsIndexError} — diff --git a/gitnexus/test/unit/drop-fts-index-error-classification.test.ts b/gitnexus/test/unit/drop-fts-index-error-classification.test.ts index 54d6d8597..74b43dc38 100644 --- a/gitnexus/test/unit/drop-fts-index-error-classification.test.ts +++ b/gitnexus/test/unit/drop-fts-index-error-classification.test.ts @@ -42,6 +42,14 @@ describe('isBenignDropFtsIndexError', () => { it('is false for an unrelated failure', () => { expect(isBenignDropFtsIndexError('Connection Exception: database is closed')).toBe(false); }); + + it('is false for a genuine failure that merely mentions "Binder exception" mid-message (anchored, not a bare substring match)', () => { + expect( + isBenignDropFtsIndexError( + 'Runtime exception: internal state corrupted while processing Binder exception: recovery failed.', + ), + ).toBe(false); + }); }); withTestLbugDB('drop-fts-index-benign-cases', (handle) => { diff --git a/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts b/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts index da3339463..e2f5c3c87 100644 --- a/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts +++ b/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts @@ -11,14 +11,57 @@ import { readFile, writeFile } from 'fs/promises'; import { execSync } from 'child_process'; import path from 'path'; -import { afterEach, describe, expect, it, vi } from 'vitest'; +import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest'; import { setupMiniRepo } from '../helpers/mini-repo.js'; import { getStoragePaths } from '../../src/storage/repo-manager.js'; import { FTS_INDEXES } from '../../src/core/search/fts-schema.js'; +import { createTempDir } from '../helpers/test-db.js'; +import { resolveAnalyzeInstallPolicy } from '../../src/core/lbug/extension-loader.js'; const ftsMustBeAvailable = process.env.GITNEXUS_REQUIRE_FTS === '1'; describe('runFullAnalysis incremental writeback — FTS drop-before-delete ordering (#2589)', () => { + let ftsAvailable = true; + let skipWarned = false; + + beforeAll(async () => { + const lbugAdapter = await import('../../src/core/lbug/lbug-adapter.js'); + // Cheap standalone probe — matches the withTestLbugDB/lbug-vector-extension + // convention of checking availability once, up front, rather than deep + // inside the (expensive) test body. + const probe = await createTempDir('gitnexus-2589-fts-probe-'); + try { + await lbugAdapter.initLbug(probe.dbPath); + ftsAvailable = await lbugAdapter.loadFTSExtension(undefined, { + policy: resolveAnalyzeInstallPolicy(), + }); + } finally { + await lbugAdapter.closeLbug(); + await probe.cleanup(); + } + }, 120_000); + + // Skip VISIBLY (ctx.skip() marks the test as skipped, not passed) when the + // extension is unavailable — silently `return`ing from inside `it()` would + // report a false pass and hide a regression in the drop-before-delete + // ordering in exactly the environments least likely to have a human notice. + beforeEach((ctx) => { + if (!ftsAvailable) { + if (ftsMustBeAvailable) { + throw new Error( + 'GITNEXUS_REQUIRE_FTS=1 but the FTS extension is unavailable — cannot verify the #2589 ordering fix.', + ); + } + if (!skipWarned) { + skipWarned = true; + console.warn( + '[incremental-fts-drop-ordering] Skipping — the LadybugDB FTS extension is unavailable.', + ); + } + ctx.skip(); + } + }); + afterEach(() => { vi.doUnmock('../../src/core/lbug/lbug-adapter.js'); vi.resetModules(); @@ -30,8 +73,7 @@ describe('runFullAnalysis incremental writeback — FTS drop-before-delete order const repo = await setupMiniRepo('gitnexus-2589-fts-order-'); try { - // First run: full rebuild, builds every FTS index for real (skipped - // gracefully below if the extension isn't available on this machine). + // First run: full rebuild, builds every FTS index for real. await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} }); // runFullAnalysis closes its own connection on return — open a fresh @@ -48,16 +90,11 @@ describe('runFullAnalysis incremental writeback — FTS drop-before-delete order const beforeChange = await showIndexNames(); await lbugAdapter.closeLbug(); - if (!FTS_INDEXES.every(({ indexName }) => beforeChange.includes(indexName))) { - if (ftsMustBeAvailable) { - throw new Error( - 'GITNEXUS_REQUIRE_FTS=1 but the first run did not build every FTS index — cannot verify the #2589 ordering fix.', - ); - } - console.warn( - '[incremental-fts-drop-ordering] Skipping — FTS extension unavailable or indexes not built.', - ); - return; + // Hard assertion, not a soft skip: the beforeEach gate already proved + // the extension loads, so every index failing to build here is a real + // bug in the full-rebuild FTS phase, not an environment gap. + for (const { indexName } of FTS_INDEXES) { + expect(beforeChange).toContain(indexName); } // Spy on the real deleteNodesForFiles, recording the FTS index list at From 86cde93652651a71766e0569cd9095f1c385ef6a Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 10:38:50 +0000 Subject: [PATCH 35/61] docs(plans): add require-node-22.18 plan Co-Authored-By: Claude Fable 5 --- ...6-07-20-gitnexus-plan-require-node-2218.md | 113 ++++++++++++++++++ 1 file changed, 113 insertions(+) create mode 100644 docs/plans/2026-07-20-gitnexus-plan-require-node-2218.md diff --git a/docs/plans/2026-07-20-gitnexus-plan-require-node-2218.md b/docs/plans/2026-07-20-gitnexus-plan-require-node-2218.md new file mode 100644 index 000000000..62e9b2e96 --- /dev/null +++ b/docs/plans/2026-07-20-gitnexus-plan-require-node-2218.md @@ -0,0 +1,113 @@ +# GitNexus Engineering Plan — Raise the gitnexus Node floor to 22.18+ + +> Task: Make `npm install` in `gitnexus/` warning-free by raising the supported Node floor to `^22.18.0 || >=24.11.0` (matching Babel 8, kept), removing the deprecated `@types/uuid` stub, and moving the CI lanes pinned below the new floor. +> Base commit: 2ea00a2b22c65073c77148d6f8303e0e31612851 (branch worktree-fix-ebadengine-babel8). +> Executed in overlay clone /home/node/gn-fix-ebadengine (the /workspace worktree is a 9p mount where the safe plan-writer's renameat2(RENAME_NOREPLACE) fails). + +## 1. Objective + +Eliminate the nine `npm warn EBADENGINE` warnings from the `@babel/*@8.x` +devDependencies and the `@types/uuid@11` deprecation warning, by adopting Node +22.18+ as the supported minimum rather than pinning Babel back to 7. This +supersedes the initial "pin Babel to 7" approach on maintainer direction: +the project will *support 22.18+* going forward. + +## 2. Current behaviour + +- `gitnexus/package.json` declares `engines: node >=22.0.0` but four + devDependencies (`@babel/generator|parser|traverse|types`) are at `^8.0.0` + (arrived via Dependabot #2518–#2520). Babel 8 declares + `engines: node ^22.18.0 || >=24.11.0`, so every dev install on Node <22.18 + warns nine times (the direct four plus five transitive Babel packages). +- `@types/uuid@11.0.0` is a deprecated stub — `uuid@14` (a runtime dep) ships + its own types, and no tsconfig `types` array references uuid. +- The only consumer of the Babel devDeps is the bench mutation oracle + (`gitnexus/bench/impact-pdg/mutation-oracle.mjs`), lazily imported by + `measure.mjs` only under `--mutation`. +- Several CI lanes pin Node below 22.18: `ci-tests.yml` `node-floor-compat` + (22.14.0) and the containment-canary job (22.16.0), + `gitnexus-review-agent.yml` (22.16.0), `gitnexus-skill-evolution.yml` + (22.16.0). Under a 22.18 floor these run below the supported minimum. + +## 3. Approach decision + +Two ways to make installs warning-free: + +- **(A) Pin Babel to 7** — keeps the floor at `>=22.0.0`; suppresses the + warning without changing what the project supports. Requires a Dependabot + ignore so the Babel 8 bump does not return. +- **(B) Raise the floor to 22.18+ and keep Babel 8** — chosen. `engines` + becomes `^22.18.0 || >=24.11.0`, matching Babel 8 exactly, so the warnings + vanish honestly and no Dependabot ignore is needed. Cost: a user-facing + raise of the minimum Node (drops 22.0–22.17), which is why it moves the + documented floor and the CI floor gate deliberately. + +The `engines` string mirrors Babel 8's own constraint (`^22.18.0` = the +22.18-and-up 22.x line; `>=24.11.0` = the 24.x line that got the relevant +backport) so that no Babel-8 EBADENGINE can reappear on any Node the project +claims to support. `>=22.18.0` would be looser but would re-warn on Node 23.x +and 24.0–24.10, which Babel 8 excludes. + +## 4. Proposed changes + +| File | Change | +| ---- | ------ | +| `gitnexus/package.json` | `engines.node` `>=22.0.0` → `^22.18.0 \|\| >=24.11.0`; remove `@types/uuid` devDependency. Babel stays `^8.0.0`. | +| `gitnexus/package-lock.json` | Mirror both edits (engines + `@types/uuid` entry removed). Hand-applied to avoid npm-version `libc` metadata churn (local npm 10.9.2 strips the `libc` platform arrays a newer npm wrote; those drive musl/glibc optional-dep resolution and must be preserved). Verified consistent via `npm ci` exit 0. | +| `CONTRIBUTING.md` | Prerequisite floor `>=22.0.0` → `^22.18.0 \|\| >=24.11.0`. | +| `.github/workflows/ci-tests.yml` | `node-floor-compat` retargeted 22.14.0 → 22.18.0 (name, comment, pin, version assertion) so the floor gate guards the *new* minimum; containment-canary pin 22.16.0 → 22.18.0. | +| `.github/workflows/gitnexus-review-agent.yml` | Pinned Node 22.16.0 → 22.18.0. | +| `.github/workflows/gitnexus-skill-evolution.yml` | Pinned Node 22.16.0 → 22.18.0. | + +CI lanes using `node-version: 22` (latest ≥22.18) or `24` are already at/above +the floor and unchanged. `CHANGELOG.md` and `package.json` `version` are +release-owned (per repo convention) and deliberately untouched. + +## 5. Implementation sequence + +1. `gitnexus/package.json` + `gitnexus/package-lock.json`: engines bump and + `@types/uuid` removal (one atomic commit). +2. CI + docs: floor-gate retarget, the three pinned-lane bumps, CONTRIBUTING + prerequisite (one atomic commit). + +Both orderings leave the tree coherent; each is `detect_changes`-gated +(config/manifest files carry no indexed symbols → empty affected set). + +## 6. Verification + +- `npm ci` in `gitnexus/` exits 0 (lockfile ↔ package.json consistency after + the hand-edit). +- On local Node 22.16.0 (now *below* the floor), the only EBADENGINE lines are + the nine Babel 8 packages plus `gitnexus` itself, all reporting the identical + `required: ^22.18.0 || >=24.11.0` — i.e. they satisfy together at ≥22.18 and + vanish by construction. No other engine warning; no `@types/uuid` + deprecation. (A true zero-warning run requires Node ≥22.18, unavailable in + this sandbox; CI's ≥22.18 lanes are the authoritative check.) +- Mutation oracle single-fixture run + (`measure.mjs --mutation --only=intra-control-branch --json`) exits 0 on + Babel 8 with a fingerprint identical to the Babel 7 run — the bench consumer + is unaffected. +- `npm run test:unit` passes. + +## 7. Risks + +- **User-facing floor raise** — dropping Node 22.0–22.17 is a support-policy + change. Deliberate and the point of this PR; belongs in the release notes at + release time (not edited here per CHANGELOG convention). +- **Lockfile hand-edit** — mitigated by the `npm ci` exit-0 consistency proof + and by preserving `libc` platform metadata (musl/Alpine native resolution). +- **node-floor-compat** — its original #2372 failure mode (`module.registerHooks` + ≥22.15 on a sub-22.15 floor) can no longer occur at a 22.18 floor; the gate + is retained, retargeted to 22.18.0, to guard the new minimum generally. + +## 8. Definition of Done + +1. `gitnexus/package.json` `engines.node` = `^22.18.0 || >=24.11.0`; Babel at + `^8.0.0`; `@types/uuid` absent. +2. `npm ci` exits 0; no `@types/uuid` deprecation; the only EBADENGINE lines on + sub-floor Node are Babel 8 + gitnexus-self, all with the new required range. +3. Mutation oracle single-fixture run exits 0 on Babel 8. +4. `npm run test:unit` passes. +5. No CI lane pins Node below 22.18; `node-floor-compat` guards 22.18.0. +6. `CONTRIBUTING.md` floor updated; `CHANGELOG.md`/`version` untouched. +7. Diff limited to the six files above (+ this plan). From c26b78d15309e4c03289e58b0cc98f569e1986d6 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 10:39:06 +0000 Subject: [PATCH 36/61] fix(deps)!: require Node >=22.18 and drop @types/uuid stub Babel 8 (devDep for the bench mutation oracle, pulled in by dependabot previous floor (>=22.0.0), so every dev install on Node <22.18 emitted nine EBADENGINE warnings. Rather than pin Babel back to 7, adopt Node 22.18+ as the supported minimum: set engines to ^22.18.0 || >=24.11.0, matching Babel 8 exactly so the warnings resolve honestly with no dependabot ignore needed. @types/uuid@11 is a deprecated stub - uuid@14 ships its own types and no tsconfig references it. Lockfile edited by hand (engines + @types/uuid entry) to preserve the libc platform metadata a newer npm wrote; verified consistent via npm ci (exit 0). BREAKING CHANGE: the gitnexus package now requires Node ^22.18.0 || >=24.11.0 (previously >=22.0.0). Node 22.0-22.17 are no longer supported. Co-Authored-By: Claude Fable 5 --- gitnexus/package-lock.json | 14 +------------- gitnexus/package.json | 3 +-- 2 files changed, 2 insertions(+), 15 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index d9462f5a6..e3d28cbd4 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -60,7 +60,6 @@ "@types/express": "^5.0.6", "@types/js-yaml": "^4.0.9", "@types/node": "^26.0.0", - "@types/uuid": "^11.0.0", "@vitest/coverage-v8": "^4.0.18", "gitnexus-shared": "file:../gitnexus-shared", "tsx": "^4.0.0", @@ -68,7 +67,7 @@ "vitest": "^4.0.18" }, "engines": { - "node": ">=22.0.0" + "node": "^22.18.0 || >=24.11.0" }, "optionalDependencies": { "@huggingface/transformers": "^4.1.0", @@ -1990,17 +1989,6 @@ "@types/node": "*" } }, - "node_modules/@types/uuid": { - "version": "11.0.0", - "resolved": "https://registry.npmjs.org/@types/uuid/-/uuid-11.0.0.tgz", - "integrity": "sha512-HVyk8nj2m+jcFRNazzqyVKiZezyhDKrGUA3jlEcg/nZ6Ms+qHwocba1Y/AaVaznJTAM9xpdFSh+ptbNrhOGvZA==", - "deprecated": "This is a stub types definition. uuid provides its own type definitions, so you do not need this installed.", - "dev": true, - "license": "MIT", - "dependencies": { - "uuid": "*" - } - }, "node_modules/@vitest/coverage-v8": { "version": "4.1.10", "resolved": "https://registry.npmjs.org/@vitest/coverage-v8/-/coverage-v8-4.1.10.tgz", diff --git a/gitnexus/package.json b/gitnexus/package.json index fbbafb3ec..387c89811 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -107,7 +107,6 @@ "@types/express": "^5.0.6", "@types/js-yaml": "^4.0.9", "@types/node": "^26.0.0", - "@types/uuid": "^11.0.0", "@vitest/coverage-v8": "^4.0.18", "gitnexus-shared": "file:../gitnexus-shared", "tsx": "^4.0.0", @@ -120,6 +119,6 @@ } }, "engines": { - "node": ">=22.0.0" + "node": "^22.18.0 || >=24.11.0" } } From eea9ac92dccead0287c2aa8bb044247f9718f834 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 10:39:18 +0000 Subject: [PATCH 37/61] ci: move Node pins to the 22.18 floor With the supported minimum raised to Node 22.18, retarget every lane and pinned runtime that sat at a lower version so nothing builds or runs the package on an unsupported (EBADENGINE-warning) Node: - ci-tests.yml: node-floor-compat 22.14 -> 22.18.0 (name, comment, pin, version assertion) so the floor gate guards the new minimum; its #2372 registerHooks failure mode cannot recur above 22.15. Containment-canary pin 22.16.0 -> 22.18.0. - gitnexus-review-agent.yml + the pinned review/canary runtime: the reproducible runtime is version-locked in lockstep across .github/{gitnexus-review-runtime,claude-canary-runtime}/package.json and their lockfiles (engines), the workflow's node-version, its two 'node --version = v22.18.0' assertions, the lockfile-engines guard, and NODE_VERSION. Moved all of them 22.16.0 -> 22.18.0. - gitnexus-skill-evolution.yml: pinned runtime 22.16.0 -> 22.18.0. - CONTRIBUTING.md prerequisite floor updated. - review-agent-workflow.test.ts, which enforces the runtime lock, updated to expect 22.18.0. Co-Authored-By: Claude Fable 5 --- .../claude-canary-runtime/package-lock.json | 2 +- .github/claude-canary-runtime/package.json | 2 +- .../gitnexus-review-runtime/package-lock.json | 2 +- .github/gitnexus-review-runtime/package.json | 2 +- .github/workflows/ci-tests.yml | 25 ++++++++++--------- .github/workflows/gitnexus-review-agent.yml | 10 ++++---- .../workflows/gitnexus-skill-evolution.yml | 2 +- CONTRIBUTING.md | 2 +- .../test/unit/review-agent-workflow.test.ts | 10 ++++---- 9 files changed, 29 insertions(+), 28 deletions(-) diff --git a/.github/claude-canary-runtime/package-lock.json b/.github/claude-canary-runtime/package-lock.json index e78392daa..7716ef93f 100644 --- a/.github/claude-canary-runtime/package-lock.json +++ b/.github/claude-canary-runtime/package-lock.json @@ -11,7 +11,7 @@ "@anthropic-ai/claude-code": "2.1.214" }, "engines": { - "node": "22.16.0" + "node": "22.18.0" } }, "node_modules/@anthropic-ai/claude-code": { diff --git a/.github/claude-canary-runtime/package.json b/.github/claude-canary-runtime/package.json index 57076d892..50820742b 100644 --- a/.github/claude-canary-runtime/package.json +++ b/.github/claude-canary-runtime/package.json @@ -3,7 +3,7 @@ "version": "0.0.0", "private": true, "engines": { - "node": "22.16.0" + "node": "22.18.0" }, "dependencies": { "@anthropic-ai/claude-code": "2.1.214" diff --git a/.github/gitnexus-review-runtime/package-lock.json b/.github/gitnexus-review-runtime/package-lock.json index e805e759b..0677011c0 100644 --- a/.github/gitnexus-review-runtime/package-lock.json +++ b/.github/gitnexus-review-runtime/package-lock.json @@ -11,7 +11,7 @@ "gitnexus": "1.6.9" }, "engines": { - "node": "22.16.0" + "node": "22.18.0" } }, "node_modules/@emnapi/runtime": { diff --git a/.github/gitnexus-review-runtime/package.json b/.github/gitnexus-review-runtime/package.json index 237310bad..de0a1dd12 100644 --- a/.github/gitnexus-review-runtime/package.json +++ b/.github/gitnexus-review-runtime/package.json @@ -3,7 +3,7 @@ "private": true, "version": "1.0.0", "engines": { - "node": "22.16.0" + "node": "22.18.0" }, "dependencies": { "gitnexus": "1.6.9" diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index c78a407b9..d9aa33a4e 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -378,15 +378,16 @@ jobs: "$PREFIX/bin/gitnexus" --version fi - # Node engines-floor gate (#2372). The embedding resolvers statically named - # `module.registerHooks`, which only exists on Node >= 22.15 / >= 23.5, so on - # the supported floor (engines: >=22.0.0) those ESM modules failed to LINK — - # a class vitest/tsx transforms structurally mask, and the default - # `node-version: 22` (resolves to latest) never hits. Build the dist on 22.x, - # then import-link every module R1 names as a load surface on a pinned 22.14 - # so a regression fails here instead of shipping to users on that Node range. + # Node engines-floor gate (#2372). A module that statically names an API + # newer than the supported floor (e.g. `module.registerHooks`, added in + # 22.15) fails to LINK on the floor — a class vitest/tsx transforms + # structurally mask, and the default `node-version: 22` (resolves to latest) + # never hits. Build the dist on 22.x, then import-link every module R1 names + # as a load surface on the pinned engines floor (22.18.0, per package.json + # `engines: ^22.18.0 || >=24.11.0`) so a regression fails here instead of + # shipping to users on the minimum supported Node. node-floor-compat: - name: node floor compat (22.14) + name: node floor compat (22.18) runs-on: ubuntu-latest timeout-minutes: 15 steps: @@ -415,14 +416,14 @@ jobs: # (so no package-manager cache is needed). - uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 with: - node-version: '22.14.0' + node-version: '22.18.0' package-manager-cache: false - - name: Import-link the built dist on Node 22.14 + - name: Import-link the built dist on Node 22.18 shell: bash run: | set -euo pipefail node --version - node --version | grep -q '^v22\.14\.' || { echo "expected Node 22.14.x" >&2; exit 1; } + node --version | grep -q '^v22\.18\.' || { echo "expected Node 22.18.x" >&2; exit 1; } for m in \ core/embeddings/runtime-install \ core/embeddings/onnxruntime-node-resolver \ @@ -556,7 +557,7 @@ jobs: persist-credentials: false - uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 with: - node-version: '22.16.0' + node-version: '22.18.0' cache: npm cache-dependency-path: | gitnexus/package-lock.json diff --git a/.github/workflows/gitnexus-review-agent.yml b/.github/workflows/gitnexus-review-agent.yml index 8582403ad..88526871e 100644 --- a/.github/workflows/gitnexus-review-agent.yml +++ b/.github/workflows/gitnexus-review-agent.yml @@ -325,7 +325,7 @@ jobs: if: steps.context.outputs.ready == 'true' uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 with: - node-version: '22.16.0' + node-version: '22.18.0' - name: Install and preflight Claude subprocess isolation id: isolation @@ -377,7 +377,7 @@ jobs: .github/claude-canary-runtime/package-lock.json \ "${runtime_dir}/package-lock.json" printf '%s\n' 'registry=https://registry.npmjs.org/' 'audit=false' 'fund=false' > "${npmrc}" - test "$(node --version)" = 'v22.16.0' + test "$(node --version)" = 'v22.18.0' test "$(uname -m)" = 'x86_64' # The trusted lock and these independent receipts pin both the thin @@ -398,7 +398,7 @@ jobs: if ( lock.lockfileVersion !== 3 || lock.packages?.['']?.dependencies?.['@anthropic-ai/claude-code'] !== '2.1.214' || - lock.packages?.['']?.engines?.node !== '22.16.0' + lock.packages?.['']?.engines?.node !== '22.18.0' ) { throw new Error('Claude runtime lock root is not exact'); } @@ -506,7 +506,7 @@ jobs: install -m 0600 .github/gitnexus-review-runtime/package.json "${runtime_dir}/package.json" install -m 0600 .github/gitnexus-review-runtime/package-lock.json "${runtime_dir}/package-lock.json" printf '%s\n' 'registry=https://registry.npmjs.org/' 'audit=false' 'fund=false' > "${npmrc}" - test "$(node --version)" = 'v22.16.0' + test "$(node --version)" = 'v22.18.0' npm ci \ --prefix "${runtime_dir}" \ --userconfig "${npmrc}" \ @@ -1241,7 +1241,7 @@ jobs: CLAUDE_CONFIG_DIR: ${{ runner.temp }}/gitnexus-review-claude-config CLAUDE_WORKING_DIR: ${{ runner.temp }}/gitnexus-review-control NPM_CONFIG_IGNORE_SCRIPTS: 'true' - NODE_VERSION: '22.16.0' + NODE_VERSION: '22.18.0' with: claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }} path_to_claude_code_executable: ${{ runner.temp }}/gitnexus-review-claude-runtime/node_modules/@anthropic-ai/claude-code/bin/claude.exe diff --git a/.github/workflows/gitnexus-skill-evolution.yml b/.github/workflows/gitnexus-skill-evolution.yml index 66e87ad18..8d9454d56 100644 --- a/.github/workflows/gitnexus-skill-evolution.yml +++ b/.github/workflows/gitnexus-skill-evolution.yml @@ -110,7 +110,7 @@ jobs: - uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0 with: - node-version: '22.16.0' + node-version: '22.18.0' cache: npm cache-dependency-path: | gitnexus/package-lock.json diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index e9922cf19..4e9e17369 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -13,7 +13,7 @@ This project uses the [PolyForm Noncommercial License 1.0.0](https://polyformpro ## Development setup -**Prerequisites:** Node.js — `gitnexus/` requires `>=22.0.0` and `gitnexus-web/` requires `^20.19.0 || >=22.12.0` (enforced via the `engines` field in each package). Use `nvm install` to match the local version. +**Prerequisites:** Node.js — `gitnexus/` requires `^22.18.0 || >=24.11.0` and `gitnexus-web/` requires `^20.19.0 || >=22.12.0` (enforced via the `engines` field in each package). Use `nvm install` to match the local version. 1. Clone the repository. 2. **Shared package:** `cd gitnexus-shared && npm install && npm run build` diff --git a/gitnexus/test/unit/review-agent-workflow.test.ts b/gitnexus/test/unit/review-agent-workflow.test.ts index faf9513d6..35657e6ea 100644 --- a/gitnexus/test/unit/review-agent-workflow.test.ts +++ b/gitnexus/test/unit/review-agent-workflow.test.ts @@ -589,12 +589,12 @@ describe('gitnexus review-agent workflow security contract', () => { } expect(runtimePackage.dependencies?.gitnexus).toBe('1.6.9'); - expect(runtimePackage.engines?.node).toBe('22.16.0'); + expect(runtimePackage.engines?.node).toBe('22.18.0'); expect(runtimeLock.packages?.['node_modules/gitnexus']?.version).toBe('1.6.9'); expect(runtimeLock.packages?.['node_modules/gitnexus']?.integrity).toMatch(/^sha512-/); expect(workflow).not.toMatch(/gitnexus@(latest|next|beta)/); - expect(workflow).toContain("node-version: '22.16.0'"); - expect(workflow).toContain('test "$(node --version)" = \'v22.16.0\''); + expect(workflow).toContain("node-version: '22.18.0'"); + expect(workflow).toContain('test "$(node --version)" = \'v22.18.0\''); expect(workflow).toContain('npm ci'); expect(workflow).not.toContain('--package-lock=false'); expect(workflow).toContain( @@ -682,7 +682,7 @@ describe('gitnexus review-agent workflow security contract', () => { '${{ runner.temp }}/gitnexus-review-claude-runtime/node_modules/@anthropic-ai/claude-code/bin/claude.exe'; expect(claudeRuntimePackage.dependencies?.['@anthropic-ai/claude-code']).toBe('2.1.214'); - expect(claudeRuntimePackage.engines?.node).toBe('22.16.0'); + expect(claudeRuntimePackage.engines?.node).toBe('22.18.0'); expect(claudeRuntimeLock.lockfileVersion).toBe(3); expect(claudeRuntimeLock.packages?.['node_modules/@anthropic-ai/claude-code']).toMatchObject({ version: '2.1.214', @@ -1119,7 +1119,7 @@ describe('gitnexus review-agent workflow security contract', () => { 'CLAUDE_CONFIG_DIR: ${{ runner.temp }}/gitnexus-review-claude-config', ); expect(analyze).toContain('CLAUDE_WORKING_DIR: ${{ runner.temp }}/gitnexus-review-control'); - expect(analyze).toContain("NODE_VERSION: '22.16.0'"); + expect(analyze).toContain("NODE_VERSION: '22.18.0'"); expect(analyze).toContain('checkout-index --all --force'); expect(analyze).toContain('find "${review_dir}" -type l -print0'); expect(analyze).toContain('Escaping copied review symlink'); From 1415bd5c2f7b7e13888eb5746cf107d86c5a2e0d Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 11:26:51 +0000 Subject: [PATCH 38/61] docs(embeddings): update engines-floor comments for the 22.18 minimum The module.registerHooks compat seam and the onnxruntime resolvers cited the old '>=22.0.0' floor as the reason their sub-22.15 fallback was reachable. With the floor now ^22.18.0 || >=24.11.0 (all >=22.15), every supported runtime exposes the API; the fallback stays as defensive handling for below-floor runtimes (engines is advisory, not engine-strict). Comments only - no behavior change. Co-Authored-By: Claude Fable 5 --- gitnexus/src/core/embeddings/node-module-compat.ts | 6 ++++-- .../src/core/embeddings/onnxruntime-common-resolver.ts | 10 +++++----- .../src/core/embeddings/onnxruntime-node-resolver.ts | 4 ++-- gitnexus/src/core/embeddings/runtime-install.ts | 2 +- gitnexus/test/unit/node-module-compat.test.ts | 5 +++-- 5 files changed, 15 insertions(+), 12 deletions(-) diff --git a/gitnexus/src/core/embeddings/node-module-compat.ts b/gitnexus/src/core/embeddings/node-module-compat.ts index b32e854e9..8f41c5d22 100644 --- a/gitnexus/src/core/embeddings/node-module-compat.ts +++ b/gitnexus/src/core/embeddings/node-module-compat.ts @@ -3,8 +3,10 @@ * * `module.registerHooks` — the synchronous ESM/CJS resolution-hook API the * embedding-stack resolvers rely on — was added in Node 22.15.0 (and 23.5.0 on - * the 23.x line). The gitnexus engines floor is `>=22.0.0`, which admits Node - * 22.0–22.14 AND 23.0–23.4, where the export is absent. + * the 23.x line). The gitnexus engines floor is `^22.18.0 || >=24.11.0`, so + * every supported runtime exposes it — but `engines` is advisory (not + * engine-strict), so a below-floor Node (22.0–22.14, or the unsupported + * 23.0–23.4 line) can still run, where the export is absent. * * In this `"type": "module"` package, a *static named* import of a missing * builtin export (`import { registerHooks } from 'node:module'`) is a diff --git a/gitnexus/src/core/embeddings/onnxruntime-common-resolver.ts b/gitnexus/src/core/embeddings/onnxruntime-common-resolver.ts index 2d494f2b7..8fd2ddf17 100644 --- a/gitnexus/src/core/embeddings/onnxruntime-common-resolver.ts +++ b/gitnexus/src/core/embeddings/onnxruntime-common-resolver.ts @@ -52,8 +52,8 @@ * per-resolution cost is a single string comparison. * * `module.registerHooks` is marked `@experimental` and requires Node >= 22.15 - * (the gitnexus engines floor is >= 22.0.0). On older runtimes it is absent and - * this is a graceful no-op: embeddings then resolve onnxruntime-common exactly + * (below the gitnexus engines floor of `^22.18.0 || >=24.11.0`). On below-floor + * runtimes it is absent and this is a graceful no-op: embeddings then resolve onnxruntime-common exactly * as before — fine on hoisted layouts. Any failure during installation is * swallowed. */ @@ -100,9 +100,9 @@ export const ensureOnnxRuntimeCommonResolvable = (): void => { attempted = true; try { - // Node < 22.15 / < 23.5 (the gitnexus engines floor is >= 22.0.0): no - // synchronous hooks API. Degrade gracefully — the import still works on - // hoisted layouts. + // Node < 22.15 / < 23.5 (below the gitnexus engines floor of + // ^22.18.0 || >=24.11.0): no synchronous hooks API. Degrade gracefully — + // the import still works on hoisted layouts. const registerHooks = getRegisterHooks(); if (typeof registerHooks !== 'function') return; diff --git a/gitnexus/src/core/embeddings/onnxruntime-node-resolver.ts b/gitnexus/src/core/embeddings/onnxruntime-node-resolver.ts index 65465eb00..f630384ca 100644 --- a/gitnexus/src/core/embeddings/onnxruntime-node-resolver.ts +++ b/gitnexus/src/core/embeddings/onnxruntime-node-resolver.ts @@ -36,8 +36,8 @@ * So CUDA-12 hosts, Windows (DirectML), macOS, and CPU-only hosts are * untouched. Idempotent; any failure is swallowed and leaves the default * resolution exactly as before. `module.registerHooks` requires Node >= 22.15 - * (the gitnexus engines floor is >= 22.0.0); on older runtimes the redirect is - * a no-op, but the default copy's CUDA major is still probed so an + * (below the gitnexus engines floor of `^22.18.0 || >=24.11.0`); on below-floor + * runtimes the redirect is a no-op, but the default copy's CUDA major is still probed so an * already-matching host (e.g. CUDA 12 + transformers' CUDA-12 build) keeps * auto-selecting the GPU. * `npm link` / symlinked local-dev checkouts are a known caveat: `resolveOurOrtNodeDir`/ diff --git a/gitnexus/src/core/embeddings/runtime-install.ts b/gitnexus/src/core/embeddings/runtime-install.ts index b7c054b0b..e52274746 100644 --- a/gitnexus/src/core/embeddings/runtime-install.ts +++ b/gitnexus/src/core/embeddings/runtime-install.ts @@ -191,7 +191,7 @@ export const ensureEmbeddingStackResolvable = (): void => { hookAttempted = true; try { - // Node < 22.15 / < 23.5 (engines floor is >= 22.0.0): no synchronous hooks + // Node < 22.15 / < 23.5 (below the engines floor of ^22.18.0 || >=24.11.0): no synchronous hooks // API. Degrade gracefully — normally-installed stacks still resolve; only // the runtime-prefix fallback is unavailable. Reachable now that the import // is a namespace access (see node-module-compat.ts) rather than a static diff --git a/gitnexus/test/unit/node-module-compat.test.ts b/gitnexus/test/unit/node-module-compat.test.ts index 6323bd836..47c5f5ba6 100644 --- a/gitnexus/test/unit/node-module-compat.test.ts +++ b/gitnexus/test/unit/node-module-compat.test.ts @@ -2,8 +2,9 @@ import { describe, it, expect, vi, afterEach } from 'vitest'; /** * Tests for the #2372 `node:module` compat seam. `module.registerHooks` was - * added in Node 22.15 / 23.5, but the engines floor is >=22.0.0, so on - * 22.0–22.14 and 23.0–23.4 the export is absent. `getRegisterHooks()` must + * added in Node 22.15 / 23.5. The engines floor is ^22.18.0 || >=24.11.0 (all + * >=22.15), but engines is advisory, so a below-floor 22.0–22.14 / 23.0–23.4 + * runtime can still run, where the export is absent. `getRegisterHooks()` must * hand back the real function when present and `undefined` when not — the value * the resolver guards degrade on. `isPrefixRuntimeLoadable()` (exported from * runtime-install.ts so CLI code never imports the compat module) is the From bb23d2998abf8e215c4058d96a1545324efd4be9 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 11:39:40 +0000 Subject: [PATCH 39/61] test(eval): update containment-job node-version assertion to 22.18.0 test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs pins the eval-containment-linux job's setup-node version; move it in lockstep with the ci-tests.yml pin bumped to the 22.18 floor. Co-Authored-By: Claude Fable 5 --- eval/tests/test_workflow_bench.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/eval/tests/test_workflow_bench.py b/eval/tests/test_workflow_bench.py index e4610e197..961c0edd4 100644 --- a/eval/tests/test_workflow_bench.py +++ b/eval/tests/test_workflow_bench.py @@ -172,7 +172,7 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs(): } assert containment["timeout-minutes"] == 20 assert containment_node_setup["with"] == { - "node-version": "22.16.0", + "node-version": "22.18.0", "cache": "npm", "cache-dependency-path": "gitnexus/package-lock.json\ngitnexus-shared/package-lock.json\n", } From 694048a987ab826d5cf819779fd8a70da4294711 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 20 Jul 2026 11:47:59 +0000 Subject: [PATCH 40/61] chore: stop tracking docs/plans (planning output stays local) Reverses the prior convention: gitnexus-plan/gitnexus-work plan documents under docs/plans/ are working artifacts and no longer travel with the PR. Drops the require-node-22.18 plan doc from tracking; the .gitignore now ignores all of docs/. The workflow_bench snapshot features scan the filesystem, not git-tracked status, so they are unaffected. Co-Authored-By: Claude Fable 5 --- .gitignore | 3 +- ...6-07-20-gitnexus-plan-require-node-2218.md | 113 ------------------ 2 files changed, 1 insertion(+), 115 deletions(-) delete mode 100644 docs/plans/2026-07-20-gitnexus-plan-require-node-2218.md diff --git a/.gitignore b/.gitignore index 60795b24f..e16544f71 100644 --- a/.gitignore +++ b/.gitignore @@ -68,9 +68,8 @@ gitnexus-web/test-results/ eval/.coverage eval/.hypothesis/ -# Local docs (docs/plans/ stays tracked — gitnexus-plan output travels with the work) +# Local docs — planning output (gitnexus-plan / gitnexus-work) stays local, not tracked docs/* -!docs/plans/ gitnexus/test/fixtures/mini-repo/*.md gitnexus/test/fixtures/mini-repo/.claude diff --git a/docs/plans/2026-07-20-gitnexus-plan-require-node-2218.md b/docs/plans/2026-07-20-gitnexus-plan-require-node-2218.md deleted file mode 100644 index 62e9b2e96..000000000 --- a/docs/plans/2026-07-20-gitnexus-plan-require-node-2218.md +++ /dev/null @@ -1,113 +0,0 @@ -# GitNexus Engineering Plan — Raise the gitnexus Node floor to 22.18+ - -> Task: Make `npm install` in `gitnexus/` warning-free by raising the supported Node floor to `^22.18.0 || >=24.11.0` (matching Babel 8, kept), removing the deprecated `@types/uuid` stub, and moving the CI lanes pinned below the new floor. -> Base commit: 2ea00a2b22c65073c77148d6f8303e0e31612851 (branch worktree-fix-ebadengine-babel8). -> Executed in overlay clone /home/node/gn-fix-ebadengine (the /workspace worktree is a 9p mount where the safe plan-writer's renameat2(RENAME_NOREPLACE) fails). - -## 1. Objective - -Eliminate the nine `npm warn EBADENGINE` warnings from the `@babel/*@8.x` -devDependencies and the `@types/uuid@11` deprecation warning, by adopting Node -22.18+ as the supported minimum rather than pinning Babel back to 7. This -supersedes the initial "pin Babel to 7" approach on maintainer direction: -the project will *support 22.18+* going forward. - -## 2. Current behaviour - -- `gitnexus/package.json` declares `engines: node >=22.0.0` but four - devDependencies (`@babel/generator|parser|traverse|types`) are at `^8.0.0` - (arrived via Dependabot #2518–#2520). Babel 8 declares - `engines: node ^22.18.0 || >=24.11.0`, so every dev install on Node <22.18 - warns nine times (the direct four plus five transitive Babel packages). -- `@types/uuid@11.0.0` is a deprecated stub — `uuid@14` (a runtime dep) ships - its own types, and no tsconfig `types` array references uuid. -- The only consumer of the Babel devDeps is the bench mutation oracle - (`gitnexus/bench/impact-pdg/mutation-oracle.mjs`), lazily imported by - `measure.mjs` only under `--mutation`. -- Several CI lanes pin Node below 22.18: `ci-tests.yml` `node-floor-compat` - (22.14.0) and the containment-canary job (22.16.0), - `gitnexus-review-agent.yml` (22.16.0), `gitnexus-skill-evolution.yml` - (22.16.0). Under a 22.18 floor these run below the supported minimum. - -## 3. Approach decision - -Two ways to make installs warning-free: - -- **(A) Pin Babel to 7** — keeps the floor at `>=22.0.0`; suppresses the - warning without changing what the project supports. Requires a Dependabot - ignore so the Babel 8 bump does not return. -- **(B) Raise the floor to 22.18+ and keep Babel 8** — chosen. `engines` - becomes `^22.18.0 || >=24.11.0`, matching Babel 8 exactly, so the warnings - vanish honestly and no Dependabot ignore is needed. Cost: a user-facing - raise of the minimum Node (drops 22.0–22.17), which is why it moves the - documented floor and the CI floor gate deliberately. - -The `engines` string mirrors Babel 8's own constraint (`^22.18.0` = the -22.18-and-up 22.x line; `>=24.11.0` = the 24.x line that got the relevant -backport) so that no Babel-8 EBADENGINE can reappear on any Node the project -claims to support. `>=22.18.0` would be looser but would re-warn on Node 23.x -and 24.0–24.10, which Babel 8 excludes. - -## 4. Proposed changes - -| File | Change | -| ---- | ------ | -| `gitnexus/package.json` | `engines.node` `>=22.0.0` → `^22.18.0 \|\| >=24.11.0`; remove `@types/uuid` devDependency. Babel stays `^8.0.0`. | -| `gitnexus/package-lock.json` | Mirror both edits (engines + `@types/uuid` entry removed). Hand-applied to avoid npm-version `libc` metadata churn (local npm 10.9.2 strips the `libc` platform arrays a newer npm wrote; those drive musl/glibc optional-dep resolution and must be preserved). Verified consistent via `npm ci` exit 0. | -| `CONTRIBUTING.md` | Prerequisite floor `>=22.0.0` → `^22.18.0 \|\| >=24.11.0`. | -| `.github/workflows/ci-tests.yml` | `node-floor-compat` retargeted 22.14.0 → 22.18.0 (name, comment, pin, version assertion) so the floor gate guards the *new* minimum; containment-canary pin 22.16.0 → 22.18.0. | -| `.github/workflows/gitnexus-review-agent.yml` | Pinned Node 22.16.0 → 22.18.0. | -| `.github/workflows/gitnexus-skill-evolution.yml` | Pinned Node 22.16.0 → 22.18.0. | - -CI lanes using `node-version: 22` (latest ≥22.18) or `24` are already at/above -the floor and unchanged. `CHANGELOG.md` and `package.json` `version` are -release-owned (per repo convention) and deliberately untouched. - -## 5. Implementation sequence - -1. `gitnexus/package.json` + `gitnexus/package-lock.json`: engines bump and - `@types/uuid` removal (one atomic commit). -2. CI + docs: floor-gate retarget, the three pinned-lane bumps, CONTRIBUTING - prerequisite (one atomic commit). - -Both orderings leave the tree coherent; each is `detect_changes`-gated -(config/manifest files carry no indexed symbols → empty affected set). - -## 6. Verification - -- `npm ci` in `gitnexus/` exits 0 (lockfile ↔ package.json consistency after - the hand-edit). -- On local Node 22.16.0 (now *below* the floor), the only EBADENGINE lines are - the nine Babel 8 packages plus `gitnexus` itself, all reporting the identical - `required: ^22.18.0 || >=24.11.0` — i.e. they satisfy together at ≥22.18 and - vanish by construction. No other engine warning; no `@types/uuid` - deprecation. (A true zero-warning run requires Node ≥22.18, unavailable in - this sandbox; CI's ≥22.18 lanes are the authoritative check.) -- Mutation oracle single-fixture run - (`measure.mjs --mutation --only=intra-control-branch --json`) exits 0 on - Babel 8 with a fingerprint identical to the Babel 7 run — the bench consumer - is unaffected. -- `npm run test:unit` passes. - -## 7. Risks - -- **User-facing floor raise** — dropping Node 22.0–22.17 is a support-policy - change. Deliberate and the point of this PR; belongs in the release notes at - release time (not edited here per CHANGELOG convention). -- **Lockfile hand-edit** — mitigated by the `npm ci` exit-0 consistency proof - and by preserving `libc` platform metadata (musl/Alpine native resolution). -- **node-floor-compat** — its original #2372 failure mode (`module.registerHooks` - ≥22.15 on a sub-22.15 floor) can no longer occur at a 22.18 floor; the gate - is retained, retargeted to 22.18.0, to guard the new minimum generally. - -## 8. Definition of Done - -1. `gitnexus/package.json` `engines.node` = `^22.18.0 || >=24.11.0`; Babel at - `^8.0.0`; `@types/uuid` absent. -2. `npm ci` exits 0; no `@types/uuid` deprecation; the only EBADENGINE lines on - sub-floor Node are Babel 8 + gitnexus-self, all with the new required range. -3. Mutation oracle single-fixture run exits 0 on Babel 8. -4. `npm run test:unit` passes. -5. No CI lane pins Node below 22.18; `node-floor-compat` guards 22.18.0. -6. `CONTRIBUTING.md` floor updated; `CHANGELOG.md`/`version` untouched. -7. Diff limited to the six files above (+ this plan). From 7666a009f058630cd55d0daceaea162719f7e1d3 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 21 Jul 2026 11:04:37 +0000 Subject: [PATCH 41/61] fix(java): resolve E.CONST.method() enum-constant receiver dispatch (#2561) Calling a method on an enum-constant receiver (E.CONST.method()) emitted no CALLS edge. The receiver "E.CONST" is a two-segment compound receiver; resolveCompoundReceiverClass walks each dotted segment via the owning class scope's typeBindings map, but enum constants had no typeBinding, so the constant segment dead-ended and no target was ever resolved. #2555/#2558 gave bodied constants a first-class synthesized E$N class with an MRO that includes the host enum; this is the receiver-side follow-up. synthesizeJavaAnonymousClassDeclarations now emits a class-scope typeBinding for every enum constant's simple name -> its E$N class (bodied) or the host enum itself (body-less), reusing the exact mechanism a field declaration uses. The generic compound-receiver chain walk then resolves E.CONST.method() with no change to any shared scope-resolution code. Bodied dispatch (EnumConst.A.hook() -> EnumConst$1.hook#0) and body-less inherited dispatch (Plain.A.m() -> Plain.m#0) are covered by new tests in the existing java-enum-constant-body fixture; both were verified to fail against the pre-fix tree. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../core/ingestion/languages/java/captures.ts | 47 +++++++++++++------ .../src/EnumConst.java | 4 ++ .../java-enum-constant-body/src/Plain.java | 13 +++++ .../test/integration/resolvers/java.test.ts | 29 ++++++++++++ 4 files changed, 79 insertions(+), 14 deletions(-) create mode 100644 gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/Plain.java diff --git a/gitnexus/src/core/ingestion/languages/java/captures.ts b/gitnexus/src/core/ingestion/languages/java/captures.ts index 3adb036c8..2212c4355 100644 --- a/gitnexus/src/core/ingestion/languages/java/captures.ts +++ b/gitnexus/src/core/ingestion/languages/java/captures.ts @@ -372,23 +372,42 @@ function synthesizeJavaAnonymousClassDeclarations(rootNode: SyntaxNode): Capture // constant's class extends its HOST ENUM (javac semantics), so the // inherits reference names the enum — giving `mroFor(E$N) ∋ E` and // keeping bare calls from the body to the enum's own helpers alive - // through the ownership gate's MRO arm. No receiver typeBinding piece: - // constants are not variable initializers; `E.A.hook()` dispatch rides - // the existing enum receiver machinery. + // through the ownership gate's MRO arm. for (const constant of rootNode.descendantsOfType('enum_constant')) { - const name = synthesizeJavaAnonymousClassName(constant); - if (name === undefined) continue; - const body = constant.childForFieldName?.('body'); - if (body === null || body === undefined || body.type !== 'class_body') continue; - out.push({ - '@declaration.class': nodeToCapture('@declaration.class', body), - '@declaration.name': syntheticCapture('@declaration.name', body, name), - }); const hostEnum = javaEnclosingEnumNameOf(constant); - if (hostEnum !== undefined) { + const bodiedName = synthesizeJavaAnonymousClassName(constant); + if (bodiedName !== undefined) { + const body = constant.childForFieldName?.('body'); + if (body !== null && body !== undefined && body.type === 'class_body') { + out.push({ + '@declaration.class': nodeToCapture('@declaration.class', body), + '@declaration.name': syntheticCapture('@declaration.name', body, bodiedName), + }); + if (hostEnum !== undefined) { + out.push({ + '@reference.inherits': nodeToCapture('@reference.inherits', body), + '@reference.name': syntheticCapture('@reference.name', body, hostEnum), + }); + } + } + } + + // Receiver dispatch (#2561): `E.CONST.method()` resolves through the + // generic compound-receiver chain walk, which looks up each dotted + // segment via the owning class scope's `typeBindings` map — the same + // mechanism a field declaration uses (`private User user;` binds + // `user` on the class scope). Binding the constant's own simple name + // there — to its synthesized `E$N` class when bodied (MRO includes E, + // so members inherited from the enum still resolve), or to the host + // enum itself when body-less — makes `E.CONST.method()` resolve with + // no changes to the shared receiver-binding machinery. + const constantNameNode = constant.childForFieldName?.('name'); + const constantType = bodiedName ?? hostEnum; + if (constantNameNode !== null && constantNameNode !== undefined && constantType !== undefined) { out.push({ - '@reference.inherits': nodeToCapture('@reference.inherits', body), - '@reference.name': syntheticCapture('@reference.name', body, hostEnum), + '@type-binding.annotation': nodeToCapture('@type-binding.annotation', constant), + '@type-binding.name': nodeToCapture('@type-binding.name', constantNameNode), + '@type-binding.type': syntheticCapture('@type-binding.type', constant, constantType), }); } } diff --git a/gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/EnumConst.java b/gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/EnumConst.java index 65292b0c5..6342aa576 100644 --- a/gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/EnumConst.java +++ b/gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/EnumConst.java @@ -21,4 +21,8 @@ class Unrelated { public void caller() { hook(); } + + public void dispatchToConstant() { + EnumConst.A.hook(); + } } diff --git a/gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/Plain.java b/gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/Plain.java new file mode 100644 index 000000000..87c750f61 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/Plain.java @@ -0,0 +1,13 @@ +public enum Plain { + A; + + public void m() { + System.out.println("plain m"); + } +} + +class PlainCaller { + public void callPlain() { + Plain.A.m(); + } +} diff --git a/gitnexus/test/integration/resolvers/java.test.ts b/gitnexus/test/integration/resolvers/java.test.ts index c7e0545c0..1eeb35e04 100644 --- a/gitnexus/test/integration/resolvers/java.test.ts +++ b/gitnexus/test/integration/resolvers/java.test.ts @@ -3055,3 +3055,32 @@ describe('Java enum constant bodies (#2555)', () => { expect(misattributed).toBeUndefined(); }, 60000); }); + +// --------------------------------------------------------------------------- +// #2561: E.CONST.method() emits no CALLS edge — the receiver-side follow-up +// to #2555. A bodied constant's receiver must resolve to its synthesized +// E$N class; a body-less constant's receiver must resolve to the host enum +// itself. +// --------------------------------------------------------------------------- + +describe('Java enum-constant receiver dispatch (#2561)', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'java-enum-constant-body'), () => {}); + }, 60000); + + it('resolves EnumConst.A.hook() to the bodied constant override (EnumConst$1.hook#0)', () => { + const calls = getRelationships(result, 'CALLS'); + const dispatch = calls.find((c) => c.source === 'dispatchToConstant' && c.target === 'hook'); + expect(dispatch).toBeDefined(); + expect(dispatch!.rel.targetId).toBe('Method:src/EnumConst.java:EnumConst$1.hook#0'); + }); + + it("resolves Plain.A.m() to the body-less constant's inherited enum method (Plain.m#0)", () => { + const calls = getRelationships(result, 'CALLS'); + const dispatch = calls.find((c) => c.source === 'callPlain' && c.target === 'm'); + expect(dispatch).toBeDefined(); + expect(dispatch!.rel.targetId).toBe('Method:src/Plain.java:Plain.m#0'); + }); +}); From d9437e6d7470756be04d5862665df69ae6a009c4 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 21 Jul 2026 12:19:51 +0000 Subject: [PATCH 42/61] test(bench): rebaseline java scope-capture fingerprint for #2561 The enum-constant receiver-dispatch fix adds one @type-binding.* capture per enum constant, so the java scope-capture fingerprint shifts (85fc7af9 -> a822cef9). Pure capture-additive drift; no bench fixtures added; scaling 1.024 < 1.5 budget. Verified `measure.mjs --check` passes for all 14 languages. Co-Authored-By: Claude Opus 4.8 (1M context) --- gitnexus/bench/scope-capture/baselines.json | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index d1413db56..0c065f52b 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -89,7 +89,7 @@ "_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0." }, "java": { - "fingerprint": "85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537", + "fingerprint": "a822cef937cb65c544987a155b05c2715f6648cbd4be30d2ad36700c4e6846c9", "scaling_budget": 1.5, "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata; same-name lexical regions use an O(ancestor-depth) ID-set lookup. Prior d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a -> 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4; scaling 0.992 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Java method-reference/SAM callable flow facts with invocation-result suppression. Prior 062d754764aaa8a6772fb90875c710502a63e3e7a300e633942381ed914faada -> d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a; scaling 1.074 < 1.5.", @@ -98,7 +98,8 @@ "_rebaselined_2522_review_fixes": "PR #2522 review fixes: get/test dropped from callableProtocolMethods. Prior 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4 -> f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67; scaling ratio re-verified within budget.", "_rebaselined_2550_instance_model": "PR #2549 (#2550): anonymous class bodies emit synthesized @declaration.class/@declaration.name (Worker$N), an @reference.inherits to the constructed type, and receiver @type-binding.* captures; six new java-* fixtures joined the corpus. Prior f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67 -> d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90; scaling 1.058 < 1.5.", "_rebaselined_2555_enum_constant_bodies": "PR for #2555: enum constant bodies emit synthesized E$N classes + @reference.inherits to the host enum; anonymous naming follows JLS 13.1 immediately-enclosing-type chains INCLUDING anonymous enclosing types (NestHost$1$1, N$1$1); six new java-* fixtures joined the corpus. Prior d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90 -> 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca; scaling 1.05 < 1.5.", - "_rebaselined_2564_record_capture": "PR for #2564: JAVA_QUERIES gained a (record_declaration name: (identifier) @name) @definition.record capture, previously entirely missing (record_declaration had no structure-phase capture at all, unlike class/interface/enum) - a record's methods existed as ownerless Method nodes with no HAS_METHOD edge. Two new java-* fixtures (java-record-methods, java-new-expr-chain-call) joined the corpus. Prior 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca -> 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537; scaling 1.059 < 1.5." + "_rebaselined_2564_record_capture": "PR for #2564: JAVA_QUERIES gained a (record_declaration name: (identifier) @name) @definition.record capture, previously entirely missing (record_declaration had no structure-phase capture at all, unlike class/interface/enum) - a record's methods existed as ownerless Method nodes with no HAS_METHOD edge. Two new java-* fixtures (java-record-methods, java-new-expr-chain-call) joined the corpus. Prior 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca -> 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537; scaling 1.059 < 1.5.", + "_rebaselined_2561_enum_constant_receiver": "PR for #2561: synthesizeJavaAnonymousClassDeclarations now emits a class-scope @type-binding.annotation/name/type per enum constant (constant simple name -> its E$N synthesized class when bodied, else the host enum) so E.CONST.method() resolves through the existing compound-receiver chain walk. Pure capture-additive (one extra type-binding match per enum_constant in the corpus); no bench fixtures added. Prior 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537 -> a822cef937cb65c544987a155b05c2715f6648cbd4be30d2ad36700c4e6846c9; scaling 1.024 < 1.5." }, "typescript": { "fingerprint": "3280b13d3f9378ab23eee31c2edc779b5a9ae1e7bb510c23a24855b44406d2f4", From 2a85425ad8b02cc2f29a8d4f4dc96ab87840f4b8 Mon Sep 17 00:00:00 2001 From: Ko <64240632+GenKoKo@users.noreply.github.com> Date: Tue, 21 Jul 2026 21:54:34 +0900 Subject: [PATCH 43/61] Merge pull request #2542 from GenKoKo/fix/worker-stdout-and-ready-timeout fix(ingestion): pipe worker stdout and make ready timeout configurable --- README.md | 1 + gitnexus/README.md | 3 +- .../src/core/ingestion/workers/worker-pool.ts | 81 +++++++++---- .../worker-pool-ready-timeout-env.test.ts | 91 +++++++++++++++ .../unit/worker-pool-stdout-forward.test.ts | 108 ++++++++++++++++++ 5 files changed, 260 insertions(+), 24 deletions(-) create mode 100644 gitnexus/test/unit/worker-pool-ready-timeout-env.test.ts create mode 100644 gitnexus/test/unit/worker-pool-stdout-forward.test.ts diff --git a/README.md b/README.md index ea40b33c2..495483cf8 100644 --- a/README.md +++ b/README.md @@ -488,6 +488,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max | `PROF_LBUG_LOAD` | unset | When `1`, emits one `[lbug-load prof]` summary line per `loadGraphToLbug` call breaking the graph-DB persistence wall into stages (`csv-emit` / `copy-nodes` / `copy-rels` / `fallback` / `total`) plus node & edge counts. Zero-cost when unset. | Attributing large-repo analyze wall time across CSV generation vs. LadybugDB `COPY` (issue #2203) — the analyze "emit" timing is the scope-resolution bucket, not this DB-write path. | | `GITNEXUS_MAX_FILE_SIZE` | `512` (KB) | Walker skip threshold in KB. Hard cap is `32768` (tree-sitter buffer ceiling). Equivalent to `--max-file-size `. | Indexing repos with intentionally-large source files (generated parsers, vendored bundles) that should still be parsed. | | `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS` | `30000` | Worker idle timeout in milliseconds before retry/fallback. Equivalent to `--worker-timeout ` × 1000. | Slow-parsing files (large minified JS, deeply-nested TS types) that legitimately need more than 30s. | +| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget in milliseconds for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. | Slow or heavily loaded hosts where a full pool cold-starting concurrently needs more than 5s, and analyze aborts with "did not report ready within 5000ms". | | `GITNEXUS_FTS_STEMMER` | `porter` | Stemmer used when rebuilding BM25/FTS indexes. Use `none` for CJK-heavy repositories, or a language stemmer such as `german`, `french`, or `spanish` for matching repository comments. Re-run `gitnexus analyze --repair-fts` after changing it. | Keyword search quality is poor for non-English comments or identifiers under English stemming. | | `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold in bytes. Equivalent to `--wal-checkpoint-threshold `. `-1` keeps LadybugDB's stock threshold (~16 MiB). Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | You need a larger or smaller WAL auto-checkpoint threshold for your analyze workload. | | `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling in bytes for every GitNexus database (analyze, MCP server, serve, group bridges). `0` restores LadybugDB's native unbounded default of 80% of system RAM; invalid values warn and fall back to the default (#2557). | A long-lived `gitnexus mcp` or a big incremental `analyze` uses too much memory, or a huge repo's working set genuinely needs a pool larger than 2 GiB. | diff --git a/gitnexus/README.md b/gitnexus/README.md index dc8ef17e5..619872b8f 100644 --- a/gitnexus/README.md +++ b/gitnexus/README.md @@ -548,7 +548,7 @@ For repositories with very large source files, `GITNEXUS_WORKER_SUB_BATCH_MAX_BY ### Worker pool resilience tuning -Three env vars expose the pool's resilience layers (respawn budget, cumulative-timeout cap, circuit breaker). Defaults are tuned for typical repos; bump them when an analyze legitimately needs more retries, or lower them to fail-fast on a known-bad shape. +Four env vars expose the pool's resilience layers (respawn budget, cumulative-timeout cap, circuit breaker, startup handshake). Defaults are tuned for typical repos; bump them when an analyze legitimately needs more retries, or lower them to fail-fast on a known-bad shape. | Variable | Default | Effect | | ----------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | @@ -556,6 +556,7 @@ Three env vars expose the pool's resilience layers (respawn budget, cumulative-t | `GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS` | `5 × subBatchTimeoutMs` | Total retry wall-time budget per job before quarantining. Bounds exponentially-growing retry waits. | | `GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD` | `max(3, poolSize)` | Per-slot consecutive deaths before the pool's circuit breaker trips. After tripping, dispatches require a fresh pool. | | `GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS` | `30000` | Max wait at pool shutdown for a retired worker still inside native code — terminated at its next JS-safe point instead of mid-native-call, which would abort the process (`Napi::Error`, #2432). | +| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. Raise it on a slow or heavily loaded host where a full pool cold-starting concurrently needs more than 5s. | | `GITNEXUS_CPP_CAPTURE_BUDGET_MS` | `20000` | Per-file wall-clock budget for C++ capture extraction; on breach the file keeps partial captures with a warning (#2432). `0` expires immediately. | ### Graph cleanup tuning diff --git a/gitnexus/src/core/ingestion/workers/worker-pool.ts b/gitnexus/src/core/ingestion/workers/worker-pool.ts index 28ec14589..9cf40a1f9 100644 --- a/gitnexus/src/core/ingestion/workers/worker-pool.ts +++ b/gitnexus/src/core/ingestion/workers/worker-pool.ts @@ -205,6 +205,19 @@ export interface WorkerPoolOptions { * created. Default `Math.max(3, poolSize)`. */ consecutiveFailureThreshold?: number; + /** + * Startup budget in milliseconds for a replacement worker to emit the + * `{type:'ready'}` handshake before the pool treats it as a startup + * crash (see {@link waitForWorkerReady}). Default 5000; also overridable + * via `GITNEXUS_WORKER_READY_TIMEOUT_MS`, mirroring + * `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS`. On a slow or heavily loaded + * host, a full pool of workers cold-starting concurrently can + * legitimately need more than 5s to load the native grammar bindings — + * without the override every slot times out and the pool misclassifies + * the slow start as a deterministic startup crash-loop, aborting the + * whole analyze. + */ + workerReadyTimeoutMs?: number; /** * Test-only injection point for the Worker constructor. When provided, * the pool uses this factory instead of `new Worker(workerUrl)`. Production @@ -406,17 +419,7 @@ const DEFAULT_TIMEOUT_BACKOFF_FACTOR = 2; const DEFAULT_MAX_RESPAWNS_PER_SLOT = 3; const DEFAULT_MAX_CUMULATIVE_TIMEOUT_FACTOR = 5; const DEFAULT_CONSECUTIVE_FAILURE_THRESHOLD_FLOOR = 3; -/** - * Bounded wait for a replacement worker to emit the `{type:'ready'}` - * handshake from `parse-worker.ts`. Trusting Node's `online` event alone - * lets a worker that crashes during top-of-script init slip past pool - * startup — the pool only notices on the first dispatch's idle timeout - * (default 30s). 5 seconds is a generous budget for parser + grammar - * imports; if the worker hasn't reported ready by then, it's almost - * certainly stuck or crashed and the pool should surface the failure - * fast rather than wait out the dispatch idle timeout. - */ -const WORKER_READY_TIMEOUT_MS = 5_000; +const DEFAULT_WORKER_READY_TIMEOUT_MS = 5_000; /** * Default upper bound on auto-resolved pool size. Past 16 workers the * dominant cost shifts from worker-side parsing to main-thread merge / @@ -547,6 +550,7 @@ interface ResolvedWorkerPoolOptions { maxCumulativeTimeoutMs: number; consecutiveFailureThreshold: number; shutdownDrainMs: number; + workerReadyTimeoutMs: number; } export function resolveWorkerPoolOptions( @@ -583,6 +587,10 @@ export function resolveWorkerPoolOptions( nonNegativeInteger(options.shutdownDrainMs) ?? nonNegativeInteger(process.env.GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS) ?? DEFAULT_SHUTDOWN_DRAIN_MS, + workerReadyTimeoutMs: + positiveInteger(options.workerReadyTimeoutMs) ?? + positiveInteger(process.env.GITNEXUS_WORKER_READY_TIMEOUT_MS) ?? + DEFAULT_WORKER_READY_TIMEOUT_MS, }; } @@ -683,6 +691,27 @@ function captureWorkerStderr(worker: Worker): void { stream.on('error', () => undefined); } +/** + * Forward a worker's piped stdout to the parent process's stdout, so worker + * logs stay visible now that the production factory spawns with + * `{ stdout: true }`. Workers with INHERITED stdout have been observed to + * crash silently during top-of-script init (exit code 1, nothing on stderr, + * roughly half of a concurrently spawned pool) on macOS 26.5 under both + * Node 22 and 26; piping stdout eliminates the crash entirely. Piping also + * matches the existing stderr handling, so worker output no longer races the + * parent's raw fd. No-op when the worker has no `stdout` stream (test + * factories). + */ +function forwardWorkerStdout(worker: Worker): void { + const stream = worker.stdout; + if (!stream) return; + stream.on('data', (chunk: Buffer | string) => { + process.stdout.write(chunk); + }); + // A stdout stream error must never crash the pool. + stream.on('error', () => undefined); +} + /** Captured stderr tail for a worker, trimmed; '' when nothing was captured. */ function workerStderrTail(worker: Worker): string { return workerStderrTails.get(worker)?.text.trim() ?? ''; @@ -722,13 +751,14 @@ function workerErrorReason(workerIndex: number, message: string, stack?: string) * (parser/grammar import failure, missing native binding) slip past * pool startup. The pool then only noticed the dead replacement on the * first dispatch's idle timeout (default 30s) — a long stall masking - * an actual crash. This handshake bounds the wait at - * {@link WORKER_READY_TIMEOUT_MS} and surfaces init failures as - * `error` / `exit` / `messageerror` events directly. `messageerror` is - * wired the same way: a V8 deserialization failure during startup is - * treated as worker death and rejects the readiness promise. + * an actual crash. This handshake bounds the wait at `readyTimeoutMs` + * (see {@link WorkerPoolOptions.workerReadyTimeoutMs}) and surfaces init + * failures as `error` / `exit` / `messageerror` events directly. + * `messageerror` is wired the same way: a V8 deserialization failure + * during startup is treated as worker death and rejects the readiness + * promise. */ -function waitForWorkerReady(worker: Worker): Promise { +function waitForWorkerReady(worker: Worker, readyTimeoutMs: number): Promise { return new Promise((resolve, reject) => { const cleanup = () => { clearTimeout(timer); @@ -781,11 +811,11 @@ function waitForWorkerReady(worker: Worker): Promise { new Error( withStderr( worker, - `Replacement worker did not report ready within ${WORKER_READY_TIMEOUT_MS}ms — likely crashed during top-of-script init`, + `Replacement worker did not report ready within ${readyTimeoutMs}ms — likely crashed during top-of-script init (slow host? raise GITNEXUS_WORKER_READY_TIMEOUT_MS)`, ), ), ); - }, WORKER_READY_TIMEOUT_MS); + }, readyTimeoutMs); worker.on('message', onMessage); worker.once('error', onError); worker.once('exit', onExit); @@ -931,6 +961,10 @@ export const createWorkerPool = ( options?.workerFactory ?? ((url: URL) => new Worker(url, { + // Piped (not inherited) stdio: stderr for crash capture (#1741), + // stdout because inherited stdout triggers silent startup crashes on + // some hosts (see forwardWorkerStdout). + stdout: true, stderr: true, workerData: workerStoreData, // The CFG visitors build per-function control-flow graphs by RECURSIVE @@ -944,10 +978,11 @@ export const createWorkerPool = ( // try/catch) and only that function's PDG is skipped, never a crash. resourceLimits: { stackSizeMb: 16 }, })); - /** Spawn + wire stderr capture in one step (used by all spawn sites). */ + /** Spawn + wire stdio capture/forwarding in one step (used by all spawn sites). */ const spawnAndCapture = (url: URL): Worker => { const worker = spawnWorker(url); captureWorkerStderr(worker); + forwardWorkerStdout(worker); return worker; }; const workers: (Worker | undefined)[] = new Array(size); @@ -1099,7 +1134,7 @@ export const createWorkerPool = ( const worker = workers[i]; if (!worker) return; // terminated mid-startup try { - await waitForWorkerReady(worker); + await waitForWorkerReady(worker, poolOptions.workerReadyTimeoutMs); anyWorkerReachedReady = true; return; // ready — slot stays in activeSlots } catch (err) { @@ -1161,7 +1196,7 @@ export const createWorkerPool = ( chunkHash?: string, ): Promise => { // Await the initial-spawn readiness gate (F13). On first dispatch - // this blocks for up to WORKER_READY_TIMEOUT_MS while every initial + // this blocks for up to poolOptions.workerReadyTimeoutMs while every initial // worker's `{type:'ready'}` handshake is checked; on subsequent // dispatches the promise is already settled and resolves // synchronously. Slots whose initial worker crashed in top-of- @@ -1360,7 +1395,7 @@ export const createWorkerPool = ( if (stopped) return false; const replacement = spawnAndCapture(workerUrl); try { - await waitForWorkerReady(replacement); + await waitForWorkerReady(replacement, poolOptions.workerReadyTimeoutMs); } catch (err) { await replacement.terminate().catch(() => undefined); logger.warn( diff --git a/gitnexus/test/unit/worker-pool-ready-timeout-env.test.ts b/gitnexus/test/unit/worker-pool-ready-timeout-env.test.ts new file mode 100644 index 000000000..f64135974 --- /dev/null +++ b/gitnexus/test/unit/worker-pool-ready-timeout-env.test.ts @@ -0,0 +1,91 @@ +/** + * `GITNEXUS_WORKER_READY_TIMEOUT_MS` overrides the worker ready budget. + * + * The 5s default is a startup budget for parser + grammar imports. On a slow + * or heavily loaded host a full pool of workers cold-starting concurrently + * can legitimately need more: without an override every slot misses the + * handshake, the identical timeout messages reproduce across respawns, and + * the pool misclassifies the slow start as a deterministic startup + * crash-loop — aborting the whole analyze. The env var mirrors + * `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS`. + * + * `resolveWorkerPoolOptions` reads the env var fresh on every + * `createWorkerPool` call, so each test just sets the env var before + * constructing the pool — no module reset needed. + */ +import { describe, expect, it, beforeEach, afterEach } from 'vitest'; +import { EventEmitter } from 'node:events'; +import path from 'node:path'; +import os from 'node:os'; +import fs from 'node:fs'; +import { pathToFileURL } from 'node:url'; +import { + createWorkerPool, + WorkerPoolInitializationError, +} from '../../src/core/ingestion/workers/worker-pool.js'; + +/** Worker double that never reports ready and never exits: a slow starter. */ +class NeverReadyWorker extends EventEmitter { + readonly stderr = new EventEmitter(); + postMessage(): void {} + async terminate(): Promise { + return 0; + } +} + +let tempDir: string; +let workerUrl: URL; +const ENV_KEY = 'GITNEXUS_WORKER_READY_TIMEOUT_MS'; +let savedEnv: string | undefined; + +beforeEach(() => { + tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-worker-ready-timeout-')); + const workerPath = path.join(tempDir, 'fake-worker.js'); + fs.writeFileSync(workerPath, '// fake'); + workerUrl = pathToFileURL(workerPath) as URL; + savedEnv = process.env[ENV_KEY]; +}); + +afterEach(() => { + if (savedEnv === undefined) delete process.env[ENV_KEY]; + else process.env[ENV_KEY] = savedEnv; + try { + fs.rmSync(tempDir, { recursive: true, force: true }); + } catch { + /* best-effort */ + } +}); + +describe('worker pool — GITNEXUS_WORKER_READY_TIMEOUT_MS override', () => { + it('applies the override to the readiness deadline and its failure message', async () => { + process.env[ENV_KEY] = '50'; + const pool = createWorkerPool(workerUrl, 1, { + workerFactory: () => new NeverReadyWorker() as unknown as Worker, + }); + + const err = await pool + .dispatch([{ path: 'a.ts', content: 'x' }]) + .catch((e: unknown) => e as InstanceType); + + expect(err).toBeInstanceOf(WorkerPoolInitializationError); + expect(err.readinessFailures.join('\n')).toContain('within 50ms'); + + await pool.terminate().catch(() => undefined); + }); + + it('falls back to the 5s default when the value is not a positive integer', async () => { + process.env[ENV_KEY] = 'not-a-number'; + const pool = createWorkerPool(workerUrl, 1, { + workerFactory: () => new NeverReadyWorker() as unknown as Worker, + }); + + const err = await pool + .dispatch([{ path: 'a.ts', content: 'x' }]) + .catch((e: unknown) => e as InstanceType); + + expect(err).toBeInstanceOf(WorkerPoolInitializationError); + expect(err.readinessFailures.join('\n')).toContain('within 5000ms'); + + await pool.terminate().catch(() => undefined); + }); +}); diff --git a/gitnexus/test/unit/worker-pool-stdout-forward.test.ts b/gitnexus/test/unit/worker-pool-stdout-forward.test.ts new file mode 100644 index 000000000..79e2654de --- /dev/null +++ b/gitnexus/test/unit/worker-pool-stdout-forward.test.ts @@ -0,0 +1,108 @@ +/** + * Worker stdout is piped and forwarded, not inherited. + * + * The production factory now spawns workers with `{ stdout: true }`: workers + * with INHERITED stdout have been observed to crash silently during + * top-of-script init (exit code 1, nothing on stderr, roughly half of a + * concurrently spawned pool) on macOS 26.5 under both Node 22 and 26. + * Piping avoids the crash, and `forwardWorkerStdout` mirrors the piped + * stream back to the parent's stdout so worker logs stay visible — the same + * tee shape `captureWorkerStderr` uses for stderr (#1741). + * + * This test injects a fake worker that writes to its `stdout` stream and + * asserts the pool forwards it to `process.stdout`; a stdout-less test + * factory must remain a no-op. + */ +import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest'; +import { EventEmitter } from 'node:events'; +import path from 'node:path'; +import os from 'node:os'; +import fs from 'node:fs'; +import { pathToFileURL } from 'node:url'; +import { createWorkerPool } from '../../src/core/ingestion/workers/worker-pool.js'; + +const WORKER_LOG_LINE = '{"level":30,"name":"gitnexus","msg":"parse-worker log line"}\n'; + +/** + * Worker double that starts cleanly and emits a log line on its piped + * `stdout` stream, mirroring a production worker spawned with + * `{ stdout: true }`. + */ +class ReadyWorkerWithStdout extends EventEmitter { + readonly stdout = new EventEmitter(); + readonly stderr = new EventEmitter(); + constructor() { + super(); + queueMicrotask(() => { + this.stdout.emit('data', Buffer.from(WORKER_LOG_LINE)); + this.emit('message', { type: 'ready' }); + }); + } + postMessage(): void {} + async terminate(): Promise { + return 0; + } +} + +/** Worker double with no stdio streams at all (typical test factory shape). */ +class ReadyWorkerWithoutStdio extends EventEmitter { + constructor() { + super(); + queueMicrotask(() => this.emit('message', { type: 'ready' })); + } + postMessage(): void {} + async terminate(): Promise { + return 0; + } +} + +let tempDir: string; +let workerUrl: URL; +let stdoutSpy: ReturnType; + +beforeEach(() => { + tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-worker-stdout-forward-')); + const workerPath = path.join(tempDir, 'fake-worker.js'); + fs.writeFileSync(workerPath, '// fake'); + workerUrl = pathToFileURL(workerPath) as URL; + // Capture the forwarded worker stdout without polluting test output. + stdoutSpy = vi.spyOn(process.stdout, 'write').mockReturnValue(true); +}); + +afterEach(() => { + stdoutSpy.mockRestore(); + try { + fs.rmSync(tempDir, { recursive: true, force: true }); + } catch { + /* best-effort */ + } +}); + +describe('worker pool — stdout forwarding', () => { + it("forwards a worker's piped stdout to the parent process stdout", async () => { + const pool = createWorkerPool(workerUrl, 1, { + workerFactory: () => new ReadyWorkerWithStdout() as unknown as Worker, + }); + + // Empty dispatch settles the initial-ready gate; the fake worker's stdout + // line is emitted in the same microtask turn as its ready handshake. + await pool.dispatch([]); + + const forwarded = stdoutSpy.mock.calls.map((c) => String(c[0])).join(''); + expect(forwarded).toContain('parse-worker log line'); + + await pool.terminate().catch(() => undefined); + }); + + it('is a no-op for workers without a stdout stream (test factories)', async () => { + const pool = createWorkerPool(workerUrl, 1, { + workerFactory: () => new ReadyWorkerWithoutStdio() as unknown as Worker, + }); + + // Must not throw while wiring stdio on a stream-less worker. + await pool.dispatch([]); + expect(pool.getStats().activeSlots).toBe(1); + + await pool.terminate().catch(() => undefined); + }); +}); From 5549403082019a3f036dea082787698363b785fd Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 21 Jul 2026 14:40:20 +0100 Subject: [PATCH 44/61] fix(eval): self-hosted skill-evolution runner + sandbox Python 3 trust fix (#2600) * fix(eval): move skill-evolution to a self-hosted runner and fix the sandbox's Python 3 trust gap GitHub-hosted runners hard-cap job execution at 6 hours, which is too short once a benchmark session actually invokes Skill/MCP tools for real (the --bare fix in #2584 means sessions no longer no-op). Move the job onto a self-hosted runner (5-day cap instead) and document the activation step in the workflow's own checklist. Validating the self-hosted run surfaced a real bug: gitnexus-plan sessions inside the bwrap sandbox failed with "planning must create or modify exactly one plan artifact; observed 0". Root cause: evidence-provenance.mjs's atomic plan-writer only trusts a Python 3 binary owned by root or by the current process. Inside this --unshare-user sandbox only the calling uid is mapped (root isn't), so the real, root-owned /usr/bin/python3 surfaces as the kernel's overflow uid and gets correctly refused as untrusted. Fix: provision a small, self-owned wrapper script (same pattern already used for shell-prefix) that execs the real interpreter, so the sandbox has a Python 3 candidate the existing trust check can actually accept -- without touching that security-sensitive validation logic at all. Also add visibility so this class of failure isn't quiet next time: report.md now shows why each row failed (error_kinds), not just resolved 0/1, and the benchmark now exits non-zero when an incumbent arm -- the currently-shipped skill -- resolves zero across every task, since that reads as a broken harness rather than a normal candidate miss. Co-Authored-By: Claude Sonnet 5 * fix(eval): close the broken_incumbent_arms zero-valid-runs gap; document runner exposure tradeoff Addresses the two MEDIUM findings from the gitnexus-review-agent on this PR (https://github.com/abhigyanpatwari/GitNexus/pull/2600#issuecomment-5033363096). broken_incumbent_arms required valid_runs > 0 before flagging an incumbent, so an incumbent that fails every run with an excluded-but-non-systemic error_kind (e.g. evidence-unverified, which the outage-streak breaker explicitly resets on rather than accumulates) never accumulated a single valid run and sailed through silently -- the exact quiet no-promotion outcome this guard exists to catch, and arguably worse than the some-runs-resolved-zero case since here nothing completed at all. aggregate() never marks an excluded/unverifiable row resolved=True, so dropping the valid_runs requirement and checking resolved == 0 alone correctly covers both cases. Added a test for exactly this all-excluded scenario, which none of the existing three did. Updated the workflow's own activation checklist to reflect what's actually true now (the gitnexus-evolution environment's branch policy and the self-hosted runner are both live, codified in infra/gitnexus-evolution/ in a companion PR) and documented the exposure-window tradeoff the review flagged: the runner is stopped between runs but not destroyed/recreated per run, so it isn't fully ephemeral. Stopping already bounds the exposure window to the job's own runtime on one day out of seven; full per-job ephemeral provisioning is a deliberate non-goal for a job that runs at most weekly, revisit if that changes. Co-Authored-By: Claude Sonnet 5 * docs(eval): remove public infra/ pointers from the activation checklist PR #2603 (the Terraform codification this checklist pointed to) got closed -- publishing the exact IAM roles, security group rules, and self-hosted runner topology for a real, live AWS account isn't safe to do in a public repo, even with no literal secrets or resource IDs in the diff. The underlying AWS/GitHub setup is unaffected and still documented privately; this just removes the now-dangling references to a directory that won't exist in this repo. Co-Authored-By: Claude Sonnet 5 * ci(actionlint): register the gitnexus-evolution self-hosted runner label actionlint rejected `runs-on: [self-hosted, linux, x64, gitnexus-evolution]` in gitnexus-skill-evolution.yml because it can't discover custom runner labels. Register it in .github/actionlint.yaml so the Workflow Lint check passes. Co-Authored-By: Claude Opus 4.8 (1M context) --------- Co-authored-by: Claude Sonnet 5 --- .github/actionlint.yaml | 5 ++ .../workflows/gitnexus-skill-evolution.yml | 28 ++++++++- eval/tests/test_proposer_sandbox.py | 22 +++++++ eval/tests/test_workflow_bench.py | 57 ++++++++++++++++++ eval/workflow_bench/proposer_sandbox.py | 21 +++++++ eval/workflow_bench/runner.py | 58 +++++++++++++++++-- 6 files changed, 183 insertions(+), 8 deletions(-) create mode 100644 .github/actionlint.yaml diff --git a/.github/actionlint.yaml b/.github/actionlint.yaml new file mode 100644 index 000000000..17f2eebe7 --- /dev/null +++ b/.github/actionlint.yaml @@ -0,0 +1,5 @@ +# Custom self-hosted runner labels actionlint can't discover on its own. +# gitnexus-evolution: the skill-evolution EC2 runner (infra/gitnexus-evolution/). +self-hosted-runner: + labels: + - gitnexus-evolution diff --git a/.github/workflows/gitnexus-skill-evolution.yml b/.github/workflows/gitnexus-skill-evolution.yml index 8d9454d56..4cf601549 100644 --- a/.github/workflows/gitnexus-skill-evolution.yml +++ b/.github/workflows/gitnexus-skill-evolution.yml @@ -12,12 +12,34 @@ # App that opens the promotion PR). The Mint-App-Token step hard-fails # without them once a promotion is detected. Verify the App installation # is scoped to this repo with only Contents: RW + Pull requests: RW. -# [ ] Create the protected Environment `gitnexus-evolution` with a +# [x] Create the protected Environment `gitnexus-evolution` with a # deployment-branch rule restricting it to `main`, and ideally scope the # three secrets above to that Environment. workflow_dispatch runs this # workflow (and eval/workflow_bench/evolve.py) from the *dispatched ref*, # so this server-side rule — not a code-side guard the branch could edit # away — is what stops a non-main branch from running with the secrets. +# [x] Register a self-hosted runner labeled `gitnexus-evolution` (a dedicated +# EC2 box works well). GitHub-hosted runners hard-cap job execution at 6 +# hours, non-configurable — too short once a benchmark session actually +# invokes Skill/MCP tools for real. Self-hosted runners cap at 5 days +# instead. This job only ever runs on schedule/workflow_dispatch, never +# on fork-PR content, so the usual public-repo self-hosted-runner risk +# doesn't apply — still keep the box dedicated to this workflow, with +# outbound-only network access, and prefer on-demand over Spot (a Spot +# reclaim mid-run loses the same way a 6-hour timeout does). Instance, +# security group, and IAM setup are documented privately, not in this +# repo — publishing the exact topology of a real, live AWS account +# isn't safe to do in a public repo even without literal secrets. +# Accepted tradeoff: the box is stopped between runs (an EventBridge +# schedule starts it ~15min before the Saturday cron and stops it 24h +# later) but is not destroyed/recreated per run, so it isn't fully +# ephemeral — a compromise between the review-flagged ideal (re-image +# between runs, bounding how long the injected model API key could +# matter if the box were ever compromised some other way) and the added +# complexity of per-job ephemeral provisioning for a job that runs at +# most weekly. Revisit if run frequency increases or the threat model +# changes; stopping already bounds the exposure window to the job's own +# runtime on 1 day out of 7. # [ ] Run workflow_dispatch once and confirm: containment preflight passes, # the benchmark completes inside the job timeout, the results artifact # uploads, and a promotion (if any) opens a well-formed PR. @@ -77,13 +99,13 @@ jobs: github.event_name == 'workflow_dispatch' || vars.GITNEXUS_EVOLUTION_ENABLED == 'true' ) - runs-on: ubuntu-latest + runs-on: [self-hosted, linux, x64, gitnexus-evolution] # Gate promotion runs on a protected Environment. An admin must attach a # deployment-branch rule (main only) and ideally scope the three secrets to # it — server-side enforcement a dispatched non-main ref cannot bypass by # editing its own workflow copy. See the activation checklist above. environment: gitnexus-evolution - timeout-minutes: 355 # ceiling just under GitHub's 360-minute hard cap + timeout-minutes: 1440 # self-hosted ceiling is 5 days (7200min); 24h is a generous margin over a single-generation serial run permissions: contents: read # The promotion PR uses a short-lived App token minted below. env: diff --git a/eval/tests/test_proposer_sandbox.py b/eval/tests/test_proposer_sandbox.py index e5ceb7b2c..70f4fe76b 100644 --- a/eval/tests/test_proposer_sandbox.py +++ b/eval/tests/test_proposer_sandbox.py @@ -19,6 +19,7 @@ from workflow_bench.process_control import ManagedProcessResult, run_managed from workflow_bench.proposer_sandbox import ( MAX_BUNDLE_BYTES, MAX_EVIDENCE_FILE_BYTES, + SANDBOX_PYTHON3, SANDBOX_SHELL_PREFIX, SANDBOX_USER_SKILLS, ReadOnlyMount, @@ -162,6 +163,27 @@ def test_sandbox_command_has_minimal_mounts_and_no_host_root_bind(tmp_path: Path ) assert probe.returncode == 0, probe.stderr assert probe.stdout == "/home/agent|/opt/claude:/usr/local/bin:/usr/bin:/bin" + + # The evidence-provenance.mjs plan-writer's PATH-scan trusts a Python 3 + # candidate only if it (and its directory) is owned by root or by the + # current process — real /usr/bin/python3 is root-owned on the host, + # which surfaces as the kernel's overflow uid inside this + # --unshare-user sandbox (root itself is never mapped in). This wrapper + # is freshly created by the host process instead, so it's trusted, and + # it must still exec through to a real, working Python 3. + python3_index = argv.index(SANDBOX_PYTHON3) + assert argv[python3_index - 2] == "--ro-bind" + python3_wrapper = Path(argv[python3_index - 1]) + assert stat.S_IMODE(python3_wrapper.stat().st_mode) == 0o500 + version = subprocess.run( + [str(python3_wrapper), "-I", "-S", "-c", "import sys; print(sys.version_info[0])"], + text=True, + capture_output=True, + check=False, + ) + assert version.returncode == 0, version.stderr + assert version.stdout.strip() == "3" + assert SANDBOX_USER_SKILLS in argv user_skills_index = argv.index(SANDBOX_USER_SKILLS) assert argv[user_skills_index - 2] == "--ro-bind" diff --git a/eval/tests/test_workflow_bench.py b/eval/tests/test_workflow_bench.py index 961c0edd4..b1222e7d2 100644 --- a/eval/tests/test_workflow_bench.py +++ b/eval/tests/test_workflow_bench.py @@ -10,6 +10,7 @@ import yaml from workflow_bench.runner import ( aggregate, + broken_incumbent_arms, build_parser, infra_error_record, normalized_model_identifier, @@ -64,6 +65,7 @@ def test_aggregate_takes_medians_and_counts_resolved(): "valid_runs": 3, "excluded_runs": 0, "transcripts_missing": 0, + "error_kinds": {}, } @@ -333,6 +335,61 @@ def test_render_report_surfaces_excluded_and_unverified_runs(): assert "no locatable session transcript" in report +def test_render_report_surfaces_why_each_row_failed(): + results = { + "t": { + "workflow": aggregate( + [record(resolved=False, error_kind="plan-evidence-invalid")], + ), + } + } + report = render_report(results) + assert "plan-evidence-invalid×1" in report + + +def test_broken_incumbent_arms_flags_an_incumbent_that_resolved_nothing(): + results = { + "t1": {"workflow": aggregate([record(resolved=False, error_kind="plan-evidence-invalid")])}, + "t2": {"workflow": aggregate([record(resolved=False, error_kind="plan-evidence-invalid")])}, + } + assert broken_incumbent_arms(results, {"workflow"}) == ["workflow"] + + +def test_broken_incumbent_arms_ignores_a_merely_underperforming_candidate(): + # The incumbent works fine; only the candidate arm fails. That's a normal, + # expected "bad candidate" outcome and must not read as a broken harness. + results = { + "t1": { + "workflow": aggregate([record(resolved=True)]), + "candidate_workflow": aggregate([record(resolved=False, error_kind="verify-failed")]), + }, + } + assert broken_incumbent_arms(results, {"workflow"}) == [] + + +def test_broken_incumbent_arms_flags_an_incumbent_with_zero_valid_runs(): + # Every run excluded via an excluded-but-non-systemic error_kind + # ("evidence-unverified"): valid_runs == 0 for every task, which the old + # `valid_runs > 0` guard let sail through silently, and which the outage + # streak breaker also doesn't catch (it resets rather than accumulates + # on this exact error_kind -- see test_systemic_outage_streak_resets_on_non_outage). + results = { + "t1": {"workflow": aggregate([record(resolved=False, error_kind="evidence-unverified")])}, + "t2": {"workflow": aggregate([record(resolved=False, error_kind="evidence-unverified")])}, + } + assert results["t1"]["workflow"]["valid_runs"] == 0 + assert broken_incumbent_arms(results, {"workflow"}) == ["workflow"] + + +def test_broken_incumbent_arms_ignores_partial_incumbent_failure(): + # Resolved in at least one task — struggling, not broken. + results = { + "t1": {"workflow": aggregate([record(resolved=False, error_kind="verify-failed")])}, + "t2": {"workflow": aggregate([record(resolved=True)])}, + } + assert broken_incumbent_arms(results, {"workflow"}) == [] + + def test_infra_error_record_captures_the_failure_and_is_excluded(): exc = subprocess.TimeoutExpired(cmd="claude -p", timeout=5) rec = infra_error_record(exc) diff --git a/eval/workflow_bench/proposer_sandbox.py b/eval/workflow_bench/proposer_sandbox.py index 254906139..731f11ad2 100644 --- a/eval/workflow_bench/proposer_sandbox.py +++ b/eval/workflow_bench/proposer_sandbox.py @@ -26,6 +26,7 @@ SANDBOX_HOME = "/home/agent" SANDBOX_TMP = "/tmp" SANDBOX_CLAUDE = "/opt/claude/claude" SANDBOX_SHELL_PREFIX = "/opt/claude/shell-prefix" +SANDBOX_PYTHON3 = "/opt/claude/python3" SANDBOX_PATH = "/opt/claude:/usr/local/bin:/usr/bin:/bin" SANDBOX_GITNEXUS = "/opt/gitnexus" SANDBOX_GITNEXUS_SHARED = "/opt/gitnexus-shared" @@ -384,6 +385,24 @@ def _create_shell_prefix_wrapper(private_root: Path) -> Path: return wrapper +def _create_python3_wrapper(private_root: Path) -> Path: + """A trusted, self-owned Python 3 launcher for evidence-provenance.mjs's atomic mover. + + /usr/bin/python3 is a real system binary, but it's root-owned on the host. + Inside this --unshare-user sandbox only the calling uid is mapped (root is + not), so root-owned files surface as the kernel's overflow uid — which + evidence-provenance.mjs's PATH-scan correctly refuses to trust. This + wrapper is freshly created by the same host process that owns + home/temp/shell-prefix, so it maps to the sandbox's own trusted uid + instead, and simply execs the real interpreter through to do the work. + """ + + wrapper = private_root / "python3" + wrapper.write_text("#!/bin/bash\nset -eu\nexec /usr/bin/python3 \"$@\"\n") + wrapper.chmod(0o500) + return wrapper + + def _resolve_executable(executable: Path | str | None, default: str) -> Path: raw = os.fspath(executable) if executable is not None else shutil.which(default) if not raw: @@ -641,6 +660,7 @@ def prepare_sandbox( directory.mkdir(mode=0o700) directory.chmod(0o700) shell_prefix = _create_shell_prefix_wrapper(private_root) + python3_wrapper = _create_python3_wrapper(private_root) # Claude may discover user-level skills below HOME. Keep the rest of HOME # writable for normal CLI state, but overlay an immutable empty skills root # so a model cannot shadow the evaluated repository/plugin skill by name. @@ -651,6 +671,7 @@ def prepare_sandbox( *read_only_mounts, ReadOnlyMount(source=user_skills, target=SANDBOX_USER_SKILLS), ReadOnlyMount(source=shell_prefix, target=SANDBOX_SHELL_PREFIX), + ReadOnlyMount(source=python3_wrapper, target=SANDBOX_PYTHON3), ) primary: BaseException | None = None try: diff --git a/eval/workflow_bench/runner.py b/eval/workflow_bench/runner.py index 2a180f43f..b502c8023 100644 --- a/eval/workflow_bench/runner.py +++ b/eval/workflow_bench/runner.py @@ -679,13 +679,21 @@ def aggregate(records: list[dict[str, Any]]) -> dict[str, Any]: # unmeasured run makes the whole median unavailable so the gate won't rank # a candidate on a cost that was never actually captured. valid_costs = [r.get("cost_usd") for r in valid] - out["cost_usd"] = None if (not valid or any(cost is None for cost in valid_costs)) else statistics.median(valid_costs) + out["cost_usd"] = ( + None if (not valid or any(cost is None for cost in valid_costs)) else statistics.median(valid_costs) + ) out["resolved"] = sum(1 for r in records if r["resolved"]) out["runs"] = len(records) out["valid_runs"] = len(valid) out["excluded_runs"] = len(records) - len(valid) out["transcripts_missing"] = sum(1 for r in records if r.get("transcript_missing")) out["class"] = records[0].get("class", "") + error_kinds: dict[str, int] = {} + for r in records: + kind = r.get("error_kind") + if kind: + error_kinds[kind] = error_kinds.get(kind, 0) + 1 + out["error_kinds"] = error_kinds return out @@ -702,6 +710,33 @@ def savings(baseline: dict[str, Any], workflow: dict[str, Any]) -> dict[str, Any return out +def broken_incumbent_arms( + results: dict[str, dict[str, dict[str, Any]]], + incumbent_arms: set[str], +) -> list[str]: + """Incumbent arms that resolved nothing across every task they ran. + + An incumbent arm is the currently-shipped, presumably-working skill: if it + resolves NOTHING across every task it ran, that reads as an environment or + harness failure (missing trusted interpreter, stale skill fingerprint, + sandbox misconfiguration), not a skill regression. A candidate merely + underperforming is a normal, expected outcome and must not trip this — + only checking incumbents keeps that distinction. + + Deliberately does NOT require valid_runs > 0 per task: an incumbent that + fails every run with an excluded-but-non-systemic error_kind (e.g. + "evidence-unverified", which the outage-streak breaker explicitly resets + on rather than accumulates) would otherwise never accumulate a single + valid run and sail through silently — the exact "quiet no-promotion" + outcome this guard exists to catch, and arguably worse than the + some-runs-resolved-zero case since here nothing completed at all. + aggregate() never marks an excluded/unverifiable row resolved=True, so + resolved == 0 alone already covers both cases. + """ + present = incumbent_arms & {arm for arms in results.values() for arm in arms} + return sorted(arm for arm in present if all(arms[arm]["resolved"] == 0 for arms in results.values() if arm in arms)) + + def _na(value: Any) -> Any: """Render an unmeasured metric as ``n/a`` instead of a misleading number.""" return "n/a" if value is None else value @@ -726,8 +761,8 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str: "efficiency, sum usage from the session transcripts instead", "(dedup events sharing one message.id).", "", - "| task | class | arm | resolved | input | cache_create | cache_read | output | cost $ | wall s | turns | churn |", - "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |", + "| task | class | arm | resolved | input | cache_create | cache_read | output | cost $ | wall s | turns | churn | errors |", + "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |", ] for task_id, arms in results.items(): for arm, agg in arms.items(): @@ -735,12 +770,14 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str: resolved_cell = f"{agg['resolved']}/{agg.get('valid_runs', agg['runs'])}" if excluded: resolved_cell += f" ({excluded} excluded)" + error_cell = ", ".join(f"{kind}×{count}" for kind, count in sorted(agg.get("error_kinds", {}).items())) lines.append( f"| {task_id} | {agg['class']} | {arm} | {resolved_cell} " f"| {agg['input_tokens']:.0f} | {agg['cache_creation_input_tokens']:.0f} " f"| {agg['cache_read_input_tokens']:.0f} | {agg['output_tokens']:.0f} " f"| {_cost_cell(agg['cost_usd'])} | {agg['duration_s']:.0f} | {agg['num_turns']:.0f} " - f"| {agg['diff_files']:.0f}/+{agg['diff_insertions']:.0f}/−{agg['diff_deletions']:.0f} |" + f"| {agg['diff_files']:.0f}/+{agg['diff_insertions']:.0f}/−{agg['diff_deletions']:.0f} " + f"| {error_cell} |" ) for arm in arms: if arm != "baseline" and "baseline" in arms: @@ -749,7 +786,7 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str: f"| {task_id} | {arms[arm]['class']} | **{arm} savings %** | — " f"| {s['input_tokens']} | {s['cache_creation_input_tokens']} " f"| {s['cache_read_input_tokens']} | {s['output_tokens']} " - f"| {_na(s['cost_usd'])} | {s['duration_s']} | — | — |" + f"| {_na(s['cost_usd'])} | {s['duration_s']} | — | — | — |" ) lines.append("") all_aggs = [agg for arms in results.values() for agg in arms.values()] @@ -1333,6 +1370,17 @@ def main() -> None: } (out_dir / "promotion.json").write_text(json.dumps(promotion, indent=2) + "\n") print(f"\n{report}\n\nWritten to {out_dir}/") + broken_incumbents = broken_incumbent_arms(results, set(CANDIDATE_ARMS.values())) + if broken_incumbents: + # Fail loudly rather than let a broken environment read as a quiet + # "no promotion, incumbent stands." + print( + f"[harness-health] incumbent arm(s) {', '.join(broken_incumbents)} resolved zero " + "tasks across every valid run — this looks like an environment/harness failure, " + "not a normal candidate miss. See the errors column in report.md and error_detail " + "in results.jsonl. Exiting non-zero rather than reporting a quiet no-promotion." + ) + raise SystemExit(1) if outage_tripped: # Non-zero exit so a driver (evolve.py) treats the partial benchmark as a # failed run and halts instead of proposing from outage-truncated evidence. From 70e0a7766c17fe056b3aec99a12b0cad86dc66aa Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 21 Jul 2026 13:51:31 +0000 Subject: [PATCH 45/61] =?UTF-8?q?fix(java):=20address=20#2561=20review=20?= =?UTF-8?q?=E2=80=94=20inherited-dispatch=20test=20+=20bodied=20fail-safe?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two gitnexus-review-agent findings on PR #2602: - MEDIUM: the bodied-constant MRO-to-host-enum path (a qualified call to an inherited, non-overridden enum method) was claimed in a comment but never tested. Add EnumConst.A.log() -> EnumConst.log#0, exercising E$N's @reference.inherits MRO arm end to end. - LOW: `bodiedName ?? hostEnum` conflated "body-less" with "name synthesis failed on a bodied constant" (reachable only on malformed/error-recovery trees), silently binding an overriding constant's receiver to the host enum — a wrong edge instead of no edge. Switch to `isBodied ? bodiedName : hostEnum` so a bodied constant binds ONLY to its E$N class, mirroring the object_creation_expression branch's skip-on-synthesis-failure. Verified output-neutral on the well-formed bench corpus. Rebaseline the java scope-capture fingerprint (a822cef9 -> d04298a9): the bench corpus IS test/fixtures/lang-resolution, so the new dispatchInherited fixture method shifts it (+6 capture groups); the logic change contributes nothing (confirmed by isolating the fixture-only fingerprint). java.test.ts 242 passed; measure.mjs --check PASS (14 languages); tsc/prettier/eslint clean. Co-Authored-By: Claude Opus 4.8 (1M context) --- gitnexus/bench/scope-capture/baselines.json | 4 +-- .../core/ingestion/languages/java/captures.ts | 31 ++++++++++++------- .../src/EnumConst.java | 4 +++ .../test/integration/resolvers/java.test.ts | 11 +++++++ 4 files changed, 36 insertions(+), 14 deletions(-) diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index 0c065f52b..cab3765a4 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -89,7 +89,7 @@ "_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0." }, "java": { - "fingerprint": "a822cef937cb65c544987a155b05c2715f6648cbd4be30d2ad36700c4e6846c9", + "fingerprint": "d04298a91beec76d0fa7099b3d71265723be60c1df688969aa954f135dd49686", "scaling_budget": 1.5, "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata; same-name lexical regions use an O(ancestor-depth) ID-set lookup. Prior d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a -> 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4; scaling 0.992 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Java method-reference/SAM callable flow facts with invocation-result suppression. Prior 062d754764aaa8a6772fb90875c710502a63e3e7a300e633942381ed914faada -> d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a; scaling 1.074 < 1.5.", @@ -99,7 +99,7 @@ "_rebaselined_2550_instance_model": "PR #2549 (#2550): anonymous class bodies emit synthesized @declaration.class/@declaration.name (Worker$N), an @reference.inherits to the constructed type, and receiver @type-binding.* captures; six new java-* fixtures joined the corpus. Prior f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67 -> d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90; scaling 1.058 < 1.5.", "_rebaselined_2555_enum_constant_bodies": "PR for #2555: enum constant bodies emit synthesized E$N classes + @reference.inherits to the host enum; anonymous naming follows JLS 13.1 immediately-enclosing-type chains INCLUDING anonymous enclosing types (NestHost$1$1, N$1$1); six new java-* fixtures joined the corpus. Prior d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90 -> 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca; scaling 1.05 < 1.5.", "_rebaselined_2564_record_capture": "PR for #2564: JAVA_QUERIES gained a (record_declaration name: (identifier) @name) @definition.record capture, previously entirely missing (record_declaration had no structure-phase capture at all, unlike class/interface/enum) - a record's methods existed as ownerless Method nodes with no HAS_METHOD edge. Two new java-* fixtures (java-record-methods, java-new-expr-chain-call) joined the corpus. Prior 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca -> 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537; scaling 1.059 < 1.5.", - "_rebaselined_2561_enum_constant_receiver": "PR for #2561: synthesizeJavaAnonymousClassDeclarations now emits a class-scope @type-binding.annotation/name/type per enum constant (constant simple name -> its E$N synthesized class when bodied, else the host enum) so E.CONST.method() resolves through the existing compound-receiver chain walk. Pure capture-additive (one extra type-binding match per enum_constant in the corpus); no bench fixtures added. Prior 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537 -> a822cef937cb65c544987a155b05c2715f6648cbd4be30d2ad36700c4e6846c9; scaling 1.024 < 1.5." + "_rebaselined_2561_enum_constant_receiver": "PR for #2561: synthesizeJavaAnonymousClassDeclarations now emits a class-scope @type-binding.annotation/name/type per enum constant (constant simple name -> its E$N synthesized class when bodied, else the host enum) so E.CONST.method() resolves through the existing compound-receiver chain walk. Two drivers of the drift, both in the java-enum-constant-body fixture (this bench's corpus IS test/fixtures/lang-resolution): (1) one extra type-binding match per enum_constant from the capture change; (2) review follow-up added a body-less Plain.java enum + EnumConst.dispatchToConstant/dispatchInherited methods (bodied-override, inherited-via-MRO, and body-less dispatch call sites). The review's fail-safe hardening (bodied constant binds ONLY to E$N, never the host enum, when name synthesis fails on a malformed tree) is output-neutral on this well-formed corpus (verified: fingerprint identical with and without it). Prior 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537 -> d04298a91beec76d0fa7099b3d71265723be60c1df688969aa954f135dd49686; scaling < 1.5." }, "typescript": { "fingerprint": "3280b13d3f9378ab23eee31c2edc779b5a9ae1e7bb510c23a24855b44406d2f4", diff --git a/gitnexus/src/core/ingestion/languages/java/captures.ts b/gitnexus/src/core/ingestion/languages/java/captures.ts index 2212c4355..62d9fe788 100644 --- a/gitnexus/src/core/ingestion/languages/java/captures.ts +++ b/gitnexus/src/core/ingestion/languages/java/captures.ts @@ -375,20 +375,19 @@ function synthesizeJavaAnonymousClassDeclarations(rootNode: SyntaxNode): Capture // through the ownership gate's MRO arm. for (const constant of rootNode.descendantsOfType('enum_constant')) { const hostEnum = javaEnclosingEnumNameOf(constant); + const bodyNode = constant.childForFieldName?.('body'); + const isBodied = bodyNode !== null && bodyNode !== undefined && bodyNode.type === 'class_body'; const bodiedName = synthesizeJavaAnonymousClassName(constant); - if (bodiedName !== undefined) { - const body = constant.childForFieldName?.('body'); - if (body !== null && body !== undefined && body.type === 'class_body') { + if (bodiedName !== undefined && isBodied) { + out.push({ + '@declaration.class': nodeToCapture('@declaration.class', bodyNode), + '@declaration.name': syntheticCapture('@declaration.name', bodyNode, bodiedName), + }); + if (hostEnum !== undefined) { out.push({ - '@declaration.class': nodeToCapture('@declaration.class', body), - '@declaration.name': syntheticCapture('@declaration.name', body, bodiedName), + '@reference.inherits': nodeToCapture('@reference.inherits', bodyNode), + '@reference.name': syntheticCapture('@reference.name', bodyNode, hostEnum), }); - if (hostEnum !== undefined) { - out.push({ - '@reference.inherits': nodeToCapture('@reference.inherits', body), - '@reference.name': syntheticCapture('@reference.name', body, hostEnum), - }); - } } } @@ -401,8 +400,16 @@ function synthesizeJavaAnonymousClassDeclarations(rootNode: SyntaxNode): Capture // so members inherited from the enum still resolve), or to the host // enum itself when body-less — makes `E.CONST.method()` resolve with // no changes to the shared receiver-binding machinery. + // + // A bodied constant binds ONLY to its `E$N` class, never the host enum: + // if name synthesis fails on a malformed/error-recovery tree (`bodiedName` + // undefined despite a real body), emit nothing rather than silently + // misattributing an OVERRIDING constant's receiver to the enum's own + // (non-overridden) method — a wrong edge is worse than no edge. Mirrors + // the `object_creation_expression` branch, which skips on synthesis + // failure. `hostEnum` is used only for genuinely body-less constants. const constantNameNode = constant.childForFieldName?.('name'); - const constantType = bodiedName ?? hostEnum; + const constantType = isBodied ? bodiedName : hostEnum; if (constantNameNode !== null && constantNameNode !== undefined && constantType !== undefined) { out.push({ '@type-binding.annotation': nodeToCapture('@type-binding.annotation', constant), diff --git a/gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/EnumConst.java b/gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/EnumConst.java index 6342aa576..abfdeb902 100644 --- a/gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/EnumConst.java +++ b/gitnexus/test/fixtures/lang-resolution/java-enum-constant-body/src/EnumConst.java @@ -25,4 +25,8 @@ class Unrelated { public void dispatchToConstant() { EnumConst.A.hook(); } + + public void dispatchInherited() { + EnumConst.A.log(); + } } diff --git a/gitnexus/test/integration/resolvers/java.test.ts b/gitnexus/test/integration/resolvers/java.test.ts index 1eeb35e04..bfaa00196 100644 --- a/gitnexus/test/integration/resolvers/java.test.ts +++ b/gitnexus/test/integration/resolvers/java.test.ts @@ -3077,6 +3077,17 @@ describe('Java enum-constant receiver dispatch (#2561)', () => { expect(dispatch!.rel.targetId).toBe('Method:src/EnumConst.java:EnumConst$1.hook#0'); }); + it("resolves EnumConst.A.log() to the host enum's inherited method via E$N's MRO (EnumConst.log#0)", () => { + // A's body overrides hook() but NOT log(); log() lives only on the enum. + // The bodied constant's receiver binds to EnumConst$1, whose MRO includes + // EnumConst (via @reference.inherits), so the qualified call reaches the + // host enum's own method — the inherited-dispatch capability the fix enables. + const calls = getRelationships(result, 'CALLS'); + const dispatch = calls.find((c) => c.source === 'dispatchInherited' && c.target === 'log'); + expect(dispatch).toBeDefined(); + expect(dispatch!.rel.targetId).toBe('Method:src/EnumConst.java:EnumConst.log#0'); + }); + it("resolves Plain.A.m() to the body-less constant's inherited enum method (Plain.m#0)", () => { const calls = getRelationships(result, 'CALLS'); const dispatch = calls.find((c) => c.source === 'callPlain' && c.target === 'm'); From 3375beec8910d4f3363f6df41e3196a7fb7323a5 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 14:29:05 +0000 Subject: [PATCH 46/61] fix(rust): strip dyn keyword when normalizing trait-object type names MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit normalizeRustTypeName/normalizeRustReturnType stripped reference sigils, pointer sigils, and smart-pointer wrappers but never the `dyn` keyword, so a `&dyn Trait`-typed receiver normalized to the literal string "dyn Trait" instead of "Trait" — an unmatchable name that silently broke every downstream receiver-type lookup for trait-object dispatch. Part of the #2604 fix (root cause has a second, independent half: abstract trait methods are invisible to scope resolution until function_signature_item is captured — next commit). --- .../core/ingestion/languages/rust/interpret.ts | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/gitnexus/src/core/ingestion/languages/rust/interpret.ts b/gitnexus/src/core/ingestion/languages/rust/interpret.ts index a53a6e1c2..73ecd264e 100644 --- a/gitnexus/src/core/ingestion/languages/rust/interpret.ts +++ b/gitnexus/src/core/ingestion/languages/rust/interpret.ts @@ -2,8 +2,23 @@ import type { CaptureMatch, ParsedImport, ParsedTypeBinding, TypeRef } from 'git const REF_PREFIX_RE = /^&\s*(mut\s+)?/; const PTR_PREFIX_RE = /^\*\s*(const|mut)?\s*/; +const DYN_PREFIX_RE = /^dyn\s+/; const ENUM_VARIANT_NAMES = new Set(['Some', 'None', 'Ok', 'Err']); +// `dyn Trait`, `&dyn Trait`, `Box` all name a trait object whose +// receiver-dispatch target is the trait itself (#2604) — strip the `dyn` +// keyword and any auto-trait/lifetime bound list (`dyn Trait + Send`) down to +// the principal trait name. Reference/pointer sigils are stripped by the +// caller first; wrapper unwrapping (Box etc.) runs before this so the +// unwrapped inner text still gets the same treatment. +function stripDynBound(t: string): string { + if (!DYN_PREFIX_RE.test(t)) return t; + t = t.replace(DYN_PREFIX_RE, ''); + const plus = t.indexOf('+'); + if (plus !== -1) t = t.slice(0, plus); + return t.trim(); +} + // ─── interpretImport ────────────────────────────────────────────────────── export function interpretRustImport(captures: CaptureMatch): ParsedImport | null { @@ -98,6 +113,7 @@ export function normalizeRustTypeName(text: string): string { const inner = extractFirstGenericArg(t); if (inner !== null) t = inner; } + t = stripDynBound(t); const bracket = t.indexOf('<'); if (bracket !== -1) t = t.slice(0, bracket); // Take last segment of qualified paths (crate::foo::Bar → Bar) @@ -158,6 +174,7 @@ function normalizeRustReturnType(text: string): string { } } } + t = stripDynBound(t); const bracket = t.indexOf('<'); if (bracket !== -1) t = t.slice(0, bracket); const lastColon = t.lastIndexOf('::'); From 57db7bc166e332b6d0a981bf6a3dd8b974356612 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 14:29:37 +0000 Subject: [PATCH 47/61] fix(rust): capture abstract trait methods for scope resolution MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit fn foo(&self) -> T; (no body) parses as function_signature_item, a grammar node distinct from function_item that RUST_SCOPE_QUERY never captured. An abstract trait method therefore had no Function scope and no declaration, so populateClassOwnedMembers never wired its ownerId to the trait's Class scope — invisible to the CALLS-edge receiver-bound resolution pass even after a receiver's type resolves to the trait correctly. Together with the previous commit's dyn-stripping fix, a call through a &dyn Trait parameter now emits a CALLS edge to the trait's method (#2604). --- gitnexus/src/core/ingestion/languages/rust/query.ts | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/gitnexus/src/core/ingestion/languages/rust/query.ts b/gitnexus/src/core/ingestion/languages/rust/query.ts index a0e75f0fa..bef3f1bd7 100644 --- a/gitnexus/src/core/ingestion/languages/rust/query.ts +++ b/gitnexus/src/core/ingestion/languages/rust/query.ts @@ -10,6 +10,7 @@ const RUST_SCOPE_QUERY = ` (enum_item) @scope.class (union_item) @scope.class (function_item) @scope.function +(function_signature_item) @scope.function (closure_expression) @scope.function (block) @scope.block (if_expression) @scope.block @@ -55,6 +56,14 @@ const RUST_SCOPE_QUERY = ` (function_item name: (identifier) @declaration.name) @declaration.function +;; Declarations — trait method signature (required method, no body, +;; e.g. fn foo(self) -> T; inside a trait body). Without this, an abstract +;; trait method is invisible to scope resolution — never owned by its +;; trait's Class scope, so a dyn Trait receiver can never dispatch to +;; it (#2604). +(function_signature_item + name: (identifier) @declaration.name) @declaration.function + ;; Declarations — struct fields (field_declaration name: (field_identifier) @declaration.name From 4e97a278d14fe4a546d5fb97a846ff02b4f311fa Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 14:31:47 +0000 Subject: [PATCH 48/61] fix(mcp): report every rename edit that apply writes (#2605) rename() reported total_edits from a partial enumeration (definition line only, one-edit-per-graph-file then break, and text search that skipped any file already covered by the graph) while the apply step does a whole-file \boldName\b global replace on every touched file. When a private symbol's definition and all its call sites live in one file, only the definition line was reported (total_edits: 1) even though apply rewrote every occurrence, in both dry-run and apply. Rebuild changes/total_edits/graph_edits/text_search_edits from one file set: classify each file to rewrite (definition + graph refs = graph confidence; rg-only files = text_search, never downgrading a graph file), then enumerate every matching line per file with apply's exact escaped global regex. The reported edit list now equals what apply writes. Apply behavior is unchanged. Adds a regression test reproducing the issue's single-file Rust case (def + 3 same-file call sites, empty graph): total_edits is 4 in both dry-run and apply, and equals the replacements that land on disk. --- gitnexus/src/mcp/local/local-backend.ts | 155 +++++++----------- gitnexus/test/unit/rename-edit-report.test.ts | 137 ++++++++++++++++ 2 files changed, 200 insertions(+), 92 deletions(-) create mode 100644 gitnexus/test/unit/rename-edit-report.test.ts diff --git a/gitnexus/src/mcp/local/local-backend.ts b/gitnexus/src/mcp/local/local-backend.ts index dac37b6ef..dacd0a4cd 100644 --- a/gitnexus/src/mcp/local/local-backend.ts +++ b/gitnexus/src/mcp/local/local-backend.ts @@ -4361,44 +4361,28 @@ export class LocalBackend { return { error: 'New name is the same as the current name.' }; } - // Step 2: Collect edits from graph (high confidence) - const changes = new Map(); - - const addEdit = ( - filePath: string, - line: number, - oldText: string, - newText: string, - confidence: string, - ) => { - if (!changes.has(filePath)) { - changes.set(filePath, { file_path: filePath, edits: [] }); - } - changes.get(filePath)!.edits.push({ line, old_text: oldText, new_text: newText, confidence }); + // Steps 2+3: Determine the set of files the apply step will rewrite, then + // enumerate every occurrence in each. The apply step (Step 4) does a + // whole-file `\boldName\b` global replace on every file in `changes`, so the + // reported edit list MUST enumerate every matching line in every such file — + // otherwise the preview under-reports what lands, and the same partial list + // comes back after apply (#2605). Building `changes` from one file set is the + // single source of truth that keeps the report and the apply provably in sync. + type RenameEdit = { + line: number; + old_text: string; + new_text: string; + confidence: 'graph' | 'text_search'; }; + const escapedOldName = oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); - // The definition itself - if (sym.filePath && sym.startLine) { - try { - const content = await fs.readFile(assertSafePath(sym.filePath), 'utf-8'); - const lines = content.split('\n'); - const lineIdx = sym.startLine - 1; - if (lineIdx >= 0 && lineIdx < lines.length && lines[lineIdx].includes(oldName)) { - const defRegex = new RegExp( - `\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, - 'g', - ); - addEdit( - sym.filePath, - sym.startLine, - lines[lineIdx].trim(), - lines[lineIdx].replace(defRegex, new_name).trim(), - 'graph', - ); - } - } catch (e) { - logQueryError('rename:read-definition', e); - } + // Classify each file to rewrite by how it was discovered. Definition and + // graph-ref files carry graph confidence; files found only by text search + // carry text_search confidence. A graph-classified file is never downgraded. + const fileConfidence = new Map(); + + if (sym.filePath) { + fileConfidence.set(sym.filePath, 'graph'); } // All incoming refs from graph (callers, importers, etc.) @@ -4408,44 +4392,13 @@ export class LocalBackend { ...(lookupResult.incoming.extends || []), ...(lookupResult.incoming.implements || []), ]; - - let graphEdits = changes.size > 0 ? 1 : 0; // count definition edit - for (const ref of allIncoming) { - if (!ref.filePath) continue; - try { - const content = await fs.readFile(assertSafePath(ref.filePath), 'utf-8'); - const lines = content.split('\n'); - for (let i = 0; i < lines.length; i++) { - if (lines[i].includes(oldName)) { - addEdit( - ref.filePath, - i + 1, - lines[i].trim(), - lines[i] - .replace( - new RegExp(`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'g'), - new_name, - ) - .trim(), - 'graph', - ); - graphEdits++; - break; // one edit per file from graph refs - } - } - } catch (e) { - logQueryError('rename:read-ref', e); + if (ref.filePath) { + fileConfidence.set(ref.filePath, 'graph'); } } - // Step 3: Text search for refs the graph might have missed - let astSearchEdits = 0; - const graphFiles = new Set( - [sym.filePath, ...allIncoming.map((r) => r.filePath)].filter(Boolean), - ); - - // Simple text search across the repo for the old name (in files not already covered by graph) + // Text search for files the graph might have missed entirely. try { const { execFileSync } = await import('child_process'); const rgArgs = [ @@ -4472,34 +4425,52 @@ export class LocalBackend { for (const file of files) { const normalizedFile = file.replace(/\\/g, '/').replace(/^\.\//, ''); - if (graphFiles.has(normalizedFile)) continue; // already covered by graph - - try { - const content = await fs.readFile(assertSafePath(normalizedFile), 'utf-8'); - const lines = content.split('\n'); - const regex = new RegExp(`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'g'); - for (let i = 0; i < lines.length; i++) { - regex.lastIndex = 0; - if (regex.test(lines[i])) { - regex.lastIndex = 0; - addEdit( - normalizedFile, - i + 1, - lines[i].trim(), - lines[i].replace(regex, new_name).trim(), - 'text_search', - ); - astSearchEdits++; - } - } - } catch (e) { - logQueryError('rename:text-search-read', e); + // Never downgrade a graph-classified file to text_search. + if (!fileConfidence.has(normalizedFile)) { + fileConfidence.set(normalizedFile, 'text_search'); } } } catch (e) { logQueryError('rename:ripgrep', e); } + // Enumerate every `\boldName\b` line in every file to rewrite, using the same + // escaped global regex the apply step uses. A file with no matching line is + // dropped (apply would write nothing to it). Because apply's per-file global + // replace rewrites exactly these lines, the reported list equals what lands. + const changes = new Map(); + let graphEdits = 0; + let astSearchEdits = 0; + + for (const [filePath, confidence] of fileConfidence) { + try { + const content = await fs.readFile(assertSafePath(filePath), 'utf-8'); + const lines = content.split('\n'); + const edits: RenameEdit[] = []; + for (let i = 0; i < lines.length; i++) { + if (!new RegExp(`\\b${escapedOldName}\\b`).test(lines[i])) { + continue; + } + edits.push({ + line: i + 1, + old_text: lines[i].trim(), + new_text: lines[i].replace(new RegExp(`\\b${escapedOldName}\\b`, 'g'), new_name).trim(), + confidence, + }); + if (confidence === 'graph') { + graphEdits++; + } else { + astSearchEdits++; + } + } + if (edits.length > 0) { + changes.set(filePath, { file_path: filePath, edits }); + } + } catch (e) { + logQueryError('rename:enumerate', e); + } + } + // Step 4: Apply or preview const allChanges = Array.from(changes.values()); const totalEdits = allChanges.reduce((sum, c) => sum + c.edits.length, 0); diff --git a/gitnexus/test/unit/rename-edit-report.test.ts b/gitnexus/test/unit/rename-edit-report.test.ts new file mode 100644 index 000000000..ba651abf4 --- /dev/null +++ b/gitnexus/test/unit/rename-edit-report.test.ts @@ -0,0 +1,137 @@ +/** + * Regression test for issue #2605: `rename` must report every edit it applies. + * + * The apply step does a whole-file `\boldName\b` global replace on each touched + * file, but the reported `changes`/`total_edits` were built from a partial + * enumeration that (a) recorded only the definition line, (b) recorded one edit + * per graph-ref file then broke, and (c) skipped text-search on any file already + * covered by the graph. When a private symbol's definition and all its call + * sites live in one file, only the definition line was reported (total_edits: 1) + * while apply rewrote every occurrence. This test drives the exact single-file + * case with an empty graph (no incoming refs) and asserts report == apply. + */ +import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; +import { promises as fs } from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; + +// Prevent onnxruntime / native search adapters from loading at import time +// (mirrors test/unit/calltool-dispatch.test.ts). We drive the private rename() +// directly, so the graph/DB/embedding layers are never exercised. +vi.mock('../../src/core/search/bm25-index.js', () => ({ + searchFTSFromLbug: vi.fn().mockResolvedValue({ results: [], ftsAvailable: true }), +})); +vi.mock('../../src/mcp/core/embedder.js', () => ({ + embedQuery: vi.fn().mockResolvedValue([]), + getEmbeddingDims: vi.fn().mockReturnValue(384), +})); + +import { LocalBackend } from '../../src/mcp/local/local-backend.js'; + +// The #2605 repro: a private free fn with exactly 4 textual occurrences of +// `rename_target` — the definition, one production call, two test calls — all +// in the same file. +const RUST_SRC = `fn rename_target(x: u32) -> u32 { + x + 1 +} + +pub fn prod_call() -> u32 { + rename_target(1) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn unit_one() { + assert_eq!(rename_target(1), 2); + } + + #[test] + fn unit_two() { + assert_eq!(rename_target(2), 3); + } +} +`; + +// 1-based occurrence lines of `rename_target` in RUST_SRC — the ground truth the +// report must match. Computed from the fixture (not hardcoded) so editing the +// snippet cannot silently desync the expectation. +const OCCURRENCE_LINES = RUST_SRC.split('\n') + .map((line, i) => (/\brename_target\b/.test(line) ? i + 1 : 0)) + .filter((n) => n > 0); + +/** Build a backend whose graph lookup returns the symbol with NO incoming refs + * (the exact condition that made the old code report only the definition). */ +function stubbedBackend(): LocalBackend { + const backend = new LocalBackend(); + vi.spyOn(backend as unknown as { ensureInitialized: () => Promise }, 'ensureInitialized').mockResolvedValue( + undefined, + ); + vi.spyOn(backend as unknown as { context: () => Promise }, 'context').mockResolvedValue({ + status: 'success', + symbol: { name: 'rename_target', filePath: 'src/lib.rs', startLine: OCCURRENCE_LINES[0] }, + incoming: { calls: [], imports: [], extends: [], implements: [] }, + }); + return backend; +} + +describe('rename edit report is faithful to apply (#2605)', () => { + let backend: LocalBackend; + let tmpDir: string; + + beforeEach(async () => { + tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'gn-2605-')); + await fs.mkdir(path.join(tmpDir, 'src')); + await fs.writeFile(path.join(tmpDir, 'src', 'lib.rs'), RUST_SRC, 'utf-8'); + backend = stubbedBackend(); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + await fs.rm(tmpDir, { recursive: true, force: true }); + }); + + it('previews every occurrence that apply will rewrite (dry_run)', async () => { + const result = await ( + backend as unknown as { rename: (r: unknown, p: unknown) => Promise } + ).rename({ repoPath: tmpDir }, { symbol_name: 'rename_target', new_name: 'renamed_fn', dry_run: true }); + + expect(result.applied).toBe(false); + expect(result.files_affected).toBe(1); + expect(result.total_edits).toBe(OCCURRENCE_LINES.length); // 4, not 1 + expect(result.graph_edits + result.text_search_edits).toBe(result.total_edits); + + const reportedLines = result.changes[0].edits + .map((e: { line: number }) => e.line) + .sort((a: number, b: number) => a - b); + expect(reportedLines).toEqual(OCCURRENCE_LINES); + + // A dry run leaves the file untouched. + const onDisk = await fs.readFile(path.join(tmpDir, 'src', 'lib.rs'), 'utf-8'); + expect(onDisk).toContain('rename_target'); + }); + + it('reports exactly what it wrote (apply)', async () => { + const result = await ( + backend as unknown as { rename: (r: unknown, p: unknown) => Promise } + ).rename({ repoPath: tmpDir }, { symbol_name: 'rename_target', new_name: 'renamed_fn', dry_run: false }); + + expect(result.applied).toBe(true); + expect(result.total_edits).toBe(OCCURRENCE_LINES.length); + + const onDisk = await fs.readFile(path.join(tmpDir, 'src', 'lib.rs'), 'utf-8'); + const renamedCount = (onDisk.match(/\brenamed_fn\b/g) || []).length; + const stragglers = (onDisk.match(/\brename_target\b/g) || []).length; + expect(renamedCount).toBe(OCCURRENCE_LINES.length); // all 4 rewritten + expect(stragglers).toBe(0); + + // The reported edit count equals the number of replacements that landed. + const reportedEdits = result.changes.reduce( + (n: number, c: { edits: unknown[] }) => n + c.edits.length, + 0, + ); + expect(reportedEdits).toBe(renamedCount); + }); +}); From 5b906c318920b8cb5ce0cb599cdb4c8fb2ac3c02 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 21 Jul 2026 15:39:56 +0100 Subject: [PATCH 49/61] fix(eval): bind the resolved node binary into the sandbox, not a hardcoded host path (#2607) Surfaced by the first real workflow_dispatch run on the self-hosted runner (https://github.com/abhigyanpatwari/GitNexus/actions/runs/29836411744): every session failed with error_kind infra-error, error_detail "bwrap: execvp /usr/local/bin/node: No such file or directory", tripping the outage-streak breaker after 5 consecutive failures. sanitized_graph.py and runner_sessions.py invoke the sandboxed graph CLI at the fixed path /usr/local/bin/node. _runtime_mount_args only binds /usr, /bin, /lib, /lib64 wholesale, so that path resolves correctly when node happens to live under /usr/local/bin on the host -- true on GitHub-hosted runner images, but not on a self-hosted runner, where actions/setup-node installs into its own tool-cache directory instead (outside all four bound trees, so invisible to the sandbox regardless of what PATH says on the host). Fix lives entirely in the mount construction: resolve `node` via shutil.which (correctly picks up wherever actions/setup-node put it, since its tool-cache dir is already on PATH by the time this runs) and bind it read-only to the same fixed sandbox path the two call sites already expect. Neither call site needed to change. Backward compatible with GitHub-hosted runners, where this resolves to the same path and binds a harmless no-op self-mount. Co-authored-by: Claude Sonnet 5 --- eval/tests/test_proposer_sandbox.py | 26 ++++++++++++++++++++++++- eval/workflow_bench/proposer_sandbox.py | 13 ++++++++++++- 2 files changed, 37 insertions(+), 2 deletions(-) diff --git a/eval/tests/test_proposer_sandbox.py b/eval/tests/test_proposer_sandbox.py index 70f4fe76b..22b4ce9b0 100644 --- a/eval/tests/test_proposer_sandbox.py +++ b/eval/tests/test_proposer_sandbox.py @@ -24,6 +24,7 @@ from workflow_bench.proposer_sandbox import ( SANDBOX_USER_SKILLS, ReadOnlyMount, SandboxError, + _runtime_mount_args, build_claude_settings, build_sandbox_environment, prepare_sandbox, @@ -191,6 +192,30 @@ def test_sandbox_command_has_minimal_mounts_and_no_host_root_bind(tmp_path: Path assert not private_root.exists() +def test_runtime_mounts_bind_the_resolved_node_wherever_path_puts_it(monkeypatch) -> None: + # sanitized_graph.py and runner_sessions.py invoke the sandboxed graph CLI + # at the fixed path /usr/local/bin/node. That's only covered by the /usr + # bind when node happens to live under /usr/local/bin on the host -- true + # on GitHub-hosted runner images, but not on a self-hosted runner where + # actions/setup-node installs into its own tool-cache directory instead + # (observed empirically: "bwrap: execvp /usr/local/bin/node: No such file + # or directory" on a fresh self-hosted runner). + monkeypatch.setattr( + "workflow_bench.proposer_sandbox.shutil.which", + lambda name: "/opt/hostedtoolcache/node/22.18.0/x64/bin/node" if name == "node" else None, + ) + args = _runtime_mount_args() + node_index = args.index("/opt/hostedtoolcache/node/22.18.0/x64/bin/node") + assert args[node_index - 1] == "--ro-bind" + assert args[node_index + 1] == "/usr/local/bin/node" + + +def test_runtime_mounts_skip_the_node_bind_when_node_is_unresolvable(monkeypatch) -> None: + monkeypatch.setattr("workflow_bench.proposer_sandbox.shutil.which", lambda name: None) + args = _runtime_mount_args() + assert "/usr/local/bin/node" not in args + + def test_stricter_prefix_freezes_evaluated_skills_and_can_unshare_network(tmp_path: Path) -> None: clone = tmp_path / "clone" skill = clone / ".claude" / "skills" / "gitnexus-work" @@ -772,4 +797,3 @@ for line in sys.stdin: assert bash_result.get("is_error") is not True, bash_result assert (clone / "bash-called").read_text() == "canary" assert (clone / "mcp-called").read_text() == "ok" - diff --git a/eval/workflow_bench/proposer_sandbox.py b/eval/workflow_bench/proposer_sandbox.py index 731f11ad2..7910bd87e 100644 --- a/eval/workflow_bench/proposer_sandbox.py +++ b/eval/workflow_bench/proposer_sandbox.py @@ -354,6 +354,17 @@ def _runtime_mount_args() -> list[str]: path = Path(raw) if path.exists(): args += ["--ro-bind", raw, raw] + # sanitized_graph.py and runner_sessions.py invoke the sandboxed graph + # CLI at the fixed path /usr/local/bin/node. That's only covered by the + # /usr bind above when node happens to live under /usr/local/bin on the + # host -- true on GitHub-hosted runner images, but not on a self-hosted + # runner where actions/setup-node installs into its own tool-cache + # directory instead. Bind whatever `node` actually resolves to on PATH + # to that same fixed sandbox path so both call sites keep working + # regardless of where the host actually put it. + node_bin = shutil.which("node") + if node_bin: + args += ["--ro-bind", node_bin, "/usr/local/bin/node"] for raw in ( "/etc/ssl", "/etc/hosts", @@ -398,7 +409,7 @@ def _create_python3_wrapper(private_root: Path) -> Path: """ wrapper = private_root / "python3" - wrapper.write_text("#!/bin/bash\nset -eu\nexec /usr/bin/python3 \"$@\"\n") + wrapper.write_text('#!/bin/bash\nset -eu\nexec /usr/bin/python3 "$@"\n') wrapper.chmod(0o500) return wrapper From 902186c4f89be7ee5cfa846ecab7be0ede2bb7bb Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 14:41:19 +0000 Subject: [PATCH 50/61] style: prettier-format rename-edit-report test (#2605) --- gitnexus/test/unit/rename-edit-report.test.ts | 17 ++++++++++++----- 1 file changed, 12 insertions(+), 5 deletions(-) diff --git a/gitnexus/test/unit/rename-edit-report.test.ts b/gitnexus/test/unit/rename-edit-report.test.ts index ba651abf4..0d0993e96 100644 --- a/gitnexus/test/unit/rename-edit-report.test.ts +++ b/gitnexus/test/unit/rename-edit-report.test.ts @@ -66,9 +66,10 @@ const OCCURRENCE_LINES = RUST_SRC.split('\n') * (the exact condition that made the old code report only the definition). */ function stubbedBackend(): LocalBackend { const backend = new LocalBackend(); - vi.spyOn(backend as unknown as { ensureInitialized: () => Promise }, 'ensureInitialized').mockResolvedValue( - undefined, - ); + vi.spyOn( + backend as unknown as { ensureInitialized: () => Promise }, + 'ensureInitialized', + ).mockResolvedValue(undefined); vi.spyOn(backend as unknown as { context: () => Promise }, 'context').mockResolvedValue({ status: 'success', symbol: { name: 'rename_target', filePath: 'src/lib.rs', startLine: OCCURRENCE_LINES[0] }, @@ -96,7 +97,10 @@ describe('rename edit report is faithful to apply (#2605)', () => { it('previews every occurrence that apply will rewrite (dry_run)', async () => { const result = await ( backend as unknown as { rename: (r: unknown, p: unknown) => Promise } - ).rename({ repoPath: tmpDir }, { symbol_name: 'rename_target', new_name: 'renamed_fn', dry_run: true }); + ).rename( + { repoPath: tmpDir }, + { symbol_name: 'rename_target', new_name: 'renamed_fn', dry_run: true }, + ); expect(result.applied).toBe(false); expect(result.files_affected).toBe(1); @@ -116,7 +120,10 @@ describe('rename edit report is faithful to apply (#2605)', () => { it('reports exactly what it wrote (apply)', async () => { const result = await ( backend as unknown as { rename: (r: unknown, p: unknown) => Promise } - ).rename({ repoPath: tmpDir }, { symbol_name: 'rename_target', new_name: 'renamed_fn', dry_run: false }); + ).rename( + { repoPath: tmpDir }, + { symbol_name: 'rename_target', new_name: 'renamed_fn', dry_run: false }, + ); expect(result.applied).toBe(true); expect(result.total_edits).toBe(OCCURRENCE_LINES.length); From 052319c9ccd80a6a85b9bbe4b9779e696feb9abc Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 14:48:54 +0000 Subject: [PATCH 51/61] test(rust): add regression coverage for trait-object dispatch (#2604) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit New minimal fixture (single trait + impl + &dyn Trait call site, no other same-named callers) proves the dyn-dispatch CALLS edge discriminates: fails against the pre-fix source (0 edges) and passes against the two preceding commits' fix (exactly 1 edge, verified via the CLI analyze pipeline against a standalone repo). The existing rust-abstract-dispatch fixture was NOT extended for this, deliberately: it already has other callers referencing the same method names (process()'s repo.find()/save()/count()), and an existing resolution fallback picks those up via simple-name matching regardless of receiver type — masking this specific defect in the in-process test-pipeline path. A dedicated, single-caller fixture keeps the regression test load-bearing. --- .../rust-dyn-trait-object/src/lib.rs | 15 ++++++++++++ .../test/integration/resolvers/rust.test.ts | 23 +++++++++++++++++++ 2 files changed, 38 insertions(+) create mode 100644 gitnexus/test/fixtures/lang-resolution/rust-dyn-trait-object/src/lib.rs diff --git a/gitnexus/test/fixtures/lang-resolution/rust-dyn-trait-object/src/lib.rs b/gitnexus/test/fixtures/lang-resolution/rust-dyn-trait-object/src/lib.rs new file mode 100644 index 000000000..d25dad83d --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/rust-dyn-trait-object/src/lib.rs @@ -0,0 +1,15 @@ +pub trait Behaviour { + fn trait_target(&self) -> u32; +} + +pub struct Impl1; + +impl Behaviour for Impl1 { + fn trait_target(&self) -> u32 { + 7 + } +} + +pub fn calls_via_dyn(b: &dyn Behaviour) -> u32 { + b.trait_target() +} diff --git a/gitnexus/test/integration/resolvers/rust.test.ts b/gitnexus/test/integration/resolvers/rust.test.ts index 9cd4f94e8..87aa6899b 100644 --- a/gitnexus/test/integration/resolvers/rust.test.ts +++ b/gitnexus/test/integration/resolvers/rust.test.ts @@ -1951,6 +1951,29 @@ describe('Rust abstract dispatch (Repository trait)', () => { }); }); +// --------------------------------------------------------------------------- +// #2604: trait-object (&dyn Trait) receiver dispatch +// --------------------------------------------------------------------------- + +describe('Rust dyn trait-object dispatch (#2604)', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(path.join(FIXTURES, 'rust-dyn-trait-object'), () => {}); + }, 60000); + + it('detects Impl1 struct and Behaviour trait', () => { + expect(getNodesByLabel(result, 'Struct')).toContain('Impl1'); + expect(getNodesByLabel(result, 'Trait')).toContain('Behaviour'); + }); + + it('emits exactly one CALLS edge from calls_via_dyn(b: &dyn Behaviour) to trait_target', () => { + const calls = getRelationships(result, 'CALLS'); + const dynCalls = calls.filter((c) => c.source === 'calls_via_dyn' && c.target === 'trait_target'); + expect(dynCalls.length).toBe(1); + }); +}); + // --------------------------------------------------------------------------- // SM-11: Rust Child extends Parent — qualified-syntax MRO // From 881c6bccc7b082eb4b44ceff0b6d30ca4a1eaa4e Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 14:49:45 +0000 Subject: [PATCH 52/61] test(rust): regenerate captures golden snapshot for function_signature_item Expected drift from the query.ts change: abstract trait methods now emit a scope + declaration capture, shifting captureGroups/digest for every rust-* fixture containing a trait with a required (bodyless) method. --- .../expected-captures.json | 32 +++++++++++-------- 1 file changed, 18 insertions(+), 14 deletions(-) diff --git a/gitnexus/test/fixtures/rust-captures-golden/expected-captures.json b/gitnexus/test/fixtures/rust-captures-golden/expected-captures.json index d977d99bf..f632f78aa 100644 --- a/gitnexus/test/fixtures/rust-captures-golden/expected-captures.json +++ b/gitnexus/test/fixtures/rust-captures-golden/expected-captures.json @@ -1,7 +1,7 @@ { "rust-abstract-dispatch/src/lib.rs": { - "captureGroups": 30, - "digest": "88309004d1ab00054f81bc55c1d058fc4ca25781162d6b94b7e7ce631a5d61b2" + "captureGroups": 34, + "digest": "973679363065ecd54c4e5128a9fab214ea27eca24f0f079c63a3c6f285e678b0" }, "rust-abstract-dispatch/src/main.rs": { "captureGroups": 21, @@ -148,8 +148,8 @@ "digest": "e0120e3f215282e68d83b4f8f5d8918945e0b3e7ce4e0128c6afd2aa43caa1c0" }, "rust-cross-module-collision/src/traits.rs": { - "captureGroups": 3, - "digest": "88eef9d92ea6e370bd8ef7fbf64c42ec622ec933fb53c56b32bc67db87fa8e03" + "captureGroups": 5, + "digest": "c7150a5052e0e2b5fd7fc21cc8ce361530fe5cc60f67ad0e2ef945bb97965b53" }, "rust-deep-field-chain/models.rs": { "captureGroups": 24, @@ -171,6 +171,10 @@ "captureGroups": 22, "digest": "c53db401a81fde2ffd5665393acb9cd605a62ec51c015c3aafb3f41c0897471f" }, + "rust-dyn-trait-object/src/lib.rs": { + "captureGroups": 23, + "digest": "720618dff6a43ab8e5b59aa354c0c448b9057dd6f2f7b3b22b13b82d53745943" + }, "rust-err-unwrap/src/error.rs": { "captureGroups": 9, "digest": "798c8e01c6e54792ba69e845248efc8abf0cba38fa3d16fb8e0d1f6dd2ad2b7e" @@ -316,8 +320,8 @@ "digest": "cd836a2a9c15ab240961d2e15f192f7e33d65eb5ebf2e1a8af2f620a47fe66ae" }, "rust-method-enrichment/src/lib.rs": { - "captureGroups": 40, - "digest": "71627a8218e32514b6945e4e310686eb631ce37451c90c9644e3f5336a37820b" + "captureGroups": 42, + "digest": "a4d9ca570fbb1ff1859a0b4f737aa3507b236f99700d37ded2c8c36518add567" }, "rust-method-enrichment/src/main.rs": { "captureGroups": 18, @@ -360,8 +364,8 @@ "digest": "141388068614e16d96f27cfdf18ac9001b9e202ce832fe10f38dab990637b3ab" }, "rust-parent-resolution/src/serializable.rs": { - "captureGroups": 3, - "digest": "f35d44f44d81e3a0be40f68ba9dbd4bde6f01659fa15b6db34a458ad460f904e" + "captureGroups": 5, + "digest": "f33bb881dd937cdd5af2eca6b0284ea79ca2d296c79217513f882dcba8a82fd8" }, "rust-parent-resolution/src/user.rs": { "captureGroups": 13, @@ -372,8 +376,8 @@ "digest": "bc8946d31db81b85d780633608fdaa7565258cd788285fa00cd6dcb0de3dd16c" }, "rust-qualified-trait/src/traits.rs": { - "captureGroups": 5, - "digest": "15be069f28f1400e4beb0b0860acb59979f78549960486f36a92f56578f05a06" + "captureGroups": 9, + "digest": "10f3bba4c2a16cdac77de0498ff506910ac09cc1a85e7daebfc5743c54e26015" }, "rust-qualified-trait/src/widget.rs": { "captureGroups": 23, @@ -492,12 +496,12 @@ "digest": "f1b9f72d74467be55a8b7679215b49bcabb4d0fced6080f752672070b32ed93d" }, "rust-traits/src/traits/clickable.rs": { - "captureGroups": 3, - "digest": "3ed5b27c172d48f83929715ba92d1030a282f9f1e29ec2fcdd3d7e9efbc54a84" + "captureGroups": 7, + "digest": "83c36832f24446fe03a07288dd394bc7494a54e5979c9b71c1dcb8b19ab54341" }, "rust-traits/src/traits/drawable.rs": { - "captureGroups": 5, - "digest": "1dca39bbc7c1b1b66f1a34730b9a5b4dba04c54ee9d2688255e0fd4e6bc48499" + "captureGroups": 9, + "digest": "cee5091f041038722f1f012394a75ba4e16870d05b2dafa37371200e785198f0" }, "rust-union/lib.rs": { "captureGroups": 10, From 67d55d7e59d552d50625b6122c630795776b3679 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 15:08:06 +0000 Subject: [PATCH 53/61] fix(storage): bump INCREMENTAL_SCHEMA_VERSION for Rust dyn-dispatch fix MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit RUST_SCOPE_QUERY gained a function_signature_item capture (previous commit) so abstract trait methods can now dispatch a CALLS edge through a &dyn Trait receiver. The incremental write set only covers changed files, so a top-up against a pre-v11 index would keep silently missing these edges for every unchanged Rust trait file — same contract as v7/v10; force a full re-analyze instead. --- gitnexus/src/storage/repo-manager.ts | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/gitnexus/src/storage/repo-manager.ts b/gitnexus/src/storage/repo-manager.ts index e34432a7a..9852e8ba4 100644 --- a/gitnexus/src/storage/repo-manager.ts +++ b/gitnexus/src/storage/repo-manager.ts @@ -437,8 +437,15 @@ export interface RepoMeta { * `Record` node and its `HAS_METHOD` edges for every unchanged record file * (same v7 contract: new nodes/edges the incremental path would otherwise * never backfill); force a full re-analyze instead. + * v11: Rust abstract trait methods (`fn foo(&self) -> T;`, no body) now get a + * scope + declaration capture (#2604): RUST_SCOPE_QUERY had no + * `function_signature_item` pattern, so a `&dyn Trait` receiver could never + * dispatch a CALLS edge to the trait's own method. Same v7/v10 contract: the + * incremental write set only covers changed files, so a top-up against a + * pre-v11 index would keep silently missing these CALLS edges for every + * unchanged Rust trait file; force a full re-analyze instead. */ -export const INCREMENTAL_SCHEMA_VERSION = 10; +export const INCREMENTAL_SCHEMA_VERSION = 11; export interface IndexedRepo { repoPath: string; From bba25b2103f70fef53d0dc16a90e7479ca75046b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 21 Jul 2026 16:09:45 +0100 Subject: [PATCH 54/61] fix(eval): bind resolved node to a fresh sandbox path (corrects #2607) (#2609) * fix(eval): bind the resolved node to a fresh sandbox path, not one under /usr #2607 bound the resolved `node` to /usr/local/bin/node, but that path lives inside the /usr tree that _runtime_mount_args already read-only-binds wholesale. The second real workflow_dispatch run on the self-hosted runner (https://github.com/abhigyanpatwari/GitNexus/actions/runs/29840270554) failed immediately in the bubblewrap preflight: "bwrap: Can't create file at /usr/local/bin/node: Read-only file system" -- bwrap can't create a new mount-point file inside a tree it already bound read-only when the real path doesn't already exist there on the host, which is exactly the self-hosted case this bind exists to fix. Introduces SANDBOX_NODE (/opt/claude/node), a fresh path outside every tree _runtime_mount_args binds, following the same pattern SANDBOX_CLAUDE and SANDBOX_PYTHON3 already use. Updates the two real call sites (sanitized_graph.py, runner_sessions.py) to use the constant instead of the hardcoded literal, so the fix can't drift out of sync with itself again, and re-exports it from runner.py alongside the other SANDBOX_* names for the real-bwrap tests that reference it directly. Adds a real-bwrap test (gated behind GITNEXUS_REQUIRE_BWRAP_CANARY, same as the existing ones) that copies a real node binary to a path outside every bound tree and actually launches bwrap against it -- an argv-construction test alone can't catch a bwrap-level "Read-only file system" error, only a real invocation can, and that's exactly the gap that let #2607's version of this fix through review looking correct. Co-Authored-By: Claude Sonnet 5 * fix(eval): don't let the new real-bwrap test's node-mock break bwrap's own resolution CI caught this immediately: the new test_real_bubblewrap_runs_node_from_outside_the_bound_trees monkeypatched shutil.which to return None for anything but "node", but prepare_sandbox's own bwrap/claude resolution (_resolve_executable) goes through shutil.which too -- so the test broke bwrap discovery before the sandbox it's supposed to exercise could even be built ("SandboxError: required executable is unavailable: bwrap"). Delegate to the real shutil.which for every other name instead of blanket-returning None. Co-Authored-By: Claude Sonnet 5 --------- Co-authored-by: Claude Sonnet 5 --- eval/tests/test_proposer_sandbox.py | 63 ++++++++++++++++++---- eval/tests/test_workflow_bench_sessions.py | 6 +-- eval/workflow_bench/proposer_sandbox.py | 20 ++++--- eval/workflow_bench/runner.py | 1 + eval/workflow_bench/runner_sessions.py | 5 +- eval/workflow_bench/sanitized_graph.py | 3 +- 6 files changed, 76 insertions(+), 22 deletions(-) diff --git a/eval/tests/test_proposer_sandbox.py b/eval/tests/test_proposer_sandbox.py index 22b4ce9b0..2c4c37259 100644 --- a/eval/tests/test_proposer_sandbox.py +++ b/eval/tests/test_proposer_sandbox.py @@ -4,6 +4,7 @@ from __future__ import annotations import json import os +import shutil import stat import subprocess import sys @@ -19,6 +20,7 @@ from workflow_bench.process_control import ManagedProcessResult, run_managed from workflow_bench.proposer_sandbox import ( MAX_BUNDLE_BYTES, MAX_EVIDENCE_FILE_BYTES, + SANDBOX_NODE, SANDBOX_PYTHON3, SANDBOX_SHELL_PREFIX, SANDBOX_USER_SKILLS, @@ -192,14 +194,19 @@ def test_sandbox_command_has_minimal_mounts_and_no_host_root_bind(tmp_path: Path assert not private_root.exists() -def test_runtime_mounts_bind_the_resolved_node_wherever_path_puts_it(monkeypatch) -> None: +def test_runtime_mounts_bind_the_resolved_node_to_a_fresh_sandbox_path(monkeypatch) -> None: # sanitized_graph.py and runner_sessions.py invoke the sandboxed graph CLI - # at the fixed path /usr/local/bin/node. That's only covered by the /usr - # bind when node happens to live under /usr/local/bin on the host -- true - # on GitHub-hosted runner images, but not on a self-hosted runner where - # actions/setup-node installs into its own tool-cache directory instead - # (observed empirically: "bwrap: execvp /usr/local/bin/node: No such file - # or directory" on a fresh self-hosted runner). + # via SANDBOX_NODE. node's real host location varies (GitHub-hosted + # runner images happen to have one under /usr/local/bin; a self-hosted + # runner's actions/setup-node installs into its own tool-cache directory + # instead), so this must bind to a FRESH sandbox path like /opt/claude/... + # rather than anywhere under /usr, /bin, /lib, or /lib64: those are + # already read-only bound by this same function, and bwrap can't create + # a new mount-point file inside an already-read-only tree when the real + # path doesn't already exist there on the host (observed empirically: + # "bwrap: Can't create file at /usr/local/bin/node: Read-only file + # system" when this bind first targeted that path on a self-hosted + # runner where node isn't really there). monkeypatch.setattr( "workflow_bench.proposer_sandbox.shutil.which", lambda name: "/opt/hostedtoolcache/node/22.18.0/x64/bin/node" if name == "node" else None, @@ -207,13 +214,14 @@ def test_runtime_mounts_bind_the_resolved_node_wherever_path_puts_it(monkeypatch args = _runtime_mount_args() node_index = args.index("/opt/hostedtoolcache/node/22.18.0/x64/bin/node") assert args[node_index - 1] == "--ro-bind" - assert args[node_index + 1] == "/usr/local/bin/node" + assert args[node_index + 1] == SANDBOX_NODE + assert not any(SANDBOX_NODE.startswith(bound + "/") for bound in ("/usr", "/bin", "/lib", "/lib64")) def test_runtime_mounts_skip_the_node_bind_when_node_is_unresolvable(monkeypatch) -> None: monkeypatch.setattr("workflow_bench.proposer_sandbox.shutil.which", lambda name: None) args = _runtime_mount_args() - assert "/usr/local/bin/node" not in args + assert SANDBOX_NODE not in args def test_stricter_prefix_freezes_evaluated_skills_and_can_unshare_network(tmp_path: Path) -> None: @@ -244,6 +252,43 @@ def test_stricter_prefix_freezes_evaluated_skills_and_can_unshare_network(tmp_pa assert prefix[user_index - 2] == "--ro-bind" +@pytest.mark.skipif( + os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1", + reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job", +) +def test_real_bubblewrap_runs_node_from_outside_the_bound_trees(tmp_path: Path, monkeypatch) -> None: + # Reproduces the self-hosted-runner failure directly: node resolved from + # a path outside /usr, /bin, /lib, /lib64 (actions/setup-node's own + # tool-cache convention) must still be reachable inside the sandbox at + # SANDBOX_NODE. A real node copied to a fresh, non-system location stands + # in for the tool-cache install; argv-construction tests alone can't + # catch a bwrap-level "Can't create file ...: Read-only file system" + # (the actual error this fix resolves), only a real bwrap invocation can. + real_node = shutil.which("node") + if not real_node: + pytest.skip("no node on PATH to relocate for this canary") + toolcache = tmp_path / "toolcache" + toolcache.mkdir() + relocated_node = toolcache / "node" + shutil.copy2(real_node, relocated_node) + relocated_node.chmod(0o755) + # Only fake "node"'s resolution -- prepare_sandbox's own bwrap/claude + # lookups (_resolve_executable) also go through shutil.which, and must + # keep resolving for real or preflight fails before the sandbox is even + # built. + real_which = shutil.which + monkeypatch.setattr( + "workflow_bench.proposer_sandbox.shutil.which", + lambda name: str(relocated_node) if name == "node" else real_which(name), + ) + + clone = tmp_path / "clone" + clone.mkdir() + with prepare_sandbox(clone=clone, claude_bin=Path(sys.executable), preflight=True) as sandbox: + result = sandbox.run([SANDBOX_NODE, "--version"], timeout=10) + assert result.ok, result.stderr_tail + + @pytest.mark.skipif( os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1", reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job", diff --git a/eval/tests/test_workflow_bench_sessions.py b/eval/tests/test_workflow_bench_sessions.py index e03104ed3..c10afa401 100644 --- a/eval/tests/test_workflow_bench_sessions.py +++ b/eval/tests/test_workflow_bench_sessions.py @@ -438,11 +438,11 @@ def test_real_bubblewrap_runtime_mount_imports_cli_without_exposing_checkout(tmp preflight=True, ) as sandbox: visibility = sandbox.run( - ["/usr/local/bin/node", "-e", visibility_script], + [runner.SANDBOX_NODE, "-e", visibility_script], timeout=10, ) imported = sandbox.run( - ["/usr/local/bin/node", runner.SANDBOX_GITNEXUS_ENTRYPOINT, "--version"], + [runner.SANDBOX_NODE, runner.SANDBOX_GITNEXUS_ENTRYPOINT, "--version"], timeout=10, ) # --version never reaches the `analyze` command, which is loaded via a @@ -451,7 +451,7 @@ def test_real_bubblewrap_runtime_mount_imports_cli_without_exposing_checkout(tmp # resolve-analyze-cmd.cjs. Require the compiled analyze module # directly so this canary actually exercises that chain. analyze_imported = sandbox.run( - ["/usr/local/bin/node", "-e", f"require('{runner.SANDBOX_GITNEXUS}/dist/cli/analyze.js')"], + [runner.SANDBOX_NODE, "-e", f"require('{runner.SANDBOX_GITNEXUS}/dist/cli/analyze.js')"], timeout=10, ) diff --git a/eval/workflow_bench/proposer_sandbox.py b/eval/workflow_bench/proposer_sandbox.py index 7910bd87e..2884eb067 100644 --- a/eval/workflow_bench/proposer_sandbox.py +++ b/eval/workflow_bench/proposer_sandbox.py @@ -27,6 +27,7 @@ SANDBOX_TMP = "/tmp" SANDBOX_CLAUDE = "/opt/claude/claude" SANDBOX_SHELL_PREFIX = "/opt/claude/shell-prefix" SANDBOX_PYTHON3 = "/opt/claude/python3" +SANDBOX_NODE = "/opt/claude/node" SANDBOX_PATH = "/opt/claude:/usr/local/bin:/usr/bin:/bin" SANDBOX_GITNEXUS = "/opt/gitnexus" SANDBOX_GITNEXUS_SHARED = "/opt/gitnexus-shared" @@ -355,16 +356,19 @@ def _runtime_mount_args() -> list[str]: if path.exists(): args += ["--ro-bind", raw, raw] # sanitized_graph.py and runner_sessions.py invoke the sandboxed graph - # CLI at the fixed path /usr/local/bin/node. That's only covered by the - # /usr bind above when node happens to live under /usr/local/bin on the - # host -- true on GitHub-hosted runner images, but not on a self-hosted - # runner where actions/setup-node installs into its own tool-cache - # directory instead. Bind whatever `node` actually resolves to on PATH - # to that same fixed sandbox path so both call sites keep working - # regardless of where the host actually put it. + # CLI via SANDBOX_NODE. Bind whatever `node` actually resolves to on PATH + # there -- true node location varies by host (GitHub-hosted runner images + # happen to have one under /usr/local/bin; a self-hosted runner's + # actions/setup-node installs into its own tool-cache directory instead). + # Target must be a fresh path like /opt/claude/... rather than anywhere + # under /usr, /bin, /lib, or /lib64: those are already read-only bound + # above, and bwrap can't create a new mount-point file inside an + # already-read-only tree when the real path doesn't already exist there + # (the exact case a self-hosted runner hits, and the reason this bind + # exists at all). node_bin = shutil.which("node") if node_bin: - args += ["--ro-bind", node_bin, "/usr/local/bin/node"] + args += ["--ro-bind", node_bin, SANDBOX_NODE] for raw in ( "/etc/ssl", "/etc/hosts", diff --git a/eval/workflow_bench/runner.py b/eval/workflow_bench/runner.py index b502c8023..d67934cd0 100644 --- a/eval/workflow_bench/runner.py +++ b/eval/workflow_bench/runner.py @@ -78,6 +78,7 @@ from .proposer_sandbox import ( SANDBOX_GITNEXUS as SANDBOX_GITNEXUS, SANDBOX_GITNEXUS_REGISTRY, SANDBOX_GITNEXUS_SHARED as SANDBOX_GITNEXUS_SHARED, + SANDBOX_NODE as SANDBOX_NODE, SANDBOX_WORKSPACE, ReadOnlyMount, SandboxError, diff --git a/eval/workflow_bench/runner_sessions.py b/eval/workflow_bench/runner_sessions.py index 53e7335ee..cc1708572 100644 --- a/eval/workflow_bench/runner_sessions.py +++ b/eval/workflow_bench/runner_sessions.py @@ -18,6 +18,7 @@ from .proposer_sandbox import ( SANDBOX_GITNEXUS, SANDBOX_GITNEXUS_REGISTRY, SANDBOX_HOME, + SANDBOX_NODE, SANDBOX_TMP, SANDBOX_WORKSPACE, SandboxError, @@ -50,6 +51,8 @@ def measured_cost(raw: Any) -> float | None: if not math.isfinite(raw) or raw < 0: return None return float(raw) + + SANDBOX_GITNEXUS_ENTRYPOINT = f"{SANDBOX_GITNEXUS}/dist/cli/index.js" SENSITIVE_EVENT_KEYS = frozenset( { @@ -113,7 +116,7 @@ def sandbox_mcp_config() -> str: "PATH=/usr/local/bin:/usr/bin:/bin", "LANG=C.UTF-8", "GIT_TERMINAL_PROMPT=0", - "/usr/local/bin/node", + SANDBOX_NODE, SANDBOX_GITNEXUS_ENTRYPOINT, "mcp", ], diff --git a/eval/workflow_bench/sanitized_graph.py b/eval/workflow_bench/sanitized_graph.py index d0e9418dd..23339b1c7 100644 --- a/eval/workflow_bench/sanitized_graph.py +++ b/eval/workflow_bench/sanitized_graph.py @@ -16,6 +16,7 @@ from .process_control import ManagedProcessError, run_managed from .proposer_sandbox import ( SANDBOX_GITNEXUS, SANDBOX_HOME, + SANDBOX_NODE, SANDBOX_WORKSPACE, ReadOnlyMount, SandboxError, @@ -248,7 +249,7 @@ def _run_graph_cli( ) -> bytes | None: command = [ *prefix, - "/usr/local/bin/node", + SANDBOX_NODE, SANDBOX_GITNEXUS_ENTRYPOINT, *arguments, ] From 54c44d91decc05ba15f6ebe9804fc4aeba562242 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 15:14:32 +0000 Subject: [PATCH 55/61] fix(mcp): reconcile rename report on partial failure; harden enumerate (#2605) Addresses gitnexus-review-agent findings on PR #2608: - MED: on a partial apply (a file's write throws), drop that file's edits from total_edits/graph_edits/text_search_edits/changes so the reported result describes what actually reached disk, not what was attempted. The comprehensive enumeration otherwise let a failing file contribute its entire line count as phantom 'applied' edits. failed_files still names every dropped file. Counts are now derived once from the reported set. - MED: hoist the word-boundary regexes out of the per-line loop (one compile each instead of one per line), reused by the apply loop. - LOW: apply loop reuses escapedOldName instead of recomputing the escape formula inline (removes a preview/apply drift risk). - Soften the in-code comment: enumeration gives per-call preview/apply consistency; the pre-existing two-read TOCTOU (external write between preview and apply) is out of scope and noted, not newly introduced. Tests: add a mixed graph-ref + text_search multi-file case (asserts per-file confidence and the never-downgrade guard, via a stubbed rg), and a partial-write-failure case (asserts only landed files are reported). Assert concrete graph_edits/text_search_edits splits, not just their sum. --- gitnexus/src/mcp/local/local-backend.ts | 78 +++++---- gitnexus/test/unit/rename-edit-report.test.ts | 165 ++++++++++++++---- 2 files changed, 177 insertions(+), 66 deletions(-) diff --git a/gitnexus/src/mcp/local/local-backend.ts b/gitnexus/src/mcp/local/local-backend.ts index dacd0a4cd..5b104dd32 100644 --- a/gitnexus/src/mcp/local/local-backend.ts +++ b/gitnexus/src/mcp/local/local-backend.ts @@ -4366,8 +4366,11 @@ export class LocalBackend { // whole-file `\boldName\b` global replace on every file in `changes`, so the // reported edit list MUST enumerate every matching line in every such file — // otherwise the preview under-reports what lands, and the same partial list - // comes back after apply (#2605). Building `changes` from one file set is the - // single source of truth that keeps the report and the apply provably in sync. + // comes back after apply (#2605). Building `changes` from one file set makes + // the preview enumerate exactly the files the apply loop rewrites, using the + // same word-boundary regex. (This is per-call consistency; the apply loop + // still re-reads each file, so an external write landing between preview and + // apply is a pre-existing gap this method does not lock against.) type RenameEdit = { line: number; old_text: string; @@ -4434,13 +4437,15 @@ export class LocalBackend { logQueryError('rename:ripgrep', e); } - // Enumerate every `\boldName\b` line in every file to rewrite, using the same - // escaped global regex the apply step uses. A file with no matching line is - // dropped (apply would write nothing to it). Because apply's per-file global - // replace rewrites exactly these lines, the reported list equals what lands. + // Enumerate every `\boldName\b` line in each file to rewrite, so the previewed + // file set is exactly the set the apply loop below rewrites. A file with no + // matching line is dropped (apply would write nothing to it). `wordTest` + // (non-global) probes each line; `wordReplace` (global) rewrites it and is + // reused by the apply loop — compiled once each rather than once per line, + // and one escaping formula serves both passes. + const wordTest = new RegExp(`\\b${escapedOldName}\\b`); + const wordReplace = new RegExp(`\\b${escapedOldName}\\b`, 'g'); const changes = new Map(); - let graphEdits = 0; - let astSearchEdits = 0; for (const [filePath, confidence] of fileConfidence) { try { @@ -4448,20 +4453,15 @@ export class LocalBackend { const lines = content.split('\n'); const edits: RenameEdit[] = []; for (let i = 0; i < lines.length; i++) { - if (!new RegExp(`\\b${escapedOldName}\\b`).test(lines[i])) { + if (!wordTest.test(lines[i])) { continue; } edits.push({ line: i + 1, old_text: lines[i].trim(), - new_text: lines[i].replace(new RegExp(`\\b${escapedOldName}\\b`, 'g'), new_name).trim(), + new_text: lines[i].replace(wordReplace, new_name).trim(), confidence, }); - if (confidence === 'graph') { - graphEdits++; - } else { - astSearchEdits++; - } } if (edits.length > 0) { changes.set(filePath, { file_path: filePath, edits }); @@ -4471,39 +4471,55 @@ export class LocalBackend { } } - // Step 4: Apply or preview - const allChanges = Array.from(changes.values()); - const totalEdits = allChanges.reduce((sum, c) => sum + c.edits.length, 0); - + // Step 4: Apply or preview. const failedFiles: string[] = []; if (!dry_run) { - // Apply edits to files - for (const change of allChanges) { + for (const change of changes.values()) { try { const fullPath = assertSafePath(change.file_path); - let content = await fs.readFile(fullPath, 'utf-8'); - const regex = new RegExp(`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'g'); - content = content.replace(regex, new_name); - await fs.writeFile(fullPath, content, 'utf-8'); + const content = await fs.readFile(fullPath, 'utf-8'); + await fs.writeFile(fullPath, content.replace(wordReplace, new_name), 'utf-8'); } catch (e) { - // A swallowed write failure must not be reported as a full success - // (#2283): record the file so the result can degrade to 'partial' - // with the unwritten files listed, rather than masquerading as done. + // A swallowed write failure must not be reported as success (#2283): + // record the file so the result degrades to 'partial'. logQueryError('rename:apply-edit', e); failedFiles.push(change.file_path); } } + // A file whose write threw did not land, so drop its edits from the + // reported result — total_edits/changes must describe what actually + // reached disk, not what was attempted (#2605: the report matches reality + // even on partial failure). failed_files still names every dropped file. + for (const f of failedFiles) { + changes.delete(f); + } + } + + // Counts derive from the reported set (dry-run: every enumerated file; + // apply: only files that landed), so the graph/text_search split always + // sums to total_edits and never overstates a partial apply. + const reported = Array.from(changes.values()); + let graphEdits = 0; + let astSearchEdits = 0; + for (const change of reported) { + for (const edit of change.edits) { + if (edit.confidence === 'graph') { + graphEdits++; + } else { + astSearchEdits++; + } + } } return { status: failedFiles.length > 0 ? 'partial' : 'success', old_name: oldName, new_name, - files_affected: allChanges.length, - total_edits: totalEdits, + files_affected: reported.length, + total_edits: graphEdits + astSearchEdits, graph_edits: graphEdits, text_search_edits: astSearchEdits, - changes: allChanges, + changes: reported, applied: !dry_run, ...(failedFiles.length > 0 && { failed_files: failedFiles }), }; diff --git a/gitnexus/test/unit/rename-edit-report.test.ts b/gitnexus/test/unit/rename-edit-report.test.ts index 0d0993e96..b7088b9e7 100644 --- a/gitnexus/test/unit/rename-edit-report.test.ts +++ b/gitnexus/test/unit/rename-edit-report.test.ts @@ -7,11 +7,13 @@ * per graph-ref file then broke, and (c) skipped text-search on any file already * covered by the graph. When a private symbol's definition and all its call * sites live in one file, only the definition line was reported (total_edits: 1) - * while apply rewrote every occurrence. This test drives the exact single-file - * case with an empty graph (no incoming refs) and asserts report == apply. + * while apply rewrote every occurrence. These tests drive the single-file repro, + * a mixed graph/text_search multi-file rename, and a partial write failure, and + * assert the report matches what apply actually writes in each case. */ import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest'; import { promises as fs } from 'node:fs'; +import fsPromises from 'fs/promises'; import os from 'node:os'; import path from 'node:path'; @@ -26,8 +28,36 @@ vi.mock('../../src/mcp/core/embedder.js', () => ({ getEmbeddingDims: vi.fn().mockReturnValue(384), })); +// rename() shells out to `rg -l` to discover text-search files. Stub it so the +// ripgrep-discovery branch is deterministic and driveable (rg is not reliably +// on PATH inside the vitest worker). Default: no hits. +const { execFileSyncMock } = vi.hoisted(() => ({ execFileSyncMock: vi.fn(() => '') })); +vi.mock('child_process', async (importActual) => { + const actual = await importActual(); + return { ...actual, execFileSync: execFileSyncMock }; +}); + import { LocalBackend } from '../../src/mcp/local/local-backend.js'; +type Incoming = { + calls: { filePath: string }[]; + imports: { filePath: string }[]; + extends: { filePath: string }[]; + implements: { filePath: string }[]; +}; +const EMPTY_INCOMING: Incoming = { calls: [], imports: [], extends: [], implements: [] }; + +type RenameResult = { + status: string; + applied: boolean; + files_affected: number; + total_edits: number; + graph_edits: number; + text_search_edits: number; + changes: { file_path: string; edits: { line: number; confidence: string }[] }[]; + failed_files?: string[]; +}; + // The #2605 repro: a private free fn with exactly 4 textual occurrences of // `rename_target` — the definition, one production call, two test calls — all // in the same file. @@ -55,16 +85,20 @@ mod tests { } `; -// 1-based occurrence lines of `rename_target` in RUST_SRC — the ground truth the -// report must match. Computed from the fixture (not hardcoded) so editing the -// snippet cannot silently desync the expectation. -const OCCURRENCE_LINES = RUST_SRC.split('\n') - .map((line, i) => (/\brename_target\b/.test(line) ? i + 1 : 0)) - .filter((n) => n > 0); +/** 1-based occurrence lines of `rename_target` in `src` — the ground truth the + * report must match. Computed (not hardcoded) so editing a fixture cannot + * silently desync the expectation. */ +function occurrenceLines(src: string): number[] { + return src + .split('\n') + .map((line, i) => (/\brename_target\b/.test(line) ? i + 1 : 0)) + .filter((n) => n > 0); +} +const OCCURRENCE_LINES = occurrenceLines(RUST_SRC); -/** Build a backend whose graph lookup returns the symbol with NO incoming refs - * (the exact condition that made the old code report only the definition). */ -function stubbedBackend(): LocalBackend { +/** Build a backend whose graph lookup returns the symbol (definition at + * src/lib.rs) with the given incoming refs. */ +function stubbedBackend(incoming: Incoming = EMPTY_INCOMING): LocalBackend { const backend = new LocalBackend(); vi.spyOn( backend as unknown as { ensureInitialized: () => Promise }, @@ -73,20 +107,32 @@ function stubbedBackend(): LocalBackend { vi.spyOn(backend as unknown as { context: () => Promise }, 'context').mockResolvedValue({ status: 'success', symbol: { name: 'rename_target', filePath: 'src/lib.rs', startLine: OCCURRENCE_LINES[0] }, - incoming: { calls: [], imports: [], extends: [], implements: [] }, + incoming, }); return backend; } +function callRename( + backend: LocalBackend, + repoPath: string, + params: Record, +): Promise { + return ( + backend as unknown as { rename: (r: unknown, p: unknown) => Promise } + ).rename({ repoPath }, { symbol_name: 'rename_target', new_name: 'renamed_fn', ...params }); +} + +const editsFor = (r: RenameResult, file: string) => + r.changes.find((c) => c.file_path === file)?.edits ?? []; + describe('rename edit report is faithful to apply (#2605)', () => { - let backend: LocalBackend; let tmpDir: string; beforeEach(async () => { + execFileSyncMock.mockReturnValue(''); // default: no ripgrep hits tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), 'gn-2605-')); await fs.mkdir(path.join(tmpDir, 'src')); await fs.writeFile(path.join(tmpDir, 'src', 'lib.rs'), RUST_SRC, 'utf-8'); - backend = stubbedBackend(); }); afterEach(async () => { @@ -95,22 +141,19 @@ describe('rename edit report is faithful to apply (#2605)', () => { }); it('previews every occurrence that apply will rewrite (dry_run)', async () => { - const result = await ( - backend as unknown as { rename: (r: unknown, p: unknown) => Promise } - ).rename( - { repoPath: tmpDir }, - { symbol_name: 'rename_target', new_name: 'renamed_fn', dry_run: true }, - ); + const result = await callRename(stubbedBackend(), tmpDir, { dry_run: true }); expect(result.applied).toBe(false); expect(result.files_affected).toBe(1); expect(result.total_edits).toBe(OCCURRENCE_LINES.length); // 4, not 1 - expect(result.graph_edits + result.text_search_edits).toBe(result.total_edits); + // Concrete split, not just the sum: all occurrences are in the definition + // file, so they are graph-confidence and text_search is zero. + expect(result.graph_edits).toBe(OCCURRENCE_LINES.length); + expect(result.text_search_edits).toBe(0); - const reportedLines = result.changes[0].edits - .map((e: { line: number }) => e.line) - .sort((a: number, b: number) => a - b); - expect(reportedLines).toEqual(OCCURRENCE_LINES); + const edits = editsFor(result, 'src/lib.rs'); + expect(edits.map((e) => e.line).sort((a, b) => a - b)).toEqual(OCCURRENCE_LINES); + expect(edits.every((e) => e.confidence === 'graph')).toBe(true); // A dry run leaves the file untouched. const onDisk = await fs.readFile(path.join(tmpDir, 'src', 'lib.rs'), 'utf-8'); @@ -118,12 +161,7 @@ describe('rename edit report is faithful to apply (#2605)', () => { }); it('reports exactly what it wrote (apply)', async () => { - const result = await ( - backend as unknown as { rename: (r: unknown, p: unknown) => Promise } - ).rename( - { repoPath: tmpDir }, - { symbol_name: 'rename_target', new_name: 'renamed_fn', dry_run: false }, - ); + const result = await callRename(stubbedBackend(), tmpDir, { dry_run: false }); expect(result.applied).toBe(true); expect(result.total_edits).toBe(OCCURRENCE_LINES.length); @@ -135,10 +173,67 @@ describe('rename edit report is faithful to apply (#2605)', () => { expect(stragglers).toBe(0); // The reported edit count equals the number of replacements that landed. - const reportedEdits = result.changes.reduce( - (n: number, c: { edits: unknown[] }) => n + c.edits.length, - 0, - ); + const reportedEdits = result.changes.reduce((n, c) => n + c.edits.length, 0); expect(reportedEdits).toBe(renamedCount); }); + + it('enumerates all occurrences across graph-ref and text-search files, keeping confidence per file', async () => { + // A graph-referencing file (not the definition) with MULTIPLE occurrences — + // the exact "one edit per file then break" bug's other original trigger. + const CALLER = 'use crate::rename_target;\nfn a() { rename_target(1); rename_target(2); }\n'; + // A file discovered only by ripgrep — the text_search branch. + const NOTES = '// see rename_target for details\n'; + await fs.writeFile(path.join(tmpDir, 'src', 'caller.rs'), CALLER, 'utf-8'); + await fs.writeFile(path.join(tmpDir, 'src', 'notes.rs'), NOTES, 'utf-8'); + // rg reports the definition file (already graph — exercises never-downgrade) + // and the text-only file. + execFileSyncMock.mockReturnValue('src/lib.rs\nsrc/notes.rs\n'); + + const backend = stubbedBackend({ + ...EMPTY_INCOMING, + calls: [{ filePath: 'src/caller.rs' }], + }); + const result = await callRename(backend, tmpDir, { dry_run: true }); + + const callerOcc = occurrenceLines(CALLER).length; // 3 + const notesOcc = occurrenceLines(NOTES).length; // 1 + + expect(result.files_affected).toBe(3); + expect(result.total_edits).toBe(OCCURRENCE_LINES.length + callerOcc + notesOcc); + // Split is concrete: definition + graph-ref file are graph; the rg-only file + // is text_search. A file reached by both graph and rg keeps graph (never + // downgraded). + expect(result.graph_edits).toBe(OCCURRENCE_LINES.length + callerOcc); + expect(result.text_search_edits).toBe(notesOcc); + + expect(editsFor(result, 'src/lib.rs').every((e) => e.confidence === 'graph')).toBe(true); + expect(editsFor(result, 'src/caller.rs').map((e) => e.confidence)).toEqual(['graph', 'graph']); + expect(editsFor(result, 'src/notes.rs').map((e) => e.confidence)).toEqual(['text_search']); + }); + + it('reports only files that landed when a write fails mid-apply (#2605 partial)', async () => { + const CALLER = 'fn a() { rename_target(1); rename_target(2); }\n'; + await fs.writeFile(path.join(tmpDir, 'src', 'caller.rs'), CALLER, 'utf-8'); + const backend = stubbedBackend({ ...EMPTY_INCOMING, calls: [{ filePath: 'src/caller.rs' }] }); + + // caller.rs write throws; lib.rs succeeds. + vi.spyOn(fsPromises, 'writeFile').mockImplementation( + async (p: Parameters[0]) => { + if (String(p).endsWith(`${path.sep}caller.rs`)) { + throw Object.assign(new Error('EACCES'), { code: 'EACCES' }); + } + }, + ); + + const result = await callRename(backend, tmpDir, { dry_run: false }); + + expect(result.status).toBe('partial'); + expect(result.failed_files).toEqual(['src/caller.rs']); + // The failed file's edits are NOT reported as applied: totals describe only + // what reached disk (lib.rs), never the attempted caller.rs occurrences. + expect(result.files_affected).toBe(1); + expect(result.total_edits).toBe(OCCURRENCE_LINES.length); + expect(result.graph_edits).toBe(OCCURRENCE_LINES.length); + expect(result.changes.map((c) => c.file_path)).toEqual(['src/lib.rs']); + }); }); From 66f11badaf8da44af9bdd64c662611928780a390 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 15:19:48 +0000 Subject: [PATCH 56/61] style: wrap long filter predicate per prettier (PR autofix) --- gitnexus/test/integration/resolvers/rust.test.ts | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/gitnexus/test/integration/resolvers/rust.test.ts b/gitnexus/test/integration/resolvers/rust.test.ts index 87aa6899b..fe970b777 100644 --- a/gitnexus/test/integration/resolvers/rust.test.ts +++ b/gitnexus/test/integration/resolvers/rust.test.ts @@ -1969,7 +1969,9 @@ describe('Rust dyn trait-object dispatch (#2604)', () => { it('emits exactly one CALLS edge from calls_via_dyn(b: &dyn Behaviour) to trait_target', () => { const calls = getRelationships(result, 'CALLS'); - const dynCalls = calls.filter((c) => c.source === 'calls_via_dyn' && c.target === 'trait_target'); + const dynCalls = calls.filter( + (c) => c.source === 'calls_via_dyn' && c.target === 'trait_target', + ); expect(dynCalls.length).toBe(1); }); }); From 00141d0da2c74c6f5f6aadb69ec9a754033f1097 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 15:32:11 +0000 Subject: [PATCH 57/61] test(bench): rebaseline rust scope-capture fingerprint for #2604 RUST_SCOPE_QUERY gained a function_signature_item capture, shifting the capture fingerprint for every bench fixture with a required trait method. Verified: node --import tsx bench/scope-capture/measure.mjs --check now passes across all 14 languages (rust scaling 1.036 < 1.5 budget). --- gitnexus/bench/scope-capture/baselines.json | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index cab3765a4..81e8b92ec 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -46,8 +46,9 @@ "_note": "#2046: F35 qualified-constructor captures now emit @reference.qualified-name + a simple-name @reference.name on `new Ns.Foo()`/`new A.B.Foo()`; namespace_declaration/file_scoped_namespace_declaration now emit @declaration.namespace name captures (feeding the non-destructive namespacePrefix sidecar for `new B.Foo()` same-tail disambiguation). + csharp-interface-only-base and csharp-namespace-qualified-ctor fixtures. Pure capture-additive + fixture-corpus drift; scaling stays linear (~1.11)." }, "rust": { - "fingerprint": "df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29", + "fingerprint": "f7742f65f14d7d6590df7f16303fc3cc9dc0c233cd80bf90c98b084933cd3846", "scaling_budget": 1.5, + "_rebaselined_dyn_trait_object_2604": "#2604: RUST_SCOPE_QUERY now captures function_signature_item (abstract trait methods, no body) as a scope + declaration, so a &dyn Trait receiver can dispatch a CALLS edge to the trait's own method. Additive capture shift across every bench fixture with a required trait method. Prior df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29 -> f7742f65f14d7d6590df7f16303fc3cc9dc0c233cd80bf90c98b084933cd3846; scaling 1.033 < 1.5.", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 65e5bca66bb1ca117949409e8fb5c80ee69d6f1b5318908eaaecf08da0482e5c -> df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29; scaling 1.065 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Rust fn-value callable flow facts with invocation/constructor-result suppression. Prior ac610bbe97666bf285923479dd7b43a2fe4c5354aae8df1bcbafdc04fb220f82 -> 65e5bca66bb1ca117949409e8fb5c80ee69d6f1b5318908eaaecf08da0482e5c; scaling 1.024 < 1.5.", "_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) \u2014 legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.", From a7bfe819ebc0eb8267f00b062643dda64a324618 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 15:42:58 +0000 Subject: [PATCH 58/61] test: update hardcoded schema-version expectations for v11 (#2604) call-summary-schema-version.test.ts pins INCREMENTAL_SCHEMA_VERSION as a literal per bump, documenting the reuse-gate boundary for each version. Update the "current" expectation to 11 and add the v10 pre-current case, matching the v7/v8/v9/v10 precedent already in the file. --- gitnexus/test/unit/call-summary-schema-version.test.ts | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/gitnexus/test/unit/call-summary-schema-version.test.ts b/gitnexus/test/unit/call-summary-schema-version.test.ts index a8240fa0f..2047754c4 100644 --- a/gitnexus/test/unit/call-summary-schema-version.test.ts +++ b/gitnexus/test/unit/call-summary-schema-version.test.ts @@ -73,8 +73,8 @@ describe('CALL_SUMMARY relation-type exclusion (U-C1)', () => { }); describe('CALL_SUMMARY incremental reuse gate (U-C5)', () => { - it('INCREMENTAL_SCHEMA_VERSION is bumped to 10 (Java record container-node re-index window)', () => { - expect(INCREMENTAL_SCHEMA_VERSION).toBe(10); + it('INCREMENTAL_SCHEMA_VERSION is bumped to 11 (Rust dyn-trait-object dispatch re-index window, #2604)', () => { + expect(INCREMENTAL_SCHEMA_VERSION).toBe(11); }); it('a pre-current stamp fails the `=== INCREMENTAL_SCHEMA_VERSION` reuse gate → forces full re-analyze', () => { @@ -112,7 +112,11 @@ describe('CALL_SUMMARY incremental reuse gate (U-C5)', () => { // (#2564) — a record's methods would keep being ownerless Method nodes // with no HAS_METHOD edge on unchanged files → must NOT reuse. expect(passesReuseGate(9)).toBe(false); + // A pre-v11 (v10) index predates the Rust dyn-trait-object dispatch fix + // (#2604) — abstract trait methods would keep being uncaptured (no + // ownerId/CALLS resolution) on unchanged Rust trait files → must NOT reuse. + expect(passesReuseGate(10)).toBe(false); // A current-version stamp passes the gate (incremental top-up eligible). - expect(passesReuseGate(10)).toBe(true); + expect(passesReuseGate(11)).toBe(true); }); }); From e18b4416c6cc7cafc48b66cc1acb31368598ea68 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 16:10:01 +0000 Subject: [PATCH 59/61] test(rust): cover Box and dyn-bound-list normalization (#2604) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit GitNexus review-agent finding: stripDynBound's documented Box, Rc/Arc, and auto-trait/lifetime bound-list (dyn Trait + Send) shapes had no test anywhere — only the bare &dyn Trait parameter case was exercised end-to-end. Add direct unit coverage on normalizeRustTypeName and (via interpretRustTypeBinding) normalizeRustReturnType for these shapes. --- .../rust/rust-dyn-type-normalization.test.ts | 61 +++++++++++++++++++ 1 file changed, 61 insertions(+) create mode 100644 gitnexus/test/unit/scope-resolution/rust/rust-dyn-type-normalization.test.ts diff --git a/gitnexus/test/unit/scope-resolution/rust/rust-dyn-type-normalization.test.ts b/gitnexus/test/unit/scope-resolution/rust/rust-dyn-type-normalization.test.ts new file mode 100644 index 000000000..3a338830e --- /dev/null +++ b/gitnexus/test/unit/scope-resolution/rust/rust-dyn-type-normalization.test.ts @@ -0,0 +1,61 @@ +import { describe, expect, it } from 'vitest'; +import type { CaptureMatch } from 'gitnexus-shared'; +import { + normalizeRustTypeName, + interpretRustTypeBinding, +} from '../../../../src/core/ingestion/languages/rust/interpret.js'; + +const RANGE = { startLine: 0, startCol: 0, endLine: 0, endCol: 0 }; + +/** Builds a minimal @type-binding.return CaptureMatch to exercise + * normalizeRustReturnType (private, only reachable through this hook). */ +function returnTypeBinding(type: string): CaptureMatch { + return { + '@type-binding.name': { name: '@type-binding.name', range: RANGE, text: 'f' }, + '@type-binding.type': { name: '@type-binding.type', range: RANGE, text: type }, + '@type-binding.return': { name: '@type-binding.return', range: RANGE, text: '' }, + }; +} + +/** + * #2604 coverage gap (GitNexus review-agent finding): stripDynBound's + * documented Box and bound-list (dyn Trait + Send) shapes had no + * test anywhere, even though the interpret.ts comment claims they're handled. + * These exercise normalizeRustTypeName/normalizeRustReturnType directly — + * stripDynBound itself is a private helper reached only through them. + */ +describe('Rust dyn-trait-object type-name normalization (#2604)', () => { + it('strips a bare dyn Trait parameter type', () => { + expect(normalizeRustTypeName('&dyn Behaviour')).toBe('Behaviour'); + expect(normalizeRustTypeName('dyn Behaviour')).toBe('Behaviour'); + }); + + it('strips dyn through Box/Rc/Arc wrappers', () => { + expect(normalizeRustTypeName('Box')).toBe('Trait'); + expect(normalizeRustTypeName('Rc')).toBe('Trait'); + expect(normalizeRustTypeName('Arc')).toBe('Trait'); + }); + + it('drops an auto-trait/lifetime bound list after dyn', () => { + expect(normalizeRustTypeName('dyn Trait + Send')).toBe('Trait'); + expect(normalizeRustTypeName("dyn Trait + Send + 'static")).toBe('Trait'); + expect(normalizeRustTypeName("Box")).toBe('Trait'); + }); + + it('truncates a dyn trait\'s own generic arguments after stripping dyn', () => { + expect(normalizeRustTypeName('dyn Iterator')).toBe('Iterator'); + }); + + it('strips dyn in return-type position, including through &', () => { + expect(interpretRustTypeBinding(returnTypeBinding('&dyn Trait'))?.rawTypeName).toBe('Trait'); + expect(interpretRustTypeBinding(returnTypeBinding('dyn Trait + Send'))?.rawTypeName).toBe( + 'Trait', + ); + }); + + it('leaves ordinary (non-dyn) type names untouched', () => { + expect(normalizeRustTypeName('Behaviour')).toBe('Behaviour'); + expect(normalizeRustTypeName('&Behaviour')).toBe('Behaviour'); + expect(normalizeRustTypeName('Box')).toBe('Behaviour'); + }); +}); From 9f57984372f6e71637ef9c3583125fdddb574400 Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Tue, 21 Jul 2026 16:12:20 +0000 Subject: [PATCH 60/61] style: fix quote style per prettier in new dyn-normalization test --- .../scope-resolution/rust/rust-dyn-type-normalization.test.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/gitnexus/test/unit/scope-resolution/rust/rust-dyn-type-normalization.test.ts b/gitnexus/test/unit/scope-resolution/rust/rust-dyn-type-normalization.test.ts index 3a338830e..273741764 100644 --- a/gitnexus/test/unit/scope-resolution/rust/rust-dyn-type-normalization.test.ts +++ b/gitnexus/test/unit/scope-resolution/rust/rust-dyn-type-normalization.test.ts @@ -42,7 +42,7 @@ describe('Rust dyn-trait-object type-name normalization (#2604)', () => { expect(normalizeRustTypeName("Box")).toBe('Trait'); }); - it('truncates a dyn trait\'s own generic arguments after stripping dyn', () => { + it("truncates a dyn trait's own generic arguments after stripping dyn", () => { expect(normalizeRustTypeName('dyn Iterator')).toBe('Iterator'); }); From eb116c8a0748139fec5d08966b29ca2b0057c06a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Tue, 21 Jul 2026 19:19:15 +0100 Subject: [PATCH 61/61] fix(eval): exclude Claude Code's own sandbox-bootstrap noise from the planning-phase check (#2615) The first fully successful real workflow_dispatch run on the self-hosted runner (https://github.com/abhigyanpatwari/GitNexus/actions/runs/29843028596) still failed: 17/18 sessions hit error_kind plan-evidence-invalid with "phase changed unauthorized workspace path(s): .claude/.cc-writes, .claude/commands, .env, .env.development, ...". Reproduced directly on the runner (SSM, matching the real sandbox settings exactly, including enableWeakerNestedSandbox): a single trivial "say OK" prompt -- no real task, no real API key even -- is enough to make Claude Code create a synthetic package.json/lockfiles/node_modules, a full set of .env variants, and .claude/agents, .claude/commands, .claude/.cc-writes in the workspace on every single session. None of this is something the model decided to write; it's Claude Code's own internal bootstrap for running inside an already-sandboxed environment, and it happens regardless of task or prompt. enforce_phase_workspace (the planning-phase boundary check: verify the plan session touched only its one plan doc) already excludes .git for exactly this class of reason -- harness/tool noise, not substantive diff. Extends the same exclusion to the empirically-observed bootstrap set. workspace_snapshot has exactly one use (this check, confirmed via every caller), so widening its exclusion list can't hide anything in some other context that actually cares about these paths changing. Co-authored-by: Claude Sonnet 5 --- eval/tests/test_runner_hardening.py | 34 +++++++++++++++++++++ eval/workflow_bench/runner_artifacts.py | 40 +++++++++++++++++++++++-- 2 files changed, 72 insertions(+), 2 deletions(-) diff --git a/eval/tests/test_runner_hardening.py b/eval/tests/test_runner_hardening.py index 8245fdf82..6905891c0 100644 --- a/eval/tests/test_runner_hardening.py +++ b/eval/tests/test_runner_hardening.py @@ -262,3 +262,37 @@ def test_phase_workspace_accepts_new_regular_review_output(tmp_path): artifact.write_text("new review") runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact) + + +def test_phase_workspace_ignores_claude_sandbox_bootstrap_noise(tmp_path): + # Reproduced empirically: Claude Code's own enableWeakerNestedSandbox + # bootstrap creates this exact set of paths on every session regardless + # of task or model output (a trivial "say OK" prompt was enough). None + # of it is something the model decided to write, so it must not read as + # an unauthorized planning-phase change. + before = runner_artifacts.workspace_snapshot(tmp_path) + (tmp_path / ".claude" / "agents").mkdir(parents=True) + (tmp_path / ".claude" / "commands").mkdir(parents=True) + (tmp_path / ".claude" / ".cc-writes").write_text("{}") + (tmp_path / ".env").write_text("") + (tmp_path / ".env.development.local").write_text("") + (tmp_path / ".npmrc").write_text("") + (tmp_path / "package.json").write_text("{}") + (tmp_path / "node_modules").mkdir() + (tmp_path / "node_modules" / ".bin").mkdir() + artifact = tmp_path / "review-output.md" + artifact.write_text("new review") + + runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact) + + +def test_phase_workspace_still_rejects_a_genuinely_unauthorized_change(tmp_path): + # The bootstrap-noise exclusion must stay narrow: an actual source-file + # edit outside the allowed artifact still has to be caught. + before = runner_artifacts.workspace_snapshot(tmp_path) + (tmp_path / "src.py").write_text("changed") + artifact = tmp_path / "review-output.md" + artifact.write_text("new review") + + with pytest.raises(ValueError, match="unauthorized workspace path"): + runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact) diff --git a/eval/workflow_bench/runner_artifacts.py b/eval/workflow_bench/runner_artifacts.py index 0b56a0c9a..da6b9ae58 100644 --- a/eval/workflow_bench/runner_artifacts.py +++ b/eval/workflow_bench/runner_artifacts.py @@ -21,6 +21,40 @@ MAX_WORKSPACE_SNAPSHOT_ENTRIES = 100_000 MAX_WORKSPACE_SNAPSHOT_PATH_BYTES = 16 * 1024 * 1024 MAX_WORKSPACE_SNAPSHOT_FILE_BYTES = 1024 * 1024 * 1024 +# Claude Code's own enableWeakerNestedSandbox bootstrap creates these paths on +# EVERY session regardless of task or model output -- reproduced empirically +# with a trivial "say OK" prompt: a synthetic package.json/lockfiles/ +# node_modules, a full set of .env variants, and .claude/agents, +# .claude/commands, .claude/.cc-writes. None of this is something the model +# decided to write, so it must not count as an "unauthorized" workspace +# change during the planning-phase boundary check (the one thing this +# snapshot is used for -- see workspace_snapshot's callers). Mirrors the +# pre-existing .git exclusion below, which is the same kind of harness/tool +# noise rather than substantive diff. +WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE = frozenset( + { + ".claude", + ".env", + ".env.development", + ".env.development.local", + ".env.local", + ".env.production", + ".env.production.local", + ".env.test", + ".env.test.local", + ".gitmodules", + ".npmrc", + ".yarnrc", + ".yarnrc.yml", + "bunfig.toml", + "node_modules", + "package-lock.json", + "package.json", + "pnpm-lock.yaml", + "yarn.lock", + } +) + IMPLEMENTATION_ARMS = frozenset( { "workflow", @@ -53,7 +87,9 @@ class VerificationResult: def workspace_snapshot(worktree: Path) -> dict[str, str]: - """Hash the workspace without following links, excluding Git internals.""" + """Hash the workspace without following links, excluding Git internals + and Claude Code's own sandbox-bootstrap noise (see + WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE).""" root = worktree.expanduser().absolute() mode = root.lstat().st_mode @@ -74,7 +110,7 @@ def workspace_snapshot(worktree: Path) -> dict[str, str]: raise ValueError(f"workspace snapshot directory is unreadable: {directory}: {exc}") from exc for entry in children: relative = relative_dir / entry.name - if relative.parts[0] == ".git": + if relative.parts[0] == ".git" or relative.parts[0] in WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE: continue entry_count += 1 path_bytes += len(relative.as_posix().encode())