From dc5c816a020e76788a50d357e6f72c3d1cb8f79a Mon Sep 17 00:00:00 2001 From: FenjuFu <92919259+FenjuFu@users.noreply.github.com> Date: Mon, 31 Aug 2026 00:12:46 +0800 Subject: [PATCH 01/26] fix(serve): protect MCP route with optional bearer auth (#3100) * fix(serve): protect MCP route with optional bearer auth Signed-off-by: FenjuFu <92919259+FenjuFu@users.noreply.github.com> * fix(serve): clarify MCP auth proxy boundaries Document the Render token incompatibility, expose serve auth in CLI help, and replace source-order assertions with live middleware coverage. Note: full test suite has pre-existing worktree failures because generated parse-worker.js is absent; targeted auth and proxy suites pass. Co-authored-by: Cursor * chore(docs): preserve existing table formatting Keep the auth clarifications focused without reformatting unrelated Markdown tables. Co-authored-by: Cursor * fix(proxy): inject backend MCP credentials Replace the consumed edge credential with the configured protocol token only for MCP routes so proxied serve authentication remains composable. Co-authored-by: Cursor --------- Signed-off-by: FenjuFu <92919259+FenjuFu@users.noreply.github.com> Co-authored-by: Gergo Magyar Co-authored-by: Cursor --- README.md | 1 + SECURITY.md | 3 +- docker-server.mjs | 22 +++- docker-server.test.mjs | 82 ++++++++++++- gitnexus/src/cli/i18n/en.ts | 2 +- gitnexus/src/cli/i18n/zh-CN.ts | 2 +- gitnexus/src/cli/index.ts | 2 +- gitnexus/src/server/api.ts | 5 +- gitnexus/src/server/mcp-http.ts | 22 +++- gitnexus/test/unit/mcp-http-transport.test.ts | 108 +++++++++++++++++- render.yaml | 4 +- 11 files changed, 239 insertions(+), 14 deletions(-) diff --git a/README.md b/README.md index 405d85201..75bdd8475 100644 --- a/README.md +++ b/README.md @@ -530,6 +530,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max | `GITNEXUS_PARSE_CHUNK_CONCURRENCY` | `2` | Number of chunks whose file contents may be read into memory in parallel while the pool dispatches the current chunk. Worker dispatch itself stays serial. | Repos large enough to chunk (multi-MB total source) where disk I/O is a measurable fraction of analyze wall-clock. | | `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. | | `GITNEXUS_AUTH_TOKEN` | unset | Bearer token required when `eval-server` binds beyond loopback. May also be read from `.env.local` or `.env`; shell values take precedence. | Exposing the evaluation HTTP tools to a container, VM, or LAN. | +| `GITNEXUS_MCP_AUTH_TOKEN` | unset | Bearer token for the dedicated `gitnexus mcp --http` server, for a **directly reachable** `gitnexus serve` `/api/mcp` route, and for the `docker-server` / web proxy in front of one. A non-loopback dedicated MCP bind requires it; `serve` enables protocol-layer MCP auth when it is set. Behind a proxy, set the **same** value on both services: the proxy spends the edge `GITNEXUS_SERVE_AUTH_TOKEN`, then replaces `Authorization` with this token on `/api/mcp` only. | Dedicated MCP, a `serve` the client can reach directly, or a proxied deploy (Render Blueprint) where the backend runs protocol-layer MCP auth — configure it on the proxy too. | | `GITNEXUS_PROFILE_DEFERRED` | unset | When `1`, emits `[deferred-profile]` timing/progress logs for the post-chunk deferred resolution band (imports → heritage → buildHeritageMap → legacy call resolution). Implied by `GITNEXUS_VERBOSE`. | Diagnosing analyze stalls in "Resolving calls (all chunks)" on large Java/Kotlin repos (issue #1741) without the full verbose ingestion noise. | | `GITNEXUS_PROFILE_DEFERRED_SLOW_MS` | `3000` (verbose) / `5000` | Per-file threshold in ms above which `processCallsFromExtracted` emits a `slow file …` log line. Parsed via `Number()`: accepts integers (`5000`), scientific notation (`2.5e3`), decimals (`.5`), and hex (`0x10`). Non-finite or non-positive values fall back to the default. | Hunting a few outlier files dominating the deferred call-resolution stage; lower to surface more, raise to focus only on the worst. | | `PROF_LBUG_LOAD` | unset | When `1`, emits one `[lbug-load prof]` summary line per `loadGraphToLbug` call breaking the graph-DB persistence wall into stages (`csv-emit` / `copy-nodes` / `copy-rels` / `fallback` / `total`) plus node & edge counts. Zero-cost when unset. | Attributing large-repo analyze wall time across CSV generation vs. LadybugDB `COPY` (issue #2203) — the analyze "emit" timing is the scope-resolution bucket, not this DB-write path. | diff --git a/SECURITY.md b/SECURITY.md index d1fbcd051..89368a1ea 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -59,7 +59,8 @@ The `render.yaml` Blueprint (see the README's **Deploy to Render**) puts `gitnex - **The generated `GITNEXUS_SERVE_AUTH_TOKEN` is the only access control.** The proxy rejects any `/api/*` request without it with a `401` before forwarding. Rotate it by editing the environment variable on the `gitnexus-web` service and redeploying. - **The CSRF guard is inert on this path.** The proxy strips `Origin` before forwarding, so the server's write-origin guard does nothing for proxied traffic — it passes `Origin`-less requests through by design. The token is not a second layer behind the guard. - **Anyone holding the token can read every indexed repo's source.** These routes carry no origin guard, and the first three carry no rate limiter either: `GET /api/repos`, `GET /api/graph`, `POST /api/query`, `GET /api/file`, `GET /api/grep`. Whoever has the token can also index and delete repositories. -- **`POST /api/mcp` rides the same path.** `serve` mounts the MCP handler via `mountMCPEndpoints`, and `createStreamableHttpHandler` is called with no `authToken` — a **pre-existing** gap in `serve` itself, not something this deploy introduces. On Render it is closed only by the edge token and the private network. A `serve` bound directly to a public interface has no such cover. +- **`POST /api/mcp` rides the same path.** When `GITNEXUS_MCP_AUTH_TOKEN` is set on the backend, `serve` protects `/api/mcp` with the same constant-time Bearer check as the dedicated HTTP MCP server, before parsing the request body. The Render Blueprint does not set a backend MCP token by default. To enable it behind the proxy, set the **same** `GITNEXUS_MCP_AUTH_TOKEN` on both the `gitnexus-web` proxy and the `gitnexus-server` backend: the proxy consumes the edge `GITNEXUS_SERVE_AUTH_TOKEN`, then replaces `Authorization` with the MCP token on `/api/mcp` (and its subpaths) only — the edge credential is never forwarded, and other `/api/*` routes stay stripped. Configuring it on the backend alone makes every proxied MCP request `401`. +- **A directly reachable `serve` still needs an explicit control.** If neither `GITNEXUS_MCP_AUTH_TOKEN` nor an authenticated edge/private-network boundary is present, `/api/mcp` is unauthenticated. Do not bind that topology to a LAN or public interface: MCP readers can access indexed source and graph context. - **Rate limits bound cost, not access.** They cap what a token holder can spend; they do not decide who gets in. Do not hand the URL out as a public demo. A token holder has read access to everything the deploy has indexed. diff --git a/docker-server.mjs b/docker-server.mjs index e3e9f68ee..3b036f7db 100644 --- a/docker-server.mjs +++ b/docker-server.mjs @@ -112,6 +112,14 @@ const upstreamOrigin = upstreamBase ? new URL(upstreamBase).origin : null; // (gitnexus/src/mcp/http-transport.ts). const authToken = process.env.GITNEXUS_SERVE_AUTH_TOKEN?.trim() || null; +// The protocol-layer credential the upstream `serve` expects on /api/mcp when it +// runs with MCP Bearer auth enabled. Set it to the SAME value on both services: +// the edge token is spent here and replaced with this one for MCP requests only +// (see proxyToUpstream). Unset — the default — means no injection, so a backend +// without MCP auth is unaffected. Blank-is-absent follows resolveAuthToken +// (gitnexus/src/mcp/http-transport.ts). Never logged. +const mcpAuthToken = process.env.GITNEXUS_MCP_AUTH_TOKEN?.trim() || null; + // Mirrors the non-loopback refusal in http-transport.ts (startMcpHttpServer), // relocated because the trust boundary is here: an unguarded `serve` behind a // private service is legitimate, an unguarded public proxy is not. @@ -341,11 +349,17 @@ async function proxyToUpstream(req, res) { // talks to this same-origin web service. delete headers.origin; delete headers.referer; - // The edge token is spent here. `serve` reads no Authorization header - // (gitnexus/src/server/mcp-http.ts mounts /api/mcp unguarded), so forwarding - // it would only copy a live credential into another service's logs. Pinned by - // test. + // The edge token is spent here and must never be forwarded: copying + // Authorization would put a live credential into another service's logs. So + // drop it unconditionally first, then — for the MCP route alone, and only + // when a backend token is configured — replace it with that separate + // protocol credential. Unset GITNEXUS_MCP_AUTH_TOKEN (the default) leaves + // every request stripped, as before. The scope is the normalized pathname, + // so a query string can't widen it and /api/mcpfoo doesn't qualify. delete headers.authorization; + const upstreamPath = upstream.pathname; + const isMcpRoute = upstreamPath === '/api/mcp' || upstreamPath.startsWith('/api/mcp/'); + if (isMcpRoute && mcpAuthToken) headers.authorization = `Bearer ${mcpAuthToken}`; headers.host = upstream.host; // Replace, never forward, the inbound chain (see clientAddressFor). const clientAddress = clientAddressFor(req); diff --git a/docker-server.test.mjs b/docker-server.test.mjs index 80e742f7e..6d2a9f6c2 100644 --- a/docker-server.test.mjs +++ b/docker-server.test.mjs @@ -271,6 +271,12 @@ it('does not inject config into static assets', async () => { const TEST_AUTH_TOKEN = 'proxy-test-token-0123456789abcdefghij'; const TEST_BEARER = `Bearer ${TEST_AUTH_TOKEN}`; +// The protocol token the upstream expects on /api/mcp. Deliberately unlike the +// edge token, so "injected the backend credential" and "forwarded the edge one" +// can never both satisfy an assertion. +const TEST_MCP_TOKEN = 'backend-mcp-token-0123456789abcdefghij'; +const TEST_MCP_BEARER = `Bearer ${TEST_MCP_TOKEN}`; + // rawRequest never sends credentials; apiRequest does. In a file whose subject // is who gets let through, no test should pass because a helper quietly // authenticated for it. @@ -376,6 +382,11 @@ async function withProxy( const proc = spawnServerWithEnv(dir, port, { GITNEXUS_UPSTREAM_URL: schemeless ? target : `http://${target}`, GITNEXUS_SERVE_AUTH_TOKEN: TEST_AUTH_TOKEN, + // An ambient GITNEXUS_MCP_AUTH_TOKEN in the developer's shell would make the + // proxy inject one on /api/mcp, so drop it: spawn omits undefined entries, + // which unsets the inherited value. A test that wants injection sets it via + // `env` below. + GITNEXUS_MCP_AUTH_TOKEN: undefined, ...env, }); proc.stderr.setEncoding('utf8'); @@ -969,8 +980,9 @@ it('forwards an /api/* request that carries the correct token', async () => { }); it('strips the Authorization header instead of forwarding the edge token', async () => { - // The token is spent at this hop. `serve` reads no Authorization header, so - // forwarding would only copy a live credential into another service's logs. + // The edge credential is spent and stripped at this hop. Forwarding it + // would copy a live credential into another service's logs. With no + // GITNEXUS_MCP_AUTH_TOKEN configured — the default — nothing replaces it. await withProxy({}, async (port, ctx) => { const res = await apiRequest(port, '/api/mcp', { method: 'POST', body: '{}' }); assert.equal(res.status, 200, 'the request itself must still be proxied'); @@ -978,6 +990,72 @@ it('strips the Authorization header instead of forwarding the edge token', async }); }); +// -- Upstream MCP token injection (GITNEXUS_MCP_AUTH_TOKEN) ----------------- +// +// A backend running protocol-layer MCP auth expects its own Bearer on +// /api/mcp, and the edge credential can't serve as one. Both services are +// configured with the same GITNEXUS_MCP_AUTH_TOKEN; this hop spends the edge +// token and substitutes the backend one, for that route only. + +// Stands in for a `serve` with MCP Bearer auth enabled: only the exact backend +// credential gets through, so a passing two-hop request proves what was sent. +const mcpBackend = (req, res) => { + if (req.headers.authorization !== TEST_MCP_BEARER) { + res.writeHead(401, { 'Content-Type': 'application/json; charset=utf-8' }); + res.end('{"error":"unauthorized"}'); + return; + } + res.writeHead(200, { 'Content-Type': 'application/json; charset=utf-8' }); + res.end('{"ok":true}'); +}; + +it('treats a blank GITNEXUS_MCP_AUTH_TOKEN as unset and still strips', async () => { + const env = { GITNEXUS_MCP_AUTH_TOKEN: ' ' }; + await withProxy({ env }, async (port, ctx) => { + const res = await apiRequest(port, '/api/mcp', { method: 'POST', body: '{}' }); + assert.equal(res.status, 200); + assert.equal(ctx.received.headers.authorization, undefined); + }); +}); + +it('replaces the edge credential with the upstream MCP token on /api/mcp', async () => { + const env = { GITNEXUS_MCP_AUTH_TOKEN: TEST_MCP_TOKEN }; + await withProxy({ upstream: mcpBackend, env }, async (port, ctx) => { + const res = await apiRequest(port, '/api/mcp', { method: 'POST', body: '{}' }); + assert.equal(res.status, 200, 'a backend that demands the MCP token must accept this hop'); + assert.equal(ctx.received.headers.authorization, TEST_MCP_BEARER); + assert.notEqual( + ctx.received.headers.authorization, + TEST_BEARER, + 'the edge credential must never be forwarded', + ); + }); +}); + +it('injects the upstream MCP token on /api/mcp subpaths and ignores the query string', async () => { + const env = { GITNEXUS_MCP_AUTH_TOKEN: TEST_MCP_TOKEN }; + await withProxy({ upstream: mcpBackend, env }, async (port, ctx) => { + for (const path of ['/api/mcp/messages', '/api/mcp?session=abc']) { + const res = await apiRequest(port, path, { method: 'POST', body: '{}' }); + assert.equal(res.status, 200, `${path} must reach the MCP backend authenticated`); + assert.equal(ctx.received.headers.authorization, TEST_MCP_BEARER, path); + } + }); +}); + +it('leaves non-MCP routes stripped when an upstream MCP token is configured', async () => { + // /api/mcpfoo shares a prefix with the MCP route but is not it, and a plain + // API route never carries a protocol credential. + const env = { GITNEXUS_MCP_AUTH_TOKEN: TEST_MCP_TOKEN }; + await withProxy({ env }, async (port, ctx) => { + for (const path of ['/api/mcpfoo', '/api/health']) { + const res = await apiRequest(port, path); + assert.equal(res.status, 200); + assert.equal(ctx.received.headers.authorization, undefined, path); + } + }); +}); + it('never gates static assets behind the token', async () => { // The UI has to load before it can prompt for a token. await withProxy({}, async (port, ctx) => { diff --git a/gitnexus/src/cli/i18n/en.ts b/gitnexus/src/cli/i18n/en.ts index 698e27d4e..f5654d1bf 100644 --- a/gitnexus/src/cli/i18n/en.ts +++ b/gitnexus/src/cli/i18n/en.ts @@ -238,7 +238,7 @@ export const en = { 'help.option.mcp.host': 'HTTP bind address (only with --http). Default: 127.0.0.1 (loopback). Use 0.0.0.0 to expose to all interfaces.', 'help.option.mcp.authToken': - 'Require this bearer token in the Authorization header (only with --http); may also be set via the GITNEXUS_MCP_AUTH_TOKEN env var. Required for a non-loopback bind (--host 0.0.0.0/::), which otherwise refuses to start.', + "Require this bearer token in the Authorization header (only with --http); may also be set via the GITNEXUS_MCP_AUTH_TOKEN env var, which also enables MCP Bearer auth on gitnexus serve's /api/mcp route. Required for a non-loopback bind (--host 0.0.0.0/::), which otherwise refuses to start.", 'help.option.force.confirmation': 'Skip confirmation prompt', 'help.option.uninstall.force': 'Apply the changes (default is a dry-run preview)', 'help.option.clean.all': 'Clean all indexed repos', diff --git a/gitnexus/src/cli/i18n/zh-CN.ts b/gitnexus/src/cli/i18n/zh-CN.ts index 24a6a2ad6..4aff202b6 100644 --- a/gitnexus/src/cli/i18n/zh-CN.ts +++ b/gitnexus/src/cli/i18n/zh-CN.ts @@ -222,7 +222,7 @@ export const zhCN = { 'help.option.mcp.host': 'HTTP 绑定地址(仅与 --http 搭配使用)。默认:127.0.0.1(回环)。使用 0.0.0.0 向所有接口开放。', 'help.option.mcp.authToken': - '要求 Authorization 头携带此 Bearer Token(仅与 --http 搭配使用);也可通过 GITNEXUS_MCP_AUTH_TOKEN 环境变量设置。非回环绑定(--host 0.0.0.0/::)时必填,否则拒绝启动。', + '要求 Authorization 头携带此 Bearer Token(仅与 --http 搭配使用);也可通过 GITNEXUS_MCP_AUTH_TOKEN 环境变量设置,该变量同时为 gitnexus serve 的 /api/mcp 路由启用 MCP Bearer 认证。非回环绑定(--host 0.0.0.0/::)时必填,否则拒绝启动。', 'help.option.force.confirmation': '跳过确认提示', 'help.option.uninstall.force': '应用更改(默认仅为预演预览)', 'help.option.clean.all': '清理所有已索引仓库', diff --git a/gitnexus/src/cli/index.ts b/gitnexus/src/cli/index.ts index 8296787af..8f6a0f2df 100644 --- a/gitnexus/src/cli/index.ts +++ b/gitnexus/src/cli/index.ts @@ -245,7 +245,7 @@ program ) .option( '--auth-token ', - 'Require this bearer token in the Authorization header (only with --http); may also be set via the GITNEXUS_MCP_AUTH_TOKEN env var. Required for a non-loopback bind (--host 0.0.0.0/::), which otherwise refuses to start.', + "Require this bearer token in the Authorization header (only with --http); may also be set via the GITNEXUS_MCP_AUTH_TOKEN env var, which also enables MCP Bearer auth on gitnexus serve's /api/mcp route. Required for a non-loopback bind (--host 0.0.0.0/::), which otherwise refuses to start.", ) .action(createLbugLazyAction(() => import('./mcp.js'), 'mcpCommand')); diff --git a/gitnexus/src/server/api.ts b/gitnexus/src/server/api.ts index bd7ab685c..fc2e7d943 100644 --- a/gitnexus/src/server/api.ts +++ b/gitnexus/src/server/api.ts @@ -39,7 +39,7 @@ import { searchFTSFromLbug } from '../core/search/bm25-index.js'; import { hybridSearch } from '../core/search/hybrid-search.js'; import { ftsDegradedWarning } from '../core/search/fts-indexes.js'; import { LocalBackend } from '../mcp/local/local-backend.js'; -import { mountMCPEndpoints } from './mcp-http.js'; +import { installServeMcpAuth, mountMCPEndpoints } from './mcp-http.js'; import { fileURLToPath } from 'url'; import { isTerminalJobStatus, JobManager, type AnalyzeJobPartialOutcome } from './analyze-job.js'; import { mountSSEProgress } from './sse-progress.js'; @@ -755,6 +755,9 @@ export const createServer = async (port: number, host: string = '127.0.0.1') => }, }), ); + // Optional protocol-layer auth for the MCP route. Keep this before the + // global body parser so rejected requests do not consume the JSON budget. + installServeMcpAuth(app); app.use(express.json({ limit: '10mb' })); // Origin guard for write routes: loopback, the server's own bound host, and diff --git a/gitnexus/src/server/mcp-http.ts b/gitnexus/src/server/mcp-http.ts index cf17bf763..74fd6b706 100644 --- a/gitnexus/src/server/mcp-http.ts +++ b/gitnexus/src/server/mcp-http.ts @@ -9,11 +9,31 @@ */ import type { Express, Request, Response } from 'express'; -import { createStreamableHttpHandler } from '../mcp/http-transport.js'; +import { + createAuthMiddleware, + createStreamableHttpHandler, + resolveAuthToken, +} from '../mcp/http-transport.js'; import type { LocalBackend } from '../mcp/local/local-backend.js'; import { createMcpRepositoryPolicy } from '../mcp/repository-policy.js'; import { logger } from '../core/logger.js'; +/** + * Protect serve's /api/mcp route when the shared MCP bearer token is configured. + * + * This middleware must be installed before Express's global JSON parser so an + * unauthenticated request body is rejected before it is parsed. The standalone + * `gitnexus mcp --http` server resolves the same environment variable. + */ +export function installServeMcpAuth(app: Express, env: NodeJS.ProcessEnv = process.env): boolean { + const authToken = resolveAuthToken(undefined, env); + if (!authToken) return false; + + app.use('/api/mcp', createAuthMiddleware(authToken)); + logger.info('Bearer authentication enabled for serve /api/mcp'); + return true; +} + export async function mountMCPEndpoints( app: Express, backend: LocalBackend, diff --git a/gitnexus/test/unit/mcp-http-transport.test.ts b/gitnexus/test/unit/mcp-http-transport.test.ts index 7cdf39f9f..d5853a63a 100644 --- a/gitnexus/test/unit/mcp-http-transport.test.ts +++ b/gitnexus/test/unit/mcp-http-transport.test.ts @@ -34,7 +34,7 @@ import { installSignalShutdown, SHUTDOWN_EXIT_CODES, } from '../../src/mcp/server.js'; -import { mountMCPEndpoints } from '../../src/server/mcp-http.js'; +import { installServeMcpAuth, mountMCPEndpoints } from '../../src/server/mcp-http.js'; // ─── Live-HTTP helpers (real req/res for SDK-touching paths) ─────────── @@ -638,6 +638,112 @@ describe('createSseHandlers', () => { // ─── mountMCPEndpoints refactor safety ─────────────────────────────── describe('mountMCPEndpoints', () => { + it.each([{}, { GITNEXUS_MCP_AUTH_TOKEN: '' }, { GITNEXUS_MCP_AUTH_TOKEN: ' ' }])( + 'does not install serve auth without a nonblank token (%j)', + (env) => { + const app = { use: vi.fn() }; + + expect(installServeMcpAuth(app as never, env)).toBe(false); + expect(app.use).not.toHaveBeenCalled(); + }, + ); + + it('installs the shared Bearer middleware for serve /api/mcp', () => { + const app = { use: vi.fn() }; + + expect(installServeMcpAuth(app as never, { GITNEXUS_MCP_AUTH_TOKEN: 'serve-secret' })).toBe( + true, + ); + expect(app.use).toHaveBeenCalledTimes(1); + expect(app.use.mock.calls[0]?.[0]).toBe('/api/mcp'); + + const middleware = app.use.mock.calls[0]?.[1] as ( + req: Request, + res: Response, + next: NextFunction, + ) => void; + const missingRes = createMockRes(); + const missingNext = vi.fn(); + middleware(createMockReq(), missingRes, missingNext); + expect(missingRes._status).toBe(401); + expect(missingNext).not.toHaveBeenCalled(); + + const wrongRes = createMockRes(); + const wrongNext = vi.fn(); + middleware(createMockReq({ authorization: 'Bearer wrong-secret' }), wrongRes, wrongNext); + expect(wrongRes._status).toBe(401); + expect(wrongNext).not.toHaveBeenCalled(); + + const validRes = createMockRes(); + const validNext = vi.fn(); + middleware(createMockReq({ authorization: 'Bearer serve-secret' }), validRes, validNext); + expect(validNext).toHaveBeenCalledOnce(); + expect(validRes._status).toBe(200); + }); + + it('wires serve MCP auth before the global JSON body parser', async () => { + const app = express(); + let parsedBodies = 0; + + expect(installServeMcpAuth(app, { GITNEXUS_MCP_AUTH_TOKEN: 'serve-secret' })).toBe(true); + app.use( + express.json({ + limit: '10mb', + verify: () => { + parsedBodies += 1; + }, + }), + ); + app.all('/api/mcp', (_req: Request, res: Response) => { + res.status(204).end(); + }); + app.post('/api/other', (req: Request, res: Response) => { + res.status(200).json(req.body); + }); + + const { port, close } = await listen(app); + const json = { 'Content-Type': 'application/json' }; + const payload = JSON.stringify({ jsonrpc: '2.0', method: 'tools/list', id: 1 }); + + try { + const missing = await request(port, 'POST', '/api/mcp', json, payload); + expect(missing.status).toBe(401); + expect(JSON.parse(missing.body)).toMatchObject({ + jsonrpc: '2.0', + error: { code: -32001, message: 'Unauthorized' }, + }); + expect(parsedBodies).toBe(0); + + const wrong = await request( + port, + 'POST', + '/api/mcp', + { ...json, Authorization: 'Bearer wrong-secret' }, + payload, + ); + expect(wrong.status).toBe(401); + expect(parsedBodies).toBe(0); + + const valid = await request( + port, + 'POST', + '/api/mcp', + { ...json, Authorization: 'Bearer serve-secret' }, + payload, + ); + expect(valid.status).toBe(204); + expect(parsedBodies).toBe(1); + + // The auth gate is scoped to /api/mcp: other routes stay unauthenticated and parsed. + const other = await request(port, 'POST', '/api/other', json, JSON.stringify({ ok: true })); + expect(other.status).toBe(200); + expect(JSON.parse(other.body)).toEqual({ ok: true }); + expect(parsedBodies).toBe(2); + } finally { + await close(); + } + }); + it('returns a cleanup function', async () => { const backend = createMockBackend(); const mockApp = { diff --git a/render.yaml b/render.yaml index 977765108..124574872 100644 --- a/render.yaml +++ b/render.yaml @@ -17,7 +17,9 @@ projects: environments: - name: production services: - # Private: no public URL. `serve` has no authentication of its own. + # Private: no public URL. `serve`'s own protocol auth (MCP Bearer) is + # optional and unset by this Blueprint; the public edge token on the + # web service below remains the access control. - type: pserv name: gitnexus-server runtime: docker From 94f67d79d5ac1e83dd0c7baa7de05c3b9b562439 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sun, 30 Aug 2026 18:07:57 +0100 Subject: [PATCH 02/26] fix(analyze): make incremental analyze skip the derived layers it can reuse (#3016) (#3102) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(analyze): make incremental analyze skip the derived layers it can reuse (#3016) A warm incremental run only ever wrote a handful of files, but it still paid for the whole graph on the way out: Leiden ran over every node, flow extraction re-derived every process, and all FTS indexes were dropped and rebuilt from scratch. On a small edit that tail dominated the run, which is why "incremental" did not feel incremental. Reuse what the previous run already derived when the write plan allows it. The pipeline holds back community detection and flow extraction whenever the persisted metadata says this run is a candidate for a surgical write; the DB keeps its Community/Process rows instead of a wipe-and-rewrite; and the FTS sweep is narrowed to the indexes the run actually has to touch. The bet is placed before the pipeline and settled after it. Any plan that turns out to need a freshly derived layer — full rebuild, escalated write, or an incremental diff with deleted files — runs the held-back phases through `runDeferredDerivedPhases`, against the same graph and phase outputs, so its output is identical to never having skipped them. Correctness details worth naming, since each one silently loses data if got wrong: - The MEMBER_OF / STEP_IN_PROCESS edges of the changed files are snapshotted before the DETACH DELETE and reattached after the subgraph load. Both endpoints are matched by explicit label: `labels(n)[0]` over an unlabelled match returns an empty string on this engine, which produced a snapshot that restored nothing. - The FTS narrowing unions three sets — what the writeback deletes (a DB probe, because a symbol the edit removed is in no fresh graph but is still a row), what it inserts (the fresh graph), and what is missing right now (else a prior escalation's dropped indexes would never come back). An unreadable index catalog withdraws the narrowing entirely. - Deletions disqualify reuse outright: persisted derived rows can reference nodes this run removes, and nothing short of re-deriving can tell which. Covered by the existing incremental suites, including the incremental-equals-force byte-equivalence test and the #2589 drop-before-delete ordering test, plus unit tests for the new helpers. Co-authored-by: Cursor * fix(analyze): address #3102 review on derived reuse and FTS narrowing Re-run Leiden/flows unless the file-hash diff is empty, restore ENTRY_POINT_OF on the preserve path, always drop class_fts before Spring synthetic Class DML, and reject seeded duplicate phase names. Prettier and exact FTS drop-ordering assertions unblock CI and pin the #2589/#3016 contract. Co-authored-by: Cursor * refactor(analyze): reuse FileHashDiff for derived-layer preserve Drop the count DTO, share phase-name uniqueness, and remove the File FTS sentinel that Class already makes unreachable. Refs #3102 Co-authored-by: Cursor * style(analyze): prettier-wrap shouldPreservePersistedDerivedGraph quality / format failed on the Pick signature wrapping. Refs #3102 Co-authored-by: Cursor --------- Co-authored-by: Gergo Magyar Co-authored-by: Cursor --- .../src/core/incremental/derived-writeback.ts | 83 ++++++++ .../src/core/incremental/subgraph-extract.ts | 13 +- .../core/ingestion/pipeline-phases/runner.ts | 55 +++-- gitnexus/src/core/ingestion/pipeline.ts | 69 ++++++- gitnexus/src/core/lbug/lbug-adapter.ts | 188 +++++++++++++++++- gitnexus/src/core/run-analyze.ts | 145 +++++++++++++- gitnexus/src/core/search/fts-indexes.ts | 39 +++- gitnexus/src/types/pipeline.ts | 12 ++ gitnexus/test/unit/fts-indexes.test.ts | 37 ++++ .../incremental-derived-writeback.test.ts | 98 +++++++++ .../incremental-fts-drop-ordering.test.ts | 26 ++- .../unit/incremental-subgraph-extract.test.ts | 13 ++ .../ingestion/pipeline-phase-registry.test.ts | 24 +++ gitnexus/test/unit/pipeline-runner.test.ts | 72 +++++++ 14 files changed, 828 insertions(+), 46 deletions(-) create mode 100644 gitnexus/src/core/incremental/derived-writeback.ts create mode 100644 gitnexus/test/unit/incremental-derived-writeback.test.ts diff --git a/gitnexus/src/core/incremental/derived-writeback.ts b/gitnexus/src/core/incremental/derived-writeback.ts new file mode 100644 index 000000000..9cec191eb --- /dev/null +++ b/gitnexus/src/core/incremental/derived-writeback.ts @@ -0,0 +1,83 @@ +/** + * Incremental derived-layer writeback helpers (#3016). + * + * The derived layers — Leiden communities, execution flows, and the FTS + * indexes — are graph-wide, so every analyze run rebuilt all three in full no + * matter how small the diff. A surgical incremental write can instead: + * - drop and rebuild only the FTS indexes whose tables hold rows in the + * write set (LadybugDB still cannot DML a table with a live FTS index — + * #2589 — so a table being written must still lose its index first); + * - leave the untouched tables' rows alone, so their indexes stay live; + * - reuse persisted Community/Process rows only when the file-hash diff is + * empty (no added, changed, or deleted files). Any content change can + * add, rename, or retarget symbols that Leiden and flow extraction + * consume — a no-deletion edit is not a validity proof. + */ +import { FTS_INDEXES } from '../search/fts-schema.js'; +import type { KnowledgeGraph } from '../graph/types.js'; +import type { FileHashDiff } from '../../storage/file-hash.js'; + +const FTS_TABLE_NAMES: ReadonlySet = new Set(FTS_INDEXES.map((i) => i.table)); + +/** The FTS-backed members of `tables`. */ +export const ftsTablesAmong = (tables: Iterable): Set => { + const out = new Set(); + for (const table of tables) { + if (FTS_TABLE_NAMES.has(table)) out.add(table); + } + return out; +}; + +/** + * Whether a surgical incremental write may reuse the persisted derived layer. + * + * Deletions disqualify it: the persisted Community/Process rows and their + * MEMBER_OF / STEP_IN_PROCESS edges can reference nodes that no longer exist + * after this run, and nothing short of re-deriving can tell which. + * + * Added or content-changed files also disqualify it: they can introduce, + * rename, or retarget symbols and CALLS edges that Leiden and flow extraction + * consume. File-deletion-only was too weak a proof that the derived graph is + * still valid. + */ +export const shouldPreservePersistedDerivedGraph = ( + diff: Pick, +): boolean => diff.deleted.length === 0 && diff.added.length === 0 && diff.changed.length === 0; + +/** + * FTS-backed node tables that the fresh graph will WRITE rows into for + * `fileSet` — the inserting half of the DML. + * + * Callers must union this with a DB probe for the deleting half + * (`nodeTablesWithRowsForFiles`): a table whose last row in these files was + * just removed by the edit has nothing here, but still holds a stale row that + * the writeback must delete, and deleting it means taking its index down too. + */ +export const incrementalFtsTablesFromGraph = ( + graph: KnowledgeGraph, + fileSet: ReadonlySet, +): Set => { + const touched = new Set(); + graph.forEachNode((n) => { + const filePath = n.properties?.filePath as string | undefined; + if (!filePath || !fileSet.has(filePath)) return; + if (FTS_TABLE_NAMES.has(n.label)) touched.add(n.label); + }); + return touched; +}; + +/** + * The node tables an incremental DETACH DELETE should target, given the FTS + * tables this run is rebuilding. + * + * Every non-FTS table (Folder, CodeElement, …) deletes as before. An FTS-backed + * table only deletes when its index is being rebuilt anyway, because deleting + * from it otherwise would mean DML against a live FTS index (#2589). + */ +export const nodeTablesForIncrementalDelete = ( + allNodeTables: readonly string[], + rebuildingFtsTables: ReadonlySet, +): string[] => + allNodeTables.filter( + (tableName) => !FTS_TABLE_NAMES.has(tableName) || rebuildingFtsTables.has(tableName), + ); diff --git a/gitnexus/src/core/incremental/subgraph-extract.ts b/gitnexus/src/core/incremental/subgraph-extract.ts index e0f0e41eb..276dba4e2 100644 --- a/gitnexus/src/core/incremental/subgraph-extract.ts +++ b/gitnexus/src/core/incremental/subgraph-extract.ts @@ -6,9 +6,9 @@ * replaced, produce a smaller KnowledgeGraph that contains: * * - Every node whose `properties.filePath` is in `toWriteSet`. - * - Every graph-wide node (Community, Process, and Spring metadata - * placeholders) — these are regenerated each run and must be fully - * rewritten. + * - Graph-wide Community/Process nodes unless `includeDerivedGraphWide` + * is false (#3016 incremental preserve). Spring metadata placeholders + * are always included. * - Every relationship where AT LEAST ONE endpoint is in the writable * set above. Relationships entirely between unchanged-file nodes * are skipped — their rows are still in the DB and re-inserting @@ -122,13 +122,18 @@ const indexNodeFilePaths = (fullGraph: KnowledgeGraph): Map => { export const extractChangedSubgraph = ( fullGraph: KnowledgeGraph, toWriteSet: ReadonlySet, + options?: { includeDerivedGraphWide?: boolean }, ): KnowledgeGraph => { const sub = createKnowledgeGraph(); const writableNodeIds = new Set(); + const includeDerivedGraphWide = options?.includeDerivedGraphWide !== false; + fullGraph.forEachNode((n: GraphNode) => { const filePath = n.properties?.filePath as string | undefined; - const include = (filePath && toWriteSet.has(filePath)) || isGraphWideNode(n); + const derivedWide = + includeDerivedGraphWide || (n.label !== 'Community' && n.label !== 'Process'); + const include = (filePath && toWriteSet.has(filePath)) || (isGraphWideNode(n) && derivedWide); if (include) { sub.addNode(n); writableNodeIds.add(n.id); diff --git a/gitnexus/src/core/ingestion/pipeline-phases/runner.ts b/gitnexus/src/core/ingestion/pipeline-phases/runner.ts index 0bfc45bd4..da8e4dd8f 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/runner.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/runner.ts @@ -16,23 +16,36 @@ import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js'; import { isDev } from '../utils/env.js'; import { logger } from '../../logger.js'; + +function assertUniquePhaseNames(phases: readonly PipelinePhase[]): void { + const seen = new Set(); + for (const phase of phases) { + if (seen.has(phase.name)) { + throw new Error(`Duplicate phase name: '${phase.name}'`); + } + seen.add(phase.name); + } +} + /** * Validate that the phases form a valid dependency graph (no cycles, all deps present). * Returns phases in topological execution order. + * + * `satisfied` names phases whose results are already available (a deferred + * follow-up run over the same context, #3016). Their edges are dropped rather + * than validated, because they are resolved by definition. */ -function topologicalSort(phases: readonly PipelinePhase[]): PipelinePhase[] { - const phaseMap = new Map(); - for (const phase of phases) { - if (phaseMap.has(phase.name)) { - throw new Error(`Duplicate phase name: '${phase.name}'`); - } - phaseMap.set(phase.name, phase); - } +function topologicalSort( + phases: readonly PipelinePhase[], + satisfied: ReadonlySet = new Set(), +): PipelinePhase[] { + assertUniquePhaseNames(phases); + const phaseMap = new Map(phases.map((p) => [p.name, p])); // Validate all deps exist for (const phase of phases) { for (const dep of phase.deps) { - if (!phaseMap.has(dep)) { + if (!phaseMap.has(dep) && !satisfied.has(dep)) { throw new Error(`Phase '${phase.name}' depends on '${dep}', which is not registered`); } } @@ -43,8 +56,9 @@ function topologicalSort(phases: readonly PipelinePhase[]): PipelinePhase[] { const reverseDeps = new Map(); for (const phase of phases) { - inDegree.set(phase.name, phase.deps.length); - for (const dep of phase.deps) { + const pendingDeps = phase.deps.filter((dep) => !satisfied.has(dep)); + inDegree.set(phase.name, pendingDeps.length); + for (const dep of pendingDeps) { let rev = reverseDeps.get(dep); if (!rev) { rev = []; @@ -143,15 +157,30 @@ function findCyclePath( * * @param phases All phases to execute (order doesn't matter — sorted internally) * @param ctx Shared pipeline context + * @param seed Results of phases that already ran against this same context, + * available to `phases` as dependencies (#3016 deferred derived + * phases). Included in the returned map. * @returns Map of phase name → PhaseResult (all completed phases) */ export async function runPipeline( phases: readonly PipelinePhase[], ctx: PipelineContext, + seed?: ReadonlyMap>, ): Promise>> { + // A seeded phase has already run against this context; re-running it would + // apply its graph writes a second time. "Already ran" is the whole meaning of + // the seed, so honour it here rather than making every caller pre-filter. + const satisfied = new Set(seed?.keys() ?? []); let sorted: PipelinePhase[]; try { - sorted = topologicalSort(phases); + // Duplicate names must be rejected on the caller-supplied list *before* + // seed-filtering. Filtering first would drop a seeded duplicate and let + // `topologicalSort` see a unique name (#3102). + assertUniquePhaseNames(phases); + sorted = topologicalSort( + phases.filter((p) => !satisfied.has(p.name)), + satisfied, + ); } catch (err) { // Emit a terminal 'error' progress event for graph-validation failures // (cycle detected, duplicate phase, missing dep) so CLI/MCP consumers see @@ -171,7 +200,7 @@ export async function runPipeline( } throw err; } - const results = new Map>(); + const results = new Map>(seed); for (const phase of sorted) { const start = Date.now(); diff --git a/gitnexus/src/core/ingestion/pipeline.ts b/gitnexus/src/core/ingestion/pipeline.ts index 050822e87..440bf689c 100644 --- a/gitnexus/src/core/ingestion/pipeline.ts +++ b/gitnexus/src/core/ingestion/pipeline.ts @@ -46,6 +46,7 @@ import { PhaseRegistry, type ScopeResolutionOutput, type PipelinePhase, + type PipelineContext, type CommunitiesOutput, type ProcessesOutput, } from './pipeline-phases/index.js'; @@ -58,6 +59,12 @@ export interface PipelineOptions { * to retain those nodes under `skipGraphPhases`. */ skipGraphPhases?: boolean; + /** + * Skip only Leiden community detection and process/flow extraction (#3016). + * MRO/DI still run. Used on warm incremental analyze so persisted + * Community/Process rows can be kept instead of wipe+rewrite. + */ + skipDerivedGraphPhases?: boolean; /** Per-advice Spring AOP candidate inspection cap. `0` disables this cap. */ springAopMaxCandidateInspectionsPerAdvice?: number; /** Aggregate Spring AOP candidate inspection cap for one analysis. `0` disables this cap. */ @@ -310,8 +317,12 @@ export function buildPhaseList(options?: PipelineOptions): PipelinePhase[] { .register(mroPhase, { enabledWhen: (o) => !o.skipGraphPhases }) .register(springAopInheritancePhase, { enabledWhen: (o) => !o.skipGraphPhases }) .register(diPhase, { enabledWhen: (o) => !o.skipGraphPhases }) - .register(communitiesPhase, { enabledWhen: (o) => !o.skipGraphPhases }) - .register(processesPhase, { enabledWhen: (o) => !o.skipGraphPhases }) + .register(communitiesPhase, { + enabledWhen: (o) => !o.skipGraphPhases && o.skipDerivedGraphPhases !== true, + }) + .register(processesPhase, { + enabledWhen: (o) => !o.skipGraphPhases && o.skipDerivedGraphPhases !== true, + }) // Normalize a missing options object once here so phase predicates above // take a required PipelineOptions and need no `?.` guard (#2080 review S1). .build(options ?? {}) @@ -351,18 +362,19 @@ export const runPipelineFromRepo = async ( } const phases = buildPhaseList(options); + const ctx: PipelineContext = { + repoPath, + graph: graphEmitSink ?? graph, + onProgress, + options, + pipelineStart, + graphEmit: graphEmitSink, + }; let graphEmitManifest: GraphEmitManifest | undefined; let results; try { - results = await runPipeline(phases, { - repoPath, - graph: graphEmitSink ?? graph, - onProgress, - options, - pipelineStart, - graphEmit: graphEmitSink, - }); + results = await runPipeline(phases, ctx); graphEmitManifest = graphEmitSink?.finalize(); } finally { // Release per-pair fds when the pipeline threw before finalize ran. @@ -412,7 +424,7 @@ export const runPipelineFromRepo = async ( }, }); - return { + const result: PipelineResult = { // The RAW graph, deliberately — NOT `graphEmitSink`. Phases above received // the sink so their reads are complete, but `loadGraphToLbug` feeds this to // `streamAllCSVsToDisk`, and the sink's complete iterator would then emit @@ -434,4 +446,39 @@ export const runPipelineFromRepo = async ( pdgEmitManifest, propertyInference, }; + + // #3016: hand back a way to run the derived phases `skipDerivedGraphPhases` + // held back. Which phases those are is answered by re-asking the registry + // with only that flag cleared — the one form of the question that stays + // correct when a different predicate (`skipGraphPhases`) also disables them, + // since then they are absent for a reason a deferred run cannot fix and the + // filter yields nothing. The sink guard mirrors the `graph` note above: a + // streaming run is a full rebuild, which never sets the skip flag, so an + // active sink here means the two got combined by mistake — and deferred + // phases writing into a finalized sink would emit past its manifest. + const deferredDerivedPhases = + options?.skipDerivedGraphPhases === true && graphEmitSink === undefined + ? buildPhaseList({ ...options, skipDerivedGraphPhases: false }).filter( + (p) => (p.name === 'communities' || p.name === 'processes') && !results.has(p.name), + ) + : []; + + if (deferredDerivedPhases.length > 0) { + result.runDeferredDerivedPhases = async () => { + const derived = await runPipeline(deferredDerivedPhases, ctx, results); + // Presence-checked for the same reason as the block above: a phase the + // registry filtered out is absent, and `getPhaseOutput` throws on absent. + if (derived.has('communities')) { + result.communityResult = getPhaseOutput( + derived, + 'communities', + ).communityResult; + } + if (derived.has('processes')) { + result.processResult = getPhaseOutput(derived, 'processes').processResult; + } + }; + } + + return result; }; diff --git a/gitnexus/src/core/lbug/lbug-adapter.ts b/gitnexus/src/core/lbug/lbug-adapter.ts index 3058c3d81..fe563aa4c 100644 --- a/gitnexus/src/core/lbug/lbug-adapter.ts +++ b/gitnexus/src/core/lbug/lbug-adapter.ts @@ -2575,7 +2575,11 @@ export const DELETE_FILES_CHUNK_SIZE = 200; */ export const deleteNodesForFiles = async ( filePaths: readonly string[], - options: { onChunk?: (filesDone: number, filesTotal: number) => void } = {}, + options: { + onChunk?: (filesDone: number, filesTotal: number) => void; + /** When set, only these node tables are DETACH DELETEd (#3016). */ + nodeTables?: readonly string[]; + } = {}, ): Promise => { if (!conn) { throw new Error('LadybugDB not initialized. Call initLbug first.'); @@ -2619,7 +2623,8 @@ export const deleteNodesForFiles = async ( ); } } - for (const tableName of NODE_TABLES) { + const tables = options.nodeTables ?? NODE_TABLES; + for (const tableName of tables) { // Community/Process are graph-wide (no filePath); the orchestrator // drops them wholesale via deleteAllCommunitiesAndProcesses. if (tableName === 'Community' || tableName === 'Process') continue; @@ -2636,6 +2641,185 @@ export const deleteNodesForFiles = async ( } }; +/** + * Which of `candidateTables` currently hold at least one row for `filePaths`. + * + * The incremental writeback uses this to decide which FTS-backed tables it is + * about to DML (#3016). It has to be a question about the DB, not about the + * freshly built graph: an edit that DELETES the last Rust trait in a file + * leaves no Trait node in the new graph, but the old row is still in the index + * and still has to be deleted — and its FTS index still has to come down first. + */ +export const nodeTablesWithRowsForFiles = async ( + filePaths: readonly string[], + candidateTables: readonly string[], +): Promise> => { + const c = conn; + if (!c) { + throw new Error('LadybugDB not initialized. Call initLbug first.'); + } + const found = new Set(); + return withConnLock(async () => { + for (const batch of chunk(filePaths, DELETE_FILES_CHUNK_SIZE)) { + const listLiteral = `[${batch.map((p) => formatCypherValue(p)).join(', ')}]`; + for (const tableName of candidateTables) { + // Graph-wide tables have no filePath column to filter on. + if (tableName === 'Community' || tableName === 'Process') continue; + if (found.has(tableName)) continue; + // determinism: probe — asks only whether the table has any row for + // these files, so which row comes back cannot change the answer. + const queryResult = await c.query( + `MATCH (n:${escapeTableName(tableName)}) WHERE n.filePath IN ${listLiteral} ` + + `RETURN n.id LIMIT 1`, + ); + try { + const result = Array.isArray(queryResult) ? queryResult[0] : queryResult; + if ((await result.getAll()).length > 0) found.add(tableName); + } finally { + await closeQueryResults(queryResult); + } + } + } + return found; + }); +}; + +/** + * The graph-wide derived edges and the node table each one points at. Both are + * produced by the derived phases (Leiden, flow extraction) rather than by + * parsing, which is why an incremental run that skips those phases has to carry + * them across the writeback itself. + */ +const DERIVED_REL_KINDS = [ + { type: 'MEMBER_OF', targetLabel: 'Community' }, + { type: 'STEP_IN_PROCESS', targetLabel: 'Process' }, + { type: 'ENTRY_POINT_OF', targetLabel: 'Process' }, +] as const; + +/** + * One MEMBER_OF / STEP_IN_PROCESS edge, carrying everything needed to recreate + * it byte-for-byte: both endpoint labels (so the re-MATCH is label-scoped + * rather than a scan of every node) and every column of the relationship + * table, `step` included — process traces order by it (`ORDER BY r.step`), so + * an edge restored without it silently scrambles the flow it belongs to. + */ +export interface DerivedRelSnapshot { + sourceId: string; + sourceLabel: string; + targetId: string; + targetLabel: string; + type: string; + confidence: number; + reason: string; + step: number; +} + +/** + * Capture the MEMBER_OF / STEP_IN_PROCESS / ENTRY_POINT_OF edges owned by `filePaths`, before a + * surgical incremental write DETACH DELETEs their file-side endpoints (#3016). + * + * Only meaningful on the write plan that keeps the persisted Community/Process + * nodes: those nodes survive the delete, but the edges tying this run's changed + * files to them do not, and the pipeline did not re-derive them. + * + * Both endpoints are matched by an EXPLICIT label — `sourceTables` on one side, + * the edge type's fixed target table on the other — so the labels come from the + * query rather than the rows. `labels(n)[0]` over an unlabelled match returns + * an empty string on this engine, which silently produced a snapshot that + * restored nothing. + * + * Read failures propagate. This runs against a warm index whose derived tables + * the caller has already established exist, so a failure here is a real fault — + * and swallowing it would drop the edges silently, which looks identical to a + * repo that genuinely has no communities. + */ +export const snapshotDerivedRelsForFiles = async ( + filePaths: readonly string[], + sourceTables: readonly string[], +): Promise => { + const c = conn; + if (!c) { + throw new Error('LadybugDB not initialized. Call initLbug first.'); + } + const out: DerivedRelSnapshot[] = []; + return withConnLock(async () => { + for (const batch of chunk(filePaths, DELETE_FILES_CHUNK_SIZE)) { + const listLiteral = `[${batch.map((p) => formatCypherValue(p)).join(', ')}]`; + for (const sourceLabel of sourceTables) { + if (sourceLabel === 'Community' || sourceLabel === 'Process') continue; + for (const { type, targetLabel } of DERIVED_REL_KINDS) { + const queryResult = await c.query( + `MATCH (n:${escapeTableName(sourceLabel)})-[r:${REL_TABLE_NAME}]->` + + `(m:${escapeTableName(targetLabel)}) ` + + `WHERE n.filePath IN ${listLiteral} AND r.type = ${formatCypherValue(type)} ` + + `RETURN n.id AS sourceId, m.id AS targetId, ` + + `r.confidence AS confidence, r.reason AS reason, r.step AS step`, + ); + try { + const result = Array.isArray(queryResult) ? queryResult[0] : queryResult; + for (const row of await result.getAll()) { + const rec = row as Record; + if (typeof rec.sourceId !== 'string' || typeof rec.targetId !== 'string') continue; + out.push({ + sourceId: rec.sourceId, + sourceLabel, + targetId: rec.targetId, + targetLabel, + type, + confidence: typeof rec.confidence === 'number' ? rec.confidence : 1.0, + reason: typeof rec.reason === 'string' ? rec.reason : '', + step: + typeof rec.step === 'number' + ? rec.step + : typeof rec.step === 'bigint' + ? Number(rec.step) + : 0, + }); + } + } finally { + await closeQueryResults(queryResult); + } + } + } + } + return out; + }); +}; + +/** + * Re-create the edges captured by `snapshotDerivedRelsForFiles`, after the + * incremental subgraph load has put their file-side endpoints back. + * + * Endpoints are matched by label + id, mirroring `fallbackRelationshipInserts`: + * an unlabelled `MATCH (a), (b)` is a cartesian product over the whole graph + * and does not finish on a real index. An endpoint the load did not restore + * simply matches nothing, so the edge is dropped rather than mis-attached. + */ +export const restoreDerivedRels = async (rels: readonly DerivedRelSnapshot[]): Promise => { + const c = conn; + if (!c) { + throw new Error('LadybugDB not initialized. Call initLbug first.'); + } + if (rels.length === 0) return; + const escapeLabel = (label: string): string => + BACKTICK_TABLES.has(label) ? `\`${label}\`` : label; + // No outer `withConnLock`: `queryAndDrain` takes the lock per statement, and + // wrapping the loop as well trips the re-entry guard in conn-lock.ts. Same + // shape as `fallbackRelationshipInserts`, the other per-edge CREATE loop. + for (const rel of rels) { + if (!NODE_TABLES.includes(rel.sourceLabel as NodeTableName)) continue; + if (!NODE_TABLES.includes(rel.targetLabel as NodeTableName)) continue; + await queryAndDrain( + c, + `MATCH (a:${escapeLabel(rel.sourceLabel)} {id: ${formatCypherValue(rel.sourceId)}}), ` + + `(b:${escapeLabel(rel.targetLabel)} {id: ${formatCypherValue(rel.targetId)}}) ` + + `CREATE (a)-[:${REL_TABLE_NAME} {type: ${formatCypherValue(rel.type)}, ` + + `confidence: ${rel.confidence}, reason: ${formatCypherValue(rel.reason)}, ` + + `step: ${rel.step}}]->(b)`, + ); + } +}; + export const getEmbeddingTableName = (): string => EMBEDDING_TABLE_NAME; /** diff --git a/gitnexus/src/core/run-analyze.ts b/gitnexus/src/core/run-analyze.ts index ad9841406..71fece179 100644 --- a/gitnexus/src/core/run-analyze.ts +++ b/gitnexus/src/core/run-analyze.ts @@ -36,6 +36,9 @@ import { closeLbugBeforeExit, loadCachedEmbeddings, deleteNodesForFiles, + nodeTablesWithRowsForFiles, + snapshotDerivedRelsForFiles, + restoreDerivedRels, ensureEmbeddingRowDmlSafe, ensureFtsRowDmlSafe, readIndexCatalogSnapshot, @@ -67,6 +70,7 @@ import { createSearchFTSIndexes, summarizeFtsIndexBuildFailures, dropSearchFTSIndexes, + missingSearchFTSIndexTables, initialiseSearchFTSStemmer, verifySearchFTSIndexes, } from './search/fts-indexes.js'; @@ -136,6 +140,13 @@ import { } from './incremental/subgraph-extract.js'; import { shadowCandidatesFor } from './incremental/shadow-candidates.js'; import { shouldEscalateIncrementalWrite } from './incremental/escalation-gate.js'; +import { + ftsTablesAmong, + incrementalFtsTablesFromGraph, + nodeTablesForIncrementalDelete, + shouldPreservePersistedDerivedGraph, +} from './incremental/derived-writeback.js'; +import { NODE_TABLES } from './lbug/schema.js'; import { loadParseCache, saveParseCache, @@ -1885,6 +1896,25 @@ async function runFullAnalysisInner( // `resolveStreamPdgEmit` — read fresh at the same point — behaves.) const streamGraphEmitActive = resolveStreamGraphEmit(options); + // #3016: hold back Leiden and flow extraction when the persisted metadata + // says this run is a candidate for a surgical incremental write, whose + // derived layer is reused rather than recomputed. Deliberately the same + // conditions as the `isIncremental` decision below MINUS the two that only + // the pipeline can answer (the analysis-feature re-check and a non-empty + // file list), so this is a superset: every run that turns out incremental + // had the phases skipped, and the runs that do not are caught by + // `runDeferredDerivedPhases` once the write plan is known. Excluded on the + // streaming path because that is a full rebuild by construction, and the + // deferred phases must not write into a finalized emit sink. + const skipDerivedGraphPhases = + !streamGraphEmitActive && + !options.force && + !!existingMeta && + !!existingMeta.fileHashes && + Object.keys(existingMeta.fileHashes).length > 0 && + repoHasGit && + !schemaFingerprintMismatch(existingMeta.schemaFingerprint); + // ── Phase 1: Full Pipeline (0–60%) ──────────────────────────────── const pipelineResult = await runPipelineFromRepo( repoPath, @@ -1927,6 +1957,7 @@ async function runFullAnalysisInner( ? resolveNativeSafeStorageDir(storagePath, 'graph-csv') : undefined, fetchWrappers: options.fetchWrappers, + skipDerivedGraphPhases, }, ); @@ -1986,6 +2017,28 @@ async function runFullAnalysisInner( ? diffFileHashes(newFileHashes, existingMeta!.fileHashes) : undefined; + // #3016: `skipDerivedGraphPhases` was decided BEFORE the pipeline, from the + // persisted metadata alone, so it can only ever be a bet that this run stays + // surgical. Settle the bet here, where `isIncremental` and the deletion set + // are both known, and pay it off by running the held-back phases whenever the + // write plan needs a freshly derived layer: + // - not incremental → full rebuild writes the whole graph, and a graph + // with no Community/Process nodes would publish an + // index with no communities and no flows; + // - added/changed/deleted files → the persisted derived layer can miss new + // symbols, keep stale memberships, or reference + // removed ids. Only an empty file-hash diff is a + // proof that Leiden/flows still match. + const preserveDerivedLayer = + skipDerivedGraphPhases && + isIncremental && + !!hashDiff && + shouldPreservePersistedDerivedGraph(hashDiff); + if (skipDerivedGraphPhases && !preserveDerivedLayer) { + progress('communities', 58, 'Detecting code communities and flows...'); + await pipelineResult.runDeferredDerivedPhases?.(); + } + // #2 atomic index publish: on a full rebuild, build the fresh DB at a temp // path and swap it over the live index in one rename at the very end, so a // concurrent MCP reader opening mid-build only ever sees the previous @@ -2196,6 +2249,7 @@ async function runFullAnalysisInner( // collapse check compares the whole in-memory graph against the whole DB, // which is only a like-for-like comparison on a full rebuild. let wroteChangedSubgraphOnly = false; + let incrementalFtsRebuildTables: Set | undefined; if (isIncremental && hashDiff) { // ── Incremental DB writeback ─────────────────────────────────── // 0. Expand the writable set with transitive importers of @@ -2490,6 +2544,14 @@ async function runFullAnalysisInner( ); if (extensionForcedRebuild || sizeForcedRebuild) { escalatedFullWrite = true; + // #3016: escalation converts this run into a wipe + full bulk COPY of + // the in-memory graph, so the derived layer the skip was betting on + // preserving has to exist in that graph after all. Same reasoning as + // the not-incremental branch above, just discovered later. + if (preserveDerivedLayer) { + progress('communities', 63, 'Detecting code communities and flows...'); + await pipelineResult.runDeferredDerivedPhases?.(); + } // Every live cause is named, not just the first: a DB can carry BOTH a // vector index and FTS indexes, and reporting one cause while the other // is equally fatal is how #2841 stayed mis-diagnosed for so long. §5.D: @@ -2701,7 +2763,53 @@ async function runFullAnalysisInner( // in between — so re-reading would only weaken the one-read invariant // the snapshot type exists to enforce. if (buildPath === lbugPath) liveIndexMutationStarted = true; - await dropSearchFTSIndexes(indexCatalogRows); + // FTS narrowing is independent of Leiden/flow reuse: even when this + // run re-derives communities, Ladybug still cannot DML a live FTS + // index (#2589), so only the tables this write set touches should + // lose their index. The probe is a question about the DB rather than + // the fresh graph — a symbol the edit DELETED is in no fresh graph + // but is still a row that has to go. + const tablesWithRows = await nodeTablesWithRowsForFiles(filesToDelete, NODE_TABLES); + // Narrowing 1 — the FTS sweep, from "every configured index" to "the + // indexes this run must touch". Three sources, and dropping any one of + // them strands something: + // - what the writeback DELETES (the probe above), because a symbol + // the edit removed is in no fresh graph but is still a row; + // - what it INSERTS (the fresh graph), because inserting under a live + // FTS index is the same #2589 hazard as deleting under one; + // - what is MISSING right now, because narrowing to the written + // tables would otherwise leave keyword search degraded forever on + // tables whose index a previous escalation dropped — the next full + // rebuild would be the only thing that ever restored them. + // An unreadable catalog proves nothing about that third set, so it + // withdraws the narrowing entirely rather than guess. + const missingFts = await missingSearchFTSIndexTables(indexCatalogRows); + const touchedFts = missingFts + ? new Set([ + ...ftsTablesAmong(tablesWithRows), + ...incrementalFtsTablesFromGraph(pipelineResult.graph, new Set(filesToDelete)), + ...missingFts, + ]) + : undefined; + // Graph-wide Spring synthetic Class nodes are DETACH DELETEd on this + // branch even when Class is not in the write set + // (`deleteSpringAutoConfigurationSyntheticClasses`). Always include + // Class so class_fts is not live across that DML (#2589), including + // when the fresh graph no longer materializes the synthetics but the + // DB still holds them. + if (touchedFts) { + touchedFts.add('Class'); + } + incrementalFtsRebuildTables = touchedFts; + // MEMBER_OF / STEP_IN_PROCESS / ENTRY_POINT_OF edges hang off the nodes + // the DETACH DELETE below removes, so preserving the Community/Process + // nodes preserves only half the layer unless these are reattached after + // the subgraph write puts the member nodes back. Only the probed tables + // can own such an edge, so they are the only ones worth scanning. + const derivedSnapshot = preserveDerivedLayer + ? await snapshotDerivedRelsForFiles(filesToDelete, [...tablesWithRows]) + : []; + await dropSearchFTSIndexes(indexCatalogRows, incrementalFtsRebuildTables); // 1b. Remove the write set's existing rows — batched (#2409): one // DETACH DELETE per table per 200-file chunk. The former per-file // loop issued a count + delete per table per FILE — ~13k @@ -2716,6 +2824,9 @@ async function runFullAnalysisInner( await deleteNodesForFiles(filesToDelete, { onChunk: (done, total) => progress('lbug', 62, `Removing rows for changed files (${done}/${total})...`), + nodeTables: incrementalFtsRebuildTables + ? nodeTablesForIncrementalDelete(NODE_TABLES, incrementalFtsRebuildTables) + : undefined, }); // Surgical path: Phase 3.5 restores exactly these files' embedding // rows (FIX 3). Sound because deleteNodesForFiles propagates errors @@ -2723,10 +2834,12 @@ async function runFullAnalysisInner( // deterministically — and this process holds the exclusive DB lock, // so no concurrent writer can disturb the derivation. deletedFilePathsForRestore = new Set(filesToDelete); - // 2. Drop graph-wide nodes (Community, Process). They'll be re-inserted - // from the fresh pipeline output below. Required for the - // "Leiden runs on the FULL graph" correctness invariant. - await deleteAllCommunitiesAndProcesses(); + if (!preserveDerivedLayer) { + // 2. Drop graph-wide nodes (Community, Process). They'll be re-inserted + // from the fresh pipeline output below. Required for the + // "Leiden runs on the FULL graph" correctness invariant. + await deleteAllCommunitiesAndProcesses(); + } // 2a. Drop INJECTS edges (DI collection injection, #2200) — their // validity is a whole-program property (a third-file change to the // interface or an implementer creates/invalidates edges between two @@ -2774,7 +2887,9 @@ async function runFullAnalysisInner( // only that. Unchanged-file rows in the DB stay untouched. Pass // the SAME effectiveWriteSet so the subgraph and the deletes // cover identical files (asymmetry would silently corrupt). - const subgraph = extractChangedSubgraph(pipelineResult.graph, effectiveWriteSet); + const subgraph = extractChangedSubgraph(pipelineResult.graph, effectiveWriteSet, { + includeDerivedGraphWide: !preserveDerivedLayer, + }); wroteChangedSubgraphOnly = true; await saveIncrementalDirtyState('load-graph', { importerExpansion, @@ -2787,6 +2902,9 @@ async function runFullAnalysisInner( const pct = Math.min(84, 65 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 19)); progress('lbug', pct, msg); }); + if (preserveDerivedLayer && derivedSnapshot.length > 0) { + await restoreDerivedRels(derivedSnapshot); + } } // Boundary drain (#2409): checkpoint at the end of the incremental @@ -2846,6 +2964,7 @@ async function runFullAnalysisInner( // pre-existing row (#2544/#2546) must not discard this run's otherwise- // successful graph/embeddings work — only keyword search degrades. const ftsResult = await buildSearchIndexesOrDegrade(executeQuery, { + tables: incrementalFtsRebuildTables, onIndexStart: options.verbose ? (table, indexName) => log(`FTS: creating ${table}.${indexName}`) : undefined, @@ -3600,8 +3719,11 @@ async function runFullAnalysisInner( files: pipelineResult.totalFileCount, nodes: stats.nodes, edges: stats.edges, - communities: pipelineResult.communityResult?.stats.totalCommunities, - processes: pipelineResult.processResult?.stats.totalProcesses, + communities: + pipelineResult.communityResult?.stats.totalCommunities ?? + existingMeta?.stats?.communities, + processes: + pipelineResult.processResult?.stats.totalProcesses ?? existingMeta?.stats?.processes, embeddings: persistedEmbeddingCount, }, capabilities: { @@ -3830,9 +3952,12 @@ async function runFullAnalysisInner( files: pipelineResult.totalFileCount, nodes: stats.nodes, edges: stats.edges, - communities: pipelineResult.communityResult?.stats.totalCommunities, + communities: + pipelineResult.communityResult?.stats.totalCommunities ?? + existingMeta?.stats?.communities, clusters: aggregatedClusterCount, - processes: pipelineResult.processResult?.stats.totalProcesses, + processes: + pipelineResult.processResult?.stats.totalProcesses ?? existingMeta?.stats?.processes, }, undefined, { diff --git a/gitnexus/src/core/search/fts-indexes.ts b/gitnexus/src/core/search/fts-indexes.ts index b53a96778..951734d4e 100644 --- a/gitnexus/src/core/search/fts-indexes.ts +++ b/gitnexus/src/core/search/fts-indexes.ts @@ -158,6 +158,12 @@ export const SUPPORTED_FTS_STEMMERS: ReadonlySet = new Set([ export interface CreateSearchFTSIndexesOptions { onIndexStart?: (table: string, indexName: string) => void; onIndexReady?: (table: string, indexName: string) => void; + /** + * When set, only these node-table names are dropped/rebuilt (#3016). + * Omit to rebuild every configured FTS index (full analyze / deleted-file + * incremental / `--repair-fts`). + */ + tables?: ReadonlySet; } let resolvedStemmer: string | undefined; @@ -219,7 +225,10 @@ export function getSearchFTSStemmer(): string { * contract, and the same one-shared-`SHOW_INDEXES`-read purpose, as the gates in * `lbug-adapter.ts`. Omit it to have the sweep read the catalog itself. */ -export async function dropSearchFTSIndexes(indexRows?: IndexCatalogSnapshot): Promise { +export async function dropSearchFTSIndexes( + indexRows?: IndexCatalogSnapshot, + tables?: ReadonlySet, +): Promise { // One catalog read for the whole sweep, decided PER CONFIGURED INDEX on // IDENTITY (#2841 cleanup review). `undefined` = the catalog could not be // read, which proves nothing — attempt every drop rather than skip a real one, @@ -240,6 +249,7 @@ export async function dropSearchFTSIndexes(indexRows?: IndexCatalogSnapshot): Pr // whether the sweep ran or not. const rows = await resolveGateRows(indexRows); for (const { table, indexName } of FTS_INDEXES) { + if (tables && !tables.has(table)) continue; // Skip only what the catalog POSITIVELY proves absent. Without this, a // machine whose FTS extension cannot load, analyzing a DB that never carried // an FTS index, pays one failed `CALL DROP_FTS_INDEX` per configured table on @@ -257,6 +267,32 @@ export async function dropSearchFTSIndexes(indexRows?: IndexCatalogSnapshot): Pr } } +/** + * The configured FTS tables whose index the catalog proves is ABSENT right now. + * + * `undefined` means the catalog could not be read, which proves nothing — the + * same fail-closed reading the sweep above applies. Callers narrowing a rebuild + * to a subset of tables (#3016) must union this in, or must not narrow at all + * when it is `undefined`: a run that rebuilds only the tables it wrote leaves + * keyword search permanently degraded on every table whose index went missing + * earlier (a prior escalation drops all of them, and only the next full rebuild + * would ever put them back). + */ +export async function missingSearchFTSIndexTables( + indexRows?: IndexCatalogSnapshot, +): Promise | undefined> { + const rows = await resolveGateRows(indexRows); + if (rows === undefined) return undefined; + const missing = new Set(); + for (const { table, indexName } of FTS_INDEXES) { + const present = rows.some( + (row) => indexRowTable(row) === table && indexRowName(row) === indexName, + ); + if (!present) missing.add(table); + } + return missing; +} + /** One configured index that could not be (re)built, and why. */ export interface FtsIndexBuildFailure { table: string; @@ -290,6 +326,7 @@ export async function createSearchFTSIndexes( const stemmer = getSearchFTSStemmer(); const failures: FtsIndexBuildFailure[] = []; for (const { table, indexName, properties } of FTS_INDEXES) { + if (options?.tables && !options.tables.has(table)) continue; options?.onIndexStart?.(table, indexName); // Drop first so the live `properties` always win. `createFTSIndex` is // idempotent-by-name (skips when the index already exists), so without the diff --git a/gitnexus/src/types/pipeline.ts b/gitnexus/src/types/pipeline.ts index 5c11f800b..960517416 100644 --- a/gitnexus/src/types/pipeline.ts +++ b/gitnexus/src/types/pipeline.ts @@ -15,6 +15,18 @@ export interface PipelineResult { totalFileCount: number; communityResult?: CommunityDetectionResult; processResult?: ProcessDetectionResult; + /** + * Runs the community/process phases that `skipDerivedGraphPhases` held back + * (#3016), against the same graph and phase outputs the pipeline already + * produced, and populates `communityResult`/`processResult` on this object. + * + * Present ONLY when those phases were skipped for that reason, so a caller + * that optimistically skipped them can still get a byte-identical derived + * layer on the paths that turn out to need one (full rebuild, escalated + * write, or an incremental run with deleted files). Absent means the phases + * either already ran or were disabled for an unrelated reason. + */ + runDeferredDerivedPhases?: () => Promise; /** * Additive diagnostics for registry-primary resolution decisions that * deliberately suppress edge emission. Empty means no diagnostic was diff --git a/gitnexus/test/unit/fts-indexes.test.ts b/gitnexus/test/unit/fts-indexes.test.ts index c3cfea224..915abc5db 100644 --- a/gitnexus/test/unit/fts-indexes.test.ts +++ b/gitnexus/test/unit/fts-indexes.test.ts @@ -33,6 +33,7 @@ const { createSearchFTSIndexes, getSearchFTSStemmer, initialiseSearchFTSStemmer, + missingSearchFTSIndexTables, } = await import('../../src/core/search/fts-indexes.js'); const { FTS_INDEXES } = await import('../../src/core/search/fts-schema.js'); const { createFTSIndex } = await import('../../src/core/lbug/lbug-adapter.js'); @@ -65,6 +66,16 @@ describe('createSearchFTSIndexes', () => { expect(calls).toEqual(expected); }); + it('rebuilds only the requested tables when options.tables is set (#3016)', async () => { + await createSearchFTSIndexes({ tables: new Set(['File', 'Function']) }); + expect(calls).toEqual([ + 'drop:File.file_fts', + 'create:File.file_fts:porter', + 'drop:Function.function_fts', + 'create:Function.function_fts:porter', + ]); + }); + it('invokes onIndexStart/onIndexReady once per index', async () => { const started: string[] = []; const ready: string[] = []; @@ -196,6 +207,32 @@ describe('buildSearchIndexesOrDegrade', () => { }); }); +describe('missingSearchFTSIndexTables (#3016)', () => { + const catalogRow = (i: { table: string; indexName: string }) => ({ + table_name: i.table, + index_name: i.indexName, + }); + + it('reports nothing missing when the catalog carries every configured index', async () => { + const missing = await missingSearchFTSIndexTables(FTS_INDEXES.map(catalogRow)); + expect(missing).toEqual(new Set()); + }); + + it('names every table when the catalog is empty (a prior escalation dropped them all)', async () => { + const missing = await missingSearchFTSIndexTables([]); + expect(missing).toEqual(new Set(FTS_INDEXES.map((i) => i.table))); + }); + + it('names only the tables whose index is absent', async () => { + const rows = FTS_INDEXES.filter((i) => i.table !== 'Function').map(catalogRow); + expect(await missingSearchFTSIndexTables(rows)).toEqual(new Set(['Function'])); + }); + + it('answers undefined when the catalog could not be read, so callers do not narrow', async () => { + expect(await missingSearchFTSIndexTables(undefined)).toBeUndefined(); + }); +}); + describe('getSearchFTSStemmer', () => { it('defaults to porter when unset', () => { expect(getSearchFTSStemmer()).toBe('porter'); diff --git a/gitnexus/test/unit/incremental-derived-writeback.test.ts b/gitnexus/test/unit/incremental-derived-writeback.test.ts new file mode 100644 index 000000000..36242eac2 --- /dev/null +++ b/gitnexus/test/unit/incremental-derived-writeback.test.ts @@ -0,0 +1,98 @@ +import { describe, expect, it } from 'vitest'; +import type { GraphNode } from 'gitnexus-shared'; +import { NODE_TABLES } from 'gitnexus-shared'; +import { createKnowledgeGraph } from '../../src/core/graph/graph.js'; +import { + ftsTablesAmong, + incrementalFtsTablesFromGraph, + nodeTablesForIncrementalDelete, + shouldPreservePersistedDerivedGraph, +} from '../../src/core/incremental/derived-writeback.js'; + +const node = (id: string, label: string, filePath: string): GraphNode => + ({ + id, + label, + properties: { filePath, name: id }, + }) as unknown as GraphNode; + +describe('shouldPreservePersistedDerivedGraph (#3016)', () => { + const empty = { deleted: [] as string[], added: [] as string[], changed: [] as string[] }; + + it('is true only when the file-hash diff is empty', () => { + expect(shouldPreservePersistedDerivedGraph(empty)).toBe(true); + }); + + it('is false when any file was deleted (old labels are not in the fresh graph)', () => { + expect(shouldPreservePersistedDerivedGraph({ ...empty, deleted: ['gone.ts'] })).toBe(false); + }); + + it('is false when a file was added (new symbols have no persisted membership)', () => { + expect(shouldPreservePersistedDerivedGraph({ ...empty, added: ['new.ts'] })).toBe(false); + }); + + it('is false when a file changed (in-file add/rename/CALLS can change Leiden/flows)', () => { + expect(shouldPreservePersistedDerivedGraph({ ...empty, changed: ['a.ts'] })).toBe(false); + }); +}); + +describe('incrementalFtsTablesFromGraph', () => { + it('returns only FTS tables that have write-set nodes', () => { + const g = createKnowledgeGraph(); + g.addNode(node('f', 'File', 'a.ts')); + g.addNode(node('fn', 'Function', 'a.ts')); + g.addNode(node('tr', 'Trait', 'b.rs')); + const touched = incrementalFtsTablesFromGraph(g, new Set(['a.ts'])); + expect([...touched].sort()).toEqual(['File', 'Function']); + }); + + it('ignores labels that are not FTS-indexed', () => { + const g = createKnowledgeGraph(); + g.addNode(node('folder', 'Folder', 'src')); + const touched = incrementalFtsTablesFromGraph(g, new Set(['src'])); + expect(touched.size).toBe(0); + }); + + it('cannot see a table whose last row the edit removed — hence the DB probe', () => { + // The graph is what the run WILL write. A trait deleted by this edit is + // absent here but still a row in the index, so on its own this answer + // would leave that row behind with a live index over it. run-analyze + // unions this with nodeTablesWithRowsForFiles for exactly that reason. + const g = createKnowledgeGraph(); + g.addNode(node('f', 'File', 'a.rs')); + const touched = incrementalFtsTablesFromGraph(g, new Set(['a.rs'])); + expect(touched.has('Trait')).toBe(false); + }); +}); + +describe('ftsTablesAmong', () => { + it('keeps the FTS-backed tables and drops the rest', () => { + expect([...ftsTablesAmong(['File', 'Folder', 'Function'])].sort()).toEqual([ + 'File', + 'Function', + ]); + }); + + it('is empty for a probe that found only non-indexed tables', () => { + expect(ftsTablesAmong(['Folder']).size).toBe(0); + }); +}); + +describe('nodeTablesForIncrementalDelete', () => { + it('keeps the FTS tables being rebuilt and drops the rest from the delete', () => { + const tables = nodeTablesForIncrementalDelete(NODE_TABLES, new Set(['File', 'Function'])); + expect(tables).toContain('File'); + expect(tables).toContain('Function'); + expect(tables).not.toContain('Trait'); + }); + + it('never withholds a non-FTS table, whatever is being rebuilt', () => { + const tables = nodeTablesForIncrementalDelete(NODE_TABLES, new Set(['File'])); + expect(tables).toContain('Folder'); + }); + + it('targets every FTS table when every FTS index is being rebuilt', () => { + const tables = nodeTablesForIncrementalDelete(NODE_TABLES, new Set(NODE_TABLES)); + expect(tables).toEqual([...NODE_TABLES]); + }); +}); diff --git a/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts b/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts index e2f5c3c87..0fa8c6af3 100644 --- a/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts +++ b/gitnexus/test/unit/incremental-fts-drop-ordering.test.ts @@ -5,8 +5,8 @@ * PREVIOUS run's index. This drives the real `runFullAnalysis` incremental * path (real git repo, real LadybugDB, real FTS extension) and asserts, * at the moment `deleteNodesForFiles` is invoked, that `SHOW_INDEXES()` - * already reports every FTS index absent — proving the drop-before-delete - * ordering end-to-end rather than only unit-testing the call sequence. + * already reports FTS indexes for tables that will be DML'd as absent + * (#2589 drop-before-delete). #3016: empty-language FTS tables may remain. */ import { readFile, writeFile } from 'fs/promises'; import { execSync } from 'child_process'; @@ -67,7 +67,7 @@ describe('runFullAnalysis incremental writeback — FTS drop-before-delete order vi.resetModules(); }); - it('SHOW_INDEXES() reports every FTS index absent by the time deleteNodesForFiles runs', async () => { + it('SHOW_INDEXES() reports the FTS indexes of every table being written as absent by the time deleteNodesForFiles runs', async () => { const lbugAdapter = await import('../../src/core/lbug/lbug-adapter.js'); const { runFullAnalysis } = await import('../../src/core/run-analyze.js'); @@ -128,8 +128,24 @@ describe('runFullAnalysis incremental writeback — FTS drop-before-delete order await runFullAnalysis(repo.dbPath, { skipAgentsMd: true }, { onProgress: () => {} }); expect(indexNamesAtDeleteTime).toBeDefined(); - for (const { indexName } of FTS_INDEXES) { - expect(indexNamesAtDeleteTime).not.toContain(indexName); + // handler.ts writes File/Function/Class/Method. The incremental write set + // also pulls importer-expanded mini-repo files (index.ts re-exports + // handler; validator.ts holds Interface + Property), so those FTS + // indexes must be down before DETACH DELETE (#2589). Every other + // configured FTS index must still be live (#3016 narrowing). + const down = new Set([ + 'file_fts', + 'function_fts', + 'class_fts', + 'method_fts', + 'interface_fts', + 'property_fts', + ]); + const configured = FTS_INDEXES.map((i) => i.indexName); + const ftsAtDelete = indexNamesAtDeleteTime!.filter((name) => configured.includes(name)); + expect([...ftsAtDelete].sort()).toEqual(configured.filter((name) => !down.has(name)).sort()); + for (const name of down) { + expect(ftsAtDelete).not.toContain(name); } } finally { await repo.cleanup(); diff --git a/gitnexus/test/unit/incremental-subgraph-extract.test.ts b/gitnexus/test/unit/incremental-subgraph-extract.test.ts index 4841bba3a..c1b3a2cf5 100644 --- a/gitnexus/test/unit/incremental-subgraph-extract.test.ts +++ b/gitnexus/test/unit/incremental-subgraph-extract.test.ts @@ -74,6 +74,19 @@ describe('extractChangedSubgraph', () => { expect(sub.nodes.map((n) => n.id).sort()).toEqual(['comm-1', 'proc-1']); }); + it('omits Community/Process when includeDerivedGraphWide is false (#3016)', () => { + const g = createKnowledgeGraph(); + g.addNode(makeFileNode('a', '/repo/a.ts')); + g.addNode(makeWideNode('comm-1', 'Community')); + g.addNode(makeWideNode('proc-1', 'Process')); + + const sub = extractChangedSubgraph(g, new Set(['/repo/a.ts']), { + includeDerivedGraphWide: false, + }); + + expect(sub.nodes.map((n) => n.id).sort()).toEqual(['a']); + }); + it('always includes Spring auto-configuration synthetic Class nodes', () => { const g = createKnowledgeGraph(); g.addNode({ diff --git a/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts b/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts index 5b8c446f9..913ba68a4 100644 --- a/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts +++ b/gitnexus/test/unit/ingestion/pipeline-phase-registry.test.ts @@ -109,6 +109,30 @@ describe('buildPhaseList parity (registry refactor, #2080)', () => { WITHOUT_GRAPH_PHASES, ); }); + + it('skipDerivedGraphPhases:true → omits communities/processes but keeps mro/di (#3016)', () => { + const names = buildPhaseList({ skipDerivedGraphPhases: true }).map((p) => p.name); + expect(names).toContain('mro'); + expect(names).toContain('di'); + expect(names).not.toContain('communities'); + expect(names).not.toContain('processes'); + }); + + it('skipDerivedGraphPhases holds back exactly the two derived phases (#3016)', () => { + // runPipelineFromRepo recovers the deferred set by diffing these two lists, + // so anything else the flag removed would be silently un-deferrable. + const skipped = buildPhaseList({ skipDerivedGraphPhases: true }).map((p) => p.name); + const full = buildPhaseList({ skipDerivedGraphPhases: false }).map((p) => p.name); + expect(full.filter((n) => !skipped.includes(n))).toEqual(['communities', 'processes']); + }); + + it('skipDerivedGraphPhases defers nothing that skipGraphPhases already removed (#3016)', () => { + const both = buildPhaseList({ skipGraphPhases: true, skipDerivedGraphPhases: true }).map( + (p) => p.name, + ); + const graphPhasesOnly = buildPhaseList({ skipGraphPhases: true }).map((p) => p.name); + expect(both).toEqual(graphPhasesOnly); + }); }); // --------------------------------------------------------------------------- diff --git a/gitnexus/test/unit/pipeline-runner.test.ts b/gitnexus/test/unit/pipeline-runner.test.ts index bddde7bd1..c29fdd7f7 100644 --- a/gitnexus/test/unit/pipeline-runner.test.ts +++ b/gitnexus/test/unit/pipeline-runner.test.ts @@ -400,6 +400,78 @@ describe('runPipeline', () => { }); }); +describe('runPipeline with seeded results (#3016 deferred derived phases)', () => { + const seeded = (name: string, output: unknown): ReadonlyMap> => + new Map([[name, { phaseName: name, output, durationMs: 0 }]]); + + it('satisfies a dependency from the seed instead of demanding the phase', async () => { + const later: PipelinePhase = { + name: 'later', + deps: ['earlier'], + async execute(_ctx, deps) { + return `${getPhaseOutput(deps, 'earlier')}+later`; + }, + }; + + const results = await runPipeline([later], makeCtx(), seeded('earlier', 'earlierOutput')); + + expect(getPhaseOutput(results, 'later')).toBe('earlierOutput+later'); + }); + + it('returns the seeded results alongside the newly run ones', async () => { + const later: PipelinePhase = { + name: 'later', + deps: ['earlier'], + execute: async () => 'x', + }; + + const results = await runPipeline([later], makeCtx(), seeded('earlier', 'earlierOutput')); + + expect([...results.keys()].sort()).toEqual(['earlier', 'later']); + }); + + it('does not re-run a seeded phase', async () => { + let ran = 0; + const earlier: PipelinePhase = { + name: 'earlier', + deps: [], + async execute() { + ran++; + return 'fresh'; + }, + }; + + await runPipeline([earlier], makeCtx(), seeded('earlier', 'seeded')); + + // The seed already carries this phase's output, so the runner must treat it + // as a duplicate registration rather than silently executing it twice. + expect(ran).toBe(0); + }); + + it('still rejects a dependency that is neither registered nor seeded', async () => { + const later: PipelinePhase = { + name: 'later', + deps: ['missing'], + execute: async () => 'x', + }; + + await expect(runPipeline([later], makeCtx(), seeded('earlier', 'e'))).rejects.toThrow( + /depends on 'missing', which is not registered/, + ); + }); + + it('rejects duplicate phase names even when one copy is also seeded', async () => { + const dup: PipelinePhase = { + name: 'earlier', + deps: [], + async execute() {}, + }; + await expect(runPipeline([dup, dup], makeCtx(), seeded('earlier', 'seed'))).rejects.toThrow( + /Duplicate phase name/, + ); + }); +}); + describe('getPhaseOutput', () => { it('retrieves typed output from dependency map', () => { const deps = new Map>(); From 72edf400871c1589ceb975ad868909389249606a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sun, 30 Aug 2026 21:50:20 +0100 Subject: [PATCH 03/26] perf(store): V8 sidecars plus hardlinked ParsedFile restore (#3099) * perf(store): add best-effort V8 sidecars beside canonical JSON caches Warm ParsedFile and parse-cache loads skip JSON.parse when a sidecar is present. JSON remains authoritative: envelope validation plus v8.deserialize decide the hit, and any failure falls back without reparsing. Co-authored-by: Cursor * fix(store): require generation bind or sidecar drop before cache overwrite A same-length JSON rewrite could accept a leftover V8 sidecar if both generation rotation and unlink failed. Refuse the new generation unless at least one of those invalidations succeeds; skip publishing a sidecar when only the drop succeeded. detect_changes --scope all: 7 files, risk low, no affected processes. tsc --noEmit clean; 115/115 relevant unit tests; cache-related integration tests pass. parse-impl-env-reads worker-ready timeout is pre-existing (same 5 failures with this change set stashed). ESLint 0 errors; remaining warnings are pre-existing and not on changed lines. Co-authored-by: Cursor * chore(autofix): apply prettier + eslint fixes via /autofix command * refactor(store): share V8 overwrite invalidation across persist paths The bind-or-drop gate lived in five writers. One helper keeps the protocol in a single place and lets bind/drop run together on the async path. detect_changes --scope all: 3 files, risk low, no affected processes. Co-authored-by: Cursor * perf(store): hardlink durable ParsedFile shards into the run store Warm restore of parsedfile-cache into parsedfile-store now publishes all four shard files via fs.link, falling back to copy-into-tmp + rename so a leftover dest hardlink can never be written through. JSON remains the canonical cache; V8 sidecars ride the same path. Co-authored-by: Cursor * perf(store): load immutable V8 shards in place, drop JSON fallback Warm analyze was still paying JSON.parse plus a restore copy. One .v8 envelope per shard and SCHEMA_BUMP 81 make a miss re-extract instead of serving a stale JSON twin. Co-authored-by: Cursor * fix(store): validate durable V8 warm-cache restores Reject incomplete or corrupt durable generations and snapshot valid shards before skipping parse workers, preserving ParsedFiles when persistence fails. Co-authored-by: Cursor * refactor(store): drop unused durable load path Load ParsedFiles only from the run-store snapshot and share one checksummed payload reader so inspect and deserialize stay consistent. Co-authored-by: Cursor --------- Co-authored-by: Gergo Magyar Co-authored-by: Cursor Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gitnexus/bench/v8-sidecar/measure.mjs | 91 ++++ .../ingestion/pipeline-phases/parse-impl.ts | 45 +- .../core/ingestion/workers/parse-worker.ts | 6 +- gitnexus/src/storage/fs-atomic.ts | 89 +++- gitnexus/src/storage/parse-cache.ts | 49 +- gitnexus/src/storage/parsedfile-store.ts | 471 ++++++------------ gitnexus/src/storage/v8-sidecar.ts | 363 ++++++++++++++ .../integration/cfg/parse-cache-mixed.test.ts | 27 +- .../test/unit/incremental-parse-cache.test.ts | 147 +++++- ...mpl-warm-cache-parsedfile-coverage.test.ts | 200 ++++++-- gitnexus/test/unit/parsedfile-store.test.ts | 306 +++++------- ...repo-manager-registry-atomic-write.test.ts | 22 +- gitnexus/test/unit/storage/fs-atomic.test.ts | 128 ++++- gitnexus/test/unit/v8-sidecar.test.ts | 204 ++++++++ 14 files changed, 1510 insertions(+), 638 deletions(-) create mode 100644 gitnexus/bench/v8-sidecar/measure.mjs create mode 100644 gitnexus/src/storage/v8-sidecar.ts create mode 100644 gitnexus/test/unit/v8-sidecar.test.ts diff --git a/gitnexus/bench/v8-sidecar/measure.mjs b/gitnexus/bench/v8-sidecar/measure.mjs new file mode 100644 index 000000000..68867b9ba --- /dev/null +++ b/gitnexus/bench/v8-sidecar/measure.mjs @@ -0,0 +1,91 @@ +#!/usr/bin/env node +/** + * Optional V8 sidecar warm-load bench (#3089). + * + * Not part of `npm test`. Measures repeated warm loads of the `.v8` ParsedFile + * shards already on disk through the production loader. Replay of identical + * shards is throughput-only — it is not unique-object scale. + * + * Copies the store into a temporary workspace first. The source cache is + * never mutated. + * + * Usage (from gitnexus/): + * node --expose-gc --import tsx bench/v8-sidecar/measure.mjs + */ +import { cp, mkdtemp, readdir, rm } from 'node:fs/promises'; +import { tmpdir } from 'node:os'; +import path from 'node:path'; +import { performance } from 'node:perf_hooks'; +import { loadParsedFilesForPaths } from '../../src/storage/parsedfile-store.ts'; +import { inspectV8Cache } from '../../src/storage/v8-sidecar.ts'; + +const srcStorage = process.argv[2]; +if (!srcStorage) { + console.error('usage: node --expose-gc --import tsx bench/v8-sidecar/measure.mjs '); + process.exit(2); +} + +const srcStoreDir = path.join(srcStorage, 'parsedfile-store'); +const benchRoot = await mkdtemp(path.join(tmpdir(), 'gnx-v8-bench-')); +const storeDir = path.join(benchRoot, 'parsedfile-store'); +const PATH_SOURCE_SHARDS = 8; +const RUNS = 3; + +try { + await cp(srcStoreDir, storeDir, { recursive: true }); + + const names = (await readdir(storeDir)) + .filter((f) => f.endsWith('.v8') && !f.includes('.v8.')) + .sort(); + if (names.length === 0) { + throw new Error( + `no .v8 ParsedFile shards in ${srcStoreDir} — run an analyze that populates the store first`, + ); + } + + const want = new Set(); + let sourceShards = 0; + for (const name of names) { + const inspected = await inspectV8Cache(path.join(storeDir, name)); + if (!inspected) continue; + sourceShards++; + for (const filePath of inspected.paths) want.add(filePath); + if (sourceShards >= PATH_SOURCE_SHARDS) break; + } + if (want.size === 0) { + throw new Error( + `no file paths readable from ${names.length} shard(s) in ${srcStoreDir} — shards may be from another Node/V8 runtime, so re-analyze with this runtime`, + ); + } + + const rss = () => Math.round(process.memoryUsage().rss / 1024 / 1024); + const heap = () => Math.round(process.memoryUsage().heapUsed / 1024 / 1024); + + const run = async (label) => { + if (typeof globalThis.gc === 'function') globalThis.gc(); + const t0 = performance.now(); + const loaded = await loadParsedFilesForPaths(benchRoot, want); + const ms = Math.round(performance.now() - t0); + if (loaded.size !== want.size) { + throw new Error(`incomplete V8 load: requested ${want.size} paths but loaded ${loaded.size}`); + } + if (typeof globalThis.gc === 'function') globalThis.gc(); + console.log( + JSON.stringify({ + label, + shards: names.length, + wantPaths: want.size, + files: loaded.size, + ms, + rssMiB: rss(), + heapUsedMiB: heap(), + }), + ); + }; + + for (let i = 1; i <= RUNS; i++) { + await run(`v8-load-${i}`); + } +} finally { + await rm(benchRoot, { recursive: true, force: true }); +} diff --git a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts index b6ca75ea0..352fd0f2b 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/parse-impl.ts @@ -36,7 +36,7 @@ import { getDurableParsedFileDir, loadDurableParsedFileIndex, prepareDurableParsedFileChunk, - restoreDurableParsedFileShard, + durableChunkHasShards, } from '../../../storage/parsedfile-store.js'; import type { ParseWorkerResult } from '../workers/parse-worker.js'; import { DEFAULT_PDG_MAX_FUNCTION_LINES } from '../cfg/collect.js'; @@ -739,15 +739,15 @@ export async function runChunkedParseAndResolve( // a sibling of the run-scoped store, NOT cleared per run. Workers write a // shard per chunk hash; on a warm parse-cache hit we restore the chunk's // shards into the run-scoped store so scope-resolution streams them without - // re-parsing. `durableHitKeys` is the prior run's index, version-gated by - // PARSE_CACHE_VERSION (a mismatch ⇒ empty ⇒ every chunk re-dispatches, which - // repopulates the durable store — never the main-thread extract fallback). + // re-parsing. `durableHitEntries` is the prior run's path-coverage index, + // version-gated by PARSE_CACHE_VERSION (a mismatch ⇒ empty ⇒ every chunk + // re-dispatches, which repopulates the durable store). const durableParsedFileDir = parsedFileStorePath !== undefined ? getDurableParsedFileDir(parsedFileStorePath) : undefined; - const durableHitKeys = + const durableHitEntries = durableParsedFileDir !== undefined ? await loadDurableParsedFileIndex(durableParsedFileDir, PARSE_CACHE_VERSION) - : new Set(); + : new Map>(); let chunkCacheHits = 0; let chunkCacheMisses = 0; let reparsedFileCount = 0; @@ -825,11 +825,14 @@ export async function runChunkedParseAndResolve( } if (chunkWorkerData.parsedFiles?.length) { if (parsedFileStorePath) { - await persistParsedFileChunk( + const wrote = await persistParsedFileChunk( parsedFileStorePath, `chunk-${chunkIdx}`, chunkWorkerData.parsedFiles, ); + if (!wrote) { + for (const item of chunkWorkerData.parsedFiles) allParsedFiles.push(item); + } } else { for (const item of chunkWorkerData.parsedFiles) allParsedFiles.push(item); } @@ -1019,8 +1022,16 @@ export async function runChunkedParseAndResolve( // store was introduced, or a pruned/version-stale shard — fall through to // a worker re-dispatch to repopulate them. NEVER let scope-resolution // re-extract on the main thread (the #1983 OOM the durable store closes). + const durableExpectedPaths = + chunkHash === null ? undefined : durableHitEntries.get(chunkHash); const durableHit = - chunkHash !== null && durableParsedFileDir !== undefined && durableHitKeys.has(chunkHash); + cachedRaw !== undefined && + cachedRaw.length > 0 && + chunkHash !== null && + durableParsedFileDir !== undefined && + parsedFileStorePath !== undefined && + durableExpectedPaths !== undefined && + (await durableChunkHasShards(parsedFileStorePath, chunkHash, durableExpectedPaths)); if (cachedRaw && cachedRaw.length > 0 && (durableHit || parsedFileStorePath === undefined)) { // Cache hit: replay cached worker output. Finalize any parked worker @@ -1053,22 +1064,8 @@ export async function runChunkedParseAndResolve( nodesCreated: graph.nodeCount, }, }); - // Restore the chunk's durable ParsedFile shards into the run-scoped - // store so scope-resolution finds full coverage with ZERO main-thread - // re-parse. A verbatim byte copy — byte-identical to a cold run. - if (durableHit && durableParsedFileDir && parsedFileStorePath && chunkHash) { - const restored = await restoreDurableParsedFileShard( - durableParsedFileDir, - parsedFileStorePath, - chunkHash, - ); - if (restored === 0) { - logger.warn( - `parsedfile-cache: durable shards missing for cached chunk ` + - `${chunkHash.slice(0, 8)} — scope-resolution will re-extract these files`, - ); - } - } + // The durable gate already snapshotted warm `.v8` shards into the + // run-scoped store for scope resolution. await applyChunkResults(chunkWorkerData, chunkIdx, chunkFiles, chunkStartMs); } else { // Cache miss: dispatch to workers, capture the raw results, store diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index 29c94e8ee..ab27cf7ac 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -3243,12 +3243,14 @@ parentPort!.on('message', (msg: WorkerIncomingMessage) => { ); } if (PARSED_FILE_STORE_STORAGE_PATH) { - persistParsedFileShardSync( + const wrote = persistParsedFileShardSync( PARSED_FILE_STORE_STORAGE_PATH, `w${threadId}-${seq}`, accumulated.parsedFiles, ); - accumulated.parsedFiles = []; + if (wrote) { + accumulated.parsedFiles = []; + } } } postResultCloneSafe(accumulated); diff --git a/gitnexus/src/storage/fs-atomic.ts b/gitnexus/src/storage/fs-atomic.ts index 7f3070614..e39053fd5 100644 --- a/gitnexus/src/storage/fs-atomic.ts +++ b/gitnexus/src/storage/fs-atomic.ts @@ -6,7 +6,7 @@ * core/group/ import (the established direction is core/group/ -> storage/, * e.g. core/group/service.ts already imports loadMeta from here). */ -import fsp from 'fs/promises'; +import { closeSync, openSync, promises as fsp, renameSync, unlinkSync, writeSync } from 'node:fs'; import { randomBytes } from 'crypto'; const RETRY_CODES = new Set(['EBUSY', 'EPERM', 'EACCES']); @@ -57,7 +57,7 @@ export async function writeFileAtomic( data: string, attempts?: number, ): Promise { - const tmpPath = `${targetPath}.tmp.${randomBytes(8).toString('hex')}`; + const tmpPath = tmpBeside(targetPath); const handle = await fsp.open(tmpPath, 'wx', 0o600); try { try { @@ -71,3 +71,88 @@ export async function writeFileAtomic( throw err; } } + +function tmpBeside(targetPath: string): string { + return `${targetPath}.tmp.${randomBytes(8).toString('hex')}`; +} + +/** + * Publish `src` at `dst` without ever writing through the inode currently + * named by `dst` (#3090). Fast path is `fs.link`. Any failure (EEXIST, EXDEV, + * EPERM, unsupported hardlinks) copies into a unique sibling and + * {@link retryRename}s over `dst`. POSIX rename of a file over an existing + * file unlinks the old name; it does not open or truncate that inode. + * + * Invariant: may replace the destination directory entry; must never modify + * the inode that entry currently references. There is no dest unlink and no + * in-place `copyFile(src, dst)`. + */ +export async function linkOrCopyFile(src: string, dst: string): Promise { + try { + await fsp.link(src, dst); + return; + } catch { + /* any link failure → publish a new inode via tmp + rename */ + } + const tmp = tmpBeside(dst); + try { + await fsp.copyFile(src, tmp); + await retryRename(tmp, dst); + } catch (err) { + await fsp.unlink(tmp).catch(() => {}); + throw err; + } +} + +/** + * Binary sibling of {@link writeFileAtomic}. Same random tmp + `'wx'` + `0o600` + * + rename contract; used for optional V8 cache sidecars that cannot go through + * a UTF-8 `writeFile`. + */ +export async function writeFileAtomicBytes( + targetPath: string, + data: Uint8Array, + attempts?: number, +): Promise { + const tmpPath = tmpBeside(targetPath); + const handle = await fsp.open(tmpPath, 'wx', 0o600); + try { + try { + await handle.writeFile(data); + } finally { + await handle.close(); + } + await retryRename(tmpPath, targetPath, attempts); + } catch (err) { + await fsp.unlink(tmpPath).catch(() => {}); + throw err; + } +} + +/** + * Sync binary publish for parse workers. Same exclusive-tmp + mode contract as + * {@link writeFileAtomicBytes}; rename is not retried (the worker is not the + * Windows multi-reader case `retryRename` exists for). + */ +export function writeFileAtomicBytesSync(targetPath: string, data: Uint8Array): void { + const tmpPath = tmpBeside(targetPath); + const fd = openSync(tmpPath, 'wx', 0o600); + try { + try { + let offset = 0; + while (offset < data.byteLength) { + offset += writeSync(fd, data, offset); + } + } finally { + closeSync(fd); + } + renameSync(tmpPath, targetPath); + } catch (err) { + try { + unlinkSync(tmpPath); + } catch { + /* leftover tmp is unlinked on the next exclusive create */ + } + throw err; + } +} diff --git a/gitnexus/src/storage/parse-cache.ts b/gitnexus/src/storage/parse-cache.ts index c99f4d87e..efd17a7f6 100644 --- a/gitnexus/src/storage/parse-cache.ts +++ b/gitnexus/src/storage/parse-cache.ts @@ -31,6 +31,7 @@ import path from 'path'; import { fileURLToPath } from 'url'; import { compareCodeUnits } from '../lib/utils.js'; import type { ParseWorkerResult } from '../core/ingestion/workers/parse-worker.js'; +import { copyV8CacheIfPresent, tryLoadV8Cache, writeV8CacheFile } from './v8-sidecar.js'; /** * Cache version composed of: @@ -663,7 +664,11 @@ import type { ParseWorkerResult } from '../core/ingestion/workers/parse-worker.j // `route-extractors/` and `workers/` module content — would close the missing- // bump axis without invalidating on unrelated churn, and is the real follow-up. // RE-CHECK AGAINST origin/main AND OPEN PRs IMMEDIATELY BEFORE MERGING. -const SCHEMA_BUMP = 80; +// 80 -> 81: ParsedFile and parse-cache shards are one immutable `.v8` envelope +// each (no JSON/path/generation siblings). A v80 index still names `.json` +// keys and would skip workers while scope-resolution found nothing — the +// #1983 main-thread reparse. origin/main at allocation is 80. +const SCHEMA_BUMP = 81; const GITNEXUS_PKG_VERSION = (() => { try { // package.json sits at gitnexus/package.json — two levels up from @@ -899,7 +904,7 @@ const getCacheIndexPath = (storagePath: string): string => path.join(getCacheDirPath(storagePath), CACHE_INDEX_FILENAME); const getCacheChunkPath = (storagePath: string, chunkHash: string): string => - path.join(getCacheDirPath(storagePath), `${chunkHash}.json`); + path.join(getCacheDirPath(storagePath), `${chunkHash}.v8`); /** * Drop fields that are not replayed by `mergeChunkResults` / parse-impl after @@ -930,9 +935,12 @@ const readParseCacheChunkFromDisk = async ( ): Promise => { if (!isValidChunkCacheKey(chunkHash)) return undefined; try { - const chunkRaw = await fs.readFile(getCacheChunkPath(storagePath, chunkHash), 'utf-8'); - const chunkData = JSON.parse(chunkRaw, mapReviver) as ParseWorkerResult[]; - return Array.isArray(chunkData) ? chunkData : undefined; + const chunkPath = getCacheChunkPath(storagePath, chunkHash); + const v8Hit = await tryLoadV8Cache(chunkPath); + if (v8Hit?.kind === 'hit' && Array.isArray(v8Hit.value)) { + return v8Hit.value as ParseWorkerResult[]; + } + return undefined; } catch { return undefined; } @@ -975,17 +983,16 @@ export const persistParseCacheChunk = async ( await fs.mkdir(cacheDir, { recursive: true }); createdCacheDirs.add(cacheDir); } - const payload = JSON.stringify(slim, mapReplacer); const chunkPath = getCacheChunkPath(cache.storagePath, chunkHash); - try { - await fs.writeFile(chunkPath, payload, 'utf-8'); - } catch (error) { - if ((error as NodeJS.ErrnoException).code !== 'ENOENT') throw error; - // Long-lived analyze --watch processes can replace the sharded cache - // directory after this process-local memo recorded it as created. + let ok = await writeV8CacheFile(chunkPath, slim); + if (!ok) { await fs.mkdir(cacheDir, { recursive: true }); createdCacheDirs.add(cacheDir); - await fs.writeFile(chunkPath, payload, 'utf-8'); + ok = await writeV8CacheFile(chunkPath, slim); + } + if (!ok) { + cache.entries.set(chunkHash, slim); + return; } cache.onDiskKeys ??= new Set(); cache.onDiskKeys.add(chunkHash); @@ -1089,25 +1096,17 @@ export const saveParseCache = async (storagePath: string, cache: ParseCache): Pr // index from what we persisted, not from the raw usedKeys snapshot. const writtenKeys: string[] = []; for (const chunkHash of keys) { - const chunkPath = path.join(tmpDir, `${chunkHash}.json`); + const chunkPath = path.join(tmpDir, `${chunkHash}.v8`); const inMemory = cache.entries.get(chunkHash); if (inMemory !== undefined) { - let payload: string; - try { - payload = JSON.stringify(inMemory, mapReplacer); - } catch { - continue; + if (await writeV8CacheFile(chunkPath, inMemory)) { + writtenKeys.push(chunkHash); } - await fs.writeFile(chunkPath, payload, 'utf-8'); - writtenKeys.push(chunkHash); continue; } const existingPath = getCacheChunkPath(storagePath, chunkHash); - try { - await fs.copyFile(existingPath, chunkPath); + if (await copyV8CacheIfPresent(existingPath, chunkPath)) { writtenKeys.push(chunkHash); - } catch { - /* shard missing — skip; next run treats as cache miss */ } } diff --git a/gitnexus/src/storage/parsedfile-store.ts b/gitnexus/src/storage/parsedfile-store.ts index 1f34ff083..51e2a9a54 100644 --- a/gitnexus/src/storage/parsedfile-store.ts +++ b/gitnexus/src/storage/parsedfile-store.ts @@ -24,11 +24,11 @@ * * ## Shape * - * `/parsedfile-store/.json` — one shard per parse chunk, - * a JSON array of `ParsedFile` serialized with the same `mapReplacer` the parse - * cache uses (Scope.bindings / Scope.typeBindings are `Map`s). The store is - * cleared at the start of each parse and after scope-resolution consumes it, so - * it never lingers and never goes stale across runs. + * `/parsedfile-store/.v8` — one shard per parse chunk, + * a V8 envelope of `ParsedFile[]` (Scope.bindings / Scope.typeBindings stay + * `Map`s). The store is cleared at the start of each parse and after + * scope-resolution consumes it, so it never lingers and never goes stale + * across runs. * * ## Durable sibling store (`parsedfile-cache/`, warm-cache coverage) * @@ -41,14 +41,14 @@ * that gap we ALSO write the worker's ParsedFiles to a second, CONTENT-ADDRESSED * store keyed by the parse chunk hash (`getDurableParsedFileDir`), which mirrors * the parse cache's lifecycle (persists across runs, pruned by `usedKeys`, - * version-tied via `PARSE_CACHE_VERSION`). On a warm hit the chunk's durable - * shards are byte-COPIED into the run-scoped store (no re-parse, no - * re-serialize → byte-identical), so scope-resolution streams them exactly as - * on a cold run. Content-addressing makes stale reuse impossible: a changed - * file changes its chunk hash, which misses BOTH stores and re-dispatches. + * version-tied via `PARSE_CACHE_VERSION`). On a warm hit the chunk's immutable + * durable shards are hardlinked (or atomically copied) into the run store after + * their envelope metadata proves complete coverage. That pins a stable snapshot + * before workers are skipped, even when another branch refreshes the shared + * durable directory concurrently. */ -import { promises as fs, mkdirSync, writeFileSync, unlinkSync } from 'node:fs'; +import { promises as fs, mkdirSync } from 'node:fs'; import path from 'node:path'; import v8 from 'node:v8'; import vm from 'node:vm'; @@ -60,7 +60,14 @@ import type { } from 'gitnexus-shared'; import { isValidReceiverChain } from '../core/ingestion/utils/receiver-chain-codec.js'; import { logger } from '../core/logger.js'; -import { mapReplacer, mapReviver } from './parse-cache.js'; +import { mapReviver } from './parse-cache.js'; +import { linkOrCopyFile } from './fs-atomic.js'; +import { + inspectV8Cache, + tryLoadV8Cache, + writeV8CacheFile, + writeV8CacheFileSync, +} from './v8-sidecar.js'; const STORE_DIRNAME = 'parsedfile-store'; const DURABLE_DIRNAME = 'parsedfile-cache'; @@ -164,24 +171,13 @@ export const clearParsedFileStore = async (storagePath: string): Promise = await fs.rm(getParsedFileStoreDir(storagePath), { recursive: true, force: true }); }; -/** - * Single source of truth for a shard's bytes. Returns `null` for an empty - * chunk (caller writes nothing). Both the async (`persistParsedFileChunk`) and - * sync (`persistParsedFileShardSync`) writers go through this so the two paths - * are guaranteed byte-identical — the shards must round-trip through the same - * `mapReviver`, and matching bytes by having both authors type the same - * `mapReplacer` call would be a coincidence, not a guarantee. - */ -const serializeParsedFileShard = (parsedFiles: readonly ParsedFile[]): string | null => { - if (parsedFiles.length === 0) return null; - return JSON.stringify(parsedFiles, mapReplacer); -}; +const isV8ShardName = (name: string): boolean => name.endsWith('.v8') && !name.includes('.v8.'); const shardPath = (storagePath: string, shardId: string): string => - path.join(getParsedFileStoreDir(storagePath), `${shardId}.json`); + path.join(getParsedFileStoreDir(storagePath), `${shardId}.v8`); -/** Sidecar listing `filePath`s in a shard; not matched by `endsWith('.json')`. */ -const shardPathsSidecarPath = (jsonPath: string): string => `${jsonPath}.paths`; +const shardFilePaths = (parsedFiles: readonly ParsedFile[]): string[] => + parsedFiles.map((pf) => pf.filePath); const LOAD_YIELD_EVERY_SHARDS = 128; @@ -191,147 +187,26 @@ const LOAD_YIELD_EVERY_SHARDS = 128; */ export const parsedFileLoadGc = { run: forceGc, - /** Raw UTF-8 JSON shard bytes between GCs (#3086). Tests may lower this. */ + /** V8 envelope bytes visited between GCs (#3086). Tests may lower this. */ byteBudget: 128 * 1024 * 1024, }; -const encodeShardPathsSidecar = (parsedFiles: readonly ParsedFile[]): string => { - const paths = parsedFiles.map((pf) => pf.filePath); - return `${paths.length}\n${paths.length === 0 ? '' : `${paths.join('\n')}\n`}`; -}; - /** - * Parse a counted NDJSON path listing. Returns `null` when the sidecar must - * not be trusted to skip the JSON shard: missing trailing newline, CR/NUL, - * a truncated listing that still ends on a complete line, or a count that - * does not match the remaining lines. - */ -const parseShardPathsSidecar = (sidecarRaw: string): string[] | null => { - if (sidecarRaw.includes('\0') || sidecarRaw.includes('\r') || !sidecarRaw.endsWith('\n')) { - return null; - } - const nl = sidecarRaw.indexOf('\n'); - if (nl < 0) return null; - const countToken = sidecarRaw.slice(0, nl); - if (!/^[0-9]+$/.test(countToken)) return null; - const count = Number(countToken); - const body = sidecarRaw.slice(nl + 1); - const listed = body === '' ? [] : body.slice(0, -1).split('\n'); - if (listed.length !== count) return null; - return listed; -}; - -/** NDJSON sidecars cannot encode paths that themselves contain CR/LF/NUL. */ -const shardPathsSidecarSafe = (parsedFiles: readonly ParsedFile[]): boolean => - parsedFiles.every((pf) => !/[\r\n\0]/.test(pf.filePath)); - -const isEnoent = (err: unknown): boolean => (err as NodeJS.ErrnoException).code === 'ENOENT'; - -const warnSidecarIo = (err: unknown, jsonPath: string, msg: string): void => { - logger.warn({ err, jsonPath }, msg); -}; - -const ignoreMissingSidecarUnlink = (err: unknown, jsonPath: string): void => { - if (isEnoent(err)) return; - warnSidecarIo( - err, - jsonPath, - 'parsedfile-store: failed to drop path sidecar; JSON remains authoritative', - ); -}; - -/** Drop a leftover listing before publishing JSON so load cannot skip new paths. */ -const dropPathSidecar = async (jsonPath: string): Promise => { - try { - await fs.unlink(shardPathsSidecarPath(jsonPath)); - } catch (err) { - ignoreMissingSidecarUnlink(err, jsonPath); - } -}; - -const dropPathSidecarSync = (jsonPath: string): void => { - try { - unlinkSync(shardPathsSidecarPath(jsonPath)); - } catch (err) { - ignoreMissingSidecarUnlink(err, jsonPath); - } -}; - -const writeShardPathsSidecar = async ( - jsonPath: string, - parsedFiles: readonly ParsedFile[], -): Promise => { - if (!shardPathsSidecarSafe(parsedFiles)) { - try { - await fs.unlink(shardPathsSidecarPath(jsonPath)); - } catch (err) { - ignoreMissingSidecarUnlink(err, jsonPath); - } - return; - } - try { - await fs.writeFile( - shardPathsSidecarPath(jsonPath), - encodeShardPathsSidecar(parsedFiles), - 'utf-8', - ); - } catch (err) { - warnSidecarIo( - err, - jsonPath, - 'parsedfile-store: path sidecar write failed; JSON shard remains authoritative', - ); - try { - await fs.unlink(shardPathsSidecarPath(jsonPath)); - } catch (unlinkErr) { - ignoreMissingSidecarUnlink(unlinkErr, jsonPath); - } - } -}; - -const writeShardPathsSidecarSync = (jsonPath: string, parsedFiles: readonly ParsedFile[]): void => { - if (!shardPathsSidecarSafe(parsedFiles)) { - try { - unlinkSync(shardPathsSidecarPath(jsonPath)); - } catch (err) { - ignoreMissingSidecarUnlink(err, jsonPath); - } - return; - } - try { - writeFileSync(shardPathsSidecarPath(jsonPath), encodeShardPathsSidecar(parsedFiles), 'utf-8'); - } catch (err) { - warnSidecarIo( - err, - jsonPath, - 'parsedfile-store: path sidecar write failed; JSON shard remains authoritative', - ); - try { - unlinkSync(shardPathsSidecarPath(jsonPath)); - } catch (unlinkErr) { - ignoreMissingSidecarUnlink(unlinkErr, jsonPath); - } - } -}; - -/** - * Write one parse chunk's `ParsedFile[]` to the store as a single shard (async). - * No-op for an empty chunk. `shardId` must be unique within a run. Used by the - * main-thread no-store-disabled fallback and any non-worker writer; the worker - * store path uses {@link persistParsedFileShardSync}. + * Write one parse chunk's `ParsedFile[]` to the store as a single `.v8` shard. + * No-op for an empty chunk. `shardId` must be unique within a run. */ export const persistParsedFileChunk = async ( storagePath: string, shardId: string, parsedFiles: readonly ParsedFile[], -): Promise => { - const payload = serializeParsedFileShard(parsedFiles); - if (payload === null) return; +): Promise => { + if (parsedFiles.length === 0) return true; await fs.mkdir(getParsedFileStoreDir(storagePath), { recursive: true }); - const dest = shardPath(storagePath, shardId); - await dropPathSidecar(dest); - await fs.writeFile(dest, payload, 'utf-8'); - await writeShardPathsSidecar(dest, parsedFiles); + return writeV8CacheFile( + shardPath(storagePath, shardId), + parsedFiles, + shardFilePaths(parsedFiles), + ); }; // Per-process set of store dirs we've already `mkdir`ed, so the sync worker @@ -341,55 +216,42 @@ const createdStoreDirs = new Set(); /** * Synchronous shard writer for use INSIDE a parse worker (#1983 parallel - * serialization). The worker is a dedicated thread, so a blocking write there - * protects the main thread, and a sync write avoids threading `async`/`await` - * through the synchronous per-file extract loop. Produces byte-identical shards - * to {@link persistParsedFileChunk} via the shared {@link serializeParsedFileShard}. - * No-op for an empty chunk. `shardId` must be globally unique for the run (the - * worker uses `w-`); a duplicate would silently overwrite. + * serialization). Returns false on write failure so the worker can keep + * ParsedFiles in the result instead of dropping them. */ export const persistParsedFileShardSync = ( storagePath: string, shardId: string, parsedFiles: readonly ParsedFile[], -): void => { - const payload = serializeParsedFileShard(parsedFiles); - if (payload === null) return; +): boolean => { + if (parsedFiles.length === 0) return true; const dir = getParsedFileStoreDir(storagePath); if (!createdStoreDirs.has(dir)) { mkdirSync(dir, { recursive: true }); createdStoreDirs.add(dir); } - const dest = shardPath(storagePath, shardId); - dropPathSidecarSync(dest); - writeFileSync(dest, payload, 'utf-8'); - writeShardPathsSidecarSync(dest, parsedFiles); + return writeV8CacheFileSync( + shardPath(storagePath, shardId), + parsedFiles, + shardFilePaths(parsedFiles), + ); +}; + +const listV8Shards = async (dir: string): Promise => { + try { + return (await fs.readdir(dir)).filter(isV8ShardName).map((name) => path.join(dir, name)); + } catch { + return []; + } }; -/** - * Stream the store and return the `ParsedFile`s whose `filePath` is in - * `wantPaths`, keyed by path. Loads one shard at a time and retains only the - * matching entries, so peak heap is bounded by (matched set) + (one shard) - * rather than the whole store. Returns an empty map when the store is absent - * (e.g. tests, or a run with no worker pool) — callers fall back to a fresh - * extract for the missing files. - */ export const loadParsedFilesForPaths = async ( storagePath: string, wantPaths: ReadonlySet, ): Promise> => { const out = new Map(); if (wantPaths.size === 0) return out; - const dir = getParsedFileStoreDir(storagePath); - let shards: string[]; - try { - shards = (await fs.readdir(dir)).filter((f) => f.endsWith('.json')); - } catch { - return out; // store absent - } - // Shared interning pool for this load — deduplicates strings ACROSS shards - // (one `int` / one repeated filePath for the whole language), which is where - // most of the saving comes from. Dropped when this function returns. + const shardPaths = await listV8Shards(getParsedFileStoreDir(storagePath)); const pool = new Map(); let droppedSites = 0; let filesWithDroppedSites = 0; @@ -411,51 +273,25 @@ export const loadParsedFilesForPaths = async ( await new Promise((resolve) => setImmediate(resolve)); } }; - for (let i = 0; i < shards.length; i++) { - const jsonName = shards[i]; - const jsonFull = path.join(dir, jsonName); - try { - const sidecarRaw = await fs.readFile(shardPathsSidecarPath(jsonFull), 'utf-8'); - // Fail closed: complete writers emit `\n` plus one path per line - // and a trailing newline, never CR. Stripping CR (or accepting a - // newline-terminated prefix) would let a truncated listing skip JSON. - const listed = parseShardPathsSidecar(sidecarRaw); - if (listed === null) { - throw new Error('corrupt sidecar'); - } - if (listed.length > 0 && !listed.some((p) => wantPaths.has(p))) { - await maybeYieldAndGc(false); - continue; - } - } catch { - // Missing or unreadable sidecar → read the shard (pre-sidecar stores). + for (const shardFull of shardPaths) { + const loaded = await tryLoadV8Cache(shardFull, pool, wantPaths); + if (loaded === undefined) { + await maybeYieldAndGc(false); + continue; } - // Per-shard def pool: a SymbolDefinition's three serialized copies live within - // a single shard (one ParsedFile), so the dedup is shard-local. A cross-shard - // pool would retain defs of files NOT in `wantPaths` (loaded-but-discarded - // shards), reintroducing the leak; per-shard drops them with the shard. - const defPool = new Map(); - const reviver = makeInterningReviver(pool, defPool); - let raw: string; - try { - raw = await fs.readFile(jsonFull, 'utf-8'); - } catch { - continue; // skip a missing shard; missing files fall back to fresh extract + if (loaded.kind === 'skip') { + bytesSinceGc += loaded.bytes; + await maybeYieldAndGc(bytesSinceGc >= parsedFileLoadGc.byteBudget); + continue; } - bytesSinceGc += Buffer.byteLength(raw, 'utf8'); + bytesSinceGc += loaded.bytes; + const parsed = Array.isArray(loaded.value) ? (loaded.value as ParsedFile[]) : undefined; const crossedBudget = bytesSinceGc >= parsedFileLoadGc.byteBudget; - let parsed: ParsedFile[] | undefined; - try { - parsed = JSON.parse(raw, reviver) as ParsedFile[]; - } catch { - parsed = undefined; - } if (Array.isArray(parsed)) { for (const pf of parsed) { if (!pf || typeof pf.filePath !== 'string' || !wantPaths.has(pf.filePath)) continue; const flow = sanitizeCallableFlowSites(pf.callableFlowSites); if (flow === undefined) { - // non-array garbage → distrust the file, re-extract rejectedFiles++; continue; } @@ -481,19 +317,12 @@ export const loadParsedFilesForPaths = async ( await maybeYieldAndGc(crossedBudget); } if (droppedSites > 0 || droppedChains > 0) { - // Facts for the dropped sites are omitted this run (the file itself is - // retained, so no re-extract happens) — surface it so a recurring drop - // on every warm load is observable rather than silent (#2522 review). logger.warn( { droppedSites, droppedChains, files: filesWithDroppedSites }, 'parsedfile-store: dropped malformed/over-bound sites at load; files retained without those facts', ); } if (rejectedFiles > 0) { - // The other half of the same defect. A rejected file silently falls back to - // a fresh extract EVERY load, so a writer that keeps minting what this - // reader keeps refusing is a permanent warm-cache miss that costs real time - // and says nothing about why. logger.warn( { rejectedFiles }, 'parsedfile-store: rejected shard entries at load (untrusted shape); those files re-extract every run', @@ -502,16 +331,6 @@ export const loadParsedFilesForPaths = async ( return out; }; -/** - * Treat the durable ParsedFile store as an untrusted serialization boundary. - * Sanitation is per-SITE, not per-file: one malformed or over-bound fact drops - * only itself (counted, logged by the caller), so a legitimately pathological - * source file cannot push its whole ParsedFile into a permanent, silent - * warm-cache-miss reparse loop (#2522 review). Only a non-array field — - * i.e. garbage that says the serialization itself is untrustworthy — rejects - * the file, and `undefined` (never emitted / no facts) passes through. - * Returns `undefined` for the reject-file case. - */ function sanitizeCallableFlowSites( value: unknown, ): { sites: readonly CallableFlowSite[] | undefined; dropped: number } | undefined { @@ -709,16 +528,15 @@ function isSafeIndex(value: unknown): boolean { // ─── Durable, content-addressed sibling store (warm-cache coverage) ────────── // -// Layout: `//-w-.json` plus a -// top-level `/index.json` = `{version, keys:[chunkHash…]}`. One -// subdir per chunk hash so a chunk's (possibly several) shards collect and -// prune as a unit, and so `readdir(/)` is O(shards-of-this-chunk), -// not O(all-history). Shards are byte-identical to run-scoped shards (same -// `serializeParsedFileShard`); restore is a verbatim copy, never a re-serialize. +// Layout: `//-w-.v8` plus a +// top-level `/index.json` that records each chunk hash's actual +// persisted file-path coverage. One subdir per chunk hash so a chunk's +// (possibly several) shards collect and prune as a unit. Warm hits snapshot +// these files into the run store before worker dispatch is skipped. interface DurableParsedFileIndex { version: string; - keys: string[]; + entries: Record; } /** Durable store dir — a sibling of `parsedfile-store/`, NEVER cleared per run. */ @@ -744,10 +562,6 @@ export const prepareDurableParsedFileChunk = async ( await fs.mkdir(dir, { recursive: true }); }; -// Per-process set of durable chunk subdirs already `mkdir`ed (mirrors -// `createdStoreDirs`) so the worker doesn't `mkdirSync` on every shard. -const createdDurableDirs = new Set(); - /** * Synchronous durable-shard writer for use INSIDE a parse worker, alongside * {@link persistParsedFileShardSync}. Writes the SAME bytes to a content-addressed @@ -757,96 +571,99 @@ const createdDurableDirs = new Set(); * uniqueness that makes the run-scoped `w-` name safe, prefixed by * content. No-op for an empty chunk. */ + +const createdDurableDirs = new Set(); + export const persistDurableParsedFileShardSync = ( durableDir: string, chunkHash: string, threadId: number, shardSeq: number, parsedFiles: readonly ParsedFile[], -): void => { - const payload = serializeParsedFileShard(parsedFiles); - if (payload === null) return; +): boolean => { + if (parsedFiles.length === 0) return true; const dir = durableChunkDir(durableDir, chunkHash); if (!createdDurableDirs.has(dir)) { mkdirSync(dir, { recursive: true }); createdDurableDirs.add(dir); } - const dest = path.join(dir, `${chunkHash}-w${threadId}-${shardSeq}.json`); - dropPathSidecarSync(dest); - writeFileSync(dest, payload, 'utf-8'); - writeShardPathsSidecarSync(dest, parsedFiles); + const dest = path.join(dir, `${chunkHash}-w${threadId}-${shardSeq}.v8`); + return writeV8CacheFileSync(dest, parsedFiles, shardFilePaths(parsedFiles)); }; /** - * Restore a cached chunk's durable shards into the run-scoped store on a warm - * hit. A verbatim byte copy (no parse, no re-serialize), so the restored - * ParsedFiles are byte-identical to a cold run and `loadParsedFilesForPaths` - * (which keys on `filePath`, not shard name) gives scope-resolution full - * coverage. The durable shard names already carry the chunk hash, so they never - * collide with the worker's run-scoped `w-` shards. Returns the number - * of shards restored (0 ⇒ no durable coverage for this chunk; caller treats it - * as a miss). + * Validate and snapshot one durable chunk into the run store. Every envelope + * must be runtime-compatible and integrity-valid, and together they must match + * the path coverage recorded when the durable index was published. Linking + * before returning pins the inodes against concurrent branch-cache rotation. */ -export const restoreDurableParsedFileShard = async ( - durableDir: string, +export const durableChunkHasShards = async ( runStoragePath: string, chunkHash: string, -): Promise => { - const src = durableChunkDir(durableDir, chunkHash); - let shards: string[]; + expectedPaths: ReadonlySet, +): Promise => { + const sourceDir = durableChunkDir(getDurableParsedFileDir(runStoragePath), chunkHash); + const shards = await listV8Shards(sourceDir); + if (shards.length === 0 || expectedPaths.size === 0) return false; + + const runDir = getParsedFileStoreDir(runStoragePath); try { - shards = (await fs.readdir(src)).filter((f) => f.endsWith('.json')); + await fs.mkdir(runDir, { recursive: true }); } catch { - return 0; // no durable shards for this chunk + return false; } - if (shards.length === 0) return 0; - const dst = getParsedFileStoreDir(runStoragePath); - await fs.mkdir(dst, { recursive: true }); - for (const name of shards) { - const srcJson = path.join(src, name); - const dstJson = path.join(dst, name); - await dropPathSidecar(dstJson); - await fs.copyFile(srcJson, dstJson); + const restored: string[] = []; + const covered = new Set(); + const rollback = async (): Promise => { + await Promise.all(restored.map((filePath) => fs.rm(filePath, { force: true }).catch(() => {}))); + return false; + }; + + for (const sourcePath of shards) { + const name = path.basename(sourcePath); + const destinationPath = path.join(runDir, name); try { - await fs.copyFile(shardPathsSidecarPath(srcJson), shardPathsSidecarPath(dstJson)); - } catch (copyErr) { - if (!isEnoent(copyErr)) { - warnSidecarIo( - copyErr, - srcJson, - 'parsedfile-store: durable path sidecar copy failed; JSON remains authoritative', - ); - continue; - } - try { - await fs.unlink(shardPathsSidecarPath(dstJson)); - } catch (err) { - ignoreMissingSidecarUnlink(err, dstJson); - } + await linkOrCopyFile(sourcePath, destinationPath); + restored.push(destinationPath); + } catch { + return rollback(); + } + const inspected = await inspectV8Cache(destinationPath); + if (!inspected) return rollback(); + for (const filePath of inspected.paths) { + if (!expectedPaths.has(filePath)) return rollback(); + covered.add(filePath); } } - return shards.length; + + if (covered.size !== expectedPaths.size) return rollback(); + return true; }; -/** - * Read the durable index and return the set of chunk hashes it vouches for, - * gated on `expectedVersion` (`PARSE_CACHE_VERSION`). A version mismatch or a - * missing/corrupt index returns the empty set — the caller then treats every - * chunk as a durable miss and re-dispatches workers (NEVER the main-thread - * `extractParsedFile` fallback), which rewrites the durable store under the new - * version. Mirrors `loadParseCache`'s version-invalidation contract. - */ export const loadDurableParsedFileIndex = async ( durableDir: string, expectedVersion: string, -): Promise> => { +): Promise>> => { try { const raw = await fs.readFile(path.join(durableDir, DURABLE_INDEX_FILENAME), 'utf-8'); - const idx = JSON.parse(raw) as DurableParsedFileIndex; - if (idx?.version !== expectedVersion || !Array.isArray(idx.keys)) return new Set(); - return new Set(idx.keys); + const idx: unknown = JSON.parse(raw); + if (!isRecord(idx) || idx.version !== expectedVersion || !isRecord(idx.entries)) { + return new Map(); + } + const entries = new Map>(); + for (const [key, paths] of Object.entries(idx.entries)) { + if ( + !Array.isArray(paths) || + paths.length === 0 || + paths.some((filePath) => typeof filePath !== 'string') + ) { + return new Map(); + } + entries.set(key, new Set(paths)); + } + return entries; } catch { - return new Set(); + return new Map(); } }; @@ -855,9 +672,9 @@ export const loadDurableParsedFileIndex = async ( * be the parse cache's surviving on-disk keys (so the two stores stay coherent: * a chunk is "cached" iff BOTH its parse-cache shard and its durable shards * exist; a quarantined chunk — no parse-cache shard — drops its durable subdir - * here and re-dispatches next run). Only subdirs with ≥1 shard are indexed - * (mirrors `saveParseCache`'s written-keys discipline — never vouch for a chunk - * hash with no backing shard). The index write is tmp+rename atomic. + * here and re-dispatches next run). Only chunks whose envelopes all validate + * are indexed, together with their exact persisted path coverage (never vouch + * for a missing/corrupt shard). The index write is tmp+rename atomic. */ export const pruneAndSaveDurableParsedFileStore = async ( durableDir: string, @@ -870,16 +687,28 @@ export const pruneAndSaveDurableParsedFileStore = async ( } catch { return; // nothing written this run } - const survivors: string[] = []; + const survivors: Record = {}; for (const name of entries) { if (name === DURABLE_INDEX_FILENAME) continue; const full = path.join(durableDir, name); if (keepKeys.has(name)) { try { - const shards = (await fs.readdir(full)).filter((f) => f.endsWith('.json')); + const shards = await listV8Shards(full); if (shards.length > 0) { - survivors.push(name); - continue; + const covered = new Set(); + let valid = true; + for (const shard of shards) { + const inspected = await inspectV8Cache(shard); + if (!inspected) { + valid = false; + break; + } + for (const filePath of inspected.paths) covered.add(filePath); + } + if (valid && covered.size > 0) { + survivors[name] = [...covered].sort(); + continue; + } } } catch { /* not a readable dir → drop below */ @@ -887,7 +716,7 @@ export const pruneAndSaveDurableParsedFileStore = async ( } await fs.rm(full, { recursive: true, force: true }); } - const idx: DurableParsedFileIndex = { version, keys: survivors }; + const idx: DurableParsedFileIndex = { version, entries: survivors }; const tmp = path.join(durableDir, `${DURABLE_INDEX_FILENAME}.tmp`); await fs.mkdir(durableDir, { recursive: true }); await fs.writeFile(tmp, JSON.stringify(idx), 'utf-8'); diff --git a/gitnexus/src/storage/v8-sidecar.ts b/gitnexus/src/storage/v8-sidecar.ts new file mode 100644 index 000000000..e3b2f92c9 --- /dev/null +++ b/gitnexus/src/storage/v8-sidecar.ts @@ -0,0 +1,363 @@ +/** + * One-file V8 cache envelope for parse-cache and ParsedFile shards. + * + * Each object-graph shard is a single immutable `.v8` file published by + * tmp+rename. JSON is not a fallback: a missing, corrupt, runtime-incompatible, + * or deserialize-failed envelope is a cache miss (re-extract / re-dispatch). + * Tiny JSON manifests (`index.json`) stay outside this module. + * + * Envelope: magic, format, Node major, V8 version, optional path listing, + * `v8.serialize` of the live graph, then SHA-256 of listing||payload. Path + * bytes live before the payload so a ParsedFile loader can skip a shard whose + * authenticated listing misses `wantPaths` without deserializing. An + * unreadable/invalid listing is fail-closed: deserialize, never skip. + */ +import { promises as fs } from 'node:fs'; +import { createHash } from 'node:crypto'; +import v8 from 'node:v8'; +import { logger } from '../core/logger.js'; +import { linkOrCopyFile, writeFileAtomicBytes, writeFileAtomicBytesSync } from './fs-atomic.js'; + +const MAGIC = Buffer.from('GNXV8CF1', 'ascii'); +/** Envelope version — independent of PARSE_CACHE_VERSION / SCHEMA_BUMP. */ +export const V8_CACHE_FORMAT = 5; +const U32 = 4; +const U16 = 2; +const PAYLOAD_HASH_LEN = 32; +const MAGIC_LEN = 8; +const FIXED_PREFIX = MAGIC_LEN + U32 + U16 + U16; // magic + format + nodeMajor + v8len + +const internString = (value: string, pool: Map): string => { + const hit = pool.get(value); + if (hit !== undefined) return hit; + pool.set(value, value); + return value; +}; + +/** + * Collapse duplicate strings in a live deserialized graph into `pool`, mutating + * in place so object identity (shared `SymbolDefinition`s, Maps) is preserved. + * Required after `v8.deserialize` of ParsedFile shards: V8 does not recreate + * the JSON reviver's cross-shard string intern, and skipping it regresses + * retained heap (~+59% measured vs interned JSON). + */ +export const internGraphStrings = (root: unknown, pool: Map): unknown => { + const seen = new WeakSet(); + const walk = (value: unknown): unknown => { + if (typeof value === 'string') return internString(value, pool); + if (value === null || typeof value !== 'object') return value; + if (ArrayBuffer.isView(value) || value instanceof ArrayBuffer) return value; + if (seen.has(value)) return value; + seen.add(value); + if (value instanceof Map) { + const entries = [...value]; + value.clear(); + for (const [k, v] of entries) value.set(walk(k), walk(v)); + return value; + } + if (value instanceof Set) { + const entries = [...value]; + value.clear(); + for (const v of entries) value.add(walk(v)); + return value; + } + if (Array.isArray(value)) { + for (let i = 0; i < value.length; i++) value[i] = walk(value[i]); + return value; + } + const rec = value as Record; + for (const key of Object.keys(rec)) { + rec[key] = walk(rec[key]); + } + return value; + }; + return walk(root); +}; + +const nodeMajor = (): number => Number.parseInt(process.versions.node.split('.')[0] ?? '0', 10); + +const isEnoent = (err: unknown): boolean => (err as NodeJS.ErrnoException).code === 'ENOENT'; + +const warnCache = (err: unknown, filePath: string, msg: string): void => { + logger.warn({ err, filePath }, msg); +}; + +/** NDJSON listing cannot encode paths that themselves contain CR/LF/NUL. */ +export const encodeCachePathListing = (paths: readonly string[]): Buffer => { + if (paths.length === 0) return Buffer.alloc(0); + if (paths.some((p) => /[\r\n\0]/.test(p))) return Buffer.alloc(0); + return Buffer.from(`${paths.length}\n${paths.join('\n')}\n`, 'utf8'); +}; + +/** + * Parse a counted NDJSON path listing. Returns `null` when it must not be + * trusted to skip the payload: missing trailing newline, CR/NUL, or a count + * that does not match the remaining lines. + */ +export const parseCachePathListing = (raw: Buffer): string[] | null => { + if (raw.byteLength === 0) return []; + let sidecarRaw: string; + try { + sidecarRaw = new TextDecoder('utf-8', { fatal: true }).decode(raw); + } catch { + return null; + } + if (sidecarRaw.includes('\0') || sidecarRaw.includes('\r') || !sidecarRaw.endsWith('\n')) { + return null; + } + const nl = sidecarRaw.indexOf('\n'); + if (nl < 0) return null; + const countToken = sidecarRaw.slice(0, nl); + if (!/^[0-9]+$/.test(countToken)) return null; + const count = Number(countToken); + const body = sidecarRaw.slice(nl + 1); + const listed = body === '' ? [] : body.slice(0, -1).split('\n'); + if (listed.length !== count) return null; + return listed; +}; + +const encodeEnvelope = (graph: unknown, paths: readonly string[]): Buffer | undefined => { + let payload: Buffer; + try { + payload = v8.serialize(graph); + } catch (err) { + warnCache(err, '', 'v8 cache: serialize failed; treating as miss on next load'); + return undefined; + } + const pathListing = encodeCachePathListing(paths); + const listed = parseCachePathListing(pathListing); + const pathCount = listed === null ? 0 : listed.length; + if (payload.byteLength > 0xffff_ffff || pathListing.byteLength > 0xffff_ffff) return undefined; + const v8ver = Buffer.from(process.versions.v8, 'utf8'); + if (v8ver.byteLength > 0xffff) return undefined; + const header = Buffer.allocUnsafe(FIXED_PREFIX + v8ver.length + U32 * 3); + MAGIC.copy(header, 0); + header.writeUInt32LE(V8_CACHE_FORMAT, MAGIC_LEN); + header.writeUInt16LE(nodeMajor(), MAGIC_LEN + U32); + header.writeUInt16LE(v8ver.length, MAGIC_LEN + U32 + U16); + v8ver.copy(header, FIXED_PREFIX); + let off = FIXED_PREFIX + v8ver.length; + header.writeUInt32LE(pathCount, off); + off += U32; + header.writeUInt32LE(pathListing.byteLength, off); + off += U32; + header.writeUInt32LE(payload.byteLength, off); + const payloadHash = createHash('sha256').update(pathListing).update(payload).digest(); + return Buffer.concat([header, pathListing, payload, payloadHash]); +}; + +type EnvelopeMeta = { + recordedNodeMajor: number; + recordedV8: string; + pathCount: number; + pathBytes: number; + payloadLen: number; + pathsOff: number; + payloadOff: number; +}; + +const decodePrefix = (buf: Buffer): EnvelopeMeta | undefined => { + if (buf.byteLength < FIXED_PREFIX) return undefined; + if (!buf.subarray(0, MAGIC_LEN).equals(MAGIC)) return undefined; + if (buf.readUInt32LE(MAGIC_LEN) !== V8_CACHE_FORMAT) return undefined; + const recordedNodeMajor = buf.readUInt16LE(MAGIC_LEN + U32); + const v8len = buf.readUInt16LE(MAGIC_LEN + U32 + U16); + const v8off = FIXED_PREFIX; + const countsOff = v8off + v8len; + if (buf.byteLength < countsOff + U32 * 3) return undefined; + const recordedV8 = buf.subarray(v8off, v8off + v8len).toString('utf8'); + const pathCount = buf.readUInt32LE(countsOff); + const pathBytes = buf.readUInt32LE(countsOff + U32); + const payloadLen = buf.readUInt32LE(countsOff + U32 * 2); + const pathsOff = countsOff + U32 * 3; + const payloadOff = pathsOff + pathBytes; + return { + recordedNodeMajor, + recordedV8, + pathCount, + pathBytes, + payloadLen, + pathsOff, + payloadOff, + }; +}; + +const runtimeCompatible = (meta: EnvelopeMeta): boolean => + meta.recordedNodeMajor === nodeMajor() && meta.recordedV8 === process.versions.v8; + +export type V8CacheHit = { kind: 'hit'; value: unknown; bytes: number }; +export type V8CacheSkip = { kind: 'skip'; bytes: number }; +export type V8CacheLoad = V8CacheHit | V8CacheSkip; +export type V8CacheInspection = { paths: readonly string[] }; + +const readExact = async ( + fh: Awaited>, + offset: number, + length: number, +): Promise => { + const buf = Buffer.allocUnsafe(length); + let got = 0; + while (got < length) { + const { bytesRead } = await fh.read(buf, got, length - got, offset + got); + if (bytesRead === 0) return undefined; + got += bytesRead; + } + return buf; +}; + +const readVerifiedPayload = async ( + fh: Awaited>, + meta: EnvelopeMeta, + pathRaw: Buffer, +): Promise => { + const payloadAndHash = await readExact(fh, meta.payloadOff, meta.payloadLen + PAYLOAD_HASH_LEN); + if (!payloadAndHash) return undefined; + const payload = payloadAndHash.subarray(0, meta.payloadLen); + const expected = payloadAndHash.subarray(meta.payloadLen); + const digest = createHash('sha256').update(pathRaw).update(payload).digest(); + return digest.equals(expected) ? payload : undefined; +}; + +/** + * Validate the immutable envelope metadata needed by the durable ParsedFile + * warm-hit gate without deserializing its payload. Atomic publication means a + * runtime-compatible envelope whose exact file length and counted path listing + * validate is a stable snapshot candidate; malformed/truncated envelopes miss. + */ +export const inspectV8Cache = async (filePath: string): Promise => { + let fh: Awaited> | undefined; + try { + fh = await fs.open(filePath, 'r'); + const st = await fh.stat(); + const prefix = await readExact(fh, 0, Math.min(st.size, FIXED_PREFIX + 256 + U32 * 3)); + if (!prefix) return undefined; + const meta = decodePrefix(prefix); + if (!meta || !runtimeCompatible(meta)) return undefined; + if (meta.payloadOff + meta.payloadLen + PAYLOAD_HASH_LEN !== st.size || meta.pathBytes === 0) { + return undefined; + } + + const pathRaw = + prefix.byteLength >= meta.payloadOff + ? Buffer.from(prefix.subarray(meta.pathsOff, meta.payloadOff)) + : await readExact(fh, meta.pathsOff, meta.pathBytes); + if (!pathRaw) return undefined; + const listed = parseCachePathListing(pathRaw); + if (listed === null || listed.length !== meta.pathCount || listed.length === 0) { + return undefined; + } + if (!(await readVerifiedPayload(fh, meta, pathRaw))) return undefined; + return { paths: listed }; + } catch (err) { + if (!isEnoent(err)) { + logger.debug({ err, filePath }, 'v8 cache: inspection failed; treating as miss'); + } + return undefined; + } finally { + await fh?.close().catch(() => {}); + } +}; + +/** + * Load a cache file. When `wantPaths` is set and a digest-verified non-empty + * path listing has no intersection, returns `{ kind: 'skip' }` without + * deserializing. An authentic listing that does not parse deserializes (fail + * closed). Envelope/runtime/digest failure returns undefined (miss). + */ +export const tryLoadV8Cache = async ( + filePath: string, + internPool?: Map, + wantPaths?: ReadonlySet, +): Promise => { + let fh: Awaited> | undefined; + try { + fh = await fs.open(filePath, 'r'); + const st = await fh.stat(); + const prefix = await readExact(fh, 0, Math.min(st.size, FIXED_PREFIX + 256 + U32 * 3)); + if (!prefix) return undefined; + const meta = decodePrefix(prefix); + if (!meta) return undefined; + if (!runtimeCompatible(meta)) return undefined; + if (meta.payloadOff + meta.payloadLen + PAYLOAD_HASH_LEN !== st.size) return undefined; + + let pathRaw = Buffer.alloc(0); + if (meta.pathBytes > 0) { + if (prefix.byteLength >= meta.payloadOff) { + pathRaw = Buffer.from(prefix.subarray(meta.pathsOff, meta.payloadOff)); + } else { + const raw = await readExact(fh, meta.pathsOff, meta.pathBytes); + if (!raw) return undefined; + pathRaw = Buffer.from(raw); + } + } + + const payload = await readVerifiedPayload(fh, meta, pathRaw); + if (!payload) return undefined; + + if (wantPaths && wantPaths.size > 0 && meta.pathBytes > 0) { + const listed = parseCachePathListing(pathRaw); + if ( + listed !== null && + listed.length === meta.pathCount && + listed.length > 0 && + !listed.some((p) => wantPaths.has(p)) + ) { + return { kind: 'skip', bytes: st.size }; + } + } + const value = v8.deserialize(payload); + if (internPool) internGraphStrings(value, internPool); + return { kind: 'hit', value, bytes: st.size }; + } catch (err) { + if (!isEnoent(err)) { + logger.debug({ err, filePath }, 'v8 cache: load failed; treating as miss'); + } + return undefined; + } finally { + await fh?.close().catch(() => {}); + } +}; + +export const writeV8CacheFile = async ( + filePath: string, + graph: unknown, + paths?: readonly string[], +): Promise => { + const blob = encodeEnvelope(graph, paths ?? []); + if (!blob) return false; + try { + await writeFileAtomicBytes(filePath, blob, 1); + return true; + } catch (err) { + warnCache(err, filePath, 'v8 cache: write failed; treating as miss'); + return false; + } +}; + +export const writeV8CacheFileSync = ( + filePath: string, + graph: unknown, + paths?: readonly string[], +): boolean => { + const blob = encodeEnvelope(graph, paths ?? []); + if (!blob) return false; + try { + writeFileAtomicBytesSync(filePath, blob); + return true; + } catch (err) { + warnCache(err, filePath, 'v8 cache: write failed; treating as miss'); + return false; + } +}; + +export const copyV8CacheIfPresent = async (srcPath: string, dstPath: string): Promise => { + try { + await linkOrCopyFile(srcPath, dstPath); + return true; + } catch (copyErr) { + if (!isEnoent(copyErr)) { + warnCache(copyErr, srcPath, 'v8 cache: copy failed; treating as miss'); + } + return false; + } +}; diff --git a/gitnexus/test/integration/cfg/parse-cache-mixed.test.ts b/gitnexus/test/integration/cfg/parse-cache-mixed.test.ts index 58f183782..b046f422a 100644 --- a/gitnexus/test/integration/cfg/parse-cache-mixed.test.ts +++ b/gitnexus/test/integration/cfg/parse-cache-mixed.test.ts @@ -5,19 +5,19 @@ import path from 'node:path'; import { getDurableParsedFileDir, persistDurableParsedFileShardSync, - restoreDurableParsedFileShard, + durableChunkHasShards, loadParsedFilesForPaths, } from '../../../src/storage/parsedfile-store.js'; import type { ParsedFile } from 'gitnexus-shared'; import type { FunctionCfg } from '../../../src/core/ingestion/cfg/types.js'; // #2082 M2 U5 — the warm/mixed cache seam for statement facts. On a warm (or -// mixed) run the unchanged chunk's ParsedFiles are BYTE-COPIED from the -// durable store instead of re-parsed (#2038); if that copy (or the store's -// interning reviver) dropped or aliased the new `bindings`/`statements` +// mixed) run the unchanged chunk's ParsedFiles are loaded from the +// durable store instead of re-parsed (#2038); if that load (or intern) +// dropped or aliased the new `bindings`/`statements` // fields, reaching-defs would silently degrade to `no-facts` for every cached // file — exactly the field-loss class the #2038 mergeChunkResults lesson -// warns about. This pins the persist → restore → load round-trip at the exact +// warns about. This pins the persist → load round-trip at the exact // seam scope-resolution consumes. const factCfg: FunctionCfg = { @@ -68,7 +68,7 @@ describe('durable ParsedFile store carries M2 statement facts (#2082 U5)', () => if (tempDir) fs.rmSync(tempDir, { recursive: true, force: true }); }); - it('persist → restore → loadParsedFilesForPaths preserves bindings + statements deep-equal', async () => { + it('persist → loadParsedFilesForPaths preserves bindings + statements deep-equal', async () => { const durableDir = getDurableParsedFileDir(tempDir); const chunkHash = 'c'.repeat(64); const files = ['src/a.ts', 'src/b.ts']; @@ -76,10 +76,9 @@ describe('durable ParsedFile store carries M2 statement facts (#2082 U5)', () => // What a worker writes at flush on a cache MISS (the cold half of a // mixed-mode run)… persistDurableParsedFileShardSync(durableDir, chunkHash, 7, 0, files.map(mkParsedFile)); - // …and what a warm HIT byte-copies into the run-scoped store. - await restoreDurableParsedFileShard(durableDir, tempDir, chunkHash); - - const loaded = await loadParsedFilesForPaths(tempDir, new Set(files)); + const wanted = new Set(files); + expect(await durableChunkHasShards(tempDir, chunkHash, wanted)).toBe(true); + const loaded = await loadParsedFilesForPaths(tempDir, wanted); expect(loaded.size).toBe(2); for (const filePath of files) { const pf = loaded.get(filePath); @@ -106,11 +105,9 @@ describe('durable ParsedFile store carries M2 statement facts (#2082 U5)', () => mkParsedFile('src/same1.ts'), mkParsedFile('src/same2.ts'), ]); - await restoreDurableParsedFileShard(durableDir, tempDir, chunkHash); - const loaded = await loadParsedFilesForPaths( - tempDir, - new Set(['src/same1.ts', 'src/same2.ts']), - ); + const wanted = new Set(['src/same1.ts', 'src/same2.ts']); + expect(await durableChunkHasShards(tempDir, chunkHash, wanted)).toBe(true); + const loaded = await loadParsedFilesForPaths(tempDir, wanted); const c1 = (loaded.get('src/same1.ts') as { cfgSideChannel?: FunctionCfg[] }).cfgSideChannel; const c2 = (loaded.get('src/same2.ts') as { cfgSideChannel?: FunctionCfg[] }).cfgSideChannel; expect(c1?.[0].bindings).toEqual(factCfg.bindings); diff --git a/gitnexus/test/unit/incremental-parse-cache.test.ts b/gitnexus/test/unit/incremental-parse-cache.test.ts index 14c4c8359..69a366c10 100644 --- a/gitnexus/test/unit/incremental-parse-cache.test.ts +++ b/gitnexus/test/unit/incremental-parse-cache.test.ts @@ -1,5 +1,5 @@ import { describe, it, expect } from 'vitest'; -import { mkdtemp, rm } from 'fs/promises'; +import { mkdtemp, rm, readdir, writeFile, readFile } from 'fs/promises'; import { tmpdir } from 'os'; import path from 'path'; import { @@ -17,6 +17,7 @@ import { slimParseWorkerResultsForCache, type ParseCache, } from '../../src/storage/parse-cache.js'; +import { writeV8CacheFile } from '../../src/storage/v8-sidecar.js'; import type { ParseWorkerResult } from '../../src/core/ingestion/workers/parse-worker.js'; const minimalResult = (overrides: Partial = {}): ParseWorkerResult => ({ @@ -243,11 +244,11 @@ describe('PARSE_CACHE_VERSION', () => { // collided, because each re-checked once and neither re-checked after the // other moved — which is why the rule is re-applied AT MERGE, not when the // number is picked. - it('pins SCHEMA_BUMP to 80 so concurrent bumps cannot silently collide (#2766, #3015, #3088)', () => { - expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(80); + it('pins SCHEMA_BUMP to 81 so concurrent bumps cannot silently collide (#2766, #3015, #3088)', () => { + expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(81); expect(PARSE_CACHE_BUCKET_COUNT).toBe(128); for (const taken of [ - 59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, + 59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, ]) { expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).not.toBe(taken); } @@ -450,12 +451,10 @@ describe('loadParseCache / saveParseCache (round-trip)', () => { }), 'utf-8', ); - await fs.writeFile( - path.join(cacheDir, `${goodKey}.json`), - JSON.stringify([minimalResult({ fileCount: 3 })]), - 'utf-8', - ); - await fs.writeFile(path.join(cacheDir, `${badKey}.json`), '{not-json', 'utf-8'); + await writeV8CacheFile(path.join(cacheDir, `${goodKey}.v8`), [ + minimalResult({ fileCount: 3 }), + ]); + await fs.writeFile(path.join(cacheDir, `${badKey}.v8`), '{not-json', 'utf-8'); const loaded = await loadParseCache(dir); expect(loaded.entries.size).toBe(0); @@ -510,7 +509,7 @@ describe('loadParseCache / saveParseCache (round-trip)', () => { await saveParseCache(dir, cache); const persisted = await fs.readdir(path.join(dir, 'parse-cache')); expect(persisted).toContain('index.json'); - expect(persisted).toContain(`${chunkKey}.json`); + expect(persisted).toContain(`${chunkKey}.v8`); const loaded = await loadParseCache(dir); const reloaded = (await loadParseCacheChunk(loaded, chunkKey))?.[0]; expect(reloaded).toBeDefined(); @@ -543,11 +542,9 @@ describe('loadParseCache / saveParseCache (round-trip)', () => { }), 'utf-8', ); - await fs.writeFile( - path.join(cacheDir, `${safeKey}.json`), - JSON.stringify([minimalResult({ fileCount: 9 })]), - 'utf-8', - ); + await writeV8CacheFile(path.join(cacheDir, `${safeKey}.v8`), [ + minimalResult({ fileCount: 9 }), + ]); const loaded = await loadParseCache(dir); expect(loaded.onDiskKeys?.size).toBe(1); const chunk = await loadParseCacheChunk(loaded, safeKey); @@ -577,7 +574,7 @@ describe('loadParseCache / saveParseCache (round-trip)', () => { const cacheDir = path.join(dir, 'parse-cache'); const names = await fs.readdir(cacheDir); expect(names).toContain('index.json'); - expect(names.filter((n) => n.endsWith('.json') && n !== 'index.json').length).toBe(3); + expect(names.filter((n) => n.endsWith('.v8')).length).toBe(3); const loaded = await loadParseCache(dir); expect(loaded.onDiskKeys?.size).toBe(3); } finally { @@ -628,8 +625,8 @@ describe('loadParseCache / saveParseCache (round-trip)', () => { usedKeys: new Set([k2]), }); const names = await fs.readdir(path.join(dir, 'parse-cache')); - expect(names).not.toContain(`${k1}.json`); - expect(names).toContain(`${k2}.json`); + expect(names).not.toContain(`${k1}.v8`); + expect(names).toContain(`${k2}.v8`); const loaded = await loadParseCache(dir); expect(loaded.onDiskKeys?.size).toBe(1); const chunk = await loadParseCacheChunk(loaded, k2); @@ -823,4 +820,116 @@ describe('loadParseCache / saveParseCache (round-trip)', () => { await rm(dir, { recursive: true, force: true }); } }); + + it('writes a V8 shard and loads it with Map-preserving semantics', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-pc-v8-')); + try { + const innerMap = new Map([ + ['k1', 'v1'], + ['k2', 'v2'], + ]); + const innerSet = new Set(['s1', 's2']); + const fake = minimalResult({ + fileCount: 9, + imports: [ + { + typeBindings: innerMap, + extras: innerSet, + } as unknown as ParseWorkerResult['imports'][number], + ], + }); + const key = 'f'.repeat(64); + const cache: ParseCache = { + version: PARSE_CACHE_VERSION, + entries: new Map(), + usedKeys: new Set([key]), + storagePath: dir, + onDiskKeys: new Set(), + }; + await persistParseCacheChunk(cache, key, [fake]); + const names = await readdir(path.join(dir, 'parse-cache')); + expect(names).toEqual(expect.arrayContaining([`${key}.v8`])); + expect(names.some((n) => n.endsWith('.json') && n !== 'index.json')).toBe(false); + const loaded = await loadParseCacheChunk(cache, key); + expect(loaded?.[0]?.fileCount).toBe(9); + const smuggled = loaded?.[0]?.imports[0] as unknown as { + typeBindings?: unknown; + extras?: unknown; + }; + expect(smuggled.typeBindings).toBeInstanceOf(Map); + expect([...(smuggled.typeBindings as Map)]).toEqual([ + ['k1', 'v1'], + ['k2', 'v2'], + ]); + expect(smuggled.extras).toBeInstanceOf(Set); + expect([...(smuggled.extras as Set)].sort()).toEqual(['s1', 's2']); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('treats a corrupt parse-cache V8 shard as a miss', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-pc-v8-fb-')); + try { + const key = 'a'.repeat(64); + const cache: ParseCache = { + version: PARSE_CACHE_VERSION, + entries: new Map(), + usedKeys: new Set([key]), + storagePath: dir, + onDiskKeys: new Set(), + }; + await persistParseCacheChunk(cache, key, [minimalResult({ fileCount: 4 })]); + await writeFile(path.join(dir, 'parse-cache', `${key}.v8`), Buffer.from([1, 2, 3])); + const loaded = await loadParseCacheChunk(cache, key); + expect(loaded).toBeUndefined(); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('saveParseCache copies an existing V8 shard', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-pc-v8-copy-')); + try { + const key = 'c'.repeat(64); + const cache: ParseCache = { + version: PARSE_CACHE_VERSION, + entries: new Map(), + usedKeys: new Set([key]), + storagePath: dir, + onDiskKeys: new Set(), + }; + await persistParseCacheChunk(cache, key, [minimalResult({ fileCount: 42 })]); + const liveV8 = await readFile(path.join(dir, 'parse-cache', `${key}.v8`)); + await saveParseCache(dir, cache); + expect(await readdir(path.join(dir, 'parse-cache'))).toEqual( + expect.arrayContaining([`${key}.v8`, 'index.json']), + ); + expect(await readFile(path.join(dir, 'parse-cache', `${key}.v8`))).toEqual(liveV8); + const loaded = await loadParseCache(dir); + expect((await loadParseCacheChunk(loaded, key))?.[0]?.fileCount).toBe(42); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('misses when the V8 shard is absent', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'gnx-pc-v8-legacy-')); + try { + const key = 'b'.repeat(64); + const cache: ParseCache = { + version: PARSE_CACHE_VERSION, + entries: new Map(), + usedKeys: new Set([key]), + storagePath: dir, + onDiskKeys: new Set(), + }; + await persistParseCacheChunk(cache, key, [minimalResult({ fileCount: 7 })]); + await rm(path.join(dir, 'parse-cache', `${key}.v8`), { force: true }); + const loaded = await loadParseCacheChunk(cache, key); + expect(loaded).toBeUndefined(); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); }); diff --git a/gitnexus/test/unit/parse-impl-warm-cache-parsedfile-coverage.test.ts b/gitnexus/test/unit/parse-impl-warm-cache-parsedfile-coverage.test.ts index 77ac0a48c..410d0c048 100644 --- a/gitnexus/test/unit/parse-impl-warm-cache-parsedfile-coverage.test.ts +++ b/gitnexus/test/unit/parse-impl-warm-cache-parsedfile-coverage.test.ts @@ -12,19 +12,19 @@ * OOM. * * The fix: workers ALSO write a durable, content-addressed ParsedFile store - * keyed by chunk hash (`parsedfile-cache/`); a warm hit BYTE-COPIES the chunk's - * durable shards into the run-scoped store so scope-resolution streams them - * exactly as on a cold run — zero re-parse, byte-identical. + * keyed by chunk hash (`parsedfile-cache/`); a warm hit LOADS those shards + * in place (no copy into the run-scoped store) so scope-resolution streams + * them exactly as on a cold run — zero re-parse, byte-identical. * * Two layers of coverage: - * (1) Store-level — the durable persist → restore → `loadParsedFilesForPaths` + * (1) Store-level — the durable persist → `loadParsedFilesForPaths` * round-trip at the EXACT seam scope-resolution consumes (phase.ts:255), * plus the index version gate and the prune-coherence rule. Build-free. * (2) Integration — a two-run `runChunkedParseAndResolve`: run #1 (all miss) * populates the durable store; run #2 (all hits) spawns NO worker and - * restores full coverage; the coherence gate re-dispatches when durable - * shards are absent; and a mixed-mode run (one file changed) hits the - * unchanged chunk while re-parsing the changed one. + * loads full coverage from durable shards; the coherence gate re-dispatches + * when durable shards are absent; and a mixed-mode run (one file changed) + * hits the unchanged chunk while re-parsing the changed one. */ import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; import fs from 'node:fs'; @@ -37,6 +37,9 @@ import { pathToFileURL } from 'node:url'; const prepareOverride = vi.hoisted(() => ({ impl: undefined as undefined | (() => Promise), })); +const persistOverride = vi.hoisted(() => ({ + impl: undefined as undefined | (() => Promise), +})); vi.mock('../../src/storage/parsedfile-store.js', async (importOriginal) => { const real = await importOriginal(); return { @@ -45,6 +48,14 @@ vi.mock('../../src/storage/parsedfile-store.js', async (importOriginal) => { prepareOverride.impl ? prepareOverride.impl() : real.prepareDurableParsedFileChunk(durableDir, chunkHash), + persistParsedFileChunk: ( + storagePath: string, + shardId: string, + parsedFiles: readonly ParsedFile[], + ) => + persistOverride.impl + ? persistOverride.impl() + : real.persistParsedFileChunk(storagePath, shardId, parsedFiles), }; }); @@ -57,9 +68,10 @@ import { } from '../../src/storage/parse-cache.js'; import { getDurableParsedFileDir, + getParsedFileStoreDir, prepareDurableParsedFileChunk, persistDurableParsedFileShardSync, - restoreDurableParsedFileShard, + durableChunkHasShards, loadParsedFilesForPaths, loadDurableParsedFileIndex, pruneAndSaveDurableParsedFileStore, @@ -92,29 +104,24 @@ describe('durable ParsedFile store — content-addressed warm-cache coverage', ( if (tempDir) fs.rmSync(tempDir, { recursive: true, force: true }); }); - it('persist → restore → loadParsedFilesForPaths gives full coverage (the warm seam)', async () => { + it('persist → loadParsedFilesForPaths gives full coverage (the warm seam)', async () => { const durableDir = getDurableParsedFileDir(tempDir); const chunkHash = 'a'.repeat(64); const files = ['src/a.ts', 'src/b.ts']; // A worker would write this at flush on a cache MISS. persistDurableParsedFileShardSync(durableDir, chunkHash, 7, 0, files.map(mkParsedFile)); - - // The run-scoped store is cleared at parse start; a warm hit restores. await clearParsedFileStore(tempDir); - const restored = await restoreDurableParsedFileShard(durableDir, tempDir, chunkHash); - expect(restored).toBe(1); - - // This is the EXACT call scope-resolution makes (phase.ts:255). Full - // coverage ⇒ preExtractedByPath has every file ⇒ no main-thread extract. - const loaded = await loadParsedFilesForPaths(tempDir, new Set(files)); + const wanted = new Set(files); + expect(await durableChunkHasShards(tempDir, chunkHash, wanted)).toBe(true); + const loaded = await loadParsedFilesForPaths(tempDir, wanted); expect([...loaded.keys()].sort()).toEqual([...files].sort()); }); - it('restore returns 0 when the chunk has no durable shards (caller re-dispatches)', async () => { - const durableDir = getDurableParsedFileDir(tempDir); - const restored = await restoreDurableParsedFileShard(durableDir, tempDir, 'b'.repeat(64)); - expect(restored).toBe(0); + it('durableChunkHasShards is false when the chunk has no durable shards', async () => { + expect(await durableChunkHasShards(tempDir, 'b'.repeat(64), new Set(['missing.ts']))).toBe( + false, + ); }); it('prepares a fresh durable generation without retaining old worker shards', async () => { @@ -129,14 +136,14 @@ describe('durable ParsedFile store — content-addressed warm-cache coverage', ( const shards = fs .readdirSync(chunkDir) - .filter((name) => name.endsWith('.json')) + .filter((name) => name.endsWith('.v8')) .sort(); - expect(shards).toEqual([`${chunkHash}-w1-0.json`, `${chunkHash}-w2-0.json`]); - await restoreDurableParsedFileShard(durableDir, tempDir, chunkHash); - const files = await loadParsedFilesForPaths( - tempDir, - new Set(['old.ts', 'new-a.ts', 'new-b.ts']), + expect(shards).toEqual([`${chunkHash}-w1-0.v8`, `${chunkHash}-w2-0.v8`]); + const wanted = new Set(['old.ts', 'new-a.ts', 'new-b.ts']); + expect(await durableChunkHasShards(tempDir, chunkHash, new Set(['new-a.ts', 'new-b.ts']))).toBe( + true, ); + const files = await loadParsedFilesForPaths(tempDir, wanted); expect([...files.keys()].sort()).toEqual(['new-a.ts', 'new-b.ts']); }); @@ -147,10 +154,10 @@ describe('durable ParsedFile store — content-addressed warm-cache coverage', ( await pruneAndSaveDurableParsedFileStore(durableDir, PARSE_CACHE_VERSION, new Set([chunkHash])); expect(await loadDurableParsedFileIndex(durableDir, PARSE_CACHE_VERSION)).toEqual( - new Set([chunkHash]), + new Map([[chunkHash, new Set(['x.ts'])]]), ); // A schema bump (different version) invalidates the whole durable store. - expect(await loadDurableParsedFileIndex(durableDir, '999+9.9.9')).toEqual(new Set()); + expect(await loadDurableParsedFileIndex(durableDir, '999+9.9.9')).toEqual(new Map()); }); it('prune keeps only keepKeys subdirs with ≥1 shard, drops the rest, and re-indexes', async () => { @@ -165,7 +172,7 @@ describe('durable ParsedFile store — content-addressed warm-cache coverage', ( expect(fs.existsSync(path.join(durableDir, keep))).toBe(true); expect(fs.existsSync(path.join(durableDir, drop))).toBe(false); expect(await loadDurableParsedFileIndex(durableDir, PARSE_CACHE_VERSION)).toEqual( - new Set([keep]), + new Map([[keep, new Set(['keep.ts'])]]), ); }); }); @@ -173,22 +180,44 @@ describe('durable ParsedFile store — content-addressed warm-cache coverage', ( // ─── Layer 2: parse-impl integration (injected worker, build-free) ─────────── // A test worker that mirrors the production flush contract: it writes a -// run-scoped shard AND a durable, content-addressed shard (when the flush -// carries a chunk hash) using the SAME directory layout as the real worker. -// Empty-scope ParsedFiles round-trip through plain JSON identically to -// `mapReplacer` (no Map fields), so the store bytes match the production path. +// run-scoped V8 shard AND a durable, content-addressed V8 shard (when the +// flush carries a chunk hash) using the SAME directory layout as the real worker. const writeStoreWorker = (workerPath: string, markerPath: string): void => { fs.writeFileSync( workerPath, ` const fs = require('node:fs'); const path = require('node:path'); +const v8 = require('node:v8'); +const { createHash } = require('node:crypto'); const { parentPort, threadId, workerData } = require('node:worker_threads'); const storePath = workerData && workerData.parsedFileStoreStoragePath; const durablePath = workerData && workerData.durableParsedFileStoragePath; let shardSeq = 0; fs.writeFileSync(${JSON.stringify(markerPath)}, 'spawned'); parentPort.postMessage({ type: 'ready' }); +const writeV8 = (filePath, graph, paths) => { + const payload = v8.serialize(graph); + const listing = paths.some((p) => /[\\r\\n\\0]/.test(p)) + ? Buffer.alloc(0) + : Buffer.from(paths.length + '\\n' + paths.join('\\n') + '\\n', 'utf8'); + const MAGIC = Buffer.from('GNXV8CF1'); + const v8ver = Buffer.from(process.versions.v8, 'utf8'); + const nodeMajor = Number.parseInt(process.versions.node.split('.')[0], 10); + const header = Buffer.allocUnsafe(16 + v8ver.length + 12); + MAGIC.copy(header, 0); + header.writeUInt32LE(5, 8); + header.writeUInt16LE(nodeMajor, 12); + header.writeUInt16LE(v8ver.length, 14); + v8ver.copy(header, 16); + let off = 16 + v8ver.length; + header.writeUInt32LE(listing.length === 0 ? 0 : paths.length, off); + header.writeUInt32LE(listing.length, off + 4); + header.writeUInt32LE(payload.length, off + 8); + fs.mkdirSync(path.dirname(filePath), { recursive: true }); + const payloadHash = createHash('sha256').update(listing).update(payload).digest(); + fs.writeFileSync(filePath, Buffer.concat([header, listing, payload, payloadHash])); +}; const reset = () => ({ nodes: [], relationships: [], symbols: [], imports: [], calls: [], assignments: [], heritage: [], routes: [], fetchCalls: [], fetchWrapperDefs: [], decoratorRoutes: [], routerIncludes: [], @@ -220,18 +249,27 @@ parentPort.on('message', (msg) => { if (msg && msg.type === 'flush') { if ((storePath || durablePath) && accumulated.parsedFiles.length > 0) { const seq = shardSeq++; - const payload = JSON.stringify(accumulated.parsedFiles); + const paths = accumulated.parsedFiles.map((pf) => pf.filePath); + let wroteStore = false; if (durablePath && typeof msg.chunkHash === 'string') { - const dir = path.join(durablePath, msg.chunkHash); - fs.mkdirSync(dir, { recursive: true }); - fs.writeFileSync(path.join(dir, msg.chunkHash + '-w' + threadId + '-' + seq + '.json'), payload); + writeV8( + path.join(durablePath, msg.chunkHash, msg.chunkHash + '-w' + threadId + '-' + seq + '.v8'), + accumulated.parsedFiles, + paths, + ); } if (storePath) { - const dir = path.join(storePath, 'parsedfile-store'); - fs.mkdirSync(dir, { recursive: true }); - fs.writeFileSync(path.join(dir, 'w' + threadId + '-' + seq + '.json'), payload); - accumulated.parsedFiles = []; + writeV8( + path.join(storePath, 'parsedfile-store', 'w' + threadId + '-' + seq + '.v8'), + accumulated.parsedFiles, + paths, + ); + wroteStore = true; } + const keepForMain = accumulated.parsedFiles.some((pf) => + pf.filePath.includes('persist-fallback') + ); + if (wroteStore && !keepForMain) accumulated.parsedFiles = []; } parentPort.postMessage({ type: 'result', data: accumulated }); accumulated = reset(); @@ -260,6 +298,8 @@ describe('parse-impl warm-cache ParsedFile coverage (#2038)', () => { }); afterEach(() => { if (tempDir) fs.rmSync(tempDir, { recursive: true, force: true }); + prepareOverride.impl = undefined; + persistOverride.impl = undefined; }); const writeFile = (rel: string, content: string): { path: string; size: number } => { @@ -329,7 +369,7 @@ describe('parse-impl warm-cache ParsedFile coverage (#2038)', () => { expect(fs.existsSync(markerPath)).toBe(true); // worker ran (miss) const chunkDir = path.join(getDurableParsedFileDir(storageDir), chunkHash); expect(fs.existsSync(chunkDir)).toBe(true); - expect(fs.readdirSync(chunkDir).filter((n) => n.endsWith('.json')).length).toBeGreaterThan(0); + expect(fs.readdirSync(chunkDir).filter((n) => n.endsWith('.v8')).length).toBeGreaterThan(0); expect(cache.usedKeys.has(chunkHash)).toBe(true); }); @@ -343,6 +383,39 @@ describe('parse-impl warm-cache ParsedFile coverage (#2038)', () => { } }); + it('retains worker ParsedFiles when the main-thread run-store write fails', async () => { + const f = writeFile( + 'src/persist-fallback.ts', + 'export function persistFallback() { return 1; }\n', + ); + persistOverride.impl = () => Promise.resolve(false); + + const result = await run(newCache(), [f]); + + expect(result.parsedFiles.map((parsed) => parsed.filePath)).toContain(f.path); + }); + + it('does not snapshot durable shards when the parse-cache payload is missing', async () => { + const f = writeFile('src/orphan.ts', 'export function orphan() { return 1; }\n'); + const chunkHash = computeChunkHash([ + { + filePath: f.path, + contentHash: fileContentHash(fs.readFileSync(path.join(repoDir, f.path), 'utf-8')), + }, + ]); + const durableDir = getDurableParsedFileDir(storageDir); + persistDurableParsedFileShardSync(durableDir, chunkHash, 1, 0, [mkParsedFile(f.path)]); + await pruneAndSaveDurableParsedFileStore(durableDir, PARSE_CACHE_VERSION, new Set([chunkHash])); + + await run(newCache(), [f]); + + const runShards = fs + .readdirSync(getParsedFileStoreDir(storageDir)) + .filter((name) => name.endsWith('.v8')); + expect(runShards.length).toBeGreaterThan(0); + expect(runShards.every((name) => !name.startsWith(chunkHash))).toBe(true); + }); + it('a repeated cache miss replaces the durable chunk generation', async () => { const f = writeFile('src/repeated.ts', 'export function repeated() { return 1; }\n'); const chunkHash = computeChunkHash([ @@ -356,11 +429,15 @@ describe('parse-impl warm-cache ParsedFile coverage (#2038)', () => { await run(newCache(), [f]); const chunkDir = path.join(getDurableParsedFileDir(storageDir), chunkHash); - const shards = fs.readdirSync(chunkDir).filter((name) => name.endsWith('.json')); + const shards = fs.readdirSync(chunkDir).filter((name) => name.endsWith('.v8')); expect(shards).toHaveLength(1); - const parsed = JSON.parse(fs.readFileSync(path.join(chunkDir, shards[0]!), 'utf-8')) as Array<{ - filePath: string; - }>; + const shard = shards[0]; + if (!shard) throw new Error('expected one durable V8 shard'); + const { tryLoadV8Cache } = await import('../../src/storage/v8-sidecar.js'); + const hit = await tryLoadV8Cache(path.join(chunkDir, shard)); + expect(hit?.kind).toBe('hit'); + if (hit?.kind !== 'hit') return; + const parsed = hit.value as Array<{ filePath: string }>; expect(parsed.map((item) => item.filePath)).toEqual(['src/repeated.ts']); }); @@ -420,6 +497,33 @@ describe('parse-impl warm-cache ParsedFile coverage (#2038)', () => { expect(fs.existsSync(markerPath)).toBe(true); }); + it('coherence gate: a parse-cache hit with a corrupt durable shard re-dispatches', async () => { + const f = writeFile('src/corrupt.ts', 'export function corrupt() { return 1; }\n'); + const cache = newCache(); + + await run(cache, [f]); + await persistCaches(cache); + const chunkHash = computeChunkHash([ + { + filePath: f.path, + contentHash: fileContentHash('export function corrupt() { return 1; }\n'), + }, + ]); + const chunkDir = path.join(getDurableParsedFileDir(storageDir), chunkHash); + const shard = fs.readdirSync(chunkDir).find((name) => name.endsWith('.v8')); + expect(shard).toBeDefined(); + if (!shard) return; + fs.writeFileSync(path.join(chunkDir, shard), Buffer.from([0, 1, 2])); + + const { loadParseCache } = await import('../../src/storage/parse-cache.js'); + const warm = await loadParseCache(storageDir); + fs.rmSync(markerPath, { force: true }); + + await run(warm as ReturnType, [f]); + + expect(fs.existsSync(markerPath)).toBe(true); + }); + it('mixed-mode: changing one file re-parses its chunk while the unchanged chunk restores', async () => { // Force one file per chunk (chunkByteBudget: 1) so a and b hash to DISTINCT // chunks — the true mixed-mode the pr-2038 mixed-mode gap warns about. @@ -446,7 +550,7 @@ describe('parse-impl warm-cache ParsedFile coverage (#2038)', () => { await run(warm as ReturnType, [a, b2], 1); - // The worker spawned (for the changed file b); a was restored from durable. + // The worker spawned (for the changed file b); a was loaded from durable. expect(fs.existsSync(markerPath)).toBe(true); // a's UNCHANGED chunk is still a hit served from the durable store. expect((warm as ReturnType).usedKeys.has(aHash)).toBe(true); diff --git a/gitnexus/test/unit/parsedfile-store.test.ts b/gitnexus/test/unit/parsedfile-store.test.ts index d799f2054..39fd4c9fe 100644 --- a/gitnexus/test/unit/parsedfile-store.test.ts +++ b/gitnexus/test/unit/parsedfile-store.test.ts @@ -1,5 +1,6 @@ import { describe, it, expect, vi } from 'vitest'; import { promises as nodeFsPromises } from 'node:fs'; +import v8 from 'node:v8'; import { mkdtemp, rm, readdir, readFile, writeFile } from 'fs/promises'; import { tmpdir } from 'os'; import path from 'path'; @@ -9,7 +10,7 @@ import { persistParsedFileChunk, persistParsedFileShardSync, persistDurableParsedFileShardSync, - restoreDurableParsedFileShard, + durableChunkHasShards, loadParsedFilesForPaths, getParsedFileStoreDir, getDurableParsedFileDir, @@ -246,24 +247,9 @@ describe('parsedfile-store', () => { const files = [makeParsedFile('a.c'), makeParsedFile('b.c')]; await persistParsedFileChunk(asyncDir, 'shard', files); persistParsedFileShardSync(syncDir, 'shard', files); - const asyncBytes = await readFile( - path.join(getParsedFileStoreDir(asyncDir), 'shard.json'), - 'utf-8', - ); - const syncBytes = await readFile( - path.join(getParsedFileStoreDir(syncDir), 'shard.json'), - 'utf-8', - ); - expect(syncBytes).toBe(asyncBytes); - const asyncPaths = await readFile( - path.join(getParsedFileStoreDir(asyncDir), 'shard.json.paths'), - 'utf-8', - ); - const syncPaths = await readFile( - path.join(getParsedFileStoreDir(syncDir), 'shard.json.paths'), - 'utf-8', - ); - expect(syncPaths).toBe(asyncPaths); + const asyncBytes = await readFile(path.join(getParsedFileStoreDir(asyncDir), 'shard.v8')); + const syncBytes = await readFile(path.join(getParsedFileStoreDir(syncDir), 'shard.v8')); + expect(syncBytes.equals(asyncBytes)).toBe(true); } finally { await rm(asyncDir, { recursive: true, force: true }); await rm(syncDir, { recursive: true, force: true }); @@ -629,116 +615,71 @@ describe('parsedfile-store receiverChain sanitation', () => { } }); - it('writes a .json.paths sidecar and skips JSON for non-intersecting shards (#3087)', async () => { - const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-sidecar-')); + it('writes one .v8 shard per chunk and skips deserialize for non-intersecting listings (#3087)', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-v8-skip-')); + const deserialize = vi.spyOn(v8, 'deserialize'); try { await persistParsedFileChunk(dir, 'chunk-0', [makeParsedFile('a.c')]); await persistParsedFileChunk(dir, 'chunk-1', [makeParsedFile('b.c')]); - const storeDir = getParsedFileStoreDir(dir); - const names = await readdir(storeDir); - expect(names.sort()).toEqual([ - 'chunk-0.json', - 'chunk-0.json.paths', - 'chunk-1.json', - 'chunk-1.json.paths', + expect((await readdir(getParsedFileStoreDir(dir))).sort()).toEqual([ + 'chunk-0.v8', + 'chunk-1.v8', ]); - const readSpy = vi.spyOn(nodeFsPromises, 'readFile'); - try { - const loaded = await loadParsedFilesForPaths(dir, new Set(['b.c'])); - expect([...loaded.keys()]).toEqual(['b.c']); - const jsonReads = readSpy.mock.calls.filter(([p]) => { - const n = String(p); - return n.endsWith('.json') && !n.endsWith('.json.paths'); - }); - expect(jsonReads).toHaveLength(1); - expect(String(jsonReads[0][0])).toMatch(/chunk-1\.json$/); - } finally { - readSpy.mockRestore(); - } + deserialize.mockClear(); + const loaded = await loadParsedFilesForPaths(dir, new Set(['b.c'])); + expect([...loaded.keys()]).toEqual(['b.c']); + expect(deserialize).toHaveBeenCalledTimes(1); } finally { + deserialize.mockRestore(); await rm(dir, { recursive: true, force: true }); } }); - it('reads a shard when its sidecar is missing or garbage (#3087)', async () => { - const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-sidecar-fb-')); + it('misses when the embedded path listing is corrupted', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-listing-fb-')); try { await persistParsedFileChunk(dir, 'ok', [makeParsedFile('a.c')]); - await persistParsedFileChunk(dir, 'bad', [makeParsedFile('b.c')]); - const storeDir = getParsedFileStoreDir(dir); - await rm(path.join(storeDir, 'ok.json.paths'), { force: true }); - await writeFile(path.join(storeDir, 'bad.json.paths'), 'not\x00valid', 'utf-8'); - const loaded = await loadParsedFilesForPaths(dir, new Set(['a.c', 'b.c'])); - expect(loaded.has('a.c')).toBe(true); - expect(loaded.has('b.c')).toBe(true); + const dest = path.join(getParsedFileStoreDir(dir), 'ok.v8'); + const buf = await readFile(dest); + const v8len = buf.readUInt16LE(14); + const pathsOff = 16 + v8len + 12; + buf[pathsOff] = 0; + await writeFile(dest, buf); + const loaded = await loadParsedFilesForPaths(dir, new Set(['a.c'])); + expect(loaded.has('a.c')).toBe(false); } finally { await rm(dir, { recursive: true, force: true }); } }); - it('reads a shard when its sidecar is truncated without a trailing newline (#3087)', async () => { - const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-sidecar-trunc-')); - try { - await persistParsedFileChunk(dir, 'ok', [makeParsedFile('wanted.c')]); - const storeDir = getParsedFileStoreDir(dir); - await writeFile(path.join(storeDir, 'ok.json.paths'), 'unrelated.c', 'utf-8'); - const loaded = await loadParsedFilesForPaths(dir, new Set(['wanted.c'])); - expect(loaded.has('wanted.c')).toBe(true); - } finally { - await rm(dir, { recursive: true, force: true }); - } - }); - - it('reads a shard when its sidecar is a newline-terminated partial listing', async () => { - const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-sidecar-partial-')); - try { - await persistParsedFileChunk(dir, 'ok', [makeParsedFile('wanted.c')]); - const storeDir = getParsedFileStoreDir(dir); - await writeFile(path.join(storeDir, 'ok.json.paths'), 'unrelated.c\n', 'utf-8'); - const loaded = await loadParsedFilesForPaths(dir, new Set(['wanted.c'])); - expect(loaded.has('wanted.c')).toBe(true); - } finally { - await rm(dir, { recursive: true, force: true }); - } - }); - - it('reads a shard when its sidecar contains CR', async () => { - const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-sidecar-cr-')); - try { - await persistParsedFileChunk(dir, 'ok', [makeParsedFile('wanted.c')]); - const storeDir = getParsedFileStoreDir(dir); - await writeFile(path.join(storeDir, 'ok.json.paths'), 'unrelated.c\r\n', 'utf-8'); - const loaded = await loadParsedFilesForPaths(dir, new Set(['wanted.c'])); - expect(loaded.has('wanted.c')).toBe(true); - } finally { - await rm(dir, { recursive: true, force: true }); - } - }); - - it('omits a sidecar when a filePath contains a newline and still loads JSON', async () => { - const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-sidecar-nl-')); + it('omits a path listing when a filePath contains a newline and still loads', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-listing-nl-')); const weird = 'weird\nname.c'; + const deserialize = vi.spyOn(v8, 'deserialize'); try { await persistParsedFileChunk(dir, 'ok', [makeParsedFile(weird)]); - const storeDir = getParsedFileStoreDir(dir); - expect(await readdir(storeDir)).toEqual(['ok.json']); - const loaded = await loadParsedFilesForPaths(dir, new Set([weird])); - expect(loaded.has(weird)).toBe(true); + expect(await readdir(getParsedFileStoreDir(dir))).toEqual(['ok.v8']); + deserialize.mockClear(); + const loaded = await loadParsedFilesForPaths(dir, new Set(['unrelated.c'])); + expect(loaded.size).toBe(0); + expect(deserialize).toHaveBeenCalledTimes(1); + expect((await loadParsedFilesForPaths(dir, new Set([weird]))).has(weird)).toBe(true); } finally { + deserialize.mockRestore(); await rm(dir, { recursive: true, force: true }); } }); - it('removes a stale sidecar when a rewritten shard is no longer listing-safe', async () => { - const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-sidecar-stale-')); + it('rewrites a shard in place when the path listing is no longer safe', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-listing-rewrite-')); const weird = 'weird\nname.c'; try { await persistParsedFileChunk(dir, 'ok', [makeParsedFile('safe.c')]); await persistParsedFileChunk(dir, 'ok', [makeParsedFile(weird)]); - const storeDir = getParsedFileStoreDir(dir); - expect(await readdir(storeDir)).toEqual(['ok.json']); - const loaded = await loadParsedFilesForPaths(dir, new Set([weird])); + expect(await readdir(getParsedFileStoreDir(dir))).toEqual(['ok.v8']); + const loaded = await loadParsedFilesForPaths(dir, new Set([weird, 'safe.c'])); expect(loaded.has(weird)).toBe(true); + expect(loaded.has('safe.c')).toBe(false); } finally { await rm(dir, { recursive: true, force: true }); } @@ -779,101 +720,128 @@ describe('parsedfile-store receiverChain sanitation', () => { } }); - it('restoreDurableParsedFileShard copies sidecars and returns JSON shard count (#3087)', async () => { - const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-restore-')); + it('restores a complete durable chunk into a stable run-store snapshot', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-durable-load-')); try { const durable = getDurableParsedFileDir(dir); persistDurableParsedFileShardSync(durable, 'abc', 1, 0, [makeParsedFile('a.c')]); - const restored = await restoreDurableParsedFileShard(durable, dir, 'abc'); - expect(restored).toBe(1); - const storeDir = getParsedFileStoreDir(dir); - expect(await readdir(storeDir)).toEqual( - expect.arrayContaining(['abc-w1-0.json', 'abc-w1-0.json.paths']), - ); - expect(await readFile(path.join(storeDir, 'abc-w1-0.json.paths'), 'utf-8')).toBe('1\na.c\n'); + expect(await durableChunkHasShards(dir, 'abc', new Set(['a.c']))).toBe(true); + await rm(path.join(durable, 'abc'), { recursive: true, force: true }); const loaded = await loadParsedFilesForPaths(dir, new Set(['a.c'])); expect(loaded.has('a.c')).toBe(true); + expect(await readdir(getParsedFileStoreDir(dir))).toEqual(['abc-w1-0.v8']); } finally { await rm(dir, { recursive: true, force: true }); } }); - it('restoreDurableParsedFileShard unlinks a stale dest sidecar when the source has none', async () => { - const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-restore-stale-')); + it('rejects a durable chunk with corrupt or incomplete shard coverage', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-durable-partial-')); try { const durable = getDurableParsedFileDir(dir); persistDurableParsedFileShardSync(durable, 'abc', 1, 0, [makeParsedFile('a.c')]); - const durableShard = path.join(durable, 'abc', 'abc-w1-0.json'); - await rm(`${durableShard}.paths`, { force: true }); - const storeDir = getParsedFileStoreDir(dir); - await nodeFsPromises.mkdir(storeDir, { recursive: true }); - await writeFile(path.join(storeDir, 'abc-w1-0.json.paths'), 'stale.c\n', 'utf-8'); - const restored = await restoreDurableParsedFileShard(durable, dir, 'abc'); - expect(restored).toBe(1); - await expect(readFile(path.join(storeDir, 'abc-w1-0.json.paths'), 'utf-8')).rejects.toThrow(); - const loaded = await loadParsedFilesForPaths(dir, new Set(['a.c'])); - expect(loaded.has('a.c')).toBe(true); + persistDurableParsedFileShardSync(durable, 'abc', 2, 0, [makeParsedFile('b.c')]); + await writeFile(path.join(durable, 'abc', 'abc-w2-0.v8'), Buffer.from([0, 1, 2])); + + expect(await durableChunkHasShards(dir, 'abc', new Set(['a.c', 'b.c']))).toBe(false); + expect(await readdir(getParsedFileStoreDir(dir))).toEqual([]); } finally { await rm(dir, { recursive: true, force: true }); } }); - it('drops a leftover sidecar before overwriting JSON so load cannot skip new paths', async () => { - const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-sidecar-rewrite-')); + it('rejects valid durable shards that do not cover every indexed path', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-durable-missing-')); try { - await persistParsedFileChunk(dir, 'ok', [makeParsedFile('stale.c')]); - const origUnlink = nodeFsPromises.unlink.bind(nodeFsPromises); - const origWrite = nodeFsPromises.writeFile.bind(nodeFsPromises); - const order: string[] = []; - const unlinkSpy = vi - .spyOn(nodeFsPromises, 'unlink') - .mockImplementation(async (p, ...rest) => { - order.push(`unlink:${path.basename(String(p))}`); - return origUnlink(p, ...rest); - }); - const writeSpy = vi - .spyOn(nodeFsPromises, 'writeFile') - .mockImplementation(async (p, data, enc) => { - order.push(`write:${path.basename(String(p))}`); - return origWrite(p, data, enc); - }); - try { - await persistParsedFileChunk(dir, 'ok', [makeParsedFile('a.c')]); - } finally { - unlinkSpy.mockRestore(); - writeSpy.mockRestore(); - } - const jsonIdx = order.indexOf('write:ok.json'); - const pathsIdx = order.indexOf('unlink:ok.json.paths'); - expect(pathsIdx).toBeGreaterThanOrEqual(0); - expect(pathsIdx).toBeLessThan(jsonIdx); - const loaded = await loadParsedFilesForPaths(dir, new Set(['a.c'])); - expect(loaded.has('a.c')).toBe(true); + const durable = getDurableParsedFileDir(dir); + persistDurableParsedFileShardSync(durable, 'abc', 1, 0, [makeParsedFile('a.c')]); + + expect(await durableChunkHasShards(dir, 'abc', new Set(['a.c', 'b.c']))).toBe(false); + expect(await readdir(getParsedFileStoreDir(dir))).toEqual([]); } finally { await rm(dir, { recursive: true, force: true }); } }); - it('persist still succeeds when the sidecar write fails', async () => { - const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-sidecar-enospc-')); + it('run-store shards overlay durable hits for the same path', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-overlay-')); try { - await persistParsedFileChunk(dir, 'ok', [makeParsedFile('stale.c')]); - const orig = nodeFsPromises.writeFile.bind(nodeFsPromises); - const spy = vi.spyOn(nodeFsPromises, 'writeFile').mockImplementation(async (p, data, enc) => { - if (String(p).endsWith('.paths')) { - throw Object.assign(new Error('ENOSPC'), { code: 'ENOSPC' }); - } - return orig(p, data, enc); - }); - try { - await persistParsedFileChunk(dir, 'ok', [makeParsedFile('a.c')]); - } finally { - spy.mockRestore(); - } + const durable = getDurableParsedFileDir(dir); + persistDurableParsedFileShardSync(durable, 'abc', 1, 0, [makeParsedFile('a.c')]); + expect(await durableChunkHasShards(dir, 'abc', new Set(['a.c']))).toBe(true); + persistParsedFileShardSync(dir, 'w1-0', [makeParsedFile('other.c')]); + persistParsedFileShardSync(dir, 'w1-1', [makeStoreEntry('a.c', { moduleScope: 'from-run' })]); + const loaded = await loadParsedFilesForPaths(dir, new Set(['a.c', 'other.c'])); + expect(loaded.get('a.c')?.moduleScope).toBe('from-run'); + expect(loaded.has('other.c')).toBe(true); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('clearParsedFileStore leaves the durable cache intact', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-durable-keep-')); + try { + const durable = getDurableParsedFileDir(dir); + persistDurableParsedFileShardSync(durable, 'abc', 1, 0, [makeParsedFile('a.c')]); + const src = path.join(durable, 'abc', 'abc-w1-0.v8'); + const before = await readFile(src); + persistParsedFileShardSync(dir, 'w1-0', [makeParsedFile('run.c')]); + await clearParsedFileStore(dir); + expect(await readFile(src)).toEqual(before); + expect(await durableChunkHasShards(dir, 'abc', new Set(['a.c']))).toBe(true); + expect((await loadParsedFilesForPaths(dir, new Set(['a.c']))).has('a.c')).toBe(true); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('round-trips Maps and shared def identity through V8 (#3089)', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-v8-id-')); + try { + const def = { + nodeId: 'Function:a.c:fn', + filePath: 'a.c', + type: 'Function' as const, + qualifiedName: 'fn', + }; + const pf = makeParsedFile('a.c'); + (pf.localDefs as unknown as object[])[0] = def; + (pf.scopes[0] as { ownedDefs: object[] }).ownedDefs = [def]; + await persistParsedFileChunk(dir, 'ok', [pf]); + const loadedFile = (await loadParsedFilesForPaths(dir, new Set(['a.c']))).get('a.c'); + expect(loadedFile).toBeDefined(); + if (!loadedFile) return; + expect(loadedFile.scopes[0].bindings).toBeInstanceOf(Map); + expect(loadedFile.localDefs[0]).toBe(loadedFile.scopes[0].ownedDefs[0]); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('treats a missing or corrupt V8 shard as a miss with no JSON fallback', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-v8-miss-')); + try { + await persistParsedFileChunk(dir, 'gone', [makeParsedFile('a.c')]); + await persistParsedFileChunk(dir, 'junk', [makeParsedFile('b.c')]); const storeDir = getParsedFileStoreDir(dir); - await expect(readFile(path.join(storeDir, 'ok.json.paths'), 'utf-8')).rejects.toThrow(); - const loaded = await loadParsedFilesForPaths(dir, new Set(['a.c'])); - expect(loaded.has('a.c')).toBe(true); + await rm(path.join(storeDir, 'gone.v8')); + await writeFile(path.join(storeDir, 'junk.v8'), Buffer.from([0, 1, 2, 3, 4])); + const loaded = await loadParsedFilesForPaths(dir, new Set(['a.c', 'b.c'])); + expect(loaded.size).toBe(0); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('returns false when the atomic V8 publish cannot replace the dest', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'pfstore-v8-blocked-')); + try { + const dest = path.join(getParsedFileStoreDir(dir), 'ok.v8'); + await nodeFsPromises.mkdir(dest, { recursive: true }); + await writeFile(path.join(dest, 'occupied'), 'x', 'utf-8'); + expect(await persistParsedFileChunk(dir, 'ok', [makeParsedFile('a.c')])).toBe(false); + expect((await loadParsedFilesForPaths(dir, new Set(['a.c']))).size).toBe(0); } finally { await rm(dir, { recursive: true, force: true }); } diff --git a/gitnexus/test/unit/repo-manager-registry-atomic-write.test.ts b/gitnexus/test/unit/repo-manager-registry-atomic-write.test.ts index a30005ed3..ba0fbe777 100644 --- a/gitnexus/test/unit/repo-manager-registry-atomic-write.test.ts +++ b/gitnexus/test/unit/repo-manager-registry-atomic-write.test.ts @@ -8,9 +8,10 @@ * startup (`LocalBackend.init` -> `refreshRepos`, nothing catches). * * Separate from repo-manager.test.ts: Vitest cannot vi.spyOn ESM namespace - * exports of fs/promises, and these tests must drive `fs.rename` itself — a - * delegating vi.mock is required (same split as repo-manager-rm-failure.test.ts - * and repo-manager-ensure-ignore-readonly.test.ts, #1549). + * exports of `node:fs` promises, and these tests must drive `retryRename`'s + * `fsp.rename` — a delegating vi.mock is required. `writeRegistry` publishes + * via `writeFileAtomic` (`node:fs`), so mocking `fs/promises` never intercepts + * the rename (same split as repo-manager-rm-failure.test.ts, #1549). */ import { describe, it, expect, beforeAll, afterAll, beforeEach, afterEach, vi } from 'vitest'; import path from 'path'; @@ -20,16 +21,17 @@ const fsCtx = vi.hoisted(() => ({ realRename: null as ((src: string, dst: string) => Promise) | null, })); -vi.mock('fs/promises', async (importOriginal) => { - const actual = await importOriginal(); - const d = actual.default; - fsCtx.realRename = d.rename.bind(d) as (src: string, dst: string) => Promise; +vi.mock('node:fs', async (importOriginal) => { + const actual = await importOriginal(); + const promises = actual.promises; + fsCtx.realRename = promises.rename.bind(promises) as (src: string, dst: string) => Promise; fsCtx.renameMock.mockImplementation((src: string, dst: string) => fsCtx.realRename!(src, dst)); return { - default: new Proxy(d, { - get(target, prop) { + ...actual, + promises: new Proxy(promises, { + get(target, prop, receiver) { if (prop === 'rename') return fsCtx.renameMock; - const v = Reflect.get(target, prop, target) as unknown; + const v = Reflect.get(target, prop, receiver) as unknown; return typeof v === 'function' ? (v as (...args: unknown[]) => unknown).bind(target) : v; }, }), diff --git a/gitnexus/test/unit/storage/fs-atomic.test.ts b/gitnexus/test/unit/storage/fs-atomic.test.ts index deff8da8c..7e8049dab 100644 --- a/gitnexus/test/unit/storage/fs-atomic.test.ts +++ b/gitnexus/test/unit/storage/fs-atomic.test.ts @@ -4,10 +4,15 @@ * source-text guards in test/unit/group/insecure-tempfile.test.ts used to * approximate by regex, for three separate copies of the sequence. */ -import { describe, it, expect, beforeEach, afterEach } from 'vitest'; -import fs from 'node:fs/promises'; +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest'; +import { promises as fs } from 'node:fs'; import path from 'node:path'; -import { writeFileAtomic } from '../../../src/storage/fs-atomic.js'; +import { + linkOrCopyFile, + writeFileAtomic, + writeFileAtomicBytes, + writeFileAtomicBytesSync, +} from '../../../src/storage/fs-atomic.js'; import { createTempDir } from '../../helpers/test-db.js'; describe('writeFileAtomic', () => { @@ -71,3 +76,120 @@ describe('writeFileAtomic', () => { expect((await fs.readdir(tmp.dbPath)).sort()).toEqual(['blocked', 'thing.json']); }); }); + +describe('writeFileAtomicBytes', () => { + let tmp: Awaited>; + let target: string; + + beforeEach(async () => { + tmp = await createTempDir('gitnexus-fs-atomic-bin-'); + target = path.join(tmp.dbPath, 'thing.bin'); + }); + + afterEach(async () => { + await tmp.cleanup(); + }); + + it('publishes binary data and leaves no tmp file behind', async () => { + const bytes = Buffer.from([0, 1, 255, 10]); + await writeFileAtomicBytes(target, bytes); + expect(Buffer.from(await fs.readFile(target))).toEqual(bytes); + expect((await fs.readdir(tmp.dbPath)).filter((f) => f !== 'thing.bin')).toEqual([]); + }); + + it('writeFileAtomicBytesSync publishes the same bytes', async () => { + const bytes = Buffer.from('hello'); + writeFileAtomicBytesSync(target, bytes); + expect(Buffer.from(await fs.readFile(target))).toEqual(bytes); + }); +}); + +describe('linkOrCopyFile (#3090)', () => { + let tmp: Awaited>; + + beforeEach(async () => { + tmp = await createTempDir('gitnexus-link-or-copy-'); + }); + + afterEach(async () => { + vi.restoreAllMocks(); + await tmp.cleanup(); + }); + + const hardlinksWork = async (): Promise => { + const a = path.join(tmp.dbPath, '.probe-a'); + const b = path.join(tmp.dbPath, '.probe-b'); + await fs.writeFile(a, 'x'); + try { + await fs.link(a, b); + return true; + } catch { + return false; + } + }; + + it('hardlinks when the filesystem allows it', async () => { + if (!(await hardlinksWork())) return; + const src = path.join(tmp.dbPath, 'src.bin'); + const dst = path.join(tmp.dbPath, 'dst.bin'); + await fs.writeFile(src, 'payload'); + await linkOrCopyFile(src, dst); + const dstFd = await fs.open(dst, 'r'); + try { + const [s, d] = await Promise.all([fs.stat(src), dstFd.stat()]); + expect(d.ino).toBe(s.ino); + expect(s.nlink).toBe(2); + expect(await dstFd.readFile('utf-8')).toBe('payload'); + } finally { + await dstFd.close(); + } + }); + + it('falls back to tmp+rename when link reports EXDEV', async () => { + const src = path.join(tmp.dbPath, 'src.bin'); + const dst = path.join(tmp.dbPath, 'dst.bin'); + await fs.writeFile(src, 'payload'); + const err = Object.assign(new Error('cross-device'), { code: 'EXDEV' }); + vi.spyOn(fs, 'link').mockRejectedValue(err); + await linkOrCopyFile(src, dst); + expect(await fs.readFile(dst, 'utf-8')).toBe('payload'); + const [s, d] = await Promise.all([fs.stat(src), fs.stat(dst)]); + if (s.ino !== 0) expect(d.ino).not.toBe(s.ino); + expect(s.nlink).toBe(1); + expect((await fs.readdir(tmp.dbPath)).filter((f) => f.includes('.tmp.'))).toEqual([]); + }); + + it('replaces an existing dest via rename without touching src', async () => { + const src = path.join(tmp.dbPath, 'src.bin'); + const dst = path.join(tmp.dbPath, 'dst.bin'); + await fs.writeFile(src, 'new-bytes'); + await fs.writeFile(dst, 'old-bytes'); + await linkOrCopyFile(src, dst); + expect(await fs.readFile(dst, 'utf-8')).toBe('new-bytes'); + expect(await fs.readFile(src, 'utf-8')).toBe('new-bytes'); + }); + + it('does not write through a dest that is already a hardlink to other durable bytes', async () => { + if (!(await hardlinksWork())) return; + const durable = path.join(tmp.dbPath, 'durable.bin'); + const dst = path.join(tmp.dbPath, 'dst.bin'); + const src = path.join(tmp.dbPath, 'src.bin'); + await fs.writeFile(durable, 'DURABLE'); + await fs.link(durable, dst); + await fs.writeFile(src, 'FRESH'); + const durableFd = await fs.open(durable, 'r'); + try { + const before = await durableFd.stat(); + vi.spyOn(fs, 'link').mockRejectedValue(Object.assign(new Error('exdev'), { code: 'EXDEV' })); + await linkOrCopyFile(src, dst); + const after = await durableFd.stat(); + expect(await durableFd.readFile('utf-8')).toBe('DURABLE'); + expect(after.ino).toBe(before.ino); + expect(after.nlink).toBe(1); + expect(await fs.readFile(dst, 'utf-8')).toBe('FRESH'); + expect((await fs.stat(dst)).ino).not.toBe(before.ino); + } finally { + await durableFd.close(); + } + }); +}); diff --git a/gitnexus/test/unit/v8-sidecar.test.ts b/gitnexus/test/unit/v8-sidecar.test.ts new file mode 100644 index 000000000..fb88d986f --- /dev/null +++ b/gitnexus/test/unit/v8-sidecar.test.ts @@ -0,0 +1,204 @@ +import { describe, it, expect } from 'vitest'; +import { mkdtemp, rm, writeFile, readFile } from 'fs/promises'; +import { tmpdir } from 'os'; +import path from 'path'; +import { + inspectV8Cache, + internGraphStrings, + parseCachePathListing, + tryLoadV8Cache, + writeV8CacheFile, + V8_CACHE_FORMAT, +} from '../../src/storage/v8-sidecar.js'; + +describe('v8 cache envelope', () => { + it('interns duplicate strings in place and keeps Map identity', () => { + const pool = new Map(); + const m = new Map([['k', 'dup']]); + const graph = { a: 'dup', b: 'dup', m }; + internGraphStrings(graph, pool); + expect(graph.a).toBe(graph.b); + expect(graph.m).toBe(m); + expect(graph.m.get('k')).toBe(graph.a); + expect([...pool.keys()].sort()).toEqual(['dup', 'k']); + expect(pool.get('dup')).toBe('dup'); + }); + + it('round-trips a live graph including Maps', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'v8cf-')); + try { + const filePath = path.join(dir, 'shard.v8'); + const graph = { n: 1, nested: { s: 'x' }, map: new Map([['a', 1]]) }; + expect(await writeV8CacheFile(filePath, graph)).toBe(true); + const hit = await tryLoadV8Cache(filePath); + expect(hit?.kind).toBe('hit'); + if (hit?.kind !== 'hit') return; + const value = hit.value as typeof graph; + expect(value.n).toBe(1); + expect(value.map).toBeInstanceOf(Map); + expect(value.map.get('a')).toBe(1); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('skips deserialize when a valid path listing misses wantPaths', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'v8cf-skip-')); + try { + const filePath = path.join(dir, 'shard.v8'); + await writeV8CacheFile(filePath, [{ filePath: 'a.c' }], ['a.c']); + const skip = await tryLoadV8Cache(filePath, undefined, new Set(['other.c'])); + expect(skip?.kind).toBe('skip'); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('misses when the path listing bytes are corrupted', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'v8cf-failclosed-')); + try { + const filePath = path.join(dir, 'shard.v8'); + await writeV8CacheFile(filePath, [{ filePath: 'a.c' }], ['a.c']); + const buf = await readFile(filePath); + const v8len = buf.readUInt16LE(14); + const pathBytesOff = 16 + v8len + 4; + const pathBytes = buf.readUInt32LE(pathBytesOff); + const pathsOff = 16 + v8len + 12; + buf.fill(0x00, pathsOff, pathsOff + Math.min(pathBytes, 1)); + await writeFile(filePath, buf); + const hit = await tryLoadV8Cache(filePath, undefined, new Set(['other.c'])); + expect(hit).toBeUndefined(); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('misses when a same-length path listing is rewritten without the digest', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'v8cf-list-tamper-')); + try { + const filePath = path.join(dir, 'shard.v8'); + await writeV8CacheFile(filePath, [{ filePath: 'a.c' }], ['a.c']); + const buf = await readFile(filePath); + const v8len = buf.readUInt16LE(14); + const pathsOff = 16 + v8len + 12; + const listing = Buffer.from('1\nb.c\n'); + listing.copy(buf, pathsOff); + await writeFile(filePath, buf); + expect(await inspectV8Cache(filePath)).toBeUndefined(); + expect(await tryLoadV8Cache(filePath, undefined, new Set(['other.c']))).toBeUndefined(); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('treats invalid UTF-8 path listing bytes as untrusted', () => { + expect(parseCachePathListing(Buffer.from([0x31, 0x0a, 0xff, 0x0a]))).toBeNull(); + }); + + it('misses when the path listing is invalid UTF-8 instead of skipping', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'v8cf-utf8-')); + try { + const filePath = path.join(dir, 'shard.v8'); + await writeV8CacheFile(filePath, [{ filePath: 'a.c' }], ['a.c']); + const buf = await readFile(filePath); + const v8len = buf.readUInt16LE(14); + const pathBytes = buf.readUInt32LE(16 + v8len + 4); + const pathsOff = 16 + v8len + 12; + buf.fill(0xff, pathsOff, pathsOff + Math.min(pathBytes, 1)); + await writeFile(filePath, buf); + const hit = await tryLoadV8Cache(filePath, undefined, new Set(['other.c'])); + expect(hit).toBeUndefined(); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('deserializes when the envelope path count disagrees with the listing', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'v8cf-count-')); + try { + const filePath = path.join(dir, 'shard.v8'); + await writeV8CacheFile(filePath, [{ filePath: 'a.c' }], ['a.c']); + const buf = await readFile(filePath); + const v8len = buf.readUInt16LE(14); + buf.writeUInt32LE(2, 16 + v8len); + await writeFile(filePath, buf); + expect(await inspectV8Cache(filePath)).toBeUndefined(); + const hit = await tryLoadV8Cache(filePath, undefined, new Set(['other.c'])); + expect(hit?.kind).toBe('hit'); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('misses when internGraphStrings throws during materialization', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'v8cf-intern-')); + try { + const filePath = path.join(dir, 'shard.v8'); + await writeV8CacheFile(filePath, { s: 'x' }); + const pool = new Map(); + pool.set = () => { + throw new Error('intern boom'); + }; + expect(await tryLoadV8Cache(filePath, pool)).toBeUndefined(); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('misses when the recorded Node major does not match this runtime', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'v8cf-compat-')); + try { + const filePath = path.join(dir, 'shard.v8'); + await writeV8CacheFile(filePath, { ok: true }); + const buf = await readFile(filePath); + buf.writeUInt16LE(1, 12); + await writeFile(filePath, buf); + expect(await tryLoadV8Cache(filePath)).toBeUndefined(); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('misses when the recorded V8 version does not match this runtime', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'v8cf-v8-compat-')); + try { + const filePath = path.join(dir, 'shard.v8'); + await writeV8CacheFile(filePath, { ok: true }); + const buf = await readFile(filePath); + buf[16] ^= 1; + await writeFile(filePath, buf); + expect(await tryLoadV8Cache(filePath)).toBeUndefined(); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('misses when the payload checksum rejects same-size corruption', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'v8cf-payload-')); + try { + const filePath = path.join(dir, 'shard.v8'); + await writeV8CacheFile(filePath, { ok: true }); + const buf = await readFile(filePath); + expect(buf.readUInt32LE(8)).toBe(V8_CACHE_FORMAT); + const v8len = buf.readUInt16LE(14); + const pathBytes = buf.readUInt32LE(16 + v8len + 4); + const payloadOff = 16 + v8len + 12 + pathBytes; + buf[payloadOff] ^= 0xff; + await writeFile(filePath, buf); + expect(await tryLoadV8Cache(filePath)).toBeUndefined(); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); + + it('misses a garbage magic without throwing', async () => { + const dir = await mkdtemp(path.join(tmpdir(), 'v8cf-magic-')); + try { + const filePath = path.join(dir, 'shard.v8'); + await writeFile(filePath, Buffer.alloc(64, 7)); + expect(await tryLoadV8Cache(filePath)).toBeUndefined(); + } finally { + await rm(dir, { recursive: true, force: true }); + } + }); +}); From 43a842724d66b4a94ac7a2044fcf26aa3f876ca1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sun, 30 Aug 2026 23:41:13 +0100 Subject: [PATCH 04/26] fix: bind Razor ViewComponent names to in-repo classes (#3104) * fix: bind Razor ViewComponent names to in-repo classes Index Component.InvokeAsync("Name") and in-repo ViewComponent("Name") as CALLS to workspace ViewComponent classes so impact sees real callers instead of an empty graph. SDK types stay unresolved. Fixes #2991 Co-authored-by: Cursor * chore(autofix): apply prettier + eslint fixes via /autofix command * fix: scan Razor and C# ViewComponent names without regex holes Use string-aware lexers so combined Name= aliases, code-block calls, this/base helpers, and escaped @@ markup match ASP.NET instead of emitting false or missing CALLS. Co-authored-by: Cursor * chore(autofix): apply prettier + eslint fixes via /autofix command * perf: skip Razor scans without ViewComponent tokens Preserve the lexer correctness fixes while avoiding per-character work for the common view that cannot contain a supported invocation. Co-authored-by: Cursor * test: gate Razor ViewComponent extractor scaling in CI Wire mixed-corpus tripwire + GITNEXUS_BENCH loader/scaling checks into the dedicated ci-tests benchmarks job so the #2991 lexer cannot regress without a wall-clock gate. Co-authored-by: Cursor * fix: read Razor views through one file handle CodeQL js/file-system-race: the size gate stat'd the path and the read re-resolved it, so a template swapped in between could be read past the size ceiling. Both now go through the same handle. Co-authored-by: Cursor --------- Co-authored-by: Gergo Magyar Co-authored-by: Cursor Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- .github/workflows/ci-tests.yml | 9 +- .../languages/csharp/razor-view-components.ts | 955 ++++++++++++++++++ .../languages/csharp/resolution-config.ts | 12 +- .../languages/csharp/scope-resolver.ts | 15 + ...rp-razor-view-components-benchmark.test.ts | 179 ++++ .../csharp-razor-view-components.test.ts | 127 +++ .../csharp/razor-view-components.test.ts | 202 ++++ 7 files changed, 1494 insertions(+), 5 deletions(-) create mode 100644 gitnexus/src/core/ingestion/languages/csharp/razor-view-components.ts create mode 100644 gitnexus/test/integration/csharp-razor-view-components-benchmark.test.ts create mode 100644 gitnexus/test/integration/csharp-razor-view-components.test.ts create mode 100644 gitnexus/test/unit/scope-resolution/csharp/razor-view-components.test.ts diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index 91d44ac45..3cf5c0aeb 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -711,16 +711,17 @@ jobs: - name: Cross-language pipeline benchmarks (GITNEXUS_BENCH, serial) if: ${{ !cancelled() }} - # cpp-adl-benchmark.test.ts is not a `*-pipeline-benchmark.test.ts` but - # belongs here for the same reason: it is skipIf-gated on GITNEXUS_BENCH, - # so it had never run in CI and the PR #1990 ADL emit-scaling guard it - # holds was dead. ~45s of test time. + # cpp-adl-benchmark.test.ts and csharp-razor-view-components-benchmark.test.ts + # are not `*-pipeline-benchmark.test.ts` files but belong here for the + # same reason: they are skipIf-gated on GITNEXUS_BENCH, so the scaling + # guards they hold never run in the main coverage job. env: GITNEXUS_BENCH: '1' run: >- npx vitest run --no-file-parallelism test/integration/cobol-pipeline-benchmark.test.ts test/integration/csharp-pipeline-benchmark.test.ts + test/integration/csharp-razor-view-components-benchmark.test.ts test/integration/cpp-adl-benchmark.test.ts test/integration/data-route-table-benchmark.test.ts test/integration/instance-ownership-pipeline-benchmark.test.ts diff --git a/gitnexus/src/core/ingestion/languages/csharp/razor-view-components.ts b/gitnexus/src/core/ingestion/languages/csharp/razor-view-components.ts new file mode 100644 index 000000000..0e7ea55a4 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/csharp/razor-view-components.ts @@ -0,0 +1,955 @@ +/** + * ASP.NET Core ViewComponent convention support. + * + * Same bound as Spring Boot DI in Java/Kotlin: do not resolve into the SDK + * (`Microsoft.AspNetCore.Mvc.ViewComponent`, `IViewComponentHelper`, + * `Component.InvokeAsync` itself). Those types live outside the workspace. + * The only hop worth taking is the framework convention that lands on an + * **in-repo** class — `InvokeAsync("Foo")` → workspace `FooViewComponent`, + * just as a Spring `@Autowired IFoo` fans out to an in-repo `@Service`, + * not to `ApplicationContext`. + * + * Razor templates are not parsed as C# (markup + code would poison + * tree-sitter-c-sharp). A small Razor state machine extracts C# islands and + * markup tag helpers; C# files use a string/comment-aware lexer so attributes + * and literals are not mistaken for helper calls. Literal names are enough + * because the target catalog is already built from parsed `.cs` classes. + */ + +import fs from 'node:fs/promises'; +import path from 'node:path'; +import { glob } from 'glob'; +import type { ParsedFile } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../../../graph/types.js'; +import { createIgnoreFilter } from '../../../../config/ignore-service.js'; +import { generateId } from '../../../../lib/utils.js'; +import { getMaxFileSizeBytes } from '../../utils/max-file-size.js'; +import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js'; +import { resolveDefGraphId } from '../../scope-resolution/graph-bridge/ids.js'; +import { definitionIdPosition } from '../../scope-resolution/utils/definition-id.js'; + +const VIEW_COMPONENT_SUFFIX = 'ViewComponent'; +const VIEW_COMPONENT_TAG_RE = /<\s*vc:([a-z][a-z0-9-]*)\b/gi; +const COMPONENT_NAME_RE = /^[A-Za-z_][A-Za-z0-9_.-]*$/; +const TYPE_MODIFIERS = new Set([ + 'public', + 'internal', + 'protected', + 'private', + 'abstract', + 'sealed', + 'partial', + 'static', + 'new', + 'file', + 'required', + 'unsafe', + 'readonly', +]); +const RAZOR_BLOCK_KEYWORDS = new Set([ + 'if', + 'for', + 'foreach', + 'while', + 'using', + 'switch', + 'try', + 'lock', + 'functions', + 'helper', + 'code', + 'section', + 'do', +]); + +export interface RazorViewComponentConfig { + /** Repo-relative `.cshtml` path → extracted invocation names. */ + readonly views: ReadonlyMap; +} + +export interface ViewComponentAliasBind { + readonly className: string; + /** 1-based line of the type declaration (including leading attributes). */ + readonly startLine: number; + /** 0-based column of the type declaration (including leading attributes). */ + readonly startCol: number; + readonly aliases: readonly string[]; +} + +class SourceCursor { + i = 0; + line = 1; + col = 0; + + constructor(readonly source: string) {} + + get length(): number { + return this.source.length; + } + + get done(): boolean { + return this.i >= this.source.length; + } + + peek(n = 0): string { + return this.source[this.i + n] ?? ''; + } + + startsWith(value: string): boolean { + return this.source.startsWith(value, this.i); + } + + snapshot(): { i: number; line: number; col: number } { + return { i: this.i, line: this.line, col: this.col }; + } + + restore(pos: { i: number; line: number; col: number }): void { + this.i = pos.i; + this.line = pos.line; + this.col = pos.col; + } + + advance(count = 1): void { + const end = Math.min(this.i + count, this.source.length); + while (this.i < end) { + const ch = this.source[this.i]!; + this.i += 1; + if (ch === '\n') { + this.line += 1; + this.col = 0; + } else { + this.col += 1; + } + } + } +} + +function isIdentStart(ch: string): boolean { + return (ch >= 'A' && ch <= 'Z') || (ch >= 'a' && ch <= 'z') || ch === '_' || ch === '@'; +} + +function isIdentPart(ch: string): boolean { + return isIdentStart(ch) || (ch >= '0' && ch <= '9'); +} + +function skipWhitespace(cur: SourceCursor): void { + while (!cur.done) { + const ch = cur.peek(); + if (ch !== ' ' && ch !== '\t' && ch !== '\n' && ch !== '\r' && ch !== '\f' && ch !== '\v') + break; + cur.advance(); + } +} + +/** Skip line comments and block comments. Returns true if a comment was consumed. */ +function skipCsharpComment(cur: SourceCursor): boolean { + if (cur.startsWith('//')) { + while (!cur.done && cur.peek() !== '\n') cur.advance(); + return true; + } + if (cur.startsWith('/*')) { + cur.advance(2); + while (!cur.done && !cur.startsWith('*/')) cur.advance(); + if (cur.startsWith('*/')) cur.advance(2); + return true; + } + return false; +} + +function skipCsharpTrivia(cur: SourceCursor): void { + for (;;) { + skipWhitespace(cur); + if (!skipCsharpComment(cur)) return; + } +} + +function skipRegularString(cur: SourceCursor, interpolated: boolean): void { + cur.advance(); // opening " + while (!cur.done) { + const ch = cur.peek(); + if (ch === '\\') { + cur.advance(2); + continue; + } + if (interpolated && ch === '{') { + if (cur.peek(1) === '{') { + cur.advance(2); + continue; + } + skipInterpolation(cur); + continue; + } + cur.advance(); + if (ch === '"') return; + } +} + +function skipVerbatimString(cur: SourceCursor, interpolated: boolean): void { + cur.advance(2); // @" + while (!cur.done) { + const ch = cur.peek(); + if (ch === '"') { + if (cur.peek(1) === '"') { + cur.advance(2); + continue; + } + cur.advance(); + return; + } + if (interpolated && ch === '{') { + if (cur.peek(1) === '{') { + cur.advance(2); + continue; + } + skipInterpolation(cur); + continue; + } + cur.advance(); + } +} + +function skipRawString(cur: SourceCursor): void { + let quoteCount = 0; + while (cur.peek() === '"') { + quoteCount += 1; + cur.advance(); + } + while (!cur.done) { + if (cur.peek() !== '"') { + cur.advance(); + continue; + } + let seen = 0; + while (cur.peek() === '"') { + seen += 1; + cur.advance(); + } + if (seen >= quoteCount) return; + } +} + +function skipInterpolation(cur: SourceCursor): void { + cur.advance(); // { + let depth = 1; + while (!cur.done && depth > 0) { + skipCsharpTrivia(cur); + if (cur.done) return; + if (skipCsharpString(cur)) continue; + const ch = cur.peek(); + if (ch === '{') depth += 1; + else if (ch === '}') depth -= 1; + cur.advance(); + } +} + +function skipCsharpString(cur: SourceCursor): boolean { + const ch = cur.peek(); + if (ch === "'") { + cur.advance(); + if (cur.peek() === '\\') cur.advance(2); + else cur.advance(); + if (cur.peek() === "'") cur.advance(); + return true; + } + if (ch === '"') { + if (cur.peek(1) === '"' && cur.peek(2) === '"') skipRawString(cur); + else skipRegularString(cur, false); + return true; + } + if (ch === '$' && cur.peek(1) === '@' && cur.peek(2) === '"') { + cur.advance(); + skipVerbatimString(cur, true); + return true; + } + if (ch === '@' && cur.peek(1) === '$' && cur.peek(2) === '"') { + cur.advance(2); + skipVerbatimString(cur, true); + return true; + } + if (ch === '@' && cur.peek(1) === '"') { + skipVerbatimString(cur, false); + return true; + } + if (ch === '$' && cur.peek(1) === '"') { + if (cur.peek(2) === '"' && cur.peek(3) === '"') { + cur.advance(); + skipRawString(cur); + } else { + cur.advance(); + skipRegularString(cur, true); + } + return true; + } + return false; +} + +function readIdent(cur: SourceCursor): string | undefined { + if (!isIdentStart(cur.peek())) return undefined; + const start = cur.i; + if (cur.peek() === '@') cur.advance(); + if (!isIdentStart(cur.peek()) && !(cur.peek() >= 'A' && cur.peek() <= 'z')) { + cur.i = start; + return undefined; + } + while (isIdentPart(cur.peek()) && cur.peek() !== '@') cur.advance(); + const raw = cur.source.slice(start, cur.i); + return raw.startsWith('@') ? raw.slice(1) : raw; +} + +function tryReadIdent(cur: SourceCursor): string | undefined { + skipCsharpTrivia(cur); + return readIdent(cur); +} + +function decodeCsharpString(cur: SourceCursor): string | undefined { + skipCsharpTrivia(cur); + const start = cur.snapshot(); + const ch = cur.peek(); + if (ch === '$') return undefined; + if (ch === '@' && cur.peek(1) === '"') { + cur.advance(2); + let value = ''; + while (!cur.done) { + if (cur.peek() === '"') { + if (cur.peek(1) === '"') { + value += '"'; + cur.advance(2); + continue; + } + cur.advance(); + return value; + } + value += cur.peek(); + cur.advance(); + } + cur.restore(start); + return undefined; + } + if (ch === '"' && cur.peek(1) === '"' && cur.peek(2) === '"') { + let quoteCount = 0; + while (cur.peek() === '"') { + quoteCount += 1; + cur.advance(); + } + const bodyStart = cur.i; + while (!cur.done) { + if (cur.peek() !== '"') { + cur.advance(); + continue; + } + const closeStart = cur.i; + let seen = 0; + while (cur.peek() === '"') { + seen += 1; + cur.advance(); + } + if (seen >= quoteCount) { + return cur.source.slice(bodyStart, closeStart); + } + } + cur.restore(start); + return undefined; + } + if (ch === '"') { + cur.advance(); + let value = ''; + while (!cur.done) { + const next = cur.peek(); + if (next === '\\') { + cur.advance(); + const esc = cur.peek(); + cur.advance(); + const map: Record = { + n: '\n', + r: '\r', + t: '\t', + '"': '"', + '\\': '\\', + '0': '\0', + }; + value += map[esc] ?? esc; + continue; + } + if (next === '"') { + cur.advance(); + return value; + } + value += next; + cur.advance(); + } + cur.restore(start); + return undefined; + } + return undefined; +} + +function skipBalanced(cur: SourceCursor, open: string, close: string): boolean { + skipCsharpTrivia(cur); + if (cur.peek() !== open) return false; + let depth = 0; + while (!cur.done) { + skipCsharpTrivia(cur); + if (cur.done) return false; + if (skipCsharpString(cur)) continue; + const ch = cur.peek(); + if (ch === open) depth += 1; + else if (ch === close) { + depth -= 1; + cur.advance(); + if (depth === 0) return true; + continue; + } + cur.advance(); + } + return false; +} + +function componentNameFromLiteral(value: string | undefined): string | undefined { + if (value === undefined || !COMPONENT_NAME_RE.test(value)) return undefined; + return value; +} + +function isViewComponentAttributeName(name: string): boolean { + return name === 'ViewComponent' || name === 'ViewComponentAttribute'; +} + +function readQualifiedTail(cur: SourceCursor): string | undefined { + skipCsharpTrivia(cur); + let name = readIdent(cur); + if (name === undefined) return undefined; + for (;;) { + skipCsharpTrivia(cur); + if (cur.peek() === '.' || (cur.peek() === ':' && cur.peek(1) === ':')) { + cur.advance(cur.peek() === ':' ? 2 : 1); + skipCsharpTrivia(cur); + const next = readIdent(cur); + if (next === undefined) return name; + name = next; + continue; + } + return name; + } +} + +function readViewComponentNameArgument(cur: SourceCursor): string | undefined { + skipCsharpTrivia(cur); + if (cur.peek() !== '(') return undefined; + cur.advance(); + let alias: string | undefined; + while (!cur.done && cur.peek() !== ')') { + skipCsharpTrivia(cur); + if (cur.peek() === ')') break; + const beforeArg = cur.snapshot(); + const ident = readIdent(cur); + skipCsharpTrivia(cur); + if (ident === 'Name' && cur.peek() === '=') { + cur.advance(); + alias = componentNameFromLiteral(decodeCsharpString(cur)); + } else { + cur.restore(beforeArg); + skipCsharpTrivia(cur); + if (cur.peek() === '"' || cur.peek() === '@') { + // Positional string arguments are not ViewComponentAttribute.Name. + skipCsharpString(cur); + } else if (cur.peek() === '(' || cur.peek() === '[' || cur.peek() === '{') { + const open = cur.peek(); + const close = open === '(' ? ')' : open === '[' ? ']' : '}'; + skipBalanced(cur, open, close); + } else { + while (!cur.done && cur.peek() !== ',' && cur.peek() !== ')') { + if (skipCsharpString(cur)) continue; + if (skipCsharpComment(cur)) continue; + cur.advance(); + } + } + } + skipCsharpTrivia(cur); + if (cur.peek() === ',') cur.advance(); + } + if (cur.peek() === ')') cur.advance(); + return alias; +} + +function collectInvokeAfterIdent( + ident: string, + cur: SourceCursor, + previous: string | undefined, + memberReceiver: string | undefined, + names: Set, +): void { + skipCsharpTrivia(cur); + const hasMvcReceiver = previous !== '.' || memberReceiver === 'this' || memberReceiver === 'base'; + if (ident === 'ViewComponent' && cur.peek() === '(') { + if (previous === '[' || previous === ',' || !hasMvcReceiver) return; + cur.advance(); + const name = componentNameFromLiteral(decodeCsharpString(cur)); + if (name !== undefined) names.add(name); + return; + } + if (ident !== 'Component' || cur.peek() !== '.' || !hasMvcReceiver) return; + const afterDot = cur.snapshot(); + cur.advance(); + skipCsharpTrivia(cur); + if (readIdent(cur) !== 'InvokeAsync') { + cur.restore(afterDot); + return; + } + skipCsharpTrivia(cur); + if (cur.peek() !== '(') return; + cur.advance(); + const name = componentNameFromLiteral(decodeCsharpString(cur)); + if (name !== undefined) names.add(name); +} + +/** In-repo C# `Component.InvokeAsync("X")` / `ViewComponent("X")` literals. */ +export function extractCsharpViewComponentInvocations(source: string): string[] { + if (!source.includes('ViewComponent') && !source.includes('InvokeAsync')) return []; + const names = new Set(); + const cur = new SourceCursor(source); + let previous: string | undefined; + let memberReceiver: string | undefined; + let squareDepth = 0; + while (!cur.done) { + skipCsharpTrivia(cur); + if (cur.done) break; + if (skipCsharpString(cur)) { + previous = 'string'; + continue; + } + const ident = readIdent(cur); + if (ident !== undefined) { + const inAttribute = squareDepth > 0; + collectInvokeAfterIdent(ident, cur, inAttribute ? '[' : previous, memberReceiver, names); + previous = ident; + memberReceiver = undefined; + continue; + } + const ch = cur.peek(); + if (ch === '[') squareDepth += 1; + else if (ch === ']' && squareDepth > 0) squareDepth -= 1; + memberReceiver = ch === '.' ? previous : undefined; + previous = ch; + cur.advance(); + } + return [...names]; +} + +function parseAttributeListBody(cur: SourceCursor): string[] { + const aliases: string[] = []; + skipCsharpTrivia(cur); + const specifier = cur.snapshot(); + const specifierName = readIdent(cur); + skipCsharpTrivia(cur); + if (specifierName !== undefined && cur.peek() === ':' && cur.peek(1) !== ':') { + cur.advance(); + } else { + cur.restore(specifier); + } + while (!cur.done && cur.peek() !== ']') { + skipCsharpTrivia(cur); + if (cur.peek() === ']') break; + const tail = readQualifiedTail(cur); + skipCsharpTrivia(cur); + if (tail !== undefined && isViewComponentAttributeName(tail) && cur.peek() === '(') { + const alias = readViewComponentNameArgument(cur); + if (alias !== undefined) aliases.push(alias); + } else if (cur.peek() === '(') { + skipBalanced(cur, '(', ')'); + } + skipCsharpTrivia(cur); + if (cur.peek() === ',') cur.advance(); + else break; + } + if (cur.peek() === ']') cur.advance(); + return aliases; +} + +/** + * Explicit `[ViewComponent(Name = "...")]` aliases keyed to the following + * class declaration. Positional constructor arguments are ignored: the MVC + * attribute only exposes `Name` as a property. + */ +export function extractViewComponentAliasBinds(source: string): ViewComponentAliasBind[] { + if (!source.includes('ViewComponent')) return []; + const binds: ViewComponentAliasBind[] = []; + const cur = new SourceCursor(source); + const pending: { startLine: number; startCol: number; aliases: string[] }[] = []; + + const flushPending = (className: string, startLine: number, startCol: number): void => { + const aliases = pending.flatMap((entry) => entry.aliases); + const start = pending[0]; + binds.push({ + className, + startLine: start?.startLine ?? startLine, + startCol: start?.startCol ?? startCol, + aliases: [...new Set(aliases)], + }); + pending.length = 0; + }; + + while (!cur.done) { + skipCsharpTrivia(cur); + if (cur.done) break; + if (skipCsharpString(cur)) continue; + const startLine = cur.line; + const startCol = cur.col; + if (cur.peek() === '[') { + cur.advance(); + const aliases = parseAttributeListBody(cur); + pending.push({ startLine, startCol, aliases }); + continue; + } + const ident = readIdent(cur); + if (ident === undefined) { + pending.length = 0; + cur.advance(); + continue; + } + if (TYPE_MODIFIERS.has(ident)) continue; + if (ident === 'class' || ident === 'record') { + let className = tryReadIdent(cur); + if (ident === 'record' && (className === 'class' || className === 'struct')) { + className = tryReadIdent(cur); + } + if (className !== undefined && pending.some((entry) => entry.aliases.length > 0)) { + flushPending(className, startLine, startCol); + } else { + pending.length = 0; + } + continue; + } + pending.length = 0; + } + return binds; +} + +/** Extract explicit `[ViewComponent(Name = "...")]` aliases by class name. */ +export function extractViewComponentAliases( + source: string, +): ReadonlyMap { + const aliases = new Map(); + for (const bind of extractViewComponentAliasBinds(source)) { + if (bind.aliases.length === 0) continue; + const existing = aliases.get(bind.className); + if (existing) { + for (const alias of bind.aliases) { + if (!existing.includes(alias)) existing.push(alias); + } + } else { + aliases.set(bind.className, [...bind.aliases]); + } + } + return aliases; +} + +function tagNameToComponentName(tagName: string): string { + return tagName + .split('-') + .filter(Boolean) + .map((part) => part[0]!.toUpperCase() + part.slice(1)) + .join(''); +} + +function collectVcTags(span: string, names: Set): void { + VIEW_COMPONENT_TAG_RE.lastIndex = 0; + for (const match of span.matchAll(VIEW_COMPONENT_TAG_RE)) { + names.add(tagNameToComponentName(match[1]!)); + } +} + +function skipRazorComment(cur: SourceCursor): boolean { + if (!cur.startsWith('@*')) return false; + cur.advance(2); + while (!cur.done && !cur.startsWith('*@')) cur.advance(); + if (cur.startsWith('*@')) cur.advance(2); + return true; +} + +function countAtRun(cur: SourceCursor): number { + let count = 0; + while (cur.peek() === '@') { + count += 1; + cur.advance(); + } + return count; +} + +function scanCsharpSpan(span: string, names: Set): void { + for (const name of extractCsharpViewComponentInvocations(span)) names.add(name); +} + +function skipOptionalParens(cur: SourceCursor): void { + skipWhitespace(cur); + if (cur.peek() === '(') skipBalanced(cur, '(', ')'); +} + +function consumeRazorCodeBlock(cur: SourceCursor, names: Set): void { + skipCsharpTrivia(cur); + skipOptionalParens(cur); + skipCsharpTrivia(cur); + if (cur.peek() !== '{') { + const start = cur.i; + while (!cur.done && cur.peek() !== '\n' && cur.peek() !== '{') { + if (skipCsharpString(cur) || skipCsharpComment(cur)) continue; + cur.advance(); + } + scanCsharpSpan(cur.source.slice(start, cur.i), names); + if (cur.peek() === '{') consumeRazorCodeBlock(cur, names); + return; + } + const bodyStart = cur.i + 1; + if (!skipBalanced(cur, '{', '}')) return; + scanCsharpSpan(cur.source.slice(bodyStart, cur.i - 1), names); +} + +function consumeImplicitExpression(cur: SourceCursor, names: Set): void { + const start = cur.i; + skipCsharpTrivia(cur); + if (cur.peek() === '(') { + const innerStart = cur.i + 1; + if (skipBalanced(cur, '(', ')')) { + scanCsharpSpan(cur.source.slice(innerStart, cur.i - 1), names); + } + return; + } + // Implicit expressions: `@await Component.InvokeAsync("X")` / `@Component.InvokeAsync(...)`. + while (!cur.done) { + skipCsharpTrivia(cur); + if (cur.done) break; + if (skipCsharpString(cur)) continue; + if (cur.peek() === '(') { + skipBalanced(cur, '(', ')'); + continue; + } + if (cur.peek() === '{') { + skipBalanced(cur, '{', '}'); + continue; + } + const ch = cur.peek(); + if (ch === '<' || ch === '\n') break; + if (ch === '@') break; + if (!isIdentPart(ch) && ch !== '.' && ch !== '?') { + if (ch === ';') cur.advance(); + break; + } + cur.advance(); + } + scanCsharpSpan(cur.source.slice(start, cur.i), names); +} + +function consumeRazorTransition(cur: SourceCursor, names: Set): void { + skipWhitespace(cur); + if (cur.peek() === '{') { + consumeRazorCodeBlock(cur, names); + return; + } + if (cur.peek() === '(') { + consumeImplicitExpression(cur, names); + return; + } + const identStart = cur.snapshot(); + const ident = readIdent(cur); + if (ident === undefined) { + consumeImplicitExpression(cur, names); + return; + } + if (ident === 'await' || ident === 'Component') { + cur.restore(identStart); + consumeImplicitExpression(cur, names); + return; + } + if (RAZOR_BLOCK_KEYWORDS.has(ident)) { + if (ident === 'section' || ident === 'helper') tryReadIdent(cur); + consumeRazorCodeBlock(cur, names); + return; + } + cur.restore(identStart); + consumeImplicitExpression(cur, names); +} + +/** Extract statically resolvable ViewComponent names from one Razor template. */ +export function extractRazorViewComponentInvocations(source: string): string[] { + // Most views do not invoke a ViewComponent. Avoid the character-by-character + // Razor scan unless one of the two supported invocation spellings is present. + // This is only a coarse gate; the state machine below still decides whether a + // token is executable markup/C# or a comment/string/escaped transition. + if (!source.includes('InvokeAsync') && !/<\s*vc:/i.test(source)) return []; + + const names = new Set(); + const cur = new SourceCursor(source); + let markupStart = 0; + const flushMarkup = (): void => { + if (cur.i > markupStart) collectVcTags(source.slice(markupStart, cur.i), names); + }; + + while (!cur.done) { + if (cur.peek() !== '@') { + cur.advance(); + continue; + } + flushMarkup(); + if (skipRazorComment(cur)) { + markupStart = cur.i; + continue; + } + const atCount = countAtRun(cur); + const leftover = atCount % 2; + if (leftover === 0) { + markupStart = cur.i; + continue; + } + consumeRazorTransition(cur, names); + markupStart = cur.i; + } + flushMarkup(); + return [...names]; +} + +/** + * Read Razor views once per C# resolution pass. The same ignore rules and file + * size ceiling as repository scanning are applied, and edge emission later + * additionally requires a live File node. This prevents ignored, oversized, + * or concurrently removed templates from entering the graph. + */ +export async function loadRazorViewComponentConfig( + repoRoot: string, +): Promise { + const ignore = await createIgnoreFilter(repoRoot); + const paths = await glob('**/*.cshtml', { + cwd: repoRoot, + nodir: true, + dot: false, + ignore, + }); + paths.sort(); + + const maxBytes = getMaxFileSizeBytes(); + const views = new Map(); + for (const rawPath of paths) { + const filePath = rawPath.replace(/\\/g, '/'); + // The size gate and the read go through one handle so both observe the same + // inode. Re-resolving the path for the read would let a template swapped in + // between them be read unchecked (CodeQL js/file-system-race). + let handle: fs.FileHandle | undefined; + try { + handle = await fs.open(path.join(repoRoot, filePath), 'r'); + const stat = await handle.stat(); + if (!stat.isFile() || stat.size > maxBytes) continue; + const source = await handle.readFile('utf8'); + views.set(filePath, extractRazorViewComponentInvocations(source)); + } catch { + // A view may disappear between glob/open/read during watch mode. + } finally { + await handle?.close().catch(() => {}); + } + } + return { views }; +} + +function addCandidate( + candidates: Map>, + invocationName: string, + targetId: string, +): void { + const key = invocationName.toLocaleLowerCase('en-US'); + const existing = candidates.get(key); + if (existing) { + existing.add(targetId); + } else { + candidates.set(key, new Set([targetId])); + } +} + +function bindAliasesForClass( + binds: readonly ViewComponentAliasBind[], + className: string, + nodeId: string, + filePath: string, +): readonly string[] | undefined { + const matches = binds.filter((bind) => bind.className === className); + if (matches.length === 0) return undefined; + if (matches.length === 1) return matches[0]!.aliases; + const pos = definitionIdPosition(nodeId, filePath); + if (pos === undefined) return undefined; + const atPosition = matches.filter( + (bind) => bind.startLine === pos.line && bind.startCol === pos.column, + ); + if (atPosition.length === 1) return atPosition[0]!.aliases; + return undefined; +} + +/** + * Emit workspace File → in-repo ViewComponent Class CALLS edges. + * + * Targets are only Class nodes produced from this repo's `.cs` files. There is + * no lookup of ASP.NET SDK types; `: ViewComponent` in source is a naming + * hint, not a resolved EXTENDS edge to `Microsoft.AspNetCore.Mvc.ViewComponent`. + * + * Ambiguous component names fail closed: two in-repo classes claiming the + * same name is not evidence for picking either one. + */ +export function emitRazorViewComponentEdges( + graph: KnowledgeGraph, + parsedFiles: readonly ParsedFile[], + nodeLookup: GraphNodeLookup, + config: RazorViewComponentConfig | undefined, + csharpSources: ReadonlyMap, +): void { + if (!config) return; + + const candidates = new Map>(); + for (const parsed of parsedFiles) { + if (!parsed.filePath.endsWith('.cs')) continue; + const source = csharpSources.get(parsed.filePath) ?? ''; + const binds = source.includes('ViewComponent') ? extractViewComponentAliasBinds(source) : []; + for (const def of parsed.localDefs) { + if (def.type !== 'Class') continue; + const className = def.qualifiedName?.split('.').pop() ?? def.nodeId.split(':').pop() ?? ''; + const conventionalName = className.endsWith(VIEW_COMPONENT_SUFFIX) + ? className.slice(0, -VIEW_COMPONENT_SUFFIX.length) + : undefined; + const explicitAliases = bindAliasesForClass(binds, className, def.nodeId, parsed.filePath); + if (!conventionalName && (explicitAliases === undefined || explicitAliases.length === 0)) { + continue; + } + + const targetId = resolveDefGraphId(parsed.filePath, def, nodeLookup); + if (!targetId || !graph.getNode(targetId)) continue; + // An explicit [ViewComponent(Name = "...")] replaces the suffix name, + // matching ASP.NET. Never register the SDK base type as a candidate. + if (explicitAliases !== undefined && explicitAliases.length > 0) { + for (const alias of explicitAliases) addCandidate(candidates, alias, targetId); + } else if (conventionalName) { + addCandidate(candidates, conventionalName, targetId); + } + } + } + + const emitFromFile = (filePath: string, invocationNames: readonly string[]): void => { + const sourceId = generateId('File', filePath); + if (!graph.getNode(sourceId)) return; + for (const invocationName of invocationNames) { + const matches = candidates.get(invocationName.toLocaleLowerCase('en-US')); + if (!matches || matches.size !== 1) continue; + const targetId = matches.values().next().value; + if (typeof targetId !== 'string' || !graph.getNode(targetId)) continue; + graph.addRelationship({ + id: generateId('CALLS', `${sourceId}:razor-view-component:${targetId}`), + sourceId, + targetId, + type: 'CALLS', + confidence: 0.9, + reason: 'aspnet-razor-view-component', + }); + } + }; + + for (const [viewPath, invocationNames] of config.views) { + emitFromFile(viewPath, invocationNames); + } + for (const [filePath, source] of csharpSources) { + if (!filePath.endsWith('.cs')) continue; + if (!source.includes('ViewComponent') && !source.includes('InvokeAsync')) continue; + emitFromFile(filePath, extractCsharpViewComponentInvocations(source)); + } +} diff --git a/gitnexus/src/core/ingestion/languages/csharp/resolution-config.ts b/gitnexus/src/core/ingestion/languages/csharp/resolution-config.ts index 9ea232c05..714be8c1f 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/resolution-config.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/resolution-config.ts @@ -12,19 +12,29 @@ import { type CSharpProjectConfig, type CSharpNamespaceEvidence, } from '../../language-config.js'; +import { + loadRazorViewComponentConfig, + type RazorViewComponentConfig, +} from './razor-view-components.js'; export interface CsharpResolutionConfig { readonly csharpConfigs: readonly CSharpProjectConfig[]; /** In-repo declared-namespace evidence gating suffix-fallback resolution (#1881). */ readonly namespaces?: CSharpNamespaceEvidence; + /** Razor views scanned for ASP.NET ViewComponent invocation conventions. */ + readonly razorViewComponents?: RazorViewComponentConfig; } export async function loadCsharpResolutionConfig( repoRoot: string, ): Promise { - const scan = await scanCSharpProject(repoRoot); + const [scan, razorViewComponents] = await Promise.all([ + scanCSharpProject(repoRoot), + loadRazorViewComponentConfig(repoRoot), + ]); return { csharpConfigs: scan.configs, namespaces: csharpScanToEvidence(scan), + razorViewComponents, }; } diff --git a/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts index 4b50efc67..6d200de9b 100644 --- a/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/csharp/scope-resolver.ts @@ -22,6 +22,7 @@ import { import { populateCsharpNamespaceSiblings } from './namespace-siblings.js'; import { loadCsharpResolutionConfig, type CsharpResolutionConfig } from './resolution-config.js'; import { unwrapCsharpElementType } from './accessor-unwrap.js'; +import { emitRazorViewComponentEdges } from './razor-view-components.js'; const csharpScopeResolver: ScopeResolver = { // Construction is keyword-prefixed: `new Service(db).doWork()` (#2708). @@ -106,6 +107,20 @@ const csharpScopeResolver: ScopeResolver = { // `IValidator` and `IValidator` are one instantiation, so the // dispatch fan-out must not read them as two (#2912). See the alias table. normalizeTypeArgument: normalizeCsharpTypeArgument, + + // Razor views stay out of the C# parser. Bind literal ViewComponent names + // only onto in-repo classes (Spring-style: skip the SDK type, hop to the + // workspace implementor). + emitPostResolutionEdges: (graph, parsedFiles, nodeLookup, _indexes, ctx) => { + const config = ctx.resolutionConfig as CsharpResolutionConfig | undefined; + emitRazorViewComponentEdges( + graph, + parsedFiles, + nodeLookup, + config?.razorViewComponents, + ctx.fileContents, + ); + }, }; /** diff --git a/gitnexus/test/integration/csharp-razor-view-components-benchmark.test.ts b/gitnexus/test/integration/csharp-razor-view-components-benchmark.test.ts new file mode 100644 index 000000000..3623c0636 --- /dev/null +++ b/gitnexus/test/integration/csharp-razor-view-components-benchmark.test.ts @@ -0,0 +1,179 @@ +/** + * C# Razor ViewComponent extractor + loader scaling guards (#2991). + * + * The coverage job always runs the tripwire (direct extractors, no worker). + * GITNEXUS_BENCH=1 additionally checks sub-quadratic scaling and that the + * production loader retains invocation names instead of view source. + * + * Run: GITNEXUS_BENCH=1 npx vitest run test/integration/csharp-razor-view-components-benchmark.test.ts + */ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { describe, expect, it } from 'vitest'; +import { + extractCsharpViewComponentInvocations, + extractRazorViewComponentInvocations, + loadRazorViewComponentConfig, +} from '../../src/core/ingestion/languages/csharp/razor-view-components.js'; + +const BENCH_ENABLED = process.env.GITNEXUS_BENCH === '1'; +const PAD = 'x'.repeat(4_000); + +function csharpMixed(i: number): string { + if (i % 10 === 0) { + return ( + 'using Microsoft.AspNetCore.Mvc;\n' + + `class C${i} {\n` + + ' async Task T() {\n' + + ' await Component.InvokeAsync("Cart");\n' + + ' }\n' + + '}\n' + + `// ${PAD}\n` + ); + } + if (i % 10 === 1) { + return `// ViewComponent decoy\nclass C${i} { string s = "InvokeAsync"; }\n// ${PAD}\n`; + } + return `class C${i} { int X => ${i}; }\n// ${PAD}\n`; +} + +function razorMixed(i: number): string { + if (i % 10 === 0) { + return `@await Component.InvokeAsync("Cart")\n\n`; + } + if (i % 10 === 1) { + return ( + '@* @await Component.InvokeAsync("Dead") *@\n' + + '@@await Component.InvokeAsync("Escaped")\n' + + '@* *@\n' + + `\n` + ); + } + return `

hello ${i}

\n\n`; +} + +function expectedHits(fileCount: number): number { + return fileCount > 0 ? Math.floor((fileCount - 1) / 10) + 1 : 0; +} + +function scanCsharp(fileCount: number): { hits: number; elapsedMs: number } { + const sources = Array.from({ length: fileCount }, (_, i) => csharpMixed(i)); + const started = performance.now(); + let hits = 0; + for (const source of sources) { + hits += extractCsharpViewComponentInvocations(source).length; + } + return { hits, elapsedMs: performance.now() - started }; +} + +function scanRazor(fileCount: number): { hits: number; elapsedMs: number } { + const sources = Array.from({ length: fileCount }, (_, i) => razorMixed(i)); + const started = performance.now(); + let hits = 0; + for (const source of sources) { + hits += extractRazorViewComponentInvocations(source).length; + } + return { hits, elapsedMs: performance.now() - started }; +} + +/** + * Direct extractor tripwire for the coverage job. Coarse budget: far above the + * measured linear path, far below a full-corpus character-by-character scan of + * every padded view without the token prefilter. + */ +describe('C# Razor ViewComponent extractor tripwire', () => { + it('scans a 400-file mixed corpus well under the O(n^2) budget', () => { + const FILE_COUNT = 400; + const BUDGET_MS = 2_000; + scanCsharp(40); + scanRazor(40); + + const csharp = scanCsharp(FILE_COUNT); + const razor = scanRazor(FILE_COUNT); + expect(csharp.hits).toBe(expectedHits(FILE_COUNT)); + expect(razor.hits).toBe(expectedHits(FILE_COUNT)); + expect(csharp.elapsedMs + razor.elapsedMs).toBeLessThan(BUDGET_MS); + }, 15_000); +}); + +describe.skipIf(!BENCH_ENABLED)('C# Razor ViewComponent extractor benchmark', () => { + it('scales sub-quadratically as mixed C# and Razor corpora grow', () => { + scanCsharp(50); + scanRazor(50); + + const csharpSmall = scanCsharp(500); + const csharpLarge = scanCsharp(2_000); + const razorSmall = scanRazor(500); + const razorLarge = scanRazor(2_000); + const ratio = 2_000 / 500; + + console.log('\nC# Razor ViewComponent extractor benchmark'); + console.log( + ` csharp files=500 wall=${csharpSmall.elapsedMs.toFixed(2)}ms hits=${csharpSmall.hits}`, + ); + console.log( + ` csharp files=2000 wall=${csharpLarge.elapsedMs.toFixed(2)}ms hits=${csharpLarge.hits}`, + ); + console.log( + ` razor files=500 wall=${razorSmall.elapsedMs.toFixed(2)}ms hits=${razorSmall.hits}`, + ); + console.log( + ` razor files=2000 wall=${razorLarge.elapsedMs.toFixed(2)}ms hits=${razorLarge.hits}`, + ); + + expect(csharpSmall.hits).toBe(expectedHits(500)); + expect(csharpLarge.hits).toBe(expectedHits(2_000)); + expect(razorSmall.hits).toBe(expectedHits(500)); + expect(razorLarge.hits).toBe(expectedHits(2_000)); + + if (csharpSmall.elapsedMs >= 5) { + expect(csharpLarge.elapsedMs / csharpSmall.elapsedMs).toBeLessThan(Math.pow(ratio, 1.5)); + } + if (razorSmall.elapsedMs >= 5) { + expect(razorLarge.elapsedMs / razorSmall.elapsedMs).toBeLessThan(Math.pow(ratio, 1.5)); + } + expect(csharpLarge.elapsedMs).toBeLessThan(5_000); + expect(razorLarge.elapsedMs).toBeLessThan(5_000); + }, 60_000); + + it('loader retains invocation names instead of view source', async () => { + const FILE_COUNT = 2_000; + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-razor-vc-bench-')); + try { + fs.mkdirSync(path.join(dir, 'Views'), { recursive: true }); + for (let i = 0; i < FILE_COUNT; i++) { + fs.writeFileSync(path.join(dir, 'Views', `v${i}.cshtml`), razorMixed(i)); + } + + await loadRazorViewComponentConfig(dir); + + const started = performance.now(); + const config = await loadRazorViewComponentConfig(dir); + const elapsedMs = performance.now() - started; + + let sourceBytes = 0; + let retainedChars = 0; + let hits = 0; + for (const names of config.views.values()) { + hits += names.length; + for (const name of names) retainedChars += name.length; + } + for (let i = 0; i < FILE_COUNT; i++) { + sourceBytes += Buffer.byteLength(razorMixed(i)); + } + + console.log( + ` loader files=${FILE_COUNT} wall=${elapsedMs.toFixed(2)}ms ` + + `source=${(sourceBytes / 1024 / 1024).toFixed(2)}MB retainedChars=${retainedChars}`, + ); + + expect(config.views.size).toBe(FILE_COUNT); + expect(hits).toBe(expectedHits(FILE_COUNT)); + expect(retainedChars).toBeLessThan(sourceBytes / 20); + expect(elapsedMs).toBeLessThan(15_000); + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } + }, 60_000); +}); diff --git a/gitnexus/test/integration/csharp-razor-view-components.test.ts b/gitnexus/test/integration/csharp-razor-view-components.test.ts new file mode 100644 index 000000000..47619cacb --- /dev/null +++ b/gitnexus/test/integration/csharp-razor-view-components.test.ts @@ -0,0 +1,127 @@ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { afterAll, beforeAll, describe, expect, it, vi } from 'vitest'; +import { + getRelationships, + runPipelineFromRepo, + writeFixtureRepo, + type PipelineResult, +} from './resolvers/helpers.js'; + +describe('C# Razor ViewComponent conventions', () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-csharp-razor-vc-')); + let result: PipelineResult; + + beforeAll(() => vi.stubEnv('GITNEXUS_WORKER_READY_TIMEOUT_MS', '60000')); + + beforeAll(async () => { + writeFixtureRepo(root, { + 'Components/SessionSummaryBarViewComponent.cs': ` + namespace Demo.Components; + public class SessionSummaryBarViewComponent : ViewComponent + { + public object Invoke() => new object(); + } + `, + 'Components/MenuViewComponent.cs': ` + namespace Demo.Components; + [ApiController, ViewComponent(Name = "AccountMenu")] + public class MenuViewComponent : ViewComponent + { + public object Invoke() => new object(); + } + `, + 'One/DuplicateViewComponent.cs': ` + namespace Demo.One; + public class DuplicateViewComponent : ViewComponent {} + `, + 'Two/DuplicateViewComponent.cs': ` + namespace Demo.Two; + public class DuplicateViewComponent : ViewComponent {} + `, + 'Views/Home/Index.cshtml': ` + @await Component.InvokeAsync("SessionSummaryBar", new { id = 1 }) + + @{ + await Component.InvokeAsync("SessionSummaryBar"); + } + @@await Component.InvokeAsync("SessionSummaryBar") + + `, + 'Views/Shared/Alias.cshtml': `@await Component.InvokeAsync("AccountMenu")`, + 'Views/Shared/Ambiguous.cshtml': `@await Component.InvokeAsync("Duplicate")`, + 'Views/Shared/Commented.cshtml': ` + @* @await Component.InvokeAsync("SessionSummaryBar") *@ + `, + 'Views/Shared/Suffix.cshtml': `@await Component.InvokeAsync("Menu")`, + 'Controllers/HomeController.cs': ` + using Microsoft.AspNetCore.Mvc; + namespace Demo.Controllers; + public class HomeController : Controller + { + public IViewComponentResult Widget() => ViewComponent("SessionSummaryBar"); + public IViewComponentResult FromBase() => base.ViewComponent("SessionSummaryBar"); + public IViewComponentResult FromThis() => this.ViewComponent("SessionSummaryBar"); + } + `, + }); + result = await runPipelineFromRepo(root, () => {}, { skipGraphPhases: true }); + }, 120000); + + afterAll(() => { + fs.rmSync(root, { recursive: true, force: true }); + }); + + it('emits File-to-Class CALLS for literal and tag-helper invocations', () => { + const calls = getRelationships(result, 'CALLS').filter( + (edge) => edge.rel.reason === 'aspnet-razor-view-component', + ); + + expect( + calls + .map((edge) => ({ + source: edge.sourceFilePath, + target: edge.target, + targetLabel: edge.targetLabel, + })) + .sort((a, b) => `${a.source}:${a.target}`.localeCompare(`${b.source}:${b.target}`)), + ).toEqual([ + { + source: 'Controllers/HomeController.cs', + target: 'SessionSummaryBarViewComponent', + targetLabel: 'Class', + }, + { + source: 'Views/Home/Index.cshtml', + target: 'MenuViewComponent', + targetLabel: 'Class', + }, + { + source: 'Views/Home/Index.cshtml', + target: 'SessionSummaryBarViewComponent', + targetLabel: 'Class', + }, + { + source: 'Views/Shared/Alias.cshtml', + target: 'MenuViewComponent', + targetLabel: 'Class', + }, + ]); + expect(calls.some((edge) => edge.target === 'ViewComponent')).toBe(false); + expect(calls.some((edge) => edge.target === 'InvokeAsync')).toBe(false); + expect(calls.some((edge) => edge.sourceFilePath === 'Components/MenuViewComponent.cs')).toBe( + false, + ); + }); + + it('fails closed for ambiguous names, Razor comments, and replaced suffixes', () => { + const razorSources = getRelationships(result, 'CALLS') + .filter((edge) => edge.rel.reason === 'aspnet-razor-view-component') + .map((edge) => edge.sourceFilePath); + + expect(razorSources).not.toContain('Views/Shared/Ambiguous.cshtml'); + expect(razorSources).not.toContain('Views/Shared/Commented.cshtml'); + expect(razorSources).not.toContain('Views/Shared/Suffix.cshtml'); + }); +}); diff --git a/gitnexus/test/unit/scope-resolution/csharp/razor-view-components.test.ts b/gitnexus/test/unit/scope-resolution/csharp/razor-view-components.test.ts new file mode 100644 index 000000000..5ade5edf0 --- /dev/null +++ b/gitnexus/test/unit/scope-resolution/csharp/razor-view-components.test.ts @@ -0,0 +1,202 @@ +import { describe, expect, it } from 'vitest'; +import { + extractCsharpViewComponentInvocations, + extractRazorViewComponentInvocations, + extractViewComponentAliasBinds, + extractViewComponentAliases, +} from '../../../../src/core/ingestion/languages/csharp/razor-view-components.js'; + +describe('Razor ViewComponent convention extraction', () => { + it('extracts literal InvokeAsync calls and ViewComponent tag helpers', () => { + const source = ` + @await Component.InvokeAsync("SessionSummaryBar", new { id = 1 }) + @Component.InvokeAsync( + "Navigation" + ) + + `; + + expect(extractRazorViewComponentInvocations(source)).toEqual([ + 'SessionSummaryBar', + 'Navigation', + 'FeaturedProduct', + ]); + }); + + it('ignores Razor comments but keeps invocations inside HTML comments', () => { + const source = ` + @* @await Component.InvokeAsync("RazorComment") *@ + + + @await Component.InvokeAsync("Visible") + `; + + expect(extractRazorViewComponentInvocations(source)).toEqual([ + 'HtmlComment', + 'SessionSummaryBar', + 'Visible', + ]); + }); + + it('does not treat plain markup text as an invocation', () => { + expect( + extractRazorViewComponentInvocations( + `

Component.InvokeAsync("NotCode")

@Html.Partial("Card")`, + ), + ).toEqual([]); + }); + + it('treats even @ runs as literals and odd leftover @ as a transition', () => { + expect( + extractRazorViewComponentInvocations(` + @@await Component.InvokeAsync("Escaped") + @@@await Component.InvokeAsync("OddTransition") + `), + ).toEqual(['OddTransition']); + }); + + it('extracts calls from Razor code islands and explicit expressions', () => { + const source = ` + @{ + await Component.InvokeAsync("InBlock"); + // @await Component.InvokeAsync("CommentedInBlock") + /* await Component.InvokeAsync("BlockComment") */ + } + @if (true) + { + await Component.InvokeAsync("InIf"); + } + @(await Component.InvokeAsync("Explicit")) + `; + expect(extractRazorViewComponentInvocations(source)).toEqual(['InBlock', 'InIf', 'Explicit']); + }); + + it('extracts in-repo C# helper calls without matching SDK Task.InvokeAsync', () => { + const source = ` + await Component.InvokeAsync("SessionSummaryBar"); + return ViewComponent("AccountMenu"); + return this.ViewComponent("FromThis"); + return base.ViewComponent("FromBase"); + await this.Component.InvokeAsync("FromThisComponent"); + await Task.InvokeAsync("NotAComponent"); + renderer.ViewComponent("UnrelatedRenderer"); + await obj.Component.InvokeAsync("UnrelatedProperty"); + `; + expect(extractCsharpViewComponentInvocations(source)).toEqual([ + 'SessionSummaryBar', + 'AccountMenu', + 'FromThis', + 'FromBase', + 'FromThisComponent', + ]); + }); + + it('does not treat string literals as helper invocations', () => { + expect( + extractCsharpViewComponentInvocations(` + const string a = "ViewComponent(\\"DocsOnly\\")"; + const string b = @"ViewComponent(""Verbatim"")"; + const string c = """ViewComponent("Raw")"""; + return ViewComponent("Visible"); + `), + ).toEqual(['Visible']); + }); + + it('does not treat positional ViewComponent attributes as helper invocations', () => { + expect( + extractCsharpViewComponentInvocations(` + [ViewComponent("Alias")] + public class MenuViewComponent : ViewComponent {} + `), + ).toEqual([]); + }); + + it('extracts named and qualified ViewComponent aliases', () => { + const source = ` + [ViewComponent(Name = "AccountMenu")] + public sealed class MenuViewComponent : ViewComponent {} + + [Microsoft.AspNetCore.Mvc.ViewComponentAttribute(Name = "Admin.Checkout")] + internal class CheckoutWidget : ViewComponent {} + `; + expect(extractViewComponentAliases(source)).toEqual( + new Map([ + ['MenuViewComponent', ['AccountMenu']], + ['CheckoutWidget', ['Admin.Checkout']], + ]), + ); + }); + + it('extracts aliases from combined attribute lists', () => { + const source = ` + [ApiController, ViewComponent(Name = "AccountMenu")] + public sealed class MenuViewComponent : ViewComponent {} + `; + expect(extractViewComponentAliases(source)).toEqual( + new Map([['MenuViewComponent', ['AccountMenu']]]), + ); + }); + + it('extracts aliases from explicit record declarations', () => { + const source = ` + [ViewComponent(Name = "AccountMenu")] + public record class MenuViewComponent : ViewComponent {} + + [ViewComponent(Name = "Checkout")] + internal record CheckoutWidget : ViewComponent {} + `; + expect(extractViewComponentAliases(source)).toEqual( + new Map([ + ['MenuViewComponent', ['AccountMenu']], + ['CheckoutWidget', ['Checkout']], + ]), + ); + }); + + it('extracts aliases when comments sit between the attribute and the class', () => { + const source = ` + [ViewComponent(Name = "AccountMenu")] + // registered name overrides the suffix + public class MenuViewComponent : ViewComponent {} + + [ViewComponent(Name = "Checkout")] + /* other attrs */ + internal class CheckoutWidget : ViewComponent {} + `; + expect(extractViewComponentAliases(source)).toEqual( + new Map([ + ['MenuViewComponent', ['AccountMenu']], + ['CheckoutWidget', ['Checkout']], + ]), + ); + }); + + it('does not treat positional constructor arguments as aliases', () => { + expect( + extractViewComponentAliases(` + [ViewComponent("AccountMenu")] + public class MenuViewComponent : ViewComponent {} + `), + ).toEqual(new Map()); + }); + + it('binds aliases to the attributed class, not a same-named sibling', () => { + const binds = extractViewComponentAliasBinds(` + namespace A { [ViewComponent(Name = "AccountMenu")] class CardViewComponent {} } + namespace B { class CardViewComponent {} } + `); + const attributed = binds.filter((bind) => bind.aliases.includes('AccountMenu')); + expect(attributed).toHaveLength(1); + expect(attributed[0]?.className).toBe('CardViewComponent'); + expect(binds.filter((bind) => bind.className === 'CardViewComponent')).toHaveLength(1); + }); + + it('ignores commented-out C# helper calls', () => { + expect( + extractCsharpViewComponentInvocations(` + // return ViewComponent("Hidden"); + return ViewComponent("Visible"); + `), + ).toEqual(['Visible']); + }); +}); From 19f6731c344f533b39b383034c4a7ef1ddadbc6e Mon Sep 17 00:00:00 2001 From: ChunxueLi <54129170+ChunxueLi@users.noreply.github.com> Date: Mon, 31 Aug 2026 07:09:21 +0800 Subject: [PATCH 05/26] feat(java): resolve SpringContextUtil.getBeans(X.class) dynamic lookups (#2886) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(java): resolve SpringContextUtil.getBeans(X.class) dynamic lookups * fix(ingestion): make Spring dynamic lookups graph-correct Capture Java and Kotlin lookups from ASTs and resolve them through scoped type bindings and transitive JVM assignability so emitted INJECTS edges are attributable, cache-safe, and production-tested. Co-authored-by: Cursor * perf(ingestion): keep Spring lookup capture linear Reuse Java and Kotlin scope-query call nodes instead of rewalking each AST, cache DI subtype closures, and enforce linear scaling with production-path benchmarks in CI. Co-authored-by: Cursor --------- Co-authored-by: Gergő Magyar Co-authored-by: Gergo Magyar Co-authored-by: Cursor --- .github/workflows/ci-tests.yml | 1 + ARCHITECTURE.md | 2 +- .../frameworks/spring/dynamic-lookups.ts | 177 +++++++++++ .../languages/java/capture-side-channel.ts | 26 ++ .../core/ingestion/languages/java/captures.ts | 13 + .../languages/java/scope-resolver.ts | 2 + .../languages/java/spring-dynamic-lookup.ts | 77 +++++ .../languages/kotlin/capture-side-channel.ts | 27 ++ .../ingestion/languages/kotlin/captures.ts | 13 + .../languages/kotlin/scope-resolver.ts | 2 + .../languages/kotlin/spring-dynamic-lookup.ts | 90 ++++++ .../src/core/ingestion/pipeline-phases/di.ts | 67 ++++- gitnexus/src/storage/parse-cache.ts | 6 +- .../spring-dynamic-lookup-benchmark.test.ts | 275 ++++++++++++++++++ .../integration/spring-dynamic-lookup.test.ts | 220 ++++++++++++++ .../test/unit/incremental-parse-cache.test.ts | 6 +- gitnexus/test/unit/ingestion/di.test.ts | 134 ++++++++- .../test/unit/spring-dynamic-lookup.test.ts | 179 ++++++++++++ 18 files changed, 1299 insertions(+), 18 deletions(-) create mode 100644 gitnexus/src/core/ingestion/frameworks/spring/dynamic-lookups.ts create mode 100644 gitnexus/src/core/ingestion/languages/java/spring-dynamic-lookup.ts create mode 100644 gitnexus/src/core/ingestion/languages/kotlin/spring-dynamic-lookup.ts create mode 100644 gitnexus/test/integration/spring-dynamic-lookup-benchmark.test.ts create mode 100644 gitnexus/test/integration/spring-dynamic-lookup.test.ts create mode 100644 gitnexus/test/unit/spring-dynamic-lookup.test.ts diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index 3cf5c0aeb..5bdcf3560 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -726,6 +726,7 @@ jobs: test/integration/data-route-table-benchmark.test.ts test/integration/instance-ownership-pipeline-benchmark.test.ts test/integration/spring-bean-resource-benchmark.test.ts + test/integration/spring-dynamic-lookup-benchmark.test.ts test/integration/rust-pipeline-benchmark.test.ts test/integration/php-pipeline-benchmark.test.ts test/integration/ruby-pipeline-benchmark.test.ts diff --git a/ARCHITECTURE.md b/ARCHITECTURE.md index 6fe547b70..b1076d1d5 100644 --- a/ARCHITECTURE.md +++ b/ARCHITECTURE.md @@ -108,7 +108,7 @@ scan → structure → [springConfig, markdown, cobol] → parse → [routes, to | `pruneLocalSymbols` | `prune-local-symbols.ts` | `scopeResolution` | Drops inert block-local `Const`/`Variable`/`Static` nodes (only a `File→DEFINES` edge) post-resolution | | `mro` | `mro.ts` | `crossFile`, `scopeResolution`, `pruneLocalSymbols`, `structure` | METHOD_OVERRIDES + METHOD_IMPLEMENTS edges | | `springAopInheritance` | `spring-aop.ts` | `springAop`, `mro` | Propagates declarative behavior through class/interface inheritance decisions | -| `di` | `di.ts` | `mro` | INJECTS edges from consumer Classes or factory Methods to provider Classes/declaration CodeElements (framework-neutral DI resolution; per-language matchers registered in `di-extractors/`) | +| `di` | `di.ts` | `mro` | INJECTS edges from consumer Classes, factory Methods, or AST-captured programmatic lookup callables to provider Classes/declaration CodeElements (framework-neutral DI resolution; per-language matchers registered in `di-extractors/`) | | `communities` | `communities.ts` | `mro`, `pruneLocalSymbols`, `structure` | Community nodes + MEMBER_OF edges (Leiden algorithm) | | `processes` | `processes.ts` | `communities`, `routes`, `tools`, `pruneLocalSymbols`, `structure` | Process nodes + STEP_IN_PROCESS edges | diff --git a/gitnexus/src/core/ingestion/frameworks/spring/dynamic-lookups.ts b/gitnexus/src/core/ingestion/frameworks/spring/dynamic-lookups.ts new file mode 100644 index 000000000..dfcfbb7c0 --- /dev/null +++ b/gitnexus/src/core/ingestion/frameworks/spring/dynamic-lookups.ts @@ -0,0 +1,177 @@ +import type { ParsedFile, Range, ScopeId, SymbolDefinition } from 'gitnexus-shared'; +import type { KnowledgeGraph } from '../../../graph/types.js'; +import type { DiInjectionMatch } from '../../di-extractors/index.js'; +import { SPRING_DI_INJECTION_SITES_PROPERTY } from '../../di-extractors/spring.js'; +import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js'; +import { + resolveCallerGraphId, + resolveDefGraphId, +} from '../../scope-resolution/graph-bridge/ids.js'; +import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js'; +import { isClassLike, lookupBindingsAt } from '../../scope-resolution/scope/walkers.js'; + +const COLLECTION_LOOKUP_METHODS = new Set(['getBeans', 'getBeansOfType']); +const SINGLE_LOOKUP_METHODS = new Set(['getBean']); + +/** + * Distinctive utility names plus conventional Spring context variable names. + * Generic locals remain recall-oriented because repositories often omit the + * third-party context type from the index; AST call/class-literal gates and + * import-aware target resolution prevent the raw-text false-positive class. + */ +const KNOWN_RECEIVERS = new Set([ + 'SpringContextUtil', + 'SpringContextHolder', + 'SpringBeanUtil', + 'ApplicationContextProvider', + 'BeanFactoryProvider', + 'ApplicationContext', + 'BeanFactory', + 'ListableBeanFactory', + 'applicationContext', + 'context', + 'ctx', + 'appContext', + 'beanFactory', +]); + +export interface SpringDynamicLookupFact { + readonly ownerScopeId: ScopeId; + readonly ownerRange: Range; + readonly receiverName: string; + readonly methodName: string; + readonly targetTypeName: string; +} + +export function springDynamicLookupCardinality( + receiverName: string, + methodName: string, +): DiInjectionMatch['cardinality'] | null { + const receiverSimpleName = receiverName.slice(receiverName.lastIndexOf('.') + 1); + if (!KNOWN_RECEIVERS.has(receiverSimpleName)) return null; + if (COLLECTION_LOOKUP_METHODS.has(methodName)) return 'collection'; + if (SINGLE_LOOKUP_METHODS.has(methodName)) return 'single'; + return null; +} + +function visibleTypeDefinitions( + fact: SpringDynamicLookupFact, + indexes: ScopeResolutionIndexes, +): readonly SymbolDefinition[] { + const simpleName = fact.targetTypeName.slice(fact.targetTypeName.lastIndexOf('.') + 1); + let scopeId: ScopeId | null = fact.ownerScopeId; + + while (scopeId !== null) { + const visible = lookupBindingsAt(scopeId, simpleName, indexes) + .map(({ def }) => def) + .filter((def) => isClassLike(def.type)) + .filter( + (def) => !fact.targetTypeName.includes('.') || def.qualifiedName === fact.targetTypeName, + ); + if (visible.length > 0) { + const unique = new Map(visible.map((def) => [def.nodeId, def])); + return [...unique.values()]; + } + scopeId = indexes.scopeTree.getScope(scopeId)?.parent ?? null; + } + + return []; +} + +function resolveTargetTypeName( + graph: KnowledgeGraph, + fact: SpringDynamicLookupFact, + callerLanguage: string | undefined, + nodeLookup: GraphNodeLookup, + indexes: ScopeResolutionIndexes, +): string | undefined { + const graphIds = new Set(); + for (const definition of visibleTypeDefinitions(fact, indexes)) { + const graphId = resolveDefGraphId(definition.filePath, definition, nodeLookup); + if (graphId === undefined) continue; + const node = graph.getNode(graphId); + if ( + (node?.label === 'Class' || + node?.label === 'Interface' || + node?.label === 'Record' || + node?.label === 'Enum') && + node.properties.language === callerLanguage + ) { + graphIds.add(graphId); + } + } + if (graphIds.size !== 1) return undefined; + + const targetId = graphIds.values().next().value; + if (targetId === undefined) return undefined; + const target = graph.getNode(targetId); + if (target === undefined) return undefined; + const qualifiedName = target.properties.qualifiedName; + return typeof qualifiedName === 'string' ? qualifiedName : target.properties.name; +} + +export interface SpringDynamicLookupMetadataAdapter { + getFacts(filePath: string): readonly SpringDynamicLookupFact[]; +} + +/** + * Attach AST-captured programmatic Spring lookups to the framework-neutral DI + * resolver. Java/Kotlin own syntax capture; this shared JVM/Spring seam owns + * import-aware type binding and metadata attachment. + */ +export function createSpringDynamicLookupMetadataAttacher( + adapter: SpringDynamicLookupMetadataAdapter, +) { + return ( + graph: KnowledgeGraph, + parsedFiles: readonly ParsedFile[], + nodeLookup: GraphNodeLookup, + indexes: ScopeResolutionIndexes, + ): void => { + for (const parsed of parsedFiles) { + for (const fact of adapter.getFacts(parsed.filePath)) { + const cardinality = springDynamicLookupCardinality(fact.receiverName, fact.methodName); + if (cardinality === null) continue; + + const callerId = resolveCallerGraphId(fact.ownerScopeId, indexes, nodeLookup, { + startLine: fact.ownerRange.startLine, + startCol: fact.ownerRange.startCol, + }); + if (callerId === undefined) continue; + const caller = graph.getNode(callerId); + if ( + caller === undefined || + (caller.label !== 'Function' && + caller.label !== 'Method' && + caller.label !== 'Constructor') + ) { + continue; + } + + const targetTypeName = resolveTargetTypeName( + graph, + fact, + caller.properties.language, + nodeLookup, + indexes, + ); + if (targetTypeName === undefined) continue; + + const match: DiInjectionMatch = { + targetTypeName, + cardinality, + edgeSource: 'site', + reason: `Spring dynamic lookup: ${fact.receiverName}.${fact.methodName}(${fact.targetTypeName})`, + }; + // Singular lookups intentionally use the shared DI selection policy: + // a unique/@Primary candidate wins; unresolved multiplicity is an + // explicit 0.5-confidence fan-out rather than a guessed runtime winner. + const existing = caller.properties[SPRING_DI_INJECTION_SITES_PROPERTY]; + caller.properties[SPRING_DI_INJECTION_SITES_PROPERTY] = [ + ...(Array.isArray(existing) ? existing : []), + match, + ]; + } + } + }; +} diff --git a/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts index 91a910fa5..0da803429 100644 --- a/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts +++ b/gitnexus/src/core/ingestion/languages/java/capture-side-channel.ts @@ -13,6 +13,7 @@ import type { JavaSpringConfigConsumerFact } from './spring-config-bindings.js'; import type { JavaSpringAopFact } from './spring-aop.js'; import type { JavaSpringConditionalFact } from './spring-conditionals.js'; import type { JavaSpringDiClassFact } from './spring-di.js'; +import type { SpringDynamicLookupFact } from '../../frameworks/spring/dynamic-lookups.js'; import type { JavaSpringNonHttpHandlerFact } from './spring-non-http-handlers.js'; export type JavaClassAnnotationFact = ClassAnnotationFact; @@ -25,6 +26,7 @@ export interface JavaCaptureSideChannel { readonly springConfigConsumers?: readonly JavaSpringConfigConsumerFact[]; readonly springConditionalFacts?: readonly JavaSpringConditionalFact[]; readonly springDiFacts?: readonly JavaSpringDiClassFact[]; + readonly springDynamicLookupFacts?: readonly SpringDynamicLookupFact[]; readonly springNonHttpHandlerFacts?: readonly JavaSpringNonHttpHandlerFact[]; } @@ -33,6 +35,7 @@ const springAopFacts = new Map(); const springConfigConsumers = new Map(); const springConditionalFacts = new Map(); const springDiFacts = new Map(); +const springDynamicLookupFacts = new Map(); const springNonHttpHandlerFacts = new Map(); /** Clear facts retained by a prior workspace pass in a long-lived process. */ @@ -42,6 +45,7 @@ export function clearJavaClassAnnotationFacts(): void { springConfigConsumers.clear(); springConditionalFacts.clear(); springDiFacts.clear(); + springDynamicLookupFacts.clear(); springNonHttpHandlerFacts.clear(); } @@ -102,6 +106,20 @@ export function getJavaSpringDiFacts(filePath: string): readonly JavaSpringDiCla return springDiFacts.get(filePath) ?? []; } +export function setJavaSpringDynamicLookupFacts( + filePath: string, + facts: readonly SpringDynamicLookupFact[], +): void { + if (facts.length === 0) springDynamicLookupFacts.delete(filePath); + else springDynamicLookupFacts.set(filePath, facts); +} + +export function getJavaSpringDynamicLookupFacts( + filePath: string, +): readonly SpringDynamicLookupFact[] { + return springDynamicLookupFacts.get(filePath) ?? []; +} + export function setJavaSpringNonHttpHandlerFacts( filePath: string, facts: readonly JavaSpringNonHttpHandlerFact[], @@ -125,6 +143,7 @@ export function collectJavaCaptureSideChannel( const configConsumers = springConfigConsumers.get(filePath) ?? []; const conditionFacts = springConditionalFacts.get(filePath) ?? []; const diFacts = springDiFacts.get(filePath) ?? []; + const dynamicLookupFacts = springDynamicLookupFacts.get(filePath) ?? []; const nonHttpHandlerFacts = springNonHttpHandlerFacts.get(filePath) ?? []; const packageFact = getJavaPackageFact(filePath); if ( @@ -133,6 +152,7 @@ export function collectJavaCaptureSideChannel( configConsumers.length === 0 && conditionFacts.length === 0 && diFacts.length === 0 && + dynamicLookupFacts.length === 0 && nonHttpHandlerFacts.length === 0 && packageFact === undefined ) { @@ -146,6 +166,7 @@ export function collectJavaCaptureSideChannel( ...(configConsumers.length > 0 ? { springConfigConsumers: configConsumers } : {}), ...(conditionFacts.length > 0 ? { springConditionalFacts: conditionFacts } : {}), ...(diFacts.length > 0 ? { springDiFacts: diFacts } : {}), + ...(dynamicLookupFacts.length > 0 ? { springDynamicLookupFacts: dynamicLookupFacts } : {}), ...(nonHttpHandlerFacts.length > 0 ? { springNonHttpHandlerFacts: nonHttpHandlerFacts } : {}), }; } @@ -169,6 +190,7 @@ export function applyJavaCaptureSideChannel(parsed: ParsedFile): void { setJavaSpringConfigConsumerFacts(parsed.filePath, []); setJavaSpringConditionalFacts(parsed.filePath, []); setJavaSpringDiFacts(parsed.filePath, []); + setJavaSpringDynamicLookupFacts(parsed.filePath, []); setJavaSpringNonHttpHandlerFacts(parsed.filePath, []); setJavaPackageFact(parsed.filePath, UNKNOWN_JVM_PACKAGE_FACT); return; @@ -190,6 +212,10 @@ export function applyJavaCaptureSideChannel(parsed: ParsedFile): void { parsed.filePath, Array.isArray(data.springDiFacts) ? data.springDiFacts : [], ); + setJavaSpringDynamicLookupFacts( + parsed.filePath, + Array.isArray(data.springDynamicLookupFacts) ? data.springDynamicLookupFacts : [], + ); setJavaSpringNonHttpHandlerFacts( parsed.filePath, Array.isArray(data.springNonHttpHandlerFacts) ? data.springNonHttpHandlerFacts : [], diff --git a/gitnexus/src/core/ingestion/languages/java/captures.ts b/gitnexus/src/core/ingestion/languages/java/captures.ts index ce5ed4b93..22083e6e5 100644 --- a/gitnexus/src/core/ingestion/languages/java/captures.ts +++ b/gitnexus/src/core/ingestion/languages/java/captures.ts @@ -39,12 +39,15 @@ import { setJavaSpringConfigConsumerFacts, setJavaSpringConditionalFacts, setJavaSpringDiFacts, + setJavaSpringDynamicLookupFacts, setJavaSpringNonHttpHandlerFacts, } from './capture-side-channel.js'; import { captureJavaPackageFact } from './package-facts.js'; import { synthesizeCallableFlowCaptures } from '../../utils/callable-flow-captures.js'; import { captureJavaSpringConfigConsumerFacts } from './spring-config-bindings.js'; import { captureJavaSpringDiClassFact, type JavaSpringDiClassFact } from './spring-di.js'; +import type { SpringDynamicLookupFact } from '../../frameworks/spring/dynamic-lookups.js'; +import { captureJavaSpringDynamicLookupFact } from './spring-dynamic-lookup.js'; import { synthesizeReceiverChainCapture } from '../../utils/receiver-chain-captures.js'; import { captureJavaSpringAopFacts, type JavaSpringAopFact } from './spring-aop.js'; import { @@ -146,6 +149,8 @@ export function emitJavaScopeCaptures( const springDiFacts: JavaSpringDiClassFact[] = []; const springNonHttpHandlerFacts: JavaSpringNonHttpHandlerFact[] = []; const springDiClassNodeIds = new Set(); + const springDynamicLookupFacts: SpringDynamicLookupFact[] = []; + const springDynamicLookupNodeIds = new Set(); for (const m of rawMatches) { const grouped: Record = {}; @@ -165,6 +170,13 @@ export function emitJavaScopeCaptures( } if (Object.keys(grouped).length === 0) continue; + const dynamicLookupNode = nodeIfType(nodeMap['@reference.call.member'], 'method_invocation'); + if (dynamicLookupNode !== null && !springDynamicLookupNodeIds.has(dynamicLookupNode.id)) { + springDynamicLookupNodeIds.add(dynamicLookupNode.id); + const fact = captureJavaSpringDynamicLookupFact(dynamicLookupNode, filePath); + if (fact !== null) springDynamicLookupFacts.push(fact); + } + const springAopTypeNode = [ nodeIfType(nodeMap['@scope.class'], 'class_declaration'), nodeIfType(nodeMap['@scope.class'], 'interface_declaration'), @@ -401,6 +413,7 @@ export function emitJavaScopeCaptures( setJavaSpringAopFacts(filePath, springAopFacts); setJavaSpringConditionalFacts(filePath, springConditionalFacts); setJavaSpringDiFacts(filePath, springDiFacts); + setJavaSpringDynamicLookupFacts(filePath, springDynamicLookupFacts); setJavaSpringNonHttpHandlerFacts(filePath, springNonHttpHandlerFacts); return [ diff --git a/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts index 0410baa6c..3f23040a6 100644 --- a/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/java/scope-resolver.ts @@ -35,6 +35,7 @@ import { attachJavaSpringConfigBindings } from './spring-config-bindings.js'; import { attachJavaSpringConditionalMetadata } from './spring-conditionals.js'; import { attachJavaSpringDiMetadata } from './spring-di.js'; import { attachJavaSpringNonHttpHandlerMetadata } from './spring-non-http-handlers.js'; +import { attachJavaSpringDynamicLookup } from './spring-dynamic-lookup.js'; import { applyJavaCaptureSideChannel, clearJavaClassAnnotationFacts, @@ -97,6 +98,7 @@ const javaScopeResolver: ScopeResolver = { attachJavaSpringDiMetadata(graph, parsedFiles, nodeLookup, indexes); attachJavaSpringNonHttpHandlerMetadata(graph, parsedFiles, nodeLookup, indexes); attachJavaSpringConfigBindings(graph, parsedFiles, nodeLookup, indexes, ctx); + attachJavaSpringDynamicLookup(graph, parsedFiles, nodeLookup, indexes); }, }; diff --git a/gitnexus/src/core/ingestion/languages/java/spring-dynamic-lookup.ts b/gitnexus/src/core/ingestion/languages/java/spring-dynamic-lookup.ts new file mode 100644 index 000000000..922264bf8 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/spring-dynamic-lookup.ts @@ -0,0 +1,77 @@ +import { makeScopeId } from 'gitnexus-shared'; +import { + createSpringDynamicLookupMetadataAttacher, + springDynamicLookupCardinality, + type SpringDynamicLookupFact, +} from '../../frameworks/spring/dynamic-lookups.js'; +import { + findAncestorBeforeBoundary, + nodeToCapture, + type SyntaxNode, +} from '../../utils/ast-helpers.js'; +import { getJavaSpringDynamicLookupFacts } from './capture-side-channel.js'; + +const CALLABLE_NODE_TYPES = new Set([ + 'method_declaration', + 'constructor_declaration', + 'compact_constructor_declaration', +]); +const NO_CALLABLE_BOUNDARIES = new Set(); + +function classLiteralTypeName(argument: SyntaxNode): string | null { + if (argument.type !== 'class_literal' || argument.namedChildCount !== 1) return null; + return argument.namedChild(0)?.text.trim() ?? null; +} + +/** Capture real Java method invocations; comments and literals are never visited as calls. */ +export function captureJavaSpringDynamicLookupFact( + node: SyntaxNode, + filePath: string, +): SpringDynamicLookupFact | null { + if (node.type !== 'method_invocation') return null; + const receiverName = node.childForFieldName('object')?.text.trim(); + const methodName = node.childForFieldName('name')?.text.trim(); + const argumentsNode = node.childForFieldName('arguments'); + if (receiverName === undefined || methodName === undefined || argumentsNode === null) return null; + if (springDynamicLookupCardinality(receiverName, methodName) === null) return null; + + const argumentsWithoutComments = argumentsNode.namedChildren.filter( + (child) => child.type !== 'line_comment' && child.type !== 'block_comment', + ); + if (argumentsWithoutComments.length !== 1) return null; + const argument = argumentsWithoutComments[0]; + if (argument === undefined) return null; + const targetTypeName = classLiteralTypeName(argument); + if (targetTypeName === null) return null; + + const owner = findAncestorBeforeBoundary(node, CALLABLE_NODE_TYPES, NO_CALLABLE_BOUNDARIES); + if (owner === null) return null; + const ownerCapture = nodeToCapture('@spring-dynamic-lookup.owner', owner); + return { + ownerScopeId: makeScopeId({ + filePath, + range: ownerCapture.range, + kind: 'Function', + }), + ownerRange: ownerCapture.range, + receiverName, + methodName, + targetTypeName, + }; +} + +/** Standalone extractor for focused tests; production reuses scope-query call nodes. */ +export function captureJavaSpringDynamicLookupFacts( + rootNode: SyntaxNode, + filePath: string, +): SpringDynamicLookupFact[] { + return rootNode + .descendantsOfType('method_invocation') + .map((node) => captureJavaSpringDynamicLookupFact(node, filePath)) + .filter((fact): fact is SpringDynamicLookupFact => fact !== null); +} + +/** Attach Java lookup facts for later resolution by the shared DI phase. */ +export const attachJavaSpringDynamicLookup = createSpringDynamicLookupMetadataAttacher({ + getFacts: getJavaSpringDynamicLookupFacts, +}); diff --git a/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts b/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts index 6ea9480a8..8c8ed35a7 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/capture-side-channel.ts @@ -49,6 +49,7 @@ import { } from '../jvm/package-facts.js'; import { getCompanionScopesForFile, markCompanionScope } from './companion-scopes.js'; import { getKotlinPackageFact, setKotlinPackageFact } from './package-facts.js'; +import type { SpringDynamicLookupFact } from '../../frameworks/spring/dynamic-lookups.js'; import type { KotlinSpringAopFact } from './spring-aop.js'; import type { KotlinSpringConditionalFact } from './spring-conditionals.js'; import type { KotlinSpringDiClassFact } from './spring-di.js'; @@ -58,6 +59,7 @@ const classAnnotations = createClassAnnotationFactStore(); const springAopFacts = new Map(); const springConditionalFacts = new Map(); const springDiFacts = new Map(); +const springDynamicLookupFacts = new Map(); const springNonHttpHandlerFacts = new Map(); /** @@ -80,6 +82,8 @@ export interface KotlinCaptureSideChannel { readonly springConditionalFacts?: readonly KotlinSpringConditionalFact[]; /** Constructor, property, and method injection syntax captured per class. */ readonly springDiFacts?: readonly KotlinSpringDiClassFact[]; + /** Programmatic Spring bean lookups captured per callable. */ + readonly springDynamicLookupFacts?: readonly SpringDynamicLookupFact[]; /** Scheduled, event, messaging, and managed-job handler syntax captured per callable. */ readonly springNonHttpHandlerFacts?: readonly KotlinSpringNonHttpHandlerFact[]; } @@ -89,6 +93,7 @@ export function clearKotlinClassAnnotationFacts(): void { springAopFacts.clear(); springConditionalFacts.clear(); springDiFacts.clear(); + springDynamicLookupFacts.clear(); springNonHttpHandlerFacts.clear(); } @@ -141,6 +146,20 @@ export function getKotlinSpringDiFacts(filePath: string): readonly KotlinSpringD return springDiFacts.get(filePath) ?? []; } +export function setKotlinSpringDynamicLookupFacts( + filePath: string, + facts: readonly SpringDynamicLookupFact[], +): void { + if (facts.length === 0) springDynamicLookupFacts.delete(filePath); + else springDynamicLookupFacts.set(filePath, facts); +} + +export function getKotlinSpringDynamicLookupFacts( + filePath: string, +): readonly SpringDynamicLookupFact[] { + return springDynamicLookupFacts.get(filePath) ?? []; +} + export function setKotlinSpringNonHttpHandlerFacts( filePath: string, facts: readonly KotlinSpringNonHttpHandlerFact[], @@ -168,6 +187,7 @@ export function collectKotlinCaptureSideChannel( const aopFacts = springAopFacts.get(filePath) ?? []; const conditionFacts = springConditionalFacts.get(filePath) ?? []; const diFacts = springDiFacts.get(filePath) ?? []; + const dynamicLookupFacts = springDynamicLookupFacts.get(filePath) ?? []; const nonHttpHandlerFacts = springNonHttpHandlerFacts.get(filePath) ?? []; const packageFact = getKotlinPackageFact(filePath); if ( @@ -176,6 +196,7 @@ export function collectKotlinCaptureSideChannel( aopFacts.length === 0 && conditionFacts.length === 0 && diFacts.length === 0 && + dynamicLookupFacts.length === 0 && nonHttpHandlerFacts.length === 0 && packageFact === undefined ) { @@ -189,6 +210,7 @@ export function collectKotlinCaptureSideChannel( ...(aopFacts.length > 0 ? { springAopFacts: aopFacts } : {}), ...(conditionFacts.length > 0 ? { springConditionalFacts: conditionFacts } : {}), ...(diFacts.length > 0 ? { springDiFacts: diFacts } : {}), + ...(dynamicLookupFacts.length > 0 ? { springDynamicLookupFacts: dynamicLookupFacts } : {}), ...(nonHttpHandlerFacts.length > 0 ? { springNonHttpHandlerFacts: nonHttpHandlerFacts } : {}), }; } @@ -215,6 +237,7 @@ export function applyKotlinCaptureSideChannel(parsed: ParsedFile): void { setKotlinSpringAopFacts(parsed.filePath, []); setKotlinSpringConditionalFacts(parsed.filePath, []); setKotlinSpringDiFacts(parsed.filePath, []); + setKotlinSpringDynamicLookupFacts(parsed.filePath, []); setKotlinSpringNonHttpHandlerFacts(parsed.filePath, []); setKotlinPackageFact(parsed.filePath, UNKNOWN_JVM_PACKAGE_FACT); return; @@ -235,6 +258,10 @@ export function applyKotlinCaptureSideChannel(parsed: ParsedFile): void { parsed.filePath, Array.isArray(data.springDiFacts) ? data.springDiFacts : [], ); + setKotlinSpringDynamicLookupFacts( + parsed.filePath, + Array.isArray(data.springDynamicLookupFacts) ? data.springDynamicLookupFacts : [], + ); setKotlinSpringNonHttpHandlerFacts( parsed.filePath, Array.isArray(data.springNonHttpHandlerFacts) ? data.springNonHttpHandlerFacts : [], diff --git a/gitnexus/src/core/ingestion/languages/kotlin/captures.ts b/gitnexus/src/core/ingestion/languages/kotlin/captures.ts index 84ec8dc4d..f3b00db03 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/captures.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/captures.ts @@ -23,11 +23,14 @@ import { setKotlinSpringAopFacts, setKotlinSpringConditionalFacts, setKotlinSpringDiFacts, + setKotlinSpringDynamicLookupFacts, setKotlinSpringNonHttpHandlerFacts, } from './capture-side-channel.js'; import { captureKotlinPackageFact } from './package-facts.js'; import { synthesizeCallableFlowCaptures } from '../../utils/callable-flow-captures.js'; import { captureKotlinSpringDiClassFact, type KotlinSpringDiClassFact } from './spring-di.js'; +import type { SpringDynamicLookupFact } from '../../frameworks/spring/dynamic-lookups.js'; +import { captureKotlinSpringDynamicLookupFact } from './spring-dynamic-lookup.js'; import { synthesizeReceiverChainCapture } from '../../utils/receiver-chain-captures.js'; import { captureKotlinSpringAopFacts, type KotlinSpringAopFact } from './spring-aop.js'; import { @@ -107,6 +110,8 @@ export function emitKotlinScopeCaptures( const springNonHttpHandlerFacts: KotlinSpringNonHttpHandlerFact[] = []; const springNonHttpHandlerTypeNodeIds = new Set(); const springDiClassNodeIds = new Set(); + const springDynamicLookupFacts: SpringDynamicLookupFact[] = []; + const springDynamicLookupNodeIds = new Set(); const returnTypes = collectKotlinReturnTypeTexts(tree.rootNode); out.push(...synthesizeKotlinLocalAssignmentBindings(tree.rootNode, returnTypes)); out.push(...synthesizeKotlinLoopBindings(tree.rootNode, returnTypes)); @@ -130,6 +135,13 @@ export function emitKotlinScopeCaptures( } if (Object.keys(grouped).length === 0) continue; + const dynamicLookupNode = nodeIfType(groupedNodes['@reference.call.member'], 'call_expression'); + if (dynamicLookupNode !== null && !springDynamicLookupNodeIds.has(dynamicLookupNode.id)) { + springDynamicLookupNodeIds.add(dynamicLookupNode.id); + const fact = captureKotlinSpringDynamicLookupFact(dynamicLookupNode, filePath); + if (fact !== null) springDynamicLookupFacts.push(fact); + } + // tree-sitter-kotlin represents both classes and interfaces with // `class_declaration`; `object_declaration` is the separate object form. const springAopTypeNode = [ @@ -357,6 +369,7 @@ export function emitKotlinScopeCaptures( setKotlinSpringAopFacts(filePath, springAopFacts); setKotlinSpringConditionalFacts(filePath, springConditionalFacts); setKotlinSpringDiFacts(filePath, springDiFacts); + setKotlinSpringDynamicLookupFacts(filePath, springDynamicLookupFacts); setKotlinSpringNonHttpHandlerFacts(filePath, springNonHttpHandlerFacts); out.push(...synthesizeCallableFlowCaptures(tree.rootNode, KOTLIN_CALLABLE_CAPTURE_OPTIONS)); return out; diff --git a/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts b/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts index 0fffeb60a..514d65d26 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/scope-resolver.ts @@ -27,6 +27,7 @@ import { clearKotlinPackageFacts } from './package-facts.js'; import { attachKotlinSpringDiMetadata } from './spring-di.js'; import { attachKotlinSpringConditionalMetadata } from './spring-conditionals.js'; import { attachKotlinSpringNonHttpHandlerMetadata } from './spring-non-http-handlers.js'; +import { attachKotlinSpringDynamicLookup } from './spring-dynamic-lookup.js'; /** * Kotlin scope resolver for RFC #909 Ring 3. @@ -148,6 +149,7 @@ export const kotlinScopeResolver: ScopeResolver = { attachKotlinSpringConditionalMetadata(graph, parsedFiles, nodeLookup, indexes); attachKotlinSpringDiMetadata(graph, parsedFiles, nodeLookup, indexes); attachKotlinSpringNonHttpHandlerMetadata(graph, parsedFiles, nodeLookup, indexes); + attachKotlinSpringDynamicLookup(graph, parsedFiles, nodeLookup, indexes); }, }; diff --git a/gitnexus/src/core/ingestion/languages/kotlin/spring-dynamic-lookup.ts b/gitnexus/src/core/ingestion/languages/kotlin/spring-dynamic-lookup.ts new file mode 100644 index 000000000..184048e07 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/kotlin/spring-dynamic-lookup.ts @@ -0,0 +1,90 @@ +import { makeScopeId } from 'gitnexus-shared'; +import { + createSpringDynamicLookupMetadataAttacher, + springDynamicLookupCardinality, + type SpringDynamicLookupFact, +} from '../../frameworks/spring/dynamic-lookups.js'; +import { + findAncestorBeforeBoundary, + nodeToCapture, + type SyntaxNode, +} from '../../utils/ast-helpers.js'; +import { getKotlinSpringDynamicLookupFacts } from './capture-side-channel.js'; + +// Kotlin emits graph callables for functions and secondary constructors. +// `init {}` / primary-constructor bodies have no independent callable node, so +// attributing their lookups to the enclosing Class would violate graph semantics. +const CALLABLE_NODE_TYPES = new Set(['function_declaration', 'secondary_constructor']); +const NO_CALLABLE_BOUNDARIES = new Set(); +const KOTLIN_CLASS_LITERAL = + /^([A-Za-z_$][A-Za-z0-9_$]*(?:\.[A-Za-z_$][A-Za-z0-9_$]*)*)::class(?:\.java)?$/; + +function navigationParts(node: SyntaxNode): { receiverName: string; methodName: string } | null { + if (node.type !== 'navigation_expression') return null; + const text = node.text.trim(); + const separator = text.lastIndexOf('.'); + if (separator <= 0 || separator === text.length - 1) return null; + return { + receiverName: text.slice(0, separator), + methodName: text.slice(separator + 1), + }; +} + +function singleClassLiteralArgument(node: SyntaxNode): string | null { + const suffix = node.namedChildren.find((child) => child.type === 'call_suffix'); + const argumentsNode = suffix?.namedChildren.find((child) => child.type === 'value_arguments'); + if (argumentsNode === undefined) return null; + const argumentsWithoutComments = argumentsNode.namedChildren.filter( + (child) => child.type !== 'line_comment' && child.type !== 'multiline_comment', + ); + if (argumentsWithoutComments.length !== 1) return null; + const value = argumentsWithoutComments[0]; + if (value?.type !== 'value_argument' || value.namedChildCount !== 1) return null; + return value.namedChild(0)?.text.trim().match(KOTLIN_CLASS_LITERAL)?.[1] ?? null; +} + +/** Capture real Kotlin calls using `Type::class` or `Type::class.java`. */ +export function captureKotlinSpringDynamicLookupFact( + node: SyntaxNode, + filePath: string, +): SpringDynamicLookupFact | null { + if (node.type !== 'call_expression') return null; + const callee = node.namedChildren.find((child) => child.type === 'navigation_expression'); + if (callee === undefined) return null; + const parts = navigationParts(callee); + if (parts === null) return null; + if (springDynamicLookupCardinality(parts.receiverName, parts.methodName) === null) return null; + const targetTypeName = singleClassLiteralArgument(node); + if (targetTypeName === null) return null; + + const owner = findAncestorBeforeBoundary(node, CALLABLE_NODE_TYPES, NO_CALLABLE_BOUNDARIES); + if (owner === null) return null; + const ownerCapture = nodeToCapture('@spring-dynamic-lookup.owner', owner); + return { + ownerScopeId: makeScopeId({ + filePath, + range: ownerCapture.range, + kind: 'Function', + }), + ownerRange: ownerCapture.range, + receiverName: parts.receiverName, + methodName: parts.methodName, + targetTypeName, + }; +} + +/** Standalone extractor for focused tests; production reuses scope-query call nodes. */ +export function captureKotlinSpringDynamicLookupFacts( + rootNode: SyntaxNode, + filePath: string, +): SpringDynamicLookupFact[] { + return rootNode + .descendantsOfType('call_expression') + .map((node) => captureKotlinSpringDynamicLookupFact(node, filePath)) + .filter((fact): fact is SpringDynamicLookupFact => fact !== null); +} + +/** Attach Kotlin lookup facts for later resolution by the shared DI phase. */ +export const attachKotlinSpringDynamicLookup = createSpringDynamicLookupMetadataAttacher({ + getFacts: getKotlinSpringDynamicLookupFacts, +}); diff --git a/gitnexus/src/core/ingestion/pipeline-phases/di.ts b/gitnexus/src/core/ingestion/pipeline-phases/di.ts index 12ee5cb3e..b8bf69bd8 100644 --- a/gitnexus/src/core/ingestion/pipeline-phases/di.ts +++ b/gitnexus/src/core/ingestion/pipeline-phases/di.ts @@ -102,6 +102,10 @@ function providerCandidates( return recognized.length > 0 ? recognized : all; } +function isConcreteTypeNode(node: GraphNode | undefined): boolean { + return node?.label === 'Class' || node?.label === 'Record' || node?.label === 'Enum'; +} + export const diPhase: PipelinePhase = { name: 'di', deps: ['mro'], @@ -160,21 +164,27 @@ export const diPhase: PipelinePhase = { }; } - const interfaceToImplementers = new Map>(); + const directSubtypes = new Map>(); const directSupertypes = new Map>(); for (const rel of ctx.graph.iterRelationshipsByType('IMPLEMENTS')) { - const set = interfaceToImplementers.get(rel.targetId) ?? new Set(); - set.add(rel.sourceId); - interfaceToImplementers.set(rel.targetId, set); + const subtypes = directSubtypes.get(rel.targetId) ?? new Set(); + subtypes.add(rel.sourceId); + directSubtypes.set(rel.targetId, subtypes); const supertypes = directSupertypes.get(rel.sourceId) ?? new Set(); supertypes.add(rel.targetId); directSupertypes.set(rel.sourceId, supertypes); } for (const rel of ctx.graph.iterRelationshipsByType('EXTENDS')) { + const subtypes = directSubtypes.get(rel.targetId) ?? new Set(); + subtypes.add(rel.sourceId); + directSubtypes.set(rel.targetId, subtypes); const supertypes = directSupertypes.get(rel.sourceId) ?? new Set(); supertypes.add(rel.targetId); directSupertypes.set(rel.sourceId, supertypes); } + const orderedDirectSubtypes = new Map( + [...directSubtypes].map(([typeId, subtypes]) => [typeId, [...subtypes].sort().reverse()]), + ); const memberToClass = new Map(); for (const relationType of ['HAS_PROPERTY', 'HAS_METHOD'] as const) { @@ -187,16 +197,45 @@ export const diPhase: PipelinePhase = { const interfacesByLanguage = new Map(); const classesByLanguage = new Map(); ctx.graph.forEachNode((node) => { - if (node.label !== 'Class' && node.label !== 'Interface') return; + const concreteType = isConcreteTypeNode(node); + if (!concreteType && node.label !== 'Interface') return; const language = node.properties.language; if (typeof language !== 'string' || !candidateLanguages.has(language)) return; - const indexes = node.label === 'Class' ? classesByLanguage : interfacesByLanguage; + const indexes = concreteType ? classesByLanguage : interfacesByLanguage; const index = indexes.get(language) ?? emptyNameIndex(); addIndexedName(index, node); indexes.set(language, index); - if (node.label === 'Class') providerNodes.set(node.id, node); + if (concreteType) providerNodes.set(node.id, node); }); + const concreteSubtypesByRoot = new Map>(); + const concreteSubtypes = (rootTypeId: string, language: string): ReadonlySet => { + const cacheKey = `${language}\0${rootTypeId}`; + const cached = concreteSubtypesByRoot.get(cacheKey); + if (cached !== undefined) return cached; + + const concrete = new Set(); + const queue = [rootTypeId]; + const visited = new Set(); + while (queue.length > 0) { + const typeId = queue.pop(); + if (typeId === undefined || visited.has(typeId)) continue; + visited.add(typeId); + const typeNode = ctx.graph.getNode(typeId); + if ( + typeNode !== undefined && + isConcreteTypeNode(typeNode) && + typeNode.properties.language === language + ) { + concrete.add(typeId); + } + const children = orderedDirectSubtypes.get(typeId) ?? []; + queue.push(...children); + } + concreteSubtypesByRoot.set(cacheKey, concrete); + return concrete; + }; + // A declaration returning a concrete class is assignable to every class or // interface that type extends/implements. Expand once per language+type and // register the declaration under those ancestor names. This keeps named @@ -300,11 +339,15 @@ export const diPhase: PipelinePhase = { continue; } - const structural = new Set(); - if (typeof classEntry === 'string') structural.add(classEntry); - if (typeof interfaceEntry === 'string') { - for (const id of interfaceToImplementers.get(interfaceEntry) ?? []) structural.add(id); - } + const rootTypeId = + typeof classEntry === 'string' + ? classEntry + : typeof interfaceEntry === 'string' + ? interfaceEntry + : undefined; + const structural = new Set( + rootTypeId === undefined ? [] : concreteSubtypes(rootTypeId, candidate.language), + ); for (const id of providedTypes.get(candidate.language)?.get(candidate.targetTypeName) ?? []) { structural.add(id); } diff --git a/gitnexus/src/storage/parse-cache.ts b/gitnexus/src/storage/parse-cache.ts index efd17a7f6..b44beb456 100644 --- a/gitnexus/src/storage/parse-cache.ts +++ b/gitnexus/src/storage/parse-cache.ts @@ -668,7 +668,11 @@ import { copyV8CacheIfPresent, tryLoadV8Cache, writeV8CacheFile } from './v8-sid // each (no JSON/path/generation siblings). A v80 index still names `.json` // keys and would skip workers while scope-resolution found nothing — the // #1983 main-thread reparse. origin/main at allocation is 80. -const SCHEMA_BUMP = 81; +// 81 -> 82: Java and Kotlin ParsedFile capture side channels now carry +// programmatic Spring lookup facts. A warm v81 cache has no such facts, so it +// would skip workers and silently omit the new INJECTS edges. origin/main at +// allocation is 81. +const SCHEMA_BUMP = 82; const GITNEXUS_PKG_VERSION = (() => { try { // package.json sits at gitnexus/package.json — two levels up from diff --git a/gitnexus/test/integration/spring-dynamic-lookup-benchmark.test.ts b/gitnexus/test/integration/spring-dynamic-lookup-benchmark.test.ts new file mode 100644 index 000000000..4e9e80e41 --- /dev/null +++ b/gitnexus/test/integration/spring-dynamic-lookup-benchmark.test.ts @@ -0,0 +1,275 @@ +/** + * Spring programmatic lookup scaling for Java and Kotlin. + * + * Always-on tripwires catch a quadratic re-walk of every invocation in a dense + * file. Gated suites measure capture and full-pipeline INJECTS resolution: + * + * GITNEXUS_BENCH=1 npx vitest run test/integration/spring-dynamic-lookup-benchmark.test.ts + */ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { describe, expect, it } from 'vitest'; +import { emitJavaScopeCaptures } from '../../src/core/ingestion/languages/java/captures.js'; +import { collectJavaCaptureSideChannel } from '../../src/core/ingestion/languages/java/capture-side-channel.js'; +import { emitKotlinScopeCaptures } from '../../src/core/ingestion/languages/kotlin/captures.js'; +import { collectKotlinCaptureSideChannel } from '../../src/core/ingestion/languages/kotlin/capture-side-channel.js'; +import { runPipelineFromRepo } from '../../src/core/ingestion/pipeline.js'; + +const BENCH_ENABLED = process.env.GITNEXUS_BENCH === '1'; +const LOOKUPS_PER_CONSUMER = 2; +// Time growth divided by input growth: linear work stays near 1. +const LINEAR_SCALING_TOLERANCE = 1.5; + +interface CaptureBenchResult { + consumers: number; + elapsedMs: number; + captureCount: number; + factCount: number; +} + +function denseJavaLookupSource(consumerCount: number): string { + const consumers = Array.from({ length: consumerCount }, (_, index) => { + return ` +class Consumer${index} { + void collect${index}() { ctx.getBeans(Parent.class); } + void single${index}() { applicationContext.getBean(Parent.class); } + void decoy${index}() { other.getName(); } + void noise${index}() { + // ctx.getBeans(Parent.class); + String example = "ctx.getBean(Parent.class)"; + } +} +`; + }).join('\n'); + + return `package com.example; +interface Parent {} +class Impl implements Parent {} +${consumers} +`; +} + +function denseKotlinLookupSource(consumerCount: number): string { + const consumers = Array.from({ length: consumerCount }, (_, index) => { + return ` +class Consumer${index} { + fun collect${index}() { ctx.getBeans(Parent::class.java) } + fun single${index}() { applicationContext.getBean(Parent::class) } + fun decoy${index}() { other.getName() } + fun noise${index}() { + // ctx.getBeans(Parent::class.java) + val example = "ctx.getBean(Parent::class.java)" + } +} +`; + }).join('\n'); + + return `package com.example +interface Parent +class Impl : Parent +${consumers} +`; +} + +function runJavaCaptureBenchmark(consumerCount: number, run: number): CaptureBenchResult { + const filePath = `src/SpringDynamicLookupBench${consumerCount}_${run}.java`; + const start = performance.now(); + const captures = emitJavaScopeCaptures(denseJavaLookupSource(consumerCount), filePath); + const elapsedMs = performance.now() - start; + const facts = collectJavaCaptureSideChannel(filePath)?.springDynamicLookupFacts ?? []; + return { + consumers: consumerCount, + elapsedMs, + captureCount: captures.length, + factCount: facts.length, + }; +} + +function runKotlinCaptureBenchmark(consumerCount: number, run: number): CaptureBenchResult { + const filePath = `src/SpringDynamicLookupBench${consumerCount}_${run}.kt`; + const start = performance.now(); + const captures = emitKotlinScopeCaptures(denseKotlinLookupSource(consumerCount), filePath); + const elapsedMs = performance.now() - start; + const facts = collectKotlinCaptureSideChannel(filePath)?.springDynamicLookupFacts ?? []; + return { + consumers: consumerCount, + elapsedMs, + captureCount: captures.length, + factCount: facts.length, + }; +} + +function assertCaptureScaling(results: readonly CaptureBenchResult[]): void { + const first = results[0]; + const last = results[results.length - 1]; + expect(last.factCount).toBe(last.consumers * LOOKUPS_PER_CONSUMER); + const sizeRatio = last.consumers / first.consumers; + if (first.elapsedMs >= 20) { + const normalizedGrowth = last.elapsedMs / first.elapsedMs / sizeRatio; + expect(normalizedGrowth).toBeLessThan(LINEAR_SCALING_TOLERANCE); + } else { + expect(last.elapsedMs).toBeLessThan(10_000); + } +} + +describe('Spring dynamic lookup capture O(n²) regression tripwire', () => { + it('captures a dense 400-consumer Java file within a coarse linear-time budget', () => { + const consumers = 400; + runJavaCaptureBenchmark(4, 0); + const result = runJavaCaptureBenchmark(consumers, 1); + expect(result.factCount).toBe(consumers * LOOKUPS_PER_CONSUMER); + expect(result.captureCount).toBeGreaterThan(consumers * 8); + expect(result.elapsedMs).toBeLessThan(10_000); + }, 30_000); + + it('captures a dense 400-consumer Kotlin file within a coarse linear-time budget', () => { + const consumers = 400; + runKotlinCaptureBenchmark(4, 0); + const result = runKotlinCaptureBenchmark(consumers, 1); + expect(result.factCount).toBe(consumers * LOOKUPS_PER_CONSUMER); + expect(result.captureCount).toBeGreaterThan(consumers * 8); + expect(result.elapsedMs).toBeLessThan(10_000); + }, 30_000); +}); + +describe.skipIf(!BENCH_ENABLED)('Java Spring dynamic lookup capture scaling', () => { + it('scales sub-quadratically as Java lookup sites grow', () => { + const scales = [100, 200, 400]; + const repetitions = 4; + const results: CaptureBenchResult[] = []; + runJavaCaptureBenchmark(8, 0); + for (const consumers of scales) { + let elapsedMs = 0; + let captureCount = 0; + let factCount = 0; + for (let run = 0; run < repetitions; run++) { + const current = runJavaCaptureBenchmark(consumers, run + 1); + elapsedMs += current.elapsedMs; + captureCount = current.captureCount; + factCount = current.factCount; + } + results.push({ consumers, elapsedMs, captureCount, factCount }); + process.stdout.write( + ` java capture n=${consumers} ×${repetitions}: ${elapsedMs.toFixed(1)}ms ` + + `(${factCount} facts, ${captureCount} captures/run)\n`, + ); + } + assertCaptureScaling(results); + }, 120_000); +}); + +describe.skipIf(!BENCH_ENABLED)('Kotlin Spring dynamic lookup capture scaling', () => { + it('scales sub-quadratically as Kotlin lookup sites grow', () => { + const scales = [100, 200, 400]; + const repetitions = 4; + const results: CaptureBenchResult[] = []; + runKotlinCaptureBenchmark(8, 0); + for (const consumers of scales) { + let elapsedMs = 0; + let captureCount = 0; + let factCount = 0; + for (let run = 0; run < repetitions; run++) { + const current = runKotlinCaptureBenchmark(consumers, run + 1); + elapsedMs += current.elapsedMs; + captureCount = current.captureCount; + factCount = current.factCount; + } + results.push({ consumers, elapsedMs, captureCount, factCount }); + process.stdout.write( + ` kotlin capture n=${consumers} ×${repetitions}: ${elapsedMs.toFixed(1)}ms ` + + `(${factCount} facts, ${captureCount} captures/run)\n`, + ); + } + assertCaptureScaling(results); + }, 120_000); +}); + +function writeJavaLookupRepo(consumerCount: number): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), `spring-dynamic-java-${consumerCount}-`)); + fs.writeFileSync( + path.join(dir, 'Parent.java'), + `package a; +public interface Parent {} +`, + ); + fs.writeFileSync( + path.join(dir, 'Impl.java'), + `package a; +public class Impl implements Parent {} +`, + ); + for (let index = 0; index < consumerCount; index++) { + fs.writeFileSync( + path.join(dir, `Consumer${index}.java`), + `package c; +import a.Parent; +class Consumer${index} { + void lookup() { ctx.getBeans(Parent.class); } +} +`, + ); + } + return dir; +} + +function writeKotlinLookupRepo(consumerCount: number): string { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), `spring-dynamic-kotlin-${consumerCount}-`)); + fs.writeFileSync(path.join(dir, 'Parent.kt'), 'package a\ninterface Parent\n'); + fs.writeFileSync(path.join(dir, 'Impl.kt'), 'package a\nclass Impl : Parent\n'); + for (let index = 0; index < consumerCount; index++) { + fs.writeFileSync( + path.join(dir, `Consumer${index}.kt`), + `package c +import a.Parent +class Consumer${index} { + fun lookup() { ctx.getBeans(Parent::class.java) } +} +`, + ); + } + return dir; +} + +async function runPipelineBenchmark( + label: string, + writeRepo: (consumerCount: number) => string, +): Promise { + const scales = [25, 50, 100]; + const results: Array<{ consumers: number; elapsedMs: number; injects: number }> = []; + + for (const consumers of scales) { + const dir = writeRepo(consumers); + try { + const start = performance.now(); + const result = await runPipelineFromRepo(dir, () => {}, {}); + const elapsedMs = performance.now() - start; + const injects = [...result.graph.iterRelationshipsByType('INJECTS')].length; + results.push({ consumers, elapsedMs, injects }); + process.stdout.write( + ` ${label} pipeline n=${consumers}: ${elapsedMs.toFixed(1)}ms (${injects} INJECTS edges)\n`, + ); + } finally { + fs.rmSync(dir, { recursive: true, force: true }); + } + } + + for (const result of results) expect(result.injects).toBe(result.consumers); + const first = results[0]; + const last = results[results.length - 1]; + const sizeRatio = last.consumers / first.consumers; + const normalizedGrowth = last.elapsedMs / first.elapsedMs / sizeRatio; + expect(normalizedGrowth).toBeLessThan(LINEAR_SCALING_TOLERANCE); +} + +describe.skipIf(!BENCH_ENABLED)('Java Spring dynamic lookup end-to-end scaling', () => { + it('keeps Java pipeline lookup resolution sub-quadratic across file counts', async () => { + await runPipelineBenchmark('java', writeJavaLookupRepo); + }, 300_000); +}); + +describe.skipIf(!BENCH_ENABLED)('Kotlin Spring dynamic lookup end-to-end scaling', () => { + it('keeps Kotlin pipeline lookup resolution sub-quadratic across file counts', async () => { + await runPipelineBenchmark('kotlin', writeKotlinLookupRepo); + }, 300_000); +}); diff --git a/gitnexus/test/integration/spring-dynamic-lookup.test.ts b/gitnexus/test/integration/spring-dynamic-lookup.test.ts new file mode 100644 index 000000000..1471901af --- /dev/null +++ b/gitnexus/test/integration/spring-dynamic-lookup.test.ts @@ -0,0 +1,220 @@ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { afterEach, describe, expect, it } from 'vitest'; +import { + getRelationships, + runPipelineFromRepo, + writeFixtureRepo, + type PipelineResult, +} from './resolvers/helpers.js'; + +const temporaryRepositories: string[] = []; + +function temporaryRepository(prefix: string): string { + const root = fs.mkdtempSync(path.join(os.tmpdir(), prefix)); + temporaryRepositories.push(root); + return root; +} + +function injectionEdges(result: PipelineResult) { + return getRelationships(result, 'INJECTS').map((edge) => ({ + source: edge.source, + target: edge.target, + type: edge.rel.type, + confidence: edge.rel.confidence, + reason: edge.rel.reason, + })); +} + +afterEach(() => { + for (const root of temporaryRepositories.splice(0)) { + fs.rmSync(root, { recursive: true, force: true }); + } +}); + +describe('Spring dynamic lookup production integration', () => { + it('resolves Java imports, assignability, concrete targets, ranges, and nested callables', async () => { + const root = temporaryRepository('gitnexus-java-spring-dynamic-'); + writeFixtureRepo(root, { + 'src/a/Parent.java': `package a; + interface Parent {} + interface Child extends Parent {} + class Impl implements Child {} + class SecondImpl implements Child {}`, + 'src/a/Base.java': 'package a; public class Base {}', + 'src/a/Concrete.java': 'package a; public class Concrete extends Base {}', + 'src/a/RecordBean.java': 'package a; public record RecordBean(String value) {}', + 'src/b/Parent.java': 'package b; public interface Parent {}', + 'src/b/OtherImpl.java': 'package b; public class OtherImpl implements Parent {}', + 'src/a/SamePackageCaller.java': `package a; + class SamePackageCaller { + void samePackageLookup() { ctx.getBeans(Parent.class); } + }`, + 'src/c/JavaCaller.java': `package c; + import a.Parent; + import a.Base; + import a.Concrete; + import a.RecordBean; + class JavaCaller { + JavaCaller() { beanFactory.getBean(Concrete.class); } + void javaCollection(){ SpringContextUtil.getBeans(Parent.class); } + void adjacentMiss(){} + void javaBase(){ applicationContext.getBeansOfType(Base.class); } + void javaSingle(){ ctx.getBean(Concrete.class); } + void javaRecord(){ ctx.getBean(RecordBean.class); } + void javaAmbiguousSingle(){ ctx.getBean(Parent.class); } + void javaOuter() { + Runnable task = new Runnable() { + public void run() { ctx.getBeans(Parent.class); } + }; + } + void javaFalsePositives() { + // ctx.getBeans(Parent.class); + /* applicationContext.getBeansOfType(Parent.class); */ + String normal = "ctx.getBean(Concrete.class)"; + String block = """ + ctx.getBeans(Parent.class) + """; + } + }`, + 'src/c/AmbiguousCaller.java': `package c; + import a.*; + import b.*; + class AmbiguousCaller { + void ambiguousLookup() { ctx.getBeans(Parent.class); } + }`, + 'src/foreign.ts': 'interface Parent {}', + }); + + const result = await runPipelineFromRepo(root, () => {}); + const edges = injectionEdges(result); + + expect(edges).toEqual( + expect.arrayContaining([ + { + source: 'javaCollection', + target: 'Impl', + type: 'INJECTS', + confidence: 0.8, + reason: 'Spring dynamic lookup: SpringContextUtil.getBeans(Parent)', + }, + { + source: 'javaCollection', + target: 'SecondImpl', + type: 'INJECTS', + confidence: 0.8, + reason: 'Spring dynamic lookup: SpringContextUtil.getBeans(Parent)', + }, + expect.objectContaining({ source: 'javaBase', target: 'Concrete', confidence: 0.8 }), + { + source: 'javaSingle', + target: 'Concrete', + type: 'INJECTS', + confidence: 0.9, + reason: 'Spring dynamic lookup: ctx.getBean(Concrete)', + }, + expect.objectContaining({ + source: 'javaRecord', + target: 'RecordBean', + confidence: 0.9, + }), + expect.objectContaining({ + source: 'javaAmbiguousSingle', + target: 'Impl', + confidence: 0.5, + }), + expect.objectContaining({ + source: 'javaAmbiguousSingle', + target: 'SecondImpl', + confidence: 0.5, + }), + expect.objectContaining({ source: 'run', target: 'Impl', confidence: 0.8 }), + expect.objectContaining({ source: 'samePackageLookup', target: 'Impl', confidence: 0.8 }), + ]), + ); + expect(edges.some((edge) => edge.source === 'adjacentMiss')).toBe(false); + expect(edges.some((edge) => edge.source === 'javaOuter')).toBe(false); + expect(edges.some((edge) => edge.source === 'javaFalsePositives')).toBe(false); + expect(edges.some((edge) => edge.source === 'ambiguousLookup')).toBe(false); + expect( + edges + .filter((edge) => edge.source === 'javaAmbiguousSingle') + .every((edge) => edge.reason.includes('ambiguous candidates: Impl, SecondImpl')), + ).toBe(true); + expect( + edges.some( + (edge) => + edge.source === 'javaCollection' && + (edge.target === 'Child' || edge.target === 'OtherImpl'), + ), + ).toBe(false); + expect(edges.some((edge) => edge.target === 'Concrete' && edge.confidence === 0.5)).toBe(false); + expect( + edges.filter((edge) => edge.source === 'JavaCaller' && edge.target === 'Concrete'), + ).toHaveLength(1); + }, 60000); + + it('resolves Kotlin class literals through the same graph semantics', async () => { + const root = temporaryRepository('gitnexus-kotlin-spring-dynamic-'); + writeFixtureRepo(root, { + 'src/a/Parent.kt': 'package a\ninterface Parent', + 'src/a/Child.kt': 'package a\ninterface Child : Parent', + 'src/a/Impl.kt': 'package a\nclass Impl : Child', + 'src/a/Concrete.kt': 'package a\nopen class Base\nclass Concrete : Base()', + 'src/c/KotlinCaller.kt': `package c + import a.Parent + import a.Base + import a.Concrete + class KotlinCaller { + constructor(marker: String) { beanFactory.getBean(Concrete::class.java) } + fun kotlinCollection(){ SpringContextUtil.getBeans(Parent::class.java) } + fun adjacentMiss(){} + fun kotlinBase(){ applicationContext.getBeansOfType(Base::class.java) } + fun kotlinSingle(){ ctx.getBean(Concrete::class) } + fun kotlinOuter() { + val task = object : Runnable { + override fun run() { ctx.getBeans(Parent::class.java) } + } + } + fun kotlinFalsePositives() { + // ctx.getBeans(Parent::class.java) + /* applicationContext.getBeansOfType(Parent::class.java) */ + val normal = "ctx.getBean(Concrete::class.java)" + val raw = ${'"""'}ctx.getBeans(Parent::class.java)${'"""'} + } + }`, + 'src/foreign.ts': 'interface Parent {}', + }); + + const result = await runPipelineFromRepo(root, () => {}); + const edges = injectionEdges(result); + + expect(edges).toEqual( + expect.arrayContaining([ + { + source: 'kotlinCollection', + target: 'Impl', + type: 'INJECTS', + confidence: 0.8, + reason: 'Spring dynamic lookup: SpringContextUtil.getBeans(Parent)', + }, + expect.objectContaining({ source: 'kotlinBase', target: 'Concrete', confidence: 0.8 }), + { + source: 'kotlinSingle', + target: 'Concrete', + type: 'INJECTS', + confidence: 0.9, + reason: 'Spring dynamic lookup: ctx.getBean(Concrete)', + }, + expect.objectContaining({ source: 'run', target: 'Impl', confidence: 0.8 }), + ]), + ); + expect(edges.some((edge) => edge.source === 'adjacentMiss')).toBe(false); + expect(edges.some((edge) => edge.source === 'kotlinOuter')).toBe(false); + expect(edges.some((edge) => edge.source === 'kotlinFalsePositives')).toBe(false); + expect( + edges.filter((edge) => edge.source === 'constructor' && edge.target === 'Concrete'), + ).toHaveLength(1); + }, 60000); +}); diff --git a/gitnexus/test/unit/incremental-parse-cache.test.ts b/gitnexus/test/unit/incremental-parse-cache.test.ts index 69a366c10..cc6c7167b 100644 --- a/gitnexus/test/unit/incremental-parse-cache.test.ts +++ b/gitnexus/test/unit/incremental-parse-cache.test.ts @@ -244,11 +244,11 @@ describe('PARSE_CACHE_VERSION', () => { // collided, because each re-checked once and neither re-checked after the // other moved — which is why the rule is re-applied AT MERGE, not when the // number is picked. - it('pins SCHEMA_BUMP to 81 so concurrent bumps cannot silently collide (#2766, #3015, #3088)', () => { - expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(81); + it('pins SCHEMA_BUMP to 82 so concurrent bumps cannot silently collide (#2766, #3015, #3088)', () => { + expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(82); expect(PARSE_CACHE_BUCKET_COUNT).toBe(128); for (const taken of [ - 59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, + 59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, 81, ]) { expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).not.toBe(taken); } diff --git a/gitnexus/test/unit/ingestion/di.test.ts b/gitnexus/test/unit/ingestion/di.test.ts index 2faa03341..28ae20b97 100644 --- a/gitnexus/test/unit/ingestion/di.test.ts +++ b/gitnexus/test/unit/ingestion/di.test.ts @@ -98,8 +98,9 @@ function addImplements( ifaceName: string, ifaceLanguage = 'java', ifaceQualifiedName?: string, + sourceLabel: NodeLabel = 'Class', ): void { - const classId = generateId('Class', className); + const classId = generateId(sourceLabel, className); const ifaceId = generateId('Interface', `${ifaceLanguage}:${ifaceQualifiedName ?? ifaceName}`); graph.addRelationship({ id: generateId('IMPLEMENTS', `${classId}->${ifaceId}`), @@ -916,6 +917,137 @@ describe('di phase', () => { expect(injectsEdges(graph)).toHaveLength(0); expect(output).toMatchObject({ injectsEdges: 0, ambiguousSkipped: 1 }); }); + + it('walks interface assignability transitively, ignores intermediate interfaces, and terminates cycles', async () => { + const graph = createKnowledgeGraph(); + const parentId = addInterface(graph, 'Parent'); + const childId = addInterface(graph, 'Child'); + const implId = addClass(graph, 'Impl', 'java'); + addImplements(graph, 'Impl', 'Child'); + graph.addRelationship({ + id: generateId('IMPLEMENTS', `${childId}->${parentId}`), + sourceId: childId, + targetId: parentId, + type: 'IMPLEMENTS', + confidence: 1, + reason: '', + }); + graph.addRelationship({ + id: generateId('IMPLEMENTS', `${parentId}->${childId}`), + sourceId: parentId, + targetId: childId, + type: 'IMPLEMENTS', + confidence: 1, + reason: 'malformed-cycle regression guard', + }); + const consumerId = addClass(graph, 'Consumer', 'java', 'Class', { + [SPRING_DI_INJECTION_SITES_PROPERTY]: [ + { + targetTypeName: 'Parent', + cardinality: 'collection', + reason: 'Spring dynamic lookup: ctx.getBeans(Parent)', + }, + ], + }); + + await diPhase.execute(makeCtx(graph), new Map()); + + expect(injectsEdges(graph)).toEqual([ + expect.objectContaining({ + sourceId: consumerId, + targetId: implId, + type: 'INJECTS', + confidence: 0.8, + }), + ]); + }); + + it('keeps records and enums as concrete interface implementers', async () => { + const graph = createKnowledgeGraph(); + addInterface(graph, 'Parent'); + const recordId = addClass(graph, 'RecordImpl', 'java', 'Record'); + const enumId = addClass(graph, 'EnumImpl', 'java', 'Enum'); + addImplements(graph, 'RecordImpl', 'Parent', 'java', undefined, 'Record'); + addImplements(graph, 'EnumImpl', 'Parent', 'java', undefined, 'Enum'); + const consumerId = addClass(graph, 'Consumer', 'java', 'Class', { + [SPRING_DI_INJECTION_SITES_PROPERTY]: [ + { + targetTypeName: 'Parent', + cardinality: 'collection', + reason: 'Spring dynamic lookup: ctx.getBeans(Parent)', + }, + ], + }); + + await diPhase.execute(makeCtx(graph), new Map()); + + expect(injectsEdges(graph)).toEqual( + expect.arrayContaining([ + expect.objectContaining({ sourceId: consumerId, targetId: recordId }), + expect.objectContaining({ sourceId: consumerId, targetId: enumId }), + ]), + ); + expect(injectsEdges(graph)).toHaveLength(2); + }); + + it('walks class inheritance and prefers concrete Spring bean candidates', async () => { + const graph = createKnowledgeGraph(); + const baseId = addClass(graph, 'Base', 'java'); + const concreteId = addClass(graph, 'Concrete', 'java', 'Class', { + [SPRING_DI_PROVIDER_PROPERTY]: { names: ['concrete'] }, + }); + graph.addRelationship({ + id: generateId('EXTENDS', `${concreteId}->${baseId}`), + sourceId: concreteId, + targetId: baseId, + type: 'EXTENDS', + confidence: 1, + reason: '', + }); + const consumerId = addClass(graph, 'Consumer', 'java', 'Class', { + [SPRING_DI_INJECTION_SITES_PROPERTY]: [ + { + targetTypeName: 'Base', + cardinality: 'collection', + reason: 'Spring dynamic lookup: applicationContext.getBeansOfType(Base)', + }, + ], + }); + + await diPhase.execute(makeCtx(graph), new Map()); + + expect(injectsEdges(graph)).toEqual([ + expect.objectContaining({ + sourceId: consumerId, + targetId: concreteId, + confidence: 0.8, + }), + ]); + }); + + it('resolves a directly requested concrete class', async () => { + const graph = createKnowledgeGraph(); + const concreteId = addClass(graph, 'Concrete', 'java'); + const consumerId = addClass(graph, 'Consumer', 'java', 'Class', { + [SPRING_DI_INJECTION_SITES_PROPERTY]: [ + { + targetTypeName: 'Concrete', + cardinality: 'single', + reason: 'Spring dynamic lookup: ctx.getBean(Concrete)', + }, + ], + }); + + await diPhase.execute(makeCtx(graph), new Map()); + + expect(injectsEdges(graph)).toEqual([ + expect.objectContaining({ + sourceId: consumerId, + targetId: concreteId, + confidence: 0.9, + }), + ]); + }); }); // --------------------------------------------------------------------------- diff --git a/gitnexus/test/unit/spring-dynamic-lookup.test.ts b/gitnexus/test/unit/spring-dynamic-lookup.test.ts new file mode 100644 index 000000000..438d78533 --- /dev/null +++ b/gitnexus/test/unit/spring-dynamic-lookup.test.ts @@ -0,0 +1,179 @@ +import { describe, expect, it } from 'vitest'; +import { getJavaParser } from '../../src/core/ingestion/languages/java/query.js'; +import { captureJavaSpringDynamicLookupFacts } from '../../src/core/ingestion/languages/java/spring-dynamic-lookup.js'; +import { getKotlinParser } from '../../src/core/ingestion/languages/kotlin/query.js'; +import { captureKotlinSpringDynamicLookupFacts } from '../../src/core/ingestion/languages/kotlin/spring-dynamic-lookup.js'; + +function javaFacts(source: string) { + return captureJavaSpringDynamicLookupFacts( + getJavaParser().parse(source).rootNode, + 'src/Example.java', + ); +} + +function kotlinFacts(source: string) { + return captureKotlinSpringDynamicLookupFacts( + getKotlinParser().parse(source).rootNode, + 'src/Example.kt', + ); +} + +describe('Java Spring dynamic lookup capture', () => { + it('captures collection and singular class-literal calls from known receivers', () => { + const facts = javaFacts(`class Example { + void load() { + SpringContextUtil.getBeans(Port.class); + ApplicationContext.getBeansOfType(com.example.Service.class); + ctx.getBean(Concrete.class); + } + }`); + + expect( + facts.map(({ receiverName, methodName, targetTypeName }) => ({ + receiverName, + methodName, + targetTypeName, + })), + ).toEqual([ + { + receiverName: 'SpringContextUtil', + methodName: 'getBeans', + targetTypeName: 'Port', + }, + { + receiverName: 'ApplicationContext', + methodName: 'getBeansOfType', + targetTypeName: 'com.example.Service', + }, + { receiverName: 'ctx', methodName: 'getBean', targetTypeName: 'Concrete' }, + ]); + }); + + it('assigns adjacent one-line calls only to their actual callable', () => { + const facts = javaFacts(`class Example { + void hit(){ ctx.getBeans(Port.class); } + void miss(){} + }`); + + expect(facts).toHaveLength(1); + expect(facts[0]?.ownerRange.startLine).toBe(2); + expect(facts[0]?.ownerRange.endLine).toBe(2); + }); + + it('assigns an anonymous-class lookup only to the nested method', () => { + const facts = javaFacts(`class Example { + void outer() { + Runnable task = new Runnable() { + public void run() { ctx.getBeans(Port.class); } + }; + } + }`); + + expect(facts).toHaveLength(1); + expect(facts[0]?.ownerRange.startLine).toBe(4); + expect(facts[0]?.ownerRange.endLine).toBe(4); + }); + + it('captures constructor lookups using normal callable semantics', () => { + const facts = javaFacts(`class Example { + Example() { beanFactory.getBean(Concrete.class); } + }`); + + expect(facts).toHaveLength(1); + expect(facts[0]?.methodName).toBe('getBean'); + }); + + it('ignores comments, Javadoc, strings, text blocks, and unsupported calls', () => { + const facts = javaFacts(`class Example { + /** + * ctx.getBeans(Port.class) + */ + void load() { + // ctx.getBeans(Port.class); + /* applicationContext.getBeansOfType(Port.class); */ + String normal = "ctx.getBean(Port.class)"; + String block = """ + ctx.getBeans(Port.class) + """; + unrelated.getBeans(Port.class); + ctx.getBean("namedBean"); + } + }`); + + expect(facts).toEqual([]); + }); +}); + +describe('Kotlin Spring dynamic lookup capture', () => { + it('captures Kotlin class literals with and without the Java bridge', () => { + const facts = kotlinFacts(`class Example { + fun load() { + SpringContextUtil.getBeans(Port::class.java) + applicationContext.getBeansOfType(com.example.Service::class.java) + ctx.getBean(Concrete::class) + } + }`); + + expect( + facts.map(({ receiverName, methodName, targetTypeName }) => ({ + receiverName, + methodName, + targetTypeName, + })), + ).toEqual([ + { + receiverName: 'SpringContextUtil', + methodName: 'getBeans', + targetTypeName: 'Port', + }, + { + receiverName: 'applicationContext', + methodName: 'getBeansOfType', + targetTypeName: 'com.example.Service', + }, + { receiverName: 'ctx', methodName: 'getBean', targetTypeName: 'Concrete' }, + ]); + }); + + it('assigns an object-expression lookup only to the nested method', () => { + const facts = kotlinFacts(`class Example { + fun outer() { + val task = object : Runnable { + override fun run() { ctx.getBeans(Port::class.java) } + } + } + }`); + + expect(facts).toHaveLength(1); + expect(facts[0]?.ownerRange.startLine).toBe(4); + expect(facts[0]?.ownerRange.endLine).toBe(4); + }); + + it('captures secondary-constructor lookups and excludes init blocks without graph callables', () => { + const facts = kotlinFacts(`class Example { + init { ctx.getBeans(Ignored::class.java) } + constructor(marker: String) { ctx.getBean(Concrete::class.java) } + }`); + + expect(facts).toHaveLength(1); + expect(facts[0]?.targetTypeName).toBe('Concrete'); + }); + + it('ignores comments, KDoc, strings, raw strings, and unsupported calls', () => { + const facts = kotlinFacts(`class Example { + /** + * ctx.getBeans(Port::class.java) + */ + fun load() { + // ctx.getBeans(Port::class.java) + /* applicationContext.getBeansOfType(Port::class.java) */ + val normal = "ctx.getBean(Port::class.java)" + val raw = ${'"""'}ctx.getBeans(Port::class.java)${'"""'} + unrelated.getBeans(Port::class.java) + ctx.getBean("namedBean") + } + }`); + + expect(facts).toEqual([]); + }); +}); From e04e1ecc6581f9503b0f3691c0f01a7a48398349 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Mon, 31 Aug 2026 16:49:43 +0100 Subject: [PATCH 06/26] fix(group): parse Maven child coordinates independently of parent POMs (#3108) * fix(group): parse Maven child coordinates independently of parent POMs Stop treating inherited parent groupId/artifactId as the child's identity so sibling repos no longer collide and workspace manifest links can resolve. Co-authored-by: Cursor * fix(group): parse Maven POMs with fast-xml-parser Replace the hand-rolled tokenizer so child identity and CDATA/namespaces stay accurate, and collect only project.dependencies so BOM, profile, and plugin entries cannot create workspace links. Co-authored-by: Cursor * fix(group): parse Gradle identity, catalogs, and named coordinates Read gradle.properties, settings.gradle, and the default libs.versions.toml catalog so workspace links work without executing Gradle, matching the static POM contract. Co-authored-by: Cursor * fix(group): resolve Kotlin Gradle DSL workspace coordinates Honor Kotlin named arguments, catalog get()/asProvider(), type-safe projects.* accessors, and ksp/kapt/commonMain configs without executing Gradle. Co-authored-by: Cursor * Address PR review feedback (#3108) Recognize Gradle group inside allprojects { } and Groovy name-first map coordinates so workspace identity and deps match common DSL forms. Co-authored-by: Cursor * chore(autofix): apply prettier + eslint fixes via /autofix command * Address PR review feedback (#3108) Restore XMLParser.parse for POMs after /autofix swapped in tree-sitter parseSourceSafe, and match underscore catalog aliases from Gradle files. Co-authored-by: Cursor --------- Co-authored-by: Gergo Magyar Co-authored-by: Cursor Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gitnexus/package-lock.json | 121 ++++ gitnexus/package.json | 1 + .../extractors/java-workspace-extractor.ts | 314 ++++++++- .../group/java-workspace-extractor.test.ts | 620 +++++++++++++++++- gitnexus/test/unit/group/sync.test.ts | 80 +++ 5 files changed, 1099 insertions(+), 37 deletions(-) diff --git a/gitnexus/package-lock.json b/gitnexus/package-lock.json index bd3ebd088..0907115e4 100644 --- a/gitnexus/package-lock.json +++ b/gitnexus/package-lock.json @@ -20,6 +20,7 @@ "cors": "^2.8.5", "express": "^5.2.1", "express-rate-limit": "^8.4.1", + "fast-xml-parser": "^5.11.1", "glob": "^13.0.6", "graphology": "^0.26.0", "graphology-indices": "^0.17.0", @@ -1397,6 +1398,18 @@ } } }, + "node_modules/@nodable/entities": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/@nodable/entities/-/entities-3.0.0.tgz", + "integrity": "sha512-8L9xFeTYKhm49xfIypoe2W5wV1m/3Z58kT+7kR9A8OyFxcPduI4VmxaUMQyKYrRjUoLLSXv6EKKID5Tvj9cUVw==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/nodable" + } + ], + "license": "MIT" + }, "node_modules/@oxc-project/types": { "version": "0.144.0", "resolved": "https://registry.npmjs.org/@oxc-project/types/-/types-0.144.0.tgz", @@ -2153,6 +2166,18 @@ "url": "https://github.com/chalk/ansi-styles?sponsor=1" } }, + "node_modules/anynum": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/anynum/-/anynum-1.0.1.tgz", + "integrity": "sha512-N6//FLET/tXYNM/F6ABca1oH6fWB+KlTt909Le28WMDBk8oaT4vY17DCrwg2MvmuqUKt3Ni4N5dGJ/EoBgcO6A==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT" + }, "node_modules/apache-arrow": { "version": "21.1.0", "resolved": "https://registry.npmjs.org/apache-arrow/-/apache-arrow-21.1.0.tgz", @@ -3021,6 +3046,45 @@ ], "license": "BSD-3-Clause" }, + "node_modules/fast-xml-builder": { + "version": "1.3.1", + "resolved": "https://registry.npmjs.org/fast-xml-builder/-/fast-xml-builder-1.3.1.tgz", + "integrity": "sha512-pIM/1n3ntFXKYrUZwW7QCK0gAW7XY+wzj1YMIV3tLDvPj/V+zTGJK5e3/4WJfwj0qWw2ElNXiTixda/R+3YSug==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "dependencies": { + "path-expression-matcher": "^1.6.2", + "xml-naming": "^0.3.0" + } + }, + "node_modules/fast-xml-parser": { + "version": "5.11.1", + "resolved": "https://registry.npmjs.org/fast-xml-parser/-/fast-xml-parser-5.11.1.tgz", + "integrity": "sha512-TBw6K/fxoQGGjCmZDw9w/ZwP3uDcnTM4YH/g+PFRWr8sbe5idXtxNN6vITh4+1ruCZaho6uBFurElsA7F0zzgw==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "dependencies": { + "@nodable/entities": "^3.0.0", + "fast-xml-builder": "^1.2.0", + "is-unsafe": "^2.0.0", + "path-expression-matcher": "^1.6.2", + "strnum": "^2.4.2", + "xml-naming": "^0.3.0" + }, + "bin": { + "fxparser": "src/cli/cli.js" + } + }, "node_modules/fdir": { "version": "6.5.0", "resolved": "https://registry.npmjs.org/fdir/-/fdir-6.5.0.tgz", @@ -3481,6 +3545,18 @@ "integrity": "sha512-hvpoI6korhJMnej285dSg6nu1+e6uxs7zG3BYAm5byqDsgJNWwxzM6z6iZiAgQR4TJ30JmBTOwqZUw3WlyH3AQ==", "license": "MIT" }, + "node_modules/is-unsafe": { + "version": "2.0.2", + "resolved": "https://registry.npmjs.org/is-unsafe/-/is-unsafe-2.0.2.tgz", + "integrity": "sha512-HgbIHPBH0KHHCcjLfGsCvhtPTVxjaAZlXjwdz7/GQC40SjSe4sfQsar8J5VFo8JOSbarkpV0OLG95bbaNd9aAQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT" + }, "node_modules/isexe": { "version": "4.0.0", "resolved": "https://registry.npmjs.org/isexe/-/isexe-4.0.0.tgz", @@ -4334,6 +4410,21 @@ "node": ">= 0.8" } }, + "node_modules/path-expression-matcher": { + "version": "1.6.2", + "resolved": "https://registry.npmjs.org/path-expression-matcher/-/path-expression-matcher-1.6.2.tgz", + "integrity": "sha512-enSlaiat05iasnzmgNxRj8reFdj3puY2QpNgP1aPIaVfT6nn9ICuPoFlKHk8EN22HcwewshO+mN2DGbkCEOtqQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "engines": { + "node": ">=14.0.0" + } + }, "node_modules/path-key": { "version": "3.1.1", "resolved": "https://registry.npmjs.org/path-key/-/path-key-3.1.1.tgz", @@ -5080,6 +5171,21 @@ "node": ">=0.10.0" } }, + "node_modules/strnum": { + "version": "2.4.2", + "resolved": "https://registry.npmjs.org/strnum/-/strnum-2.4.2.tgz", + "integrity": "sha512-rDG3Ah4TV0k1hWvLSzkZtMmLN9+eS+h3knq4MP6A42Y3Yh5qGNnOUs1jJkoSr8FG5dsL28c7KgkIBzSEykqtuw==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "dependencies": { + "anynum": "^1.0.1" + } + }, "node_modules/supports-color": { "version": "7.2.0", "resolved": "https://registry.npmjs.org/supports-color/-/supports-color-7.2.0.tgz", @@ -5765,6 +5871,21 @@ "integrity": "sha512-l4Sp/DRseor9wL6EvV2+TuQn63dMkPjZ/sp9XkghTEbV9KlPS1xUsZ3u7/IQO4wxtcFB4bgpQPRcR3QCvezPcQ==", "license": "ISC" }, + "node_modules/xml-naming": { + "version": "0.3.0", + "resolved": "https://registry.npmjs.org/xml-naming/-/xml-naming-0.3.0.tgz", + "integrity": "sha512-ghig2TBE/H11aOVgmahA3MhimvkBr6JIYknH/Dhdk10nXwdbIqBJsbfMxpvFPG8bAw77gN29aQWvKpmVoPlvPQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "engines": { + "node": ">=16.0.0" + } + }, "node_modules/y18n": { "version": "5.0.8", "resolved": "https://registry.npmjs.org/y18n/-/y18n-5.0.8.tgz", diff --git a/gitnexus/package.json b/gitnexus/package.json index cf511b17b..85a03befa 100644 --- a/gitnexus/package.json +++ b/gitnexus/package.json @@ -66,6 +66,7 @@ "cors": "^2.8.5", "express": "^5.2.1", "express-rate-limit": "^8.4.1", + "fast-xml-parser": "^5.11.1", "glob": "^13.0.6", "graphology": "^0.26.0", "graphology-indices": "^0.17.0", diff --git a/gitnexus/src/core/group/extractors/java-workspace-extractor.ts b/gitnexus/src/core/group/extractors/java-workspace-extractor.ts index b6beed71c..fcbb06507 100644 --- a/gitnexus/src/core/group/extractors/java-workspace-extractor.ts +++ b/gitnexus/src/core/group/extractors/java-workspace-extractor.ts @@ -1,5 +1,6 @@ import fs from 'node:fs/promises'; import path from 'node:path'; +import { XMLParser } from 'fast-xml-parser'; import type { CypherExecutor } from '../contract-extractor.js'; import type { GroupManifestLink, ContractRole } from '../types.js'; import { shouldIgnorePath, loadIgnoreRules } from '../../../config/ignore-service.js'; @@ -20,6 +21,21 @@ interface ImportedSymbol { filePath: string; } +type XmlNode = Record; + +// POMs are static metadata. Parse hierarchy with a real XML parser, but do not +// invoke Maven or resolve the effective model. Properties, profiles, and remote +// parent resolution remain outside this extractor's deterministic boundary. +const pomParser = new XMLParser({ + ignoreAttributes: true, + removeNSPrefix: true, + trimValues: true, + parseTagValue: false, + processEntities: false, + ignoreDeclaration: true, + ignorePiTags: true, +}); + async function parseJavaManifest( repoPath: string, ): Promise<{ groupId: string; artifactId: string; deps: string[] } | null> { @@ -28,14 +44,15 @@ async function parseJavaManifest( const content = await fs.readFile(pomPath, 'utf-8'); return parsePom(content); } catch { - // fall through to Gradle + // Missing pom.xml — fall through to Gradle. } + const gradleSidecars = await readGradleSidecars(repoPath); for (const name of ['build.gradle.kts', 'build.gradle']) { const gradlePath = path.join(repoPath, name); try { const content = await fs.readFile(gradlePath, 'utf-8'); - return parseGradle(content, repoPath); + return parseGradle(content, repoPath, gradleSidecars); } catch { continue; } @@ -44,59 +61,286 @@ async function parseJavaManifest( return null; } -function parsePom(content: string): { groupId: string; artifactId: string; deps: string[] } | null { - const projectGroupMatch = content.match(/]*>[\s\S]*?([^<]+)<\/groupId>/); - const projectArtifactMatch = content.match( - /]*>[\s\S]*?([^<]+)<\/artifactId>/, - ); - if (!projectGroupMatch || !projectArtifactMatch) return null; +interface GradleSidecars { + propertiesGroup?: string; + rootProjectName?: string; + catalogLibraries: Map; + catalogBundles: Map; +} - const groupId = projectGroupMatch[1].trim(); - const artifactId = projectArtifactMatch[1].trim(); +async function readIfPresent(filePath: string): Promise { + try { + return await fs.readFile(filePath, 'utf-8'); + } catch { + return undefined; + } +} - const deps: string[] = []; - const depBlocks = content.matchAll(/\s*([\s\S]*?)<\/dependency>/g); - for (const block of depBlocks) { - const gMatch = block[1].match(/([^<]+)<\/groupId>/); - const aMatch = block[1].match(/([^<]+)<\/artifactId>/); - if (gMatch && aMatch) { - deps.push(`${gMatch[1].trim()}:${aMatch[1].trim()}`); +async function readGradleSidecars(repoPath: string): Promise { + const [properties, settingsKts, settingsGroovy, catalog] = await Promise.all([ + readIfPresent(path.join(repoPath, 'gradle.properties')), + readIfPresent(path.join(repoPath, 'settings.gradle.kts')), + readIfPresent(path.join(repoPath, 'settings.gradle')), + readIfPresent(path.join(repoPath, 'gradle', 'libs.versions.toml')), + ]); + + const sidecars: GradleSidecars = { + catalogLibraries: new Map(), + catalogBundles: new Map(), + }; + + const groupMatch = properties?.match(/(?:^|\n)\s*group\s*=\s*([^\s#]+)/); + if (groupMatch) sidecars.propertiesGroup = groupMatch[1]; + + const settings = settingsKts ?? settingsGroovy; + const nameMatch = settings?.match(/rootProject\.name\s*=\s*['"]([^'"]+)['"]/); + if (nameMatch) sidecars.rootProjectName = nameMatch[1]; + + if (catalog) { + const parsed = parseGradleVersionCatalog(catalog); + sidecars.catalogLibraries = parsed.libraries; + sidecars.catalogBundles = parsed.bundles; + } + + return sidecars; +} + +function catalogAccessors(alias: string): string[] { + const dotted = alias.replace(/[-_]/g, '.'); + const camel = alias.replace(/[-_]+([A-Za-z0-9])/g, (_, char: string) => char.toUpperCase()); + return [...new Set([alias, dotted, camel])]; +} + +function projectAccessorToArtifactId(accessor: string): string { + const last = accessor.split('.').pop()!; + return last.replace(/[A-Z]/g, (char) => `-${char.toLowerCase()}`).replace(/^-/, ''); +} + +function moduleToGa(module: string): string | undefined { + const parts = module.split(':'); + return parts.length >= 2 ? `${parts[0]}:${parts[1]}` : undefined; +} + +function parseInlineTomlTable(rhs: string): Record { + const fields: Record = {}; + for (const match of rhs.matchAll(/([A-Za-z0-9_-]+)\s*=\s*['"]([^'"]+)['"]/g)) { + fields[match[1]] = match[2]; + } + return fields; +} + +/** Default Gradle catalog (`gradle/libs.versions.toml`) — aliases only, no version resolution. */ +function parseGradleVersionCatalog(toml: string): { + libraries: Map; + bundles: Map; +} { + const libraries = new Map(); + const bundles = new Map(); + let section: 'libraries' | 'bundles' | 'other' = 'other'; + + const addLibrary = (alias: string, ga: string) => { + for (const accessor of catalogAccessors(alias)) libraries.set(accessor, ga); + }; + + for (const raw of toml.split(/\r?\n/)) { + const line = raw.replace(/#.*$/, '').trim(); + if (!line) continue; + const header = line.match(/^\[([^\]]+)\]$/); + if (header) { + const name = header[1]; + section = + name === 'libraries' || name.endsWith('.libraries') + ? 'libraries' + : name === 'bundles' || name.endsWith('.bundles') + ? 'bundles' + : 'other'; + continue; + } + + if (section === 'libraries') { + const dottedModule = line.match(/^([A-Za-z0-9._-]+)\.module\s*=\s*['"]([^'"]+)['"]$/); + if (dottedModule) { + const ga = moduleToGa(dottedModule[2]); + if (ga) addLibrary(dottedModule[1], ga); + continue; + } + const assignment = line.match(/^([A-Za-z0-9._-]+)\s*=\s*(.+)$/); + if (!assignment) continue; + const alias = assignment[1]; + const rhs = assignment[2].trim(); + const quoted = rhs.match(/^['"]([^'"]+)['"]$/); + if (quoted) { + const ga = moduleToGa(quoted[1]); + if (ga) addLibrary(alias, ga); + continue; + } + const table = parseInlineTomlTable(rhs); + const ga = table.module + ? moduleToGa(table.module) + : table.group && table.name + ? `${table.group}:${table.name}` + : undefined; + if (ga) addLibrary(alias, ga); + continue; + } + + if (section === 'bundles') { + const assignment = line.match(/^([A-Za-z0-9._-]+)\s*=\s*\[([^\]]*)\]$/); + if (!assignment) continue; + const members = [...assignment[2].matchAll(/['"]([^'"]+)['"]/g)].map((match) => match[1]); + for (const accessor of catalogAccessors(assignment[1])) bundles.set(accessor, members); } } + return { libraries, bundles }; +} + +const GRADLE_GROUP_PATTERNS = [ + /(?:^|[\n{;])\s*(?:rootProject\.)?group\s*=\s*['"]([^'"]+)['"]/, + /(?:^|[\n{;])\s*group\s+['"]([^'"]+)['"]/, +]; + +const GRADLE_COORD_CONFIGS = + 'implementation|api|compileOnly|runtimeOnly|testImplementation|testApi|testCompileOnly|compile|kapt|ksp|commonMainImplementation|commonMainApi'; + +const CATALOG_ALIAS = '([A-Za-z0-9_]+(?:\\.[A-Za-z0-9_]+)*)(?:\\.get\\(\\)|\\.asProvider\\(\\))?'; + +function gradleDepRe(suffix: string): RegExp { + return new RegExp(`(?:${GRADLE_COORD_CONFIGS})\\s*${suffix}`, 'g'); +} + +function parseGradleGroup(content: string): string | undefined { + for (const pattern of GRADLE_GROUP_PATTERNS) { + const match = content.match(pattern); + if (match?.[1]) return match[1]; + } + return undefined; +} + +function asXmlNode(value: unknown): XmlNode | undefined { + return value !== null && typeof value === 'object' && !Array.isArray(value) + ? (value as XmlNode) + : undefined; +} + +function xmlText(value: unknown): string | undefined { + if (typeof value === 'string' || typeof value === 'number') { + const text = String(value).trim(); + return text || undefined; + } + const nested = asXmlNode(value)?.['#text']; + if (nested === undefined) return undefined; + return xmlText(nested); +} + +function xmlChildText(node: XmlNode | undefined, name: string): string | undefined { + return node ? xmlText(node[name]) : undefined; +} + +function asList(value: unknown): unknown[] { + if (value === undefined || value === null) return []; + return Array.isArray(value) ? value : [value]; +} + +/** Direct project dependencies only — not BOM, profiles, or plugin classpath. */ +function collectProjectDependencies(project: XmlNode, deps: string[]): void { + const dependencies = asXmlNode(project.dependencies); + if (!dependencies) return; + for (const dep of asList(dependencies.dependency)) { + const depNode = asXmlNode(dep); + const groupId = xmlChildText(depNode, 'groupId'); + const artifactId = xmlChildText(depNode, 'artifactId'); + if (groupId && artifactId) deps.push(`${groupId}:${artifactId}`); + } +} + +function parsePom(content: string): { groupId: string; artifactId: string; deps: string[] } | null { + let parsed: unknown; + try { + // parseSourceSafe guards tree-sitter's Windows SIGSEGV by switching to a + // chunked input callback above 16 KB; XMLParser only accepts XML text, so + // routing POMs through it silently yields an empty document. + // eslint-disable-next-line gitnexus/require-safe-parse + parsed = pomParser.parse(content); + } catch { + return null; + } + + const project = asXmlNode(asXmlNode(parsed)?.project); + if (!project) return null; + + // Maven inherits groupId from , but artifactId is always the + // project's own direct child and must never fall back to parent.artifactId. + const groupId = + xmlChildText(project, 'groupId') ?? xmlChildText(asXmlNode(project.parent), 'groupId'); + const artifactId = xmlChildText(project, 'artifactId'); + if (!groupId || !artifactId) return null; + + const deps: string[] = []; + collectProjectDependencies(project, deps); return { groupId, artifactId, deps: [...new Set(deps)] }; } function parseGradle( content: string, repoPath: string, + sidecars: GradleSidecars = { catalogLibraries: new Map(), catalogBundles: new Map() }, ): { groupId: string; artifactId: string; deps: string[] } | null { - const groupMatch = content.match(/group\s*=\s*['"]([^'"]+)['"]/); - const dirName = path.basename(repoPath); - const groupId = groupMatch ? groupMatch[1] : ''; + // Static text + default catalog file. Do not execute Gradle. + const groupId = parseGradleGroup(content) ?? sidecars.propertiesGroup ?? ''; if (!groupId) return null; - const artifactId = dirName; + const artifactId = sidecars.rootProjectName ?? path.basename(repoPath); + const { catalogLibraries, catalogBundles } = sidecars; const deps: string[] = []; - // implementation("group:artifact:version") or api("group:artifact:version") - const depMatches = content.matchAll( - /(?:implementation|api|compileOnly|runtimeOnly)\s*\(\s*['"]([^'"]+)['"]\s*\)/g, + const pushCatalogAlias = (alias: string) => { + const ga = catalogLibraries.get(alias); + if (ga) deps.push(ga); + }; + + const namedPattern = gradleDepRe( + `(?:\\(\\s*)?(?:group\\s*=\\s*['"](?[^'"]+)['"]\\s*,\\s*name\\s*=\\s*['"](?[^'"]+)['"]|name\\s*=\\s*['"](?[^'"]+)['"]\\s*,\\s*group\\s*=\\s*['"](?[^'"]+)['"]|group:\\s*['"](?[^'"]+)['"]\\s*,\\s*name:\\s*['"](?[^'"]+)['"]|name:\\s*['"](?[^'"]+)['"]\\s*,\\s*group:\\s*['"](?[^'"]+)['"])`, ); - for (const m of depMatches) { - const parts = m[1].split(':'); - if (parts.length >= 2) { - deps.push(`${parts[0]}:${parts[1]}`); + for (const match of content.matchAll(namedPattern)) { + const group = + match.groups?.group1 ?? match.groups?.group2 ?? match.groups?.group3 ?? match.groups?.group4; + const name = + match.groups?.name1 ?? match.groups?.name2 ?? match.groups?.name3 ?? match.groups?.name4; + if (group && name) deps.push(`${group}:${name}`); + } + + for (const match of content.matchAll( + gradleDepRe(`(?:\\(\\s*)?libs(?:\\.libraries)?\\.(?!bundles\\.|plugins\\.)${CATALOG_ALIAS}`), + )) { + pushCatalogAlias(match[1]); + } + + for (const match of content.matchAll( + gradleDepRe(`(?:\\(\\s*)?libs\\.bundles\\.${CATALOG_ALIAS}`), + )) { + for (const member of catalogBundles.get(match[1]) ?? []) { + for (const accessor of catalogAccessors(member)) pushCatalogAlias(accessor); } } - // implementation(project(":subproject")) - const projDeps = content.matchAll( - /(?:implementation|api)\s*\(\s*project\s*\(\s*['"]([^'"]+)['"]\s*\)\s*\)/g, - ); - for (const m of projDeps) { - const subName = m[1].replace(/^:/, ''); - deps.push(`${groupId}:${subName}`); + for (const match of content.matchAll(gradleDepRe(`\\(\\s*projects\\.([A-Za-z][A-Za-z0-9.]*)`))) { + deps.push(`${groupId}:${projectAccessorToArtifactId(match[1])}`); + } + + for (const match of content.matchAll( + gradleDepRe(`(?:\\(\\s*['"]([^'"]+)['"]\\s*\\)|['"]([^'"]+)['"])`), + )) { + const coord = match[1] ?? match[2]; + if (!coord) continue; + const parts = coord.split(':'); + if (parts.length >= 2) deps.push(`${parts[0]}:${parts[1]}`); + } + + for (const match of content.matchAll( + gradleDepRe(`(?:\\(\\s*)?project\\s*\\(\\s*['"]([^'"]+)['"]\\s*\\)`), + )) { + deps.push(`${groupId}:${match[1].replace(/^:/, '')}`); } return { groupId, artifactId, deps: [...new Set(deps)] }; diff --git a/gitnexus/test/unit/group/java-workspace-extractor.test.ts b/gitnexus/test/unit/group/java-workspace-extractor.test.ts index 09233ab12..fa8c3729f 100644 --- a/gitnexus/test/unit/group/java-workspace-extractor.test.ts +++ b/gitnexus/test/unit/group/java-workspace-extractor.test.ts @@ -31,6 +31,24 @@ describe('JavaWorkspaceExtractor', () => { return `${g}${a}${depXml}`; }; + const inheritedPomTemplate = (artifactId: string, deps: string[] = []) => { + const depXml = deps + .map((d) => { + const [gid, aid] = d.split(':'); + return `${gid}${aid}`; + }) + .join('\n'); + return ` + + com.example + parent + 1 + + ${artifactId} + ${depXml} + `; + }; + it('discovers cross-project imports via Maven pom.xml', async () => { await writeFile('models/pom.xml', pomTemplate('com.acme', 'models')); await writeFile( @@ -62,6 +80,280 @@ describe('JavaWorkspaceExtractor', () => { }); }); + it('keeps child artifacts distinct when independent repositories share a Maven parent', async () => { + await writeFile('parent/pom.xml', pomTemplate('com.example', 'parent')); + await writeFile('shared-lib/pom.xml', inheritedPomTemplate('shared-lib')); + await writeFile( + 'shared-lib/src/main/java/com/example/shared/lib/SharedType.java', + 'package com.example.shared.lib;\npublic class SharedType {}\n', + ); + + await writeFile( + 'service-a/pom.xml', + inheritedPomTemplate('service-a', ['com.example:shared-lib']), + ); + await writeFile( + 'service-a/src/main/java/com/example/service/a/App.java', + 'package com.example.service.a;\nimport com.example.shared.lib.SharedType;\npublic class App {}\n', + ); + + await writeFile( + 'service-b/pom.xml', + inheritedPomTemplate('service-b', ['com.example:shared-lib']), + ); + await writeFile( + 'service-b/src/main/kotlin/com/example/service/b/App.kt', + 'package com.example.service.b\nimport com.example.shared.lib.SharedType\nclass App\n', + ); + + const repos = { + parent: 'parent', + 'shared-lib': 'shared-lib', + 'service-a': 'service-a', + 'service-b': 'service-b', + }; + const repoPaths = new Map( + Object.keys(repos).map((groupPath) => [groupPath, path.join(tmpDir, groupPath)]), + ); + + const result = await extractJavaWorkspaceLinks(repos, repoPaths); + + expect(result.discoveredProjects.size).toBe(4); + expect(result.discoveredProjects.get('parent')?.artifactId).toBe('parent'); + expect(result.discoveredProjects.get('shared-lib')?.artifactId).toBe('shared-lib'); + expect(result.discoveredProjects.get('service-a')?.artifactId).toBe('service-a'); + expect(result.discoveredProjects.get('service-b')?.artifactId).toBe('service-b'); + expect(result.links).toEqual([ + { + from: 'shared-lib', + to: 'service-a', + type: 'custom', + contract: 'shared-lib::SharedType', + role: 'provider', + }, + { + from: 'shared-lib', + to: 'service-b', + type: 'custom', + contract: 'shared-lib::SharedType', + role: 'provider', + }, + ]); + }); + + async function extractNamed(names: string[]) { + return extractJavaWorkspaceLinks( + Object.fromEntries(names.map((name) => [name, name])), + new Map(names.map((name) => [name, path.join(tmpDir, name)])), + ); + } + + it('skips POMs that cannot yield a child artifact identity', async () => { + await writeFile('broken/pom.xml', ''); + await writeFile('empty/pom.xml', ''); + await writeFile( + 'parent-only/pom.xml', + ` + + com.example + parent + 1 + + `, + ); + await writeFile('no-group/pom.xml', 'orphan'); + + const result = await extractNamed(['broken', 'not-maven', 'empty', 'parent-only', 'no-group']); + + expect(result.discoveredProjects.size).toBe(0); + expect(result.links).toHaveLength(0); + expect(result.discoveredProjects.has('parent-only')).toBe(false); + }); + + it('ignores dependencyManagement, profiles, and plugin dependencies for workspace links', async () => { + await writeFile('shared-lib/pom.xml', pomTemplate('com.example', 'shared-lib')); + await writeFile( + 'shared-lib/src/main/java/com/example/shared/lib/SharedType.java', + 'package com.example.shared.lib;\npublic class SharedType {}\n', + ); + + await writeFile( + 'bom-consumer/pom.xml', + ` + com.example + bom-consumer + + + + com.example + shared-lib + 1 + + + + `, + ); + await writeFile( + 'bom-consumer/src/main/java/com/example/bom/App.java', + 'package com.example.bom;\nimport com.example.shared.lib.SharedType;\npublic class App {}\n', + ); + + await writeFile( + 'profile-consumer/pom.xml', + ` + com.example + profile-consumer + + + extra + + + com.example + shared-lib + + + + + `, + ); + await writeFile( + 'profile-consumer/src/main/kotlin/com/example/profile/App.kt', + 'package com.example.profile\nimport com.example.shared.lib.SharedType\nclass App\n', + ); + + await writeFile( + 'plugin-consumer/pom.xml', + ` + com.example + plugin-consumer + + + + org.apache.maven.plugins + maven-compiler-plugin + + + com.example + shared-lib + + + + + + `, + ); + await writeFile( + 'plugin-consumer/src/main/java/com/example/plugin/App.java', + 'package com.example.plugin;\nimport com.example.shared.lib.SharedType;\npublic class App {}\n', + ); + + const result = await extractNamed([ + 'shared-lib', + 'bom-consumer', + 'profile-consumer', + 'plugin-consumer', + ]); + + expect(result.discoveredProjects.get('bom-consumer')?.deps).toEqual([]); + expect(result.discoveredProjects.get('profile-consumer')?.deps).toEqual([]); + expect(result.discoveredProjects.get('plugin-consumer')?.deps).toEqual([]); + expect(result.links).toEqual([]); + }); + + it('uses an explicit child groupId instead of its Maven parent groupId', async () => { + await writeFile( + 'service/pom.xml', + ` + + com.parent + parent + 1 + + com.child + service + `, + ); + + const result = await extractJavaWorkspaceLinks( + { service: 'service' }, + new Map([['service', path.join(tmpDir, 'service')]]), + ); + + expect(result.discoveredProjects.get('service')).toMatchObject({ + groupId: 'com.child', + artifactId: 'service', + }); + }); + + it('parses namespaced POM coordinates and CDATA text with a real XML parser', async () => { + await writeFile( + 'lib/pom.xml', + ` + + + com.parent + parent + + + + + com.acme + models + + + `, + ); + await writeFile('models/pom.xml', pomTemplate('com.acme', 'models')); + await writeFile( + 'models/src/main/java/com/acme/models/User.java', + 'package com.acme.models;\npublic class User {}\n', + ); + await writeFile( + 'lib/src/main/java/com/parent/shared/lib/App.java', + 'package com.parent.shared.lib;\nimport com.acme.models.User;\npublic class App {}\n', + ); + + const result = await extractJavaWorkspaceLinks( + { lib: 'lib', models: 'models' }, + new Map([ + ['lib', path.join(tmpDir, 'lib')], + ['models', path.join(tmpDir, 'models')], + ]), + ); + + expect(result.discoveredProjects.get('lib')).toMatchObject({ + groupId: 'com.parent', + artifactId: 'shared-lib', + }); + expect(result.links).toEqual([ + { + from: 'models', + to: 'lib', + type: 'custom', + contract: 'models::User', + role: 'provider', + }, + ]); + }); + + it('still rejects genuinely duplicate effective Maven coordinates', async () => { + await writeFile('first/pom.xml', inheritedPomTemplate('shared-lib')); + await writeFile('second/pom.xml', inheritedPomTemplate('shared-lib')); + + const result = await extractJavaWorkspaceLinks( + { first: 'first', second: 'second' }, + new Map([ + ['first', path.join(tmpDir, 'first')], + ['second', path.join(tmpDir, 'second')], + ]), + ); + + expect(result.discoveredProjects.size).toBe(1); + expect(result.discoveredProjects.has('first')).toBe(true); + expect(result.discoveredProjects.has('second')).toBe(false); + }); + it('handles Gradle build files', async () => { await writeFile('core/build.gradle.kts', 'group = "com.acme"\nversion = "1.0"\n'); await writeFile( @@ -74,8 +366,8 @@ describe('JavaWorkspaceExtractor', () => { 'group = "com.acme"\nversion = "1.0"\ndependencies {\n implementation("com.acme:core:1.0")\n}\n', ); await writeFile( - 'svc/src/main/java/com/acme/svc/App.java', - 'package com.acme.svc;\nimport com.acme.core.Config;\npublic class App {}\n', + 'svc/src/main/kotlin/com/acme/svc/App.kt', + 'package com.acme.svc\nimport com.acme.core.Config\nclass App\n', ); const repos = { core: 'core', svc: 'svc' }; @@ -118,6 +410,330 @@ describe('JavaWorkspaceExtractor', () => { expect(result.links[0].contract).toBe('common::Entity'); }); + it('reads Gradle group from gradle.properties and Groovy setter syntax', async () => { + await writeFile('core/gradle.properties', 'group=com.acme\n'); + await writeFile('core/build.gradle', 'plugins { id "java" }\n'); + await writeFile( + 'core/src/main/java/com/acme/core/Config.java', + 'package com.acme.core;\npublic class Config {}\n', + ); + + await writeFile( + 'svc/build.gradle', + "group 'com.acme'\ndependencies {\n testImplementation 'com.acme:core:1.0'\n}\n", + ); + await writeFile( + 'svc/src/test/kotlin/com/acme/svc/AppTest.kt', + 'package com.acme.svc\nimport com.acme.core.Config\nclass AppTest\n', + ); + + const result = await extractNamed(['core', 'svc']); + + expect(result.discoveredProjects.get('core')).toMatchObject({ + groupId: 'com.acme', + artifactId: 'core', + }); + expect(result.links).toEqual([ + { + from: 'core', + to: 'svc', + type: 'custom', + contract: 'core::Config', + role: 'provider', + }, + ]); + }); + + it('reads Gradle group assigned inside allprojects { }', async () => { + await writeFile('core/build.gradle', 'allprojects { group = "com.acme" }\n'); + await writeFile( + 'core/src/main/java/com/acme/core/Config.java', + 'package com.acme.core;\npublic class Config {}\n', + ); + await writeFile( + 'svc/build.gradle', + 'allprojects { group = "com.acme" }\ndependencies {\n implementation "com.acme:core:1.0"\n}\n', + ); + await writeFile( + 'svc/src/main/java/com/acme/svc/App.java', + 'package com.acme.svc;\nimport com.acme.core.Config;\npublic class App {}\n', + ); + + const result = await extractNamed(['core', 'svc']); + + expect(result.discoveredProjects.get('core')).toMatchObject({ + groupId: 'com.acme', + artifactId: 'core', + }); + expect(result.links).toEqual([ + { + from: 'core', + to: 'svc', + type: 'custom', + contract: 'core::Config', + role: 'provider', + }, + ]); + }); + + it('uses settings.gradle rootProject.name as the Gradle artifactId', async () => { + await writeFile('checkout/settings.gradle.kts', 'rootProject.name = "shared-core"\n'); + await writeFile('checkout/build.gradle.kts', 'group = "com.acme"\n'); + await writeFile( + 'checkout/src/main/java/com/acme/shared/core/Flag.java', + 'package com.acme.shared.core;\npublic class Flag {}\n', + ); + await writeFile( + 'app/build.gradle.kts', + 'group = "com.acme"\ndependencies {\n implementation("com.acme:shared-core:1.0")\n}\n', + ); + await writeFile( + 'app/src/main/kotlin/com/acme/app/Main.kt', + 'package com.acme.app\nimport com.acme.shared.core.Flag\nclass Main\n', + ); + + const result = await extractJavaWorkspaceLinks( + { checkout: 'checkout', app: 'app' }, + new Map([ + ['checkout', path.join(tmpDir, 'checkout')], + ['app', path.join(tmpDir, 'app')], + ]), + ); + + expect(result.discoveredProjects.get('checkout')?.artifactId).toBe('shared-core'); + expect(result.links).toEqual([ + { + from: 'checkout', + to: 'app', + type: 'custom', + contract: 'shared-core::Flag', + role: 'provider', + }, + ]); + }); + + it('resolves Gradle version-catalog and named group/name coordinates', async () => { + await writeFile('shared-lib/build.gradle.kts', 'group = "com.example"\n'); + await writeFile( + 'shared-lib/src/main/java/com/example/shared/lib/SharedType.java', + 'package com.example.shared.lib;\npublic class SharedType {}\n', + ); + await writeFile('models/build.gradle.kts', 'group = "com.example"\n'); + await writeFile( + 'models/src/main/java/com/example/models/User.java', + 'package com.example.models;\npublic class User {}\n', + ); + + await writeFile( + 'app/gradle/libs.versions.toml', + `[libraries] +shared-lib = { module = "com.example:shared-lib", version = "1.0" } +models = { group = "com.example", name = "models", version.ref = "unused" } + +[bundles] +workspace = ["shared-lib", "models"] +`, + ); + await writeFile( + 'app/build.gradle.kts', + `group = "com.example" +dependencies { + implementation(libs.shared.lib) + implementation(libs.bundles.workspace) +} +`, + ); + await writeFile( + 'app/src/main/kotlin/com/example/app/App.kt', + 'package com.example.app\nimport com.example.shared.lib.SharedType\nimport com.example.models.User\nclass App\n', + ); + + await writeFile( + 'named/build.gradle', + `group = 'com.example' +dependencies { + implementation group: 'com.example', name: 'shared-lib', version: '1.0' +} +`, + ); + await writeFile( + 'named-first/build.gradle', + `group = 'com.example' +dependencies { + implementation name: 'shared-lib', group: 'com.example', version: '1.0' +} +`, + ); + await writeFile( + 'named/src/main/java/com/example/named/App.java', + 'package com.example.named;\nimport com.example.shared.lib.SharedType;\npublic class App {}\n', + ); + await writeFile( + 'named-first/src/main/java/com/example/namedfirst/App.java', + 'package com.example.namedfirst;\nimport com.example.shared.lib.SharedType;\npublic class App {}\n', + ); + + const result = await extractNamed(['shared-lib', 'models', 'app', 'named', 'named-first']); + + expect(result.discoveredProjects.get('app')?.deps.sort()).toEqual([ + 'com.example:models', + 'com.example:shared-lib', + ]); + expect(result.links).toEqual( + expect.arrayContaining([ + expect.objectContaining({ + from: 'shared-lib', + to: 'app', + contract: 'shared-lib::SharedType', + }), + expect.objectContaining({ from: 'models', to: 'app', contract: 'models::User' }), + expect.objectContaining({ + from: 'shared-lib', + to: 'named', + contract: 'shared-lib::SharedType', + }), + expect.objectContaining({ + from: 'shared-lib', + to: 'named-first', + contract: 'shared-lib::SharedType', + }), + ]), + ); + expect(result.links).toHaveLength(4); + }); + + it('resolves version-catalog aliases that contain underscores', async () => { + await writeFile('shared-lib/build.gradle.kts', 'group = "com.example"\n'); + await writeFile( + 'shared-lib/src/main/java/com/example/shared/lib/SharedType.java', + 'package com.example.shared.lib;\npublic class SharedType {}\n', + ); + await writeFile( + 'app/gradle/libs.versions.toml', + `[libraries] +foo_bar = { module = "com.example:shared-lib", version = "1.0" } +`, + ); + await writeFile( + 'app/build.gradle.kts', + `group = "com.example" +dependencies { + implementation(libs.foo_bar) +} +`, + ); + await writeFile( + 'app/src/main/kotlin/com/example/app/App.kt', + 'package com.example.app\nimport com.example.shared.lib.SharedType\nclass App\n', + ); + + const result = await extractNamed(['shared-lib', 'app']); + + expect(result.discoveredProjects.get('app')?.deps).toEqual(['com.example:shared-lib']); + expect(result.links).toEqual([ + expect.objectContaining({ + from: 'shared-lib', + to: 'app', + contract: 'shared-lib::SharedType', + }), + ]); + }); + + it('resolves Kotlin DSL named args, catalog get(), and type-safe project accessors', async () => { + await writeFile('shared-lib/build.gradle.kts', 'group = "com.example"\n'); + await writeFile( + 'shared-lib/src/main/kotlin/com/example/shared/lib/SharedType.kt', + 'package com.example.shared.lib\nclass SharedType\n', + ); + await writeFile('models/build.gradle.kts', 'group = "com.example"\n'); + await writeFile( + 'models/src/main/kotlin/com/example/models/User.kt', + 'package com.example.models\ndata class User(val id: Int)\n', + ); + + await writeFile( + 'app/gradle/libs.versions.toml', + '[libraries]\nmodels = { group = "com.example", name = "models" }\n', + ); + await writeFile( + 'app/build.gradle.kts', + `group = "com.example" +kotlin { + sourceSets { + commonMain { + dependencies { + implementation(name = "shared-lib", group = "com.example", version = "1.0") + implementation(projects.sharedLib) + implementation(libs.models.get()) + ksp(libs.models) + } + } + } +} +`, + ); + await writeFile( + 'app/src/commonMain/kotlin/com/example/app/App.kt', + 'package com.example.app\nimport com.example.shared.lib.SharedType as ST\nimport com.example.models.User\nclass App(val user: User, val shared: ST)\n', + ); + + const result = await extractNamed(['shared-lib', 'models', 'app']); + + expect(result.discoveredProjects.get('app')?.deps.sort()).toEqual([ + 'com.example:models', + 'com.example:shared-lib', + ]); + expect(result.links).toEqual( + expect.arrayContaining([ + expect.objectContaining({ + from: 'shared-lib', + to: 'app', + contract: 'shared-lib::SharedType', + }), + expect.objectContaining({ from: 'models', to: 'app', contract: 'models::User' }), + ]), + ); + expect(result.links).toHaveLength(2); + }); + + it('discovers Maven and Gradle projects together without changing Gradle identity', async () => { + await writeFile('shared-lib/pom.xml', pomTemplate('com.example', 'shared-lib')); + await writeFile( + 'shared-lib/src/main/java/com/example/shared/lib/SharedType.java', + 'package com.example.shared.lib;\npublic class SharedType {}\n', + ); + await writeFile( + 'gradle-app/build.gradle.kts', + 'group = "com.example"\ndependencies {\n implementation("com.example:shared-lib:1.0")\n}\n', + ); + await writeFile( + 'gradle-app/src/main/kotlin/com/example/gradle/app/App.kt', + 'package com.example.gradle.app\nimport com.example.shared.lib.SharedType\nclass App\n', + ); + + const result = await extractJavaWorkspaceLinks( + { 'shared-lib': 'shared-lib', 'gradle-app': 'gradle-app' }, + new Map([ + ['shared-lib', path.join(tmpDir, 'shared-lib')], + ['gradle-app', path.join(tmpDir, 'gradle-app')], + ]), + ); + + expect(result.discoveredProjects.get('gradle-app')).toMatchObject({ + groupId: 'com.example', + artifactId: 'gradle-app', + }); + expect(result.links).toEqual([ + { + from: 'shared-lib', + to: 'gradle-app', + type: 'custom', + contract: 'shared-lib::SharedType', + role: 'provider', + }, + ]); + }); + it('handles static imports', async () => { await writeFile('lib/pom.xml', pomTemplate('com.acme', 'lib')); await writeFile( diff --git a/gitnexus/test/unit/group/sync.test.ts b/gitnexus/test/unit/group/sync.test.ts index 95bdd96d7..91b22f15f 100644 --- a/gitnexus/test/unit/group/sync.test.ts +++ b/gitnexus/test/unit/group/sync.test.ts @@ -838,6 +838,86 @@ service OrderService { expect(manifestLinks[0].to.repo).toBe('parser/mathlex'); }); + it('builds Maven manifest links when independent repositories share a parent POM', async () => { + tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-sync-ws-maven-parent-')); + + const parentCoordinates = ` + com.example + parent + 1 + `; + const childPom = (artifactId: string, dependency = '') => ` + ${parentCoordinates} + ${artifactId} + ${dependency} + `; + const sharedDependency = + 'com.exampleshared-lib'; + + writeFileSync( + 'parent/pom.xml', + 'com.exampleparentpom', + ); + writeFileSync('shared-lib/pom.xml', childPom('shared-lib')); + writeFileSync('service-a/pom.xml', childPom('service-a', sharedDependency)); + writeFileSync( + 'service-a/src/main/java/com/example/service/a/App.java', + 'package com.example.service.a;\nimport com.example.shared.lib.SharedType;\npublic class App {}\n', + ); + writeFileSync('service-b/pom.xml', childPom('service-b', sharedDependency)); + writeFileSync( + 'service-b/src/main/kotlin/com/example/service/b/App.kt', + 'package com.example.service.b\nimport com.example.shared.lib.SharedType\nclass App\n', + ); + + const repoPaths = ['parent', 'shared-lib', 'service-a', 'service-b']; + const mockEntries: RegistryEntry[] = repoPaths.map((repoPath) => ({ + name: repoPath, + path: path.join(tmpDir, repoPath), + storagePath: path.join(tmpDir, repoPath, '.gitnexus'), + indexedAt: '', + lastCommit: '', + })); + + const repoManager = await import('../../../src/storage/repo-manager.js'); + vi.spyOn(repoManager, 'readRegistry').mockResolvedValue(mockEntries); + + const config = makeWsConfig( + { + parent: 'parent', + 'libs/shared-lib': 'shared-lib', + 'services/service-a': 'service-a', + 'services/service-b': 'service-b', + }, + true, + ); + + const result = await syncGroup(config, { + extractorOverride: async () => [], + skipWrite: true, + }); + + const manifestLinks = result.crossLinks.filter((link) => link.matchType === 'manifest'); + expect( + manifestLinks.map((link) => ({ + from: link.from.repo, + to: link.to.repo, + contractId: link.contractId, + })), + ).toEqual([ + { + from: 'services/service-a', + to: 'libs/shared-lib', + contractId: 'custom::shared-lib::SharedType', + }, + { + from: 'services/service-b', + to: 'libs/shared-lib', + contractId: 'custom::shared-lib::SharedType', + }, + ]); + }); + it('workspace_deps: false skips workspace extraction entirely', async () => { tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-sync-ws-off-')); From 4aa6bddd0a78135136d29d8440eb29613097f616 Mon Sep 17 00:00:00 2001 From: ChunxueLi <54129170+ChunxueLi@users.noreply.github.com> Date: Tue, 1 Sep 2026 01:55:52 +0800 Subject: [PATCH 07/26] feat(jvm): synthesize Lombok and Kotlin JVM accessor methods (#2885) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(java): synthesize Lombok @Data/@Getter/@Setter accessor methods * fix(lombok): resolve class identity by AST node id, not simple name Root-cause fix for the bot review's name-ambiguity findings: 1. Cross-file collision: the owner map was rebuilt per file from result.symbols, which accumulates across the whole language group — a later Java file with the same simple class name resolved to the earlier file's class node. The map is now filled INSIDE the capture loop (per-file scope) and keyed by the class_declaration AST node id (SyntaxNode.id), which is unique by construction. 2. Same-tail nested classes (Outer.A vs Other.A): a name-keyed map overwrote one with the other; AST-node-id keys cannot collide. 3. Synthesized method ids now follow the SAME convention real nested member ids use (keyed by the class's own simple name, matching findEnclosingClassInfo().className), so call resolution can hit synthesized accessors exactly like hand-written ones. 4. Lombok semantics: setters are no longer generated for final fields (Lombok never emits those) and @Setter(AccessLevel.NONE) now suppresses setters, symmetric to the existing getter suppression. Also tightens two vacuous test loops flagged by the bot (empty-array for..of passed trivially): counts are asserted before property loops, and a new regression test pins distinct owners for same-tailed nested classes plus the real id convention for nested accessors. * feat(java): synthesize Lombok accessors via provider hook and scope dual-path Replace the worker language===Java branch with LanguageProvider.synthesizeStructureMembers, align MethodRegistry ownership through scope captures, and bump parse-cache schema to 83 so warm caches cannot replay pre-synthesis worker output. Co-authored-by: Cursor * test(java): cover Lombok synthesis semantics, cache replay, and CI bench Add unit/integration matrices (including durable cold/warm/historical parse-cache), a permanent no-Lombok vs Lombok-heavy harness with fingerprint budgets, and a CI --check step. Document that Kotlin→Java member CALLS remains a pre-existing gap. Co-authored-by: Cursor * refactor(lombok): drop dead state and redundant scans from accessor synthesis Collapse Lombok import provenance into one compilation-unit scan with a cached wildcard flag, remove unused planned-accessor fields and the duplicate @Data enable flag, and plan scope captures without wrapping a fake Parser.Tree. Co-authored-by: Cursor * Address PR review feedback (#2885) Give each Lombok accessor a unique scope range so multi-declarator fields do not share @scope.function IDs, and type the owner map as ReadonlyMap to match the provider hook. Co-authored-by: Cursor * fix(bench): pin the real Lombok synthesis fingerprint (#2885) The committed baseline held a fingerprint no revision of this branch ever produced, so the CI guard failed on every push. Re-pin it to the value the synthesizer deterministically emits and correct the method count the comment claims (800 x 4 x 2 = 6400, not 12800). Co-authored-by: Cursor * feat(kotlin): synthesize JVM accessors using shared beanspec helpers (#2885) Kotlin val/var properties now emit the same JavaBeans get/set Methods as Lombok, via jvm/beanspec + jvm/synthetic-accessors. SCHEMA_BUMP 84 invalidates warm caches that would omit those callables. Co-authored-by: Cursor * fix(kotlin): match kotlinc JVM accessor ABI (#2885) Emit custom getters, preserve is-prefix names, and convert synthetic graph lines to 0-based so same-name accessors resolve to the owner. Co-authored-by: Cursor * Address PR review feedback (#2885) Restrict Lombok provenance to lombok/experimental FQNs and match Kotlin existing methods by exact JVM name. Co-authored-by: Cursor * refactor(jvm): consolidate accessor synthesis (#2885) Keep language-specific discovery in Java and Kotlin adapters while centralizing owner orchestration, collision policy, graph emission, and captures. Co-authored-by: Cursor * fix(jvm): align accessor synthesis with compiler ABI (#2885) Match Lombok and kotlinc provenance, companion owners, and collision arity so mixed-JVM CALLS bind to the Methods compilers actually emit. Co-authored-by: Cursor * Address PR review feedback (#2885) Mark Kotlin interface accessors abstract, pin the Lombok case-fold collision test, and document the non-lowercase is-prefix rule. Co-authored-by: Cursor * Address PR review feedback (#2885) Honor explicit Getter/Setter over @Data regardless of order, and let field @Accessors replace class-level fluent/chain. Co-authored-by: Cursor * fix(ci): pin Kotlin scope-capture fingerprint after interface accessors (#2885) Invalidate warm parse cache so interface property Methods are not replayed as concrete. Co-authored-by: Cursor --------- Co-authored-by: ChunxueLi Co-authored-by: Gergő Magyar Co-authored-by: Gergo Magyar Co-authored-by: Cursor --- .github/workflows/ci-tests.yml | 14 + .../java-lombok-synthesis/baselines.json | 8 + .../bench/java-lombok-synthesis/measure.mjs | 124 +++ .../bench/kotlin-jvm-accessors/baselines.json | 8 + .../bench/kotlin-jvm-accessors/measure.mjs | 121 +++ gitnexus/bench/lib/identity-guard.mjs | 62 ++ gitnexus/bench/scope-capture/baselines.json | 10 +- .../src/core/ingestion/language-provider.ts | 48 ++ gitnexus/src/core/ingestion/languages/java.ts | 3 + .../core/ingestion/languages/java/captures.ts | 2 + .../languages/java/lombok-synthesizer.ts | 539 +++++++++++++ .../languages/jvm/accessor-synthesis.ts | 316 ++++++++ .../core/ingestion/languages/jvm/beanspec.ts | 49 ++ .../src/core/ingestion/languages/kotlin.ts | 2 + .../ingestion/languages/kotlin/captures.ts | 2 + .../languages/kotlin/lombok-synthesizer.ts | 512 ++++++++++++ .../src/core/ingestion/scope-extractor.ts | 1 + .../core/ingestion/workers/parse-worker.ts | 29 + gitnexus/src/storage/parse-cache.ts | 15 +- .../integration/resolvers/java-lombok.test.ts | 258 +++++++ .../resolvers/kotlin-jvm-accessors.test.ts | 121 +++ .../test/integration/resolvers/kotlin.test.ts | 29 +- .../test/unit/incremental-parse-cache.test.ts | 5 +- .../unit/kotlin-lombok-synthesizer.test.ts | 420 ++++++++++ gitnexus/test/unit/lombok-synthesizer.test.ts | 726 ++++++++++++++++++ 25 files changed, 3416 insertions(+), 8 deletions(-) create mode 100644 gitnexus/bench/java-lombok-synthesis/baselines.json create mode 100644 gitnexus/bench/java-lombok-synthesis/measure.mjs create mode 100644 gitnexus/bench/kotlin-jvm-accessors/baselines.json create mode 100644 gitnexus/bench/kotlin-jvm-accessors/measure.mjs create mode 100644 gitnexus/bench/lib/identity-guard.mjs create mode 100644 gitnexus/src/core/ingestion/languages/java/lombok-synthesizer.ts create mode 100644 gitnexus/src/core/ingestion/languages/jvm/accessor-synthesis.ts create mode 100644 gitnexus/src/core/ingestion/languages/jvm/beanspec.ts create mode 100644 gitnexus/src/core/ingestion/languages/kotlin/lombok-synthesizer.ts create mode 100644 gitnexus/test/integration/resolvers/java-lombok.test.ts create mode 100644 gitnexus/test/integration/resolvers/kotlin-jvm-accessors.test.ts create mode 100644 gitnexus/test/unit/kotlin-lombok-synthesizer.test.ts create mode 100644 gitnexus/test/unit/lombok-synthesizer.test.ts diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index 5bdcf3560..9739f9028 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -509,6 +509,20 @@ jobs: run: node --import tsx bench/callable-value-flow/measure.mjs --check working-directory: gitnexus + - name: Java Lombok accessor synthesis guards (#2885) + if: ${{ !cancelled() }} + # Build-free: no-Lombok vs Lombok-heavy corpora; fingerprint over + # synthetic Method ids; scaling + widening overhead budgets. + run: node --import tsx bench/java-lombok-synthesis/measure.mjs --check + working-directory: gitnexus + + - name: Kotlin JVM accessor synthesis guards (#2885) + if: ${{ !cancelled() }} + # Build-free: no-property vs data-class corpora; fingerprint over + # synthetic Method ids; scaling + widening overhead budgets. + run: node --import tsx bench/kotlin-jvm-accessors/measure.mjs --check + working-directory: gitnexus + - name: Re-export closure scaling guards (#2864) # Build-free: asserts buildReexportClosures stays linear in chain depth # and within an absolute ceiling on a wide package corpus. #2864 changed diff --git a/gitnexus/bench/java-lombok-synthesis/baselines.json b/gitnexus/bench/java-lombok-synthesis/baselines.json new file mode 100644 index 000000000..9e14f8f3c --- /dev/null +++ b/gitnexus/bench/java-lombok-synthesis/baselines.json @@ -0,0 +1,8 @@ +{ + "_comment": "Baselines for bench/java-lombok-synthesis/measure.mjs --check (#2885). fingerprint is sha256 over synthetic Method node ids on the lombok_large corpus (800 @Data entities × 4 fields × 2 accessors = 6400 methods). no_lombok arm must emit 0 methods. Budgets are timing gates with CI headroom.", + "fingerprint": "b935d6894d32de7594d5887bb62af6ade2b66b19d6846700a05ef3baf1ed1eb1", + "scaling_budget": 1.6, + "_scaling_note": "(t_large/t_small)/(800/250) on the lombok arm. Measured ~1.01.", + "widening_overhead_budget": 2.5, + "_widening_overhead_note": "lombok_large_ms / no_lombok_large_ms using an unannotated, shape-equivalent four-field control. Measured about 1.24; budget guards against a pathological feature-arm regression." +} diff --git a/gitnexus/bench/java-lombok-synthesis/measure.mjs b/gitnexus/bench/java-lombok-synthesis/measure.mjs new file mode 100644 index 000000000..512af1e12 --- /dev/null +++ b/gitnexus/bench/java-lombok-synthesis/measure.mjs @@ -0,0 +1,124 @@ +/** + * Build-free throughput + identity bench for Java Lombok accessor synthesis. + * + * Arms: + * - no_lombok: unannotated fields (shape-equivalent control) — synthesizer no-ops + * - lombok_heavy: @Data classes (feature path) + * + * Times synthesizeLombokAccessors over N separate files (not one giant buffer). + * + * Usage: + * node --import tsx bench/java-lombok-synthesis/measure.mjs + * node --import tsx bench/java-lombok-synthesis/measure.mjs --check + */ +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import Parser from 'tree-sitter'; +import Java from 'tree-sitter-java'; +import { synthesizeLombokAccessors } from '../../src/core/ingestion/languages/java/lombok-synthesizer.ts'; +import { + fingerprintIds, + minSample, + runBaselineCheck, + runMethodCountCheck, +} from '../lib/identity-guard.mjs'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); + +const SMALL = 250; +const LARGE = 800; +const REPS = 15; +const WARMUP = 5; + +function entitySource(i, mode) { + if (mode === 'lombok') { + return `import lombok.Data; +@Data +public class Entity${i} { + private String id; + private String name; + private boolean active; + private Long amount; +} +`; + } + return `public class Entity${i} { + private String id; + private String name; + private boolean active; + private Long amount; +} +`; +} + +function ownerMap(tree, filePath) { + const map = new Map(); + const walk = (node) => { + if (node.type === 'class_declaration') { + const name = node.childForFieldName('name')?.text; + if (name) map.set(node.id, `Class:${filePath}:${name}`); + } + for (const c of node.children) walk(c); + }; + walk(tree.rootNode); + return map; +} + +function prepare(mode, fileCount) { + const files = []; + for (let i = 0; i < fileCount; i++) { + const parser = new Parser(); + parser.setLanguage(Java); + const filePath = `bench/${mode}/Entity${i}.java`; + const tree = parser.parse(entitySource(i, mode)); + files.push({ tree, filePath, owners: ownerMap(tree, filePath) }); + } + return files; +} + +function runAll(files) { + const nodes = []; + for (const f of files) { + const result = synthesizeLombokAccessors(f.tree, f.filePath, f.owners); + for (const n of result.nodes) nodes.push(n.id); + } + return nodes; +} + +function measure(mode, fileCount) { + const files = prepare(mode, fileCount); + const { last, ms } = minSample(() => runAll(files), WARMUP, REPS); + return { + files: fileCount, + ms, + methods: last.length, + fingerprint: fingerprintIds(last), + }; +} + +const report = { + no_lombok_small: measure('bare', SMALL), + no_lombok_large: measure('bare', LARGE), + lombok_small: measure('lombok', SMALL), + lombok_large: measure('lombok', LARGE), +}; +report.scaling_ratio = Number( + (report.lombok_large.ms / report.lombok_small.ms / (LARGE / SMALL)).toFixed(3), +); +report.widening_overhead = Number( + (report.lombok_large.ms / Math.max(report.no_lombok_large.ms, 0.001)).toFixed(3), +); +report.fingerprint = report.lombok_large.fingerprint; + +runMethodCountCheck(report, { + no_lombok_large: 0, + lombok_large: 6400, +}); + +if (!process.argv.includes('--check')) { + console.log(JSON.stringify(report, null, 2)); + process.exit(0); +} + +runBaselineCheck(report, BASELINE_PATH); diff --git a/gitnexus/bench/kotlin-jvm-accessors/baselines.json b/gitnexus/bench/kotlin-jvm-accessors/baselines.json new file mode 100644 index 000000000..6ef877719 --- /dev/null +++ b/gitnexus/bench/kotlin-jvm-accessors/baselines.json @@ -0,0 +1,8 @@ +{ + "_comment": "Baselines for bench/kotlin-jvm-accessors/measure.mjs --check (#2885). fingerprint is sha256 over synthetic Method node ids on the data_large corpus (800 data classes × 4 vars × 2 accessors = 6400 methods). no_props arm uses @JvmField so kotlinc and the synthesizer emit 0 accessor methods. Budgets are timing gates with CI headroom.", + "fingerprint": "18e4f295a437a747c486699e8ec5d310d9bde54437d9a96356a1b1bf8442b0ef", + "scaling_budget": 1.6, + "_scaling_note": "(t_large/t_small)/(800/250) on the data-class arm. Measured ~1.02.", + "widening_overhead_budget": 2.5, + "_widening_overhead_note": "data_large_ms / no_props_large_ms. The @JvmField control preserves four property declarations without accessors; budget guards against a pathological synthesis-arm regression." +} diff --git a/gitnexus/bench/kotlin-jvm-accessors/measure.mjs b/gitnexus/bench/kotlin-jvm-accessors/measure.mjs new file mode 100644 index 000000000..0edbfb1e4 --- /dev/null +++ b/gitnexus/bench/kotlin-jvm-accessors/measure.mjs @@ -0,0 +1,121 @@ +/** + * Build-free throughput + identity bench for Kotlin JVM accessor synthesis. + * + * Arms: + * - no_props: @JvmField properties with no JVM accessors (control) + * - data_class: data class constructor properties (feature path) + * + * Usage: + * node --import tsx bench/kotlin-jvm-accessors/measure.mjs + * node --import tsx bench/kotlin-jvm-accessors/measure.mjs --check + */ +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import Parser from 'tree-sitter'; +import { SupportedLanguages } from 'gitnexus-shared'; +import { getLanguageGrammar } from '../../src/core/tree-sitter/parser-loader.ts'; +import { synthesizeLombokAccessors } from '../../src/core/ingestion/languages/kotlin/lombok-synthesizer.ts'; +import { + fingerprintIds, + minSample, + runBaselineCheck, + runMethodCountCheck, +} from '../lib/identity-guard.mjs'; + +const __dirname = path.dirname(fileURLToPath(import.meta.url)); +const BASELINE_PATH = path.resolve(__dirname, 'baselines.json'); + +const SMALL = 250; +const LARGE = 800; +const REPS = 15; +const WARMUP = 5; + +function entitySource(i, mode) { + if (mode === 'data') { + return `data class Entity${i}(var id: String, var name: String, var active: Boolean, var amount: Long) +`; + } + // @JvmField suppresses accessors in kotlinc and in the synthesizer while + // retaining the same four property declarations as the feature arm. + return `class Entity${i} { + @JvmField var id: String = "" + @JvmField var name: String = "" + @JvmField var active: Boolean = false + @JvmField var amount: Long = 0 +} +`; +} + +function ownerMap(tree, filePath) { + const map = new Map(); + const walk = (node) => { + if (node.type === 'class_declaration' || node.type === 'object_declaration') { + const name = + node.childForFieldName('name')?.text ?? + node.namedChildren.find((c) => c.type === 'type_identifier')?.text; + if (name) map.set(node.id, `Class:${filePath}:${name}`); + } + for (const c of node.children) walk(c); + }; + walk(tree.rootNode); + return map; +} + +function prepare(mode, fileCount) { + const files = []; + const lang = getLanguageGrammar(SupportedLanguages.Kotlin); + for (let i = 0; i < fileCount; i++) { + const parser = new Parser(); + parser.setLanguage(lang); + const filePath = `bench/${mode}/Entity${i}.kt`; + const tree = parser.parse(entitySource(i, mode)); + files.push({ tree, filePath, owners: ownerMap(tree, filePath), parser }); + } + return files; +} + +function runAll(files) { + const nodes = []; + for (const f of files) { + const result = synthesizeLombokAccessors(f.tree, f.filePath, f.owners); + for (const n of result.nodes) nodes.push(n.id); + } + return nodes; +} + +function measure(mode, fileCount) { + const files = prepare(mode, fileCount); + const { last, ms } = minSample(() => runAll(files), WARMUP, REPS); + return { + files: fileCount, + ms, + methods: last.length, + fingerprint: fingerprintIds(last), + }; +} + +const report = { + no_props_small: measure('hand', SMALL), + no_props_large: measure('hand', LARGE), + data_small: measure('data', SMALL), + data_large: measure('data', LARGE), +}; +report.scaling_ratio = Number( + (report.data_large.ms / report.data_small.ms / (LARGE / SMALL)).toFixed(3), +); +report.widening_overhead = Number( + (report.data_large.ms / Math.max(report.no_props_large.ms, 0.001)).toFixed(3), +); +report.fingerprint = report.data_large.fingerprint; + +runMethodCountCheck(report, { + no_props_large: 0, + data_large: 6400, +}); + +if (!process.argv.includes('--check')) { + console.log(JSON.stringify(report, null, 2)); + process.exit(0); +} + +runBaselineCheck(report, BASELINE_PATH); diff --git a/gitnexus/bench/lib/identity-guard.mjs b/gitnexus/bench/lib/identity-guard.mjs new file mode 100644 index 000000000..b73a1a241 --- /dev/null +++ b/gitnexus/bench/lib/identity-guard.mjs @@ -0,0 +1,62 @@ +/** + * Shared fingerprint + --check for JVM accessor synthesis benches. + */ +import fs from 'node:fs'; +import crypto from 'node:crypto'; + +export function fingerprintIds(ids) { + return crypto + .createHash('sha256') + .update([...ids].sort().join('\n')) + .digest('hex'); +} + +export function minSample(run, warmup, reps) { + for (let w = 0; w < warmup; w++) run(); + const samples = []; + let last; + for (let r = 0; r < reps; r++) { + const t0 = performance.now(); + last = run(); + samples.push(performance.now() - t0); + } + return { last, ms: Math.min(...samples) }; +} + +export function runMethodCountCheck(report, expectedCounts) { + const errors = []; + for (const [arm, expected] of Object.entries(expectedCounts)) { + const actual = report[arm]?.methods; + if (actual !== expected) { + errors.push(`${arm}.methods ${String(actual)} != ${expected}`); + } + } + if (errors.length) { + console.error(JSON.stringify({ report, errors }, null, 2)); + process.exit(1); + } +} + +export function runBaselineCheck(report, baselinePath) { + const baseline = JSON.parse(fs.readFileSync(baselinePath, 'utf-8')); + const errors = []; + if (report.fingerprint !== baseline.fingerprint) { + errors.push(`fingerprint drift: ${report.fingerprint} != ${baseline.fingerprint}`); + } + if (report.scaling_ratio > baseline.scaling_budget) { + errors.push(`scaling_ratio ${report.scaling_ratio} > ${baseline.scaling_budget}`); + } + if ( + baseline.widening_overhead_budget !== undefined && + report.widening_overhead > baseline.widening_overhead_budget + ) { + errors.push( + `widening_overhead ${report.widening_overhead} > ${baseline.widening_overhead_budget}`, + ); + } + if (errors.length) { + console.error(JSON.stringify({ report, errors }, null, 2)); + process.exit(1); + } + console.log(JSON.stringify({ ok: true, report }, null, 2)); +} diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index d8831f822..7dcd85c15 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -209,8 +209,10 @@ "_rebaselined_blind_spots_2856": "#2856 blind-spots series: the JS/TS SCOPE queries gained capture rules, so fingerprint drift is expected and additive. Verified before re-baselining by diffing the capture-name sets in both scope queries against origin/main: TypeScript gained exactly @reference.read.identifier (A2 bare-identifier reads in value positions) and @reference.type (R2-2 type references, so a declared contract stops reporting incoming:{}); JavaScript gained exactly @reference.read.identifier, @reference.read.destructured (R2-1c) and @reference.write.property-key (R2-1b record-construction writes). NOTHING was removed on either side \u2014 the delta is a pure superset, which is the check that no existing capture moved. capture_groups_small/large are unchanged (4503/14403) because those measure the SYNTHETIC scaling source, which this branch does not touch; only the fixture-corpus count moves. capture_groups_fp 2097 -> 2338 and fixture_count 146 -> 151 from 21 new lang-resolution fixtures. Scaling stayed linear and inside budget: typescript 1.116 < 1.5, javascript 1.010 < 1.5. Prior typescript ed92588e0fc7b28b3a0174339ac378b4dd85965fe007db1208dea97a65ce0571 -> f66a3e6f1e096431e7046505129a627deaa00ca0de5bc846b080591b397248f7; prior javascript 806f70ad3cce5fc849f6d06a08ace8a95f92a1ea84a2418fddabb1eef5846594 -> 2026993b81b873839dd2ef8797d9c14d9c48516b2b57b05ac17d8d43f2f4eba3." }, "kotlin": { - "fingerprint": "f98e7e936afbce0e99588285cfc603bf945fd58c5de45271860509a5d90eb832", + "fingerprint": "aeafc7a87402c933786ef582b7c98683b1822b78fa909e605cb97552867fa0d5", "scaling_budget": 1.5, + "_rebaselined_interface_abstract_2885": "#2885: Kotlin interface property accessors stay in the capture set (groups still 5753/18403 and capture_groups_fp 2563) but Method isAbstract is now true for body-less interface properties, which changes accessor-plan identity in the fixture digest. Prior 82ae5e1f750580383344d4c84c400a290474528cd502be4af8cd56705819a683 -> aeafc7a87402c933786ef582b7c98683b1822b78fa909e605cb97552867fa0d5; CI scaling 0.838 < 1.5.", + "_rebaselined_jvm_property_accessors_2885": "#2885: Kotlin val/var properties now emit JVM getter/setter scope and declaration captures, including data-class constructor properties and custom accessors. Synthetic scaling counts move 4753/15203 -> 5753/18403; fixture-corpus groups move 2367 -> 2563. Accessor declaration sidecars use the canonical @declaration.qualified_name key, preserve same-name owner identity, follow JvmAbi is-prefix naming, and suppress @JvmName-renamed accessors until their custom names are modeled. Prior f98e7e936afbce0e99588285cfc603bf945fd58c5de45271860509a5d90eb832 -> 82ae5e1f750580383344d4c84c400a290474528cd502be4af8cd56705819a683; scaling 0.869 < 1.5.", "_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior bddba25d5a88152bbbee8d70e82c944b5302accb4b625df782adb1d4f7a7ac12 -> e856951c2a779163d555dadc8e1bf59304a86caed78ac1f450d9caa2b50f63d1; scaling 1.090 < 1.5.", "_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Kotlin callable-reference flow facts with invocation-result suppression. Prior 4900431791f2b9280009deb2b82659c26ead8aa6fb8731190a7c505dec5a9041 -> bddba25d5a88152bbbee8d70e82c944b5302accb4b625df782adb1d4f7a7ac12; scaling 0.880 < 1.5.", "_added": "#1951: bench coverage added (was ungated); scale source heritage-bearing (: Base()); js/kotlin O(n^2) findNodeAtRange-per-match fixed to threaded captured node, now linear.", @@ -223,9 +225,9 @@ "_rebaselined_2766_receiver_chain_wire_v2": "#2766: receiver-chain wire format v1 -> v2 (name-free `await` / `index` step kinds). The VERSION prefix is part of every emitted `@reference.receiver-chain` capture, so every chain-minting language's capture text changed. WIRE-FORMAT CHANGE, NOT A CAPTURE-SET CHANGE: the same chains are minted for the same sites, spelled `2|\u2026` instead of `1|\u2026`. Exactly the 12 chain-minting languages drifted; c, cobol and dart did not, which is the check that this is the prefix and not a capture regression. Accompanied by SCHEMA_BUMP 34 -> 37 and INCREMENTAL_SCHEMA_VERSION 28 -> 31 so a stale index is rejected rather than replaying chains a v2 decoder refuses. Prior d3c4d2fa0d82d248a2299cfc888b067187ad1faf2c87a97f93c6ed835eefc3f1 -> c1f0cc9058ab11b7cd6fc8b440deb6db2b2f530f2eb21178923e68a3d0796c4b.", "_rebaselined_2766_await_subscript_emission": "#2766: extractMixedChain now walks THROUGH await and subscript nodes and peels transparent wrappers at loop entry, so sites whose receiver is `repos[0]` or `(await f())` mint a receiver chain where they previously minted none. EMISSION CHANGE: more sites carry `@reference.receiver-chain`; no existing chain changed shape. Only go and kotlin drifted of 15 \u2014 the two whose fixture corpora contain such receivers. Prior c1f0cc9058ab11b7cd6fc8b440deb6db2b2f530f2eb21178923e68a3d0796c4b -> efd5dbf80ffcd3bab2834d1010f6fe2b239dcc5d58229938dea9cff8d0f380f2.", "_rebaselined_2960_declared_package_fixture": "#2960 adds four Kotlin declared-package import-resolution fixture files. This is fixture-corpus growth only: fixture_count 137 -> 141 and capture_groups_fp 2334 -> 2367; the synthetic capture counts remain 4753/15203, no Kotlin scope-capture query or implementation changed, and package resolution runs after capture. Other language fingerprints matched their baselines in the same CI run. Prior a184f8ff0ae40d246db855b63f7ff26bda3afac03e5f4c76e4593c7e2cefce54 -> f98e7e936afbce0e99588285cfc603bf945fd58c5de45271860509a5d90eb832.", - "capture_groups_small": 4753, - "capture_groups_large": 15203, - "capture_groups_fp": 2367, + "capture_groups_small": 5753, + "capture_groups_large": 18403, + "capture_groups_fp": 2563, "fixture_count": 141 } } diff --git a/gitnexus/src/core/ingestion/language-provider.ts b/gitnexus/src/core/ingestion/language-provider.ts index 2775cf07b..a835c2105 100644 --- a/gitnexus/src/core/ingestion/language-provider.ts +++ b/gitnexus/src/core/ingestion/language-provider.ts @@ -402,6 +402,54 @@ interface LanguageProviderConfig { filePath: string, ) => SharedSpringType[]; + /** + * Optional post-capture emission of synthetic structure members (nodes, + * symbols, ownership edges) that have no AST method node — e.g. Lombok + * accessors. Called once per file after the capture loop, at the same + * post-capture site as {@link extractDecoratorRoutes}. + * + * `classOwnersByNodeId` maps in-memory tree-sitter node ids of type + * declarations materialized in THIS file's capture loop to their graph + * node ids. Keys are never persisted; they exist only for the duration + * of the worker pass. + * + * Default: undefined (no synthetic structure members). + */ + readonly synthesizeStructureMembers?: ( + tree: Parser.Tree, + filePath: string, + classOwnersByNodeId: ReadonlyMap, + ) => { + nodes: ReadonlyArray<{ + id: string; + label: string; + properties: Record; + }>; + symbols: ReadonlyArray<{ + filePath: string; + name: string; + nodeId: string; + type: string; + ownerId?: string; + parameterCount?: number; + requiredParameterCount?: number; + parameterTypes?: string[]; + returnType?: string; + visibility?: string; + isStatic?: boolean; + isAbstract?: boolean; + isFinal?: boolean; + }>; + relationships: ReadonlyArray<{ + id: string; + sourceId: string; + targetId: string; + type: string; + confidence: number; + reason: string; + }>; + }; + /** * Harvest this file's module-level string constants (#2391 core, #2980 Java * parity) into the language-agnostic {@link ModuleConstants} shape, so the diff --git a/gitnexus/src/core/ingestion/languages/java.ts b/gitnexus/src/core/ingestion/languages/java.ts index 0fddb6657..25b1daa18 100644 --- a/gitnexus/src/core/ingestion/languages/java.ts +++ b/gitnexus/src/core/ingestion/languages/java.ts @@ -38,6 +38,7 @@ import { javaRecordMethodExtractor, shouldSkipJavaRecordComponentDefinition, } from './java/record-components.js'; +import { synthesizeLombokAccessors } from './java/lombok-synthesizer.js'; import { emitJavaScopeCaptures, interpretJavaImport, @@ -222,6 +223,8 @@ export const javaProvider = defineLanguage({ extractDecoratorRoutes: extractSpringRoutes, extractRouteInheritanceTypes: extractSpringTypes, + synthesizeStructureMembers: synthesizeLombokAccessors, + // ── #2980: constant harvest + qualified-ref fold for non-literal mapping // paths (`@PostMapping(ApiPaths.SAVE_V1)`) — kept behind provider hooks so // the shared ingestion layers stay language-agnostic. The heuristic is diff --git a/gitnexus/src/core/ingestion/languages/java/captures.ts b/gitnexus/src/core/ingestion/languages/java/captures.ts index 22083e6e5..2c3c2c281 100644 --- a/gitnexus/src/core/ingestion/languages/java/captures.ts +++ b/gitnexus/src/core/ingestion/languages/java/captures.ts @@ -59,6 +59,7 @@ import { type JavaSpringNonHttpHandlerFact, } from './spring-non-http-handlers.js'; import { synthesizeJavaRecordComponentAccessorCaptures } from './record-components.js'; +import { synthesizeLombokAccessorCaptures } from './lombok-synthesizer.js'; /** Declaration anchors that carry function-like arity metadata. */ const FUNCTION_DECL_TAGS = ['@declaration.method', '@declaration.constructor'] as const; @@ -422,6 +423,7 @@ export function emitJavaScopeCaptures( ...synthesizeJavaExplicitConstructorReferences(tree.rootNode), ...synthesizeJavaAnonymousClassDeclarations(tree.rootNode), ...synthesizeJavaRecordComponentAccessorCaptures(tree.rootNode), + ...synthesizeLombokAccessorCaptures(tree.rootNode), ...synthesizeCallableFlowCaptures(tree.rootNode, JAVA_CALLABLE_CAPTURE_OPTIONS), ]; } diff --git a/gitnexus/src/core/ingestion/languages/java/lombok-synthesizer.ts b/gitnexus/src/core/ingestion/languages/java/lombok-synthesizer.ts new file mode 100644 index 000000000..3f2eeab9a --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/java/lombok-synthesizer.ts @@ -0,0 +1,539 @@ +/** + * Lombok accessor synthesizer for Java. + * + * Lombok generates getters/setters at compile time. They are absent from the + * AST, so calls like `obj.getOrderId()` on a `@Data` class would otherwise + * leave unresolved CALLS edges. This module walks the tree-sitter Java AST + * and synthesizes Method graph members for the accessors Lombok would emit + * under the supported subset. + * + * ## Supported subset (v1) + * - Proven `lombok.Data` / `lombok.Getter` / `lombok.Setter` (FQN or import). + * - Class- or field-level enable; `AccessLevel.NONE` disables. + * - Default JavaBeans naming; primitive `boolean isX` → `isX` / `setX`. + * - Access levels PUBLIC/PROTECTED/PRIVATE/PACKAGE. + * - `@Accessors(chain=true)` modeled as setter return = declaring type. + * - `@Accessors(fluent=true)` / `prefix=…`: omit affected accessors (names + * cannot be proven without full Lombok config). + * - External `lombok.config`: unsupported (may change semantics invisibly). + * + * ## Identity + * Owner lookup uses in-memory AST node ids only. Method ids are derived from + * the stable declaring-owner graph key (the Class node id's name segment), + * never from persisted tree-sitter node ids. + */ + +import type Parser from 'tree-sitter'; +import type { CaptureMatch } from 'gitnexus-shared'; +import { jvmGetterName, jvmSetterName } from '../jvm/beanspec.js'; +import { + createExistingMethodIndex, + createJvmAccessorSynthesis, + hasExistingMethod, + rememberExistingMethodRange, + type ExistingMethodIndex, + type PlannedJvmAccessor, + type PlannedJvmAccessorOwner, + type SyntheticAccessorResult, + type SyntheticVisibility, +} from '../jvm/accessor-synthesis.js'; + +const JAVA_TYPE_DECLS = new Set([ + 'class_declaration', + 'enum_declaration', + 'interface_declaration', + 'record_declaration', +]); + +// ── Public result types (ParsedSymbol / ParsedNode compatible) ──────────── + +export type LombokVisibility = SyntheticVisibility; +export type SyntheticSymbol = SyntheticAccessorResult['symbols'][number]; +export type SyntheticNode = SyntheticAccessorResult['nodes'][number]; +export type SyntheticRelationship = SyntheticAccessorResult['relationships'][number]; +export type LombokSynthesisResult = SyntheticAccessorResult; +export type PlannedLombokAccessor = PlannedJvmAccessor; + +export interface AccessorConfig { + enabled: boolean; + visibility: LombokVisibility; +} + +interface AccessorsOptions { + /** When true, JavaBeans get/set/is prefixes are not used — omit (unsupported). */ + fluent: boolean; + /** When true, field prefixes alter base names — omit (unsupported). */ + hasPrefix: boolean; + /** When true, setters return the declaring type instead of void. */ + chain: boolean; +} + +interface LombokField { + name: string; + type: string; + isStatic: boolean; + isFinal: boolean; + startLine: number; + endLine: number; + declaratorNode: Parser.SyntaxNode; + fieldGetter: AccessorConfig | null; + fieldSetter: AccessorConfig | null; + accessors: AccessorsOptions; + accessorsPresent: boolean; +} + +interface LombokClass { + node: Parser.SyntaxNode; + name: string; + classGetter: AccessorConfig | null; + classSetter: AccessorConfig | null; + classAccessors: AccessorsOptions; + fields: LombokField[]; + existingMethods: ExistingMethodIndex; +} + +const LOMBOK_ANNOTATION_PACKAGE = new Map([ + ['Data', 'lombok'], + ['Getter', 'lombok'], + ['Setter', 'lombok'], + ['Accessors', 'lombok.experimental'], + ['Tolerate', 'lombok.experimental'], +]); + +export function getterName(fieldName: string, fieldType: string): string { + return jvmGetterName(fieldName, fieldType === 'boolean'); +} + +export function setterName(fieldName: string, fieldType: string): string { + return jvmSetterName(fieldName, fieldType === 'boolean'); +} + +// ── Provenance / imports ────────────────────────────────────────────────── + +function annotationSimpleName(nameText: string): string { + return nameText.split('.').pop() ?? nameText; +} + +interface LombokImportIndex { + bySimple: Map; + starPackages: Set; + shadowedSimpleNames: Set; +} + +/** + * Compilation-unit imports only — Java `import` is never nested in a type body. + */ +function collectLombokImports(root: Parser.SyntaxNode): LombokImportIndex { + const bySimple = new Map(); + const starPackages = new Set(); + const shadowedSimpleNames = new Set(); + for (const child of root.children) { + if (!JAVA_TYPE_DECLS.has(child.type) && child.type !== 'annotation_type_declaration') continue; + const name = child.childForFieldName('name')?.text; + if (name) shadowedSimpleNames.add(name); + } + for (const child of root.children) { + if (child.type !== 'import_declaration') continue; + if (/^import\s+static\b/.test(child.text)) continue; + const text = child.text + .replace(/^import\s+/, '') + .replace(/;\s*$/, '') + .replace(/\/\*[\s\S]*?\*\//g, '') + .replace(/\s+/g, '') + .trim(); + if (text === 'lombok.*') { + starPackages.add('lombok'); + } else if (text === 'lombok.experimental.*') { + starPackages.add('lombok.experimental'); + } else if (!text.endsWith('.*')) { + bySimple.set(annotationSimpleName(text), text); + } + } + return { bySimple, starPackages, shadowedSimpleNames }; +} + +function isProvenLombokAnnotation(nameText: string, imports: LombokImportIndex): boolean { + const simple = annotationSimpleName(nameText); + const packageName = LOMBOK_ANNOTATION_PACKAGE.get(simple); + if (packageName === undefined) return false; + if (nameText.includes('.')) return nameText === `${packageName}.${simple}`; + const imported = imports.bySimple.get(simple); + if (imported !== undefined) return imported === `${packageName}.${simple}`; + if (imports.shadowedSimpleNames.has(simple)) return false; + return imports.starPackages.has(packageName); +} + +// ── AccessLevel / Accessors structural parse ────────────────────────────── + +function parseAccessLevelToken(text: string): LombokVisibility | 'none' | null { + const simple = annotationSimpleName(text.trim()); + switch (simple) { + case 'PUBLIC': + return 'public'; + case 'PROTECTED': + return 'protected'; + case 'PRIVATE': + return 'private'; + case 'PACKAGE': + case 'MODULE': // treated as package-private for graph metadata + return 'package'; + case 'NONE': + return 'none'; + default: + return null; + } +} + +function findAccessLevelInAnnotation(ann: Parser.SyntaxNode): LombokVisibility | 'none' | null { + // Positional: @Getter(AccessLevel.PROTECTED) or @Getter(lombok.AccessLevel.NONE) + // Named: @Getter(value = AccessLevel.PRIVATE) + const stack: Parser.SyntaxNode[] = [...ann.children]; + while (stack.length > 0) { + const n = stack.pop(); + if (!n) break; + if (n.type === 'field_access' || n.type === 'identifier') { + const level = parseAccessLevelToken(n.text); + if (level !== null) return level; + } + for (const c of n.children) stack.push(c); + } + return null; +} + +function defaultAccessors(): AccessorsOptions { + return { fluent: false, hasPrefix: false, chain: false }; +} + +function parseAccessorsAnnotation(ann: Parser.SyntaxNode): AccessorsOptions { + const opts = defaultAccessors(); + const stack: Parser.SyntaxNode[] = [...ann.children]; + while (stack.length > 0) { + const n = stack.pop(); + if (!n) break; + if (n.type === 'element_value_pair') { + const key = + n.childForFieldName('key')?.text ?? n.children.find((c) => c.type === 'identifier')?.text; + const valueNode = + n.childForFieldName('value') ?? + n.children.find( + (c) => + c.type === 'true' || c.type === 'false' || c.type === 'element_value_array_initializer', + ); + if (key === 'fluent' && (valueNode?.type === 'true' || valueNode?.type === 'false')) { + opts.fluent = valueNode.type === 'true'; + } + if (key === 'chain' && (valueNode?.type === 'true' || valueNode?.type === 'false')) { + opts.chain = valueNode.type === 'true'; + } + if (key === 'prefix') opts.hasPrefix = true; + } + for (const c of n.children) stack.push(c); + } + const text = ann.text; + if (/\bprefix\s*=/.test(text)) opts.hasPrefix = true; + if (/\bfluent\s*=\s*true\b/.test(text)) opts.fluent = true; + if (/\bfluent\s*=\s*false\b/.test(text)) opts.fluent = false; + if (/\bchain\s*=\s*true\b/.test(text)) opts.chain = true; + if (/\bchain\s*=\s*false\b/.test(text)) opts.chain = false; + return opts; +} + +interface ParsedAnnotations { + getter: AccessorConfig | null; + setter: AccessorConfig | null; + accessors: AccessorsOptions; + accessorsPresent: boolean; + tolerate: boolean; +} + +function parseModifierAnnotations( + modifiersNode: Parser.SyntaxNode | null, + imports: LombokImportIndex, +): ParsedAnnotations { + const result: ParsedAnnotations = { + getter: null, + setter: null, + accessors: defaultAccessors(), + accessorsPresent: false, + tolerate: false, + }; + if (!modifiersNode) return result; + + for (const child of modifiersNode.children) { + if (child.type !== 'marker_annotation' && child.type !== 'annotation') continue; + const nameNode = child.childForFieldName('name'); + const nameText = nameNode?.text ?? ''; + if (!isProvenLombokAnnotation(nameText, imports)) continue; + const simple = annotationSimpleName(nameText); + + if (simple === 'Tolerate') { + result.tolerate = true; + continue; + } + if (simple === 'Accessors') { + result.accessors = parseAccessorsAnnotation(child); + result.accessorsPresent = true; + continue; + } + if (simple === 'Data') { + result.getter ??= { enabled: true, visibility: 'public' }; + result.setter ??= { enabled: true, visibility: 'public' }; + continue; + } + if (simple === 'Getter' || simple === 'Setter') { + const level = child.type === 'annotation' ? findAccessLevelInAnnotation(child) : null; + const cfg: AccessorConfig = + level === 'none' + ? { enabled: false, visibility: 'public' } + : { enabled: true, visibility: level ?? 'public' }; + if (simple === 'Getter') result.getter = cfg; + else result.setter = cfg; + } + } + return result; +} + +function mergeAccessors( + classOpts: AccessorsOptions, + fieldOpts: AccessorsOptions, + fieldAccessorsPresent: boolean, +): AccessorsOptions { + return fieldAccessorsPresent ? fieldOpts : classOpts; +} + +function effectiveAccessor( + classCfg: AccessorConfig | null, + fieldCfg: AccessorConfig | null, +): AccessorConfig | null { + if (fieldCfg !== null) return fieldCfg; + return classCfg; +} + +// ── Field / method collection ───────────────────────────────────────────── + +function parseFieldDeclaration( + fieldNode: Parser.SyntaxNode, + imports: LombokImportIndex, +): LombokField[] { + const typeNode = fieldNode.childForFieldName('type'); + const fieldType = typeNode?.text ?? 'Object'; + const modifiers = fieldNode.children.find((c) => c.type === 'modifiers') ?? null; + let isStatic = false; + let isFinal = false; + if (modifiers) { + for (const mod of modifiers.children) { + if (mod.text === 'static') isStatic = true; + else if (mod.text === 'final') isFinal = true; + } + } + const fieldAnn = parseModifierAnnotations(modifiers, imports); + + const declarators: Parser.SyntaxNode[] = []; + const declaratorField = fieldNode.childForFieldName('declarator'); + if (declaratorField) declarators.push(declaratorField); + for (const child of fieldNode.children) { + if (child.type === 'variable_declarator' && child !== declaratorField) { + declarators.push(child); + } + } + + const startLine = fieldNode.startPosition.row + 1; + const endLine = fieldNode.endPosition.row + 1; + const out: LombokField[] = []; + for (const declaratorNode of declarators) { + const nameNode = declaratorNode.childForFieldName('name'); + if (!nameNode) continue; + out.push({ + name: nameNode.text, + type: fieldType, + isStatic, + isFinal, + startLine, + endLine, + declaratorNode, + fieldGetter: fieldAnn.getter, + fieldSetter: fieldAnn.setter, + accessors: fieldAnn.accessors, + accessorsPresent: fieldAnn.accessorsPresent, + }); + } + return out; +} + +function methodArityRange(methodNode: Parser.SyntaxNode): { min: number; max: number } { + const params = methodNode.childForFieldName('parameters'); + if (!params) return { min: 0, max: 0 }; + let count = 0; + for (const child of params.namedChildren) { + if (child.type === 'spread_parameter') return { min: count, max: Number.POSITIVE_INFINITY }; + if (child.type === 'formal_parameter') count += 1; + } + return { min: count, max: count }; +} + +function collectExistingMethods( + classBody: Parser.SyntaxNode | null, + imports: LombokImportIndex, +): ExistingMethodIndex { + const index = createExistingMethodIndex('case-folded'); + if (!classBody) return index; + const scan = (container: Parser.SyntaxNode): void => { + for (const child of container.children) { + if (child.type === 'enum_body_declarations') { + scan(child); + continue; + } + if (child.type !== 'method_declaration') continue; + const mods = child.children.find((c) => c.type === 'modifiers') ?? null; + const ann = parseModifierAnnotations(mods, imports); + if (ann.tolerate) continue; + const nameNode = child.childForFieldName('name'); + if (!nameNode) continue; + const arity = methodArityRange(child); + rememberExistingMethodRange(index, nameNode.text, arity.min, arity.max); + } + }; + scan(classBody); + return index; +} + +const TYPE_BODIES = new Set(['class_body', 'enum_body']); + +function findTypeBody(node: Parser.SyntaxNode): Parser.SyntaxNode | null { + return node.children.find((c) => TYPE_BODIES.has(c.type)) ?? null; +} + +function findLombokClasses(root: Parser.SyntaxNode, imports: LombokImportIndex): LombokClass[] { + const classes: LombokClass[] = []; + + function walk(node: Parser.SyntaxNode): void { + if (node.type === 'class_declaration' || node.type === 'enum_declaration') { + const modifiers = node.children.find((c) => c.type === 'modifiers') ?? null; + const classAnn = parseModifierAnnotations(modifiers, imports); + const nameNode = node.childForFieldName('name'); + const className = nameNode?.text ?? ''; + if (className) { + const body = findTypeBody(node); + const fields: LombokField[] = []; + if (body) { + const collectFields = (container: Parser.SyntaxNode): void => { + for (const child of container.children) { + if (child.type === 'field_declaration') { + for (const f of parseFieldDeclaration(child, imports)) { + if (f.isStatic) continue; + fields.push(f); + } + } else if (child.type === 'enum_body_declarations') { + collectFields(child); + } + } + }; + collectFields(body); + } + + const anyFieldEnable = fields.some( + (f) => f.fieldGetter?.enabled === true || f.fieldSetter?.enabled === true, + ); + const classEnable = classAnn.getter?.enabled === true || classAnn.setter?.enabled === true; + + // Class-level NONE alone is not enable — getter/setter configs may be disabled + if (classEnable || anyFieldEnable) { + classes.push({ + node, + name: className, + classGetter: classAnn.getter, + classSetter: classAnn.setter, + classAccessors: classAnn.accessors, + fields, + existingMethods: collectExistingMethods(body, imports), + }); + } + } + } + for (const child of node.children) walk(child); + } + + walk(root); + return classes; +} + +function planAccessors(cls: LombokClass): PlannedLombokAccessor[] { + const planned: PlannedLombokAccessor[] = []; + for (const field of cls.fields) { + const accessors = mergeAccessors(cls.classAccessors, field.accessors, field.accessorsPresent); + // fluent/prefix change names — omit rather than invent wrong names + if (accessors.fluent || accessors.hasPrefix) continue; + + const getterCfg = effectiveAccessor(cls.classGetter, field.fieldGetter); + const setterCfg = effectiveAccessor(cls.classSetter, field.fieldSetter); + + if (getterCfg?.enabled) { + const gName = getterName(field.name, field.type); + if (!hasExistingMethod(cls.existingMethods, gName, 0)) { + planned.push({ + kind: 'getter', + name: gName, + returnType: field.type, + parameterTypes: [], + visibility: getterCfg.visibility, + isStatic: false, + isAbstract: false, + startLine: field.startLine, + endLine: field.endLine, + declaratorNode: field.declaratorNode, + }); + } + } + + if (setterCfg?.enabled && !field.isFinal) { + const sName = setterName(field.name, field.type); + if (!hasExistingMethod(cls.existingMethods, sName, 1)) { + // chain=true → setter returns declaring type; never emit void in that case + const returnType = accessors.chain ? cls.name : 'void'; + planned.push({ + kind: 'setter', + name: sName, + returnType, + parameterTypes: [field.type], + visibility: setterCfg.visibility, + isStatic: false, + isAbstract: false, + startLine: field.startLine, + endLine: field.endLine, + declaratorNode: field.declaratorNode, + }); + } + } + } + return planned; +} + +function planLombokAccessorOwners(root: Parser.SyntaxNode): PlannedJvmAccessorOwner[] { + const imports = collectLombokImports(root); + return findLombokClasses(root, imports).map((cls) => ({ + node: cls.node, + name: cls.name, + accessors: planAccessors(cls), + })); +} + +const lombokAccessorSynthesis = createJvmAccessorSynthesis({ + language: 'java', + synthetic: 'lombok', + planOwners: planLombokAccessorOwners, +}); + +// ── Main API ────────────────────────────────────────────────────────────── + +export function synthesizeLombokAccessors( + tree: Parser.Tree, + filePath: string, + classOwnersById: ReadonlyMap, +): LombokSynthesisResult { + return lombokAccessorSynthesis.synthesize(tree, filePath, classOwnersById); +} + +/** Scope captures for Lombok accessors (dual-path parity with record components). */ +export function synthesizeLombokAccessorCaptures(rootNode: Parser.SyntaxNode): CaptureMatch[] { + return lombokAccessorSynthesis.captures(rootNode); +} diff --git a/gitnexus/src/core/ingestion/languages/jvm/accessor-synthesis.ts b/gitnexus/src/core/ingestion/languages/jvm/accessor-synthesis.ts new file mode 100644 index 000000000..e1cb89253 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/jvm/accessor-synthesis.ts @@ -0,0 +1,316 @@ +/** + * Shared planning orchestration and emission for synthetic JVM accessors. + * + * Language adapters discover accessor plans. This module owns method-collision + * policy, graph emission, and scope captures without naming any language. + */ +import type Parser from 'tree-sitter'; +import type { Capture, CaptureMatch } from 'gitnexus-shared'; +import { toZeroBasedLine } from '../../utils/line-base.js'; + +export type SyntheticVisibility = 'public' | 'protected' | 'private' | 'package'; +export type MethodNameMatching = 'exact' | 'case-folded'; + +export interface ExistingMethodIndex { + readonly matching: MethodNameMatching; + readonly aritiesByName: Map>; + readonly arityRangesByName: Map>; +} + +export function createExistingMethodIndex(matching: MethodNameMatching): ExistingMethodIndex { + return { matching, aritiesByName: new Map(), arityRangesByName: new Map() }; +} + +function methodKey(index: ExistingMethodIndex, name: string): string { + return index.matching === 'case-folded' ? name.toLowerCase() : name; +} + +export function rememberExistingMethod( + index: ExistingMethodIndex, + name: string, + arity: number, +): void { + const key = methodKey(index, name); + let arities = index.aritiesByName.get(key); + if (!arities) { + arities = new Set(); + index.aritiesByName.set(key, arities); + } + arities.add(arity); +} + +export function rememberExistingMethodRange( + index: ExistingMethodIndex, + name: string, + min: number, + max: number, +): void { + if (min === max) { + rememberExistingMethod(index, name, min); + return; + } + const key = methodKey(index, name); + const ranges = index.arityRangesByName.get(key) ?? []; + ranges.push({ min, max }); + index.arityRangesByName.set(key, ranges); +} + +export function hasExistingMethod( + index: ExistingMethodIndex, + name: string, + arity: number, +): boolean { + const key = methodKey(index, name); + if (index.aritiesByName.get(key)?.has(arity) === true) return true; + return ( + index.arityRangesByName.get(key)?.some((range) => range.min <= arity && arity <= range.max) === + true + ); +} + +export interface SyntheticAccessorSymbol { + filePath: string; + name: string; + nodeId: string; + type: 'Method'; + ownerId: string; + qualifiedName: string; + parameterCount: number; + requiredParameterCount: number; + parameterTypes: string[]; + returnType: string; + visibility: SyntheticVisibility; + isStatic: boolean; + isAbstract: boolean; + isFinal: boolean; +} + +export interface SyntheticAccessorNode { + id: string; + label: 'Method'; + properties: { + name: string; + filePath: string; + startLine: number; + endLine: number; + language: string; + isExported: boolean; + synthetic: string; + visibility: SyntheticVisibility; + isStatic: boolean; + returnType: string; + parameterTypes: string[]; + parameterCount: number; + qualifiedName: string; + }; +} + +export interface SyntheticAccessorRelationship { + id: string; + sourceId: string; + targetId: string; + type: 'HAS_METHOD'; + confidence: number; + reason: string; +} + +export interface SyntheticAccessorResult { + symbols: SyntheticAccessorSymbol[]; + nodes: SyntheticAccessorNode[]; + relationships: SyntheticAccessorRelationship[]; +} + +export interface PlannedJvmAccessor { + kind: 'getter' | 'setter'; + name: string; + returnType: string; + parameterTypes: string[]; + visibility: SyntheticVisibility; + isStatic: boolean; + isAbstract: boolean; + startLine: number; + endLine: number; + declaratorNode: Parser.SyntaxNode; +} + +export interface PlannedJvmAccessorOwner { + node: Parser.SyntaxNode; + name: string; + accessors: readonly PlannedJvmAccessor[]; +} + +interface JvmAccessorSynthesisConfig { + language: string; + synthetic: string; + planOwners(rootNode: Parser.SyntaxNode): readonly PlannedJvmAccessorOwner[]; +} + +export interface JvmAccessorSynthesis { + synthesize( + tree: Parser.Tree, + filePath: string, + classOwnersById: ReadonlyMap, + ): SyntheticAccessorResult; + captures(rootNode: Parser.SyntaxNode): CaptureMatch[]; +} + +export function createJvmAccessorSynthesis( + config: JvmAccessorSynthesisConfig, +): JvmAccessorSynthesis { + return { + synthesize(tree, filePath, classOwnersById) { + const result = emptySyntheticAccessorResult(); + for (const owner of config.planOwners(tree.rootNode)) { + const ownerId = classOwnersById.get(owner.node.id); + if (!ownerId) continue; + emitPlannedAccessors({ + planned: owner.accessors, + filePath, + ownerId, + idPrefix: ownerIdNamePrefix(ownerId, filePath, owner.name), + language: config.language, + synthetic: config.synthetic, + result, + }); + } + return result; + }, + captures(rootNode) { + return capturesForPlannedAccessors(config.planOwners(rootNode)); + }, + }; +} + +function emptySyntheticAccessorResult(): SyntheticAccessorResult { + return { symbols: [], nodes: [], relationships: [] }; +} + +function ownerIdNamePrefix(ownerId: string, filePath: string, fallback: string): string { + const needle = `Class:${filePath}:`; + if (ownerId.startsWith(needle)) return ownerId.slice(needle.length); + const enumNeedle = `Enum:${filePath}:`; + if (ownerId.startsWith(enumNeedle)) return ownerId.slice(enumNeedle.length); + const ifaceNeedle = `Interface:${filePath}:`; + if (ownerId.startsWith(ifaceNeedle)) return ownerId.slice(ifaceNeedle.length); + return fallback; +} + +export function jvmTypeSimpleName(node: Parser.SyntaxNode): string | undefined { + const named = node.childForFieldName('name')?.text; + if (named) return named; + for (const child of node.namedChildren) { + if (child.type === 'type_identifier' || child.type === 'simple_identifier') return child.text; + } + return undefined; +} + +function emitPlannedAccessors(args: { + planned: readonly PlannedJvmAccessor[]; + filePath: string; + ownerId: string; + idPrefix: string; + language: string; + synthetic: string; + result: SyntheticAccessorResult; +}): void { + const emittedIds = new Set(); + for (const acc of args.planned) { + const arity = acc.parameterTypes.length; + const qualifiedName = `${args.idPrefix}.${acc.name}`; + const nodeId = `Method:${args.filePath}:${qualifiedName}#${arity}`; + if (emittedIds.has(nodeId)) continue; + emittedIds.add(nodeId); + args.result.nodes.push({ + id: nodeId, + label: 'Method', + properties: { + name: acc.name, + filePath: args.filePath, + startLine: toZeroBasedLine(acc.startLine), + endLine: toZeroBasedLine(acc.endLine), + language: args.language, + isExported: false, + synthetic: args.synthetic, + visibility: acc.visibility, + isStatic: acc.isStatic, + returnType: acc.returnType, + parameterTypes: acc.parameterTypes, + parameterCount: arity, + qualifiedName, + }, + }); + args.result.symbols.push({ + filePath: args.filePath, + name: acc.name, + nodeId, + type: 'Method', + ownerId: args.ownerId, + qualifiedName, + parameterCount: arity, + requiredParameterCount: arity, + parameterTypes: acc.parameterTypes, + returnType: acc.returnType, + visibility: acc.visibility, + isStatic: acc.isStatic, + isAbstract: acc.isAbstract, + isFinal: false, + }); + args.result.relationships.push({ + id: `HAS_METHOD:${args.ownerId}->${nodeId}`, + sourceId: args.ownerId, + targetId: nodeId, + type: 'HAS_METHOD', + confidence: 1.0, + reason: acc.kind === 'getter' ? `${args.synthetic}-getter` : `${args.synthetic}-setter`, + }); + } +} + +function accessorCapture(name: string, acc: PlannedJvmAccessor, text: string): Capture { + const node = acc.declaratorNode; + const startLine = node.startPosition.row + 1; + const startCol = node.startPosition.column; + const endLine = node.endPosition.row + 1; + const endCol = acc.kind === 'getter' ? node.endPosition.column : startCol; + return { name, range: { startLine, startCol, endLine, endCol }, text }; +} + +function capturesForPlannedAccessors(owners: readonly PlannedJvmAccessorOwner[]): CaptureMatch[] { + const captures: CaptureMatch[] = []; + for (const owner of owners) { + const enclosing = owner.name; + const emitted = new Set(); + for (const acc of owner.accessors) { + const arity = String(acc.parameterTypes.length); + const qualifiedName = `${enclosing}.${acc.name}`; + const identity = `${qualifiedName}#${arity}`; + if (emitted.has(identity)) continue; + emitted.add(identity); + captures.push({ + '@scope.function': accessorCapture('@scope.function', acc, acc.name), + }); + captures.push({ + '@declaration.method': accessorCapture('@declaration.method', acc, acc.name), + '@declaration.name': accessorCapture('@declaration.name', acc, acc.name), + '@declaration.qualified_name': accessorCapture( + '@declaration.qualified_name', + acc, + qualifiedName, + ), + '@declaration.parameter-count': accessorCapture('@declaration.parameter-count', acc, arity), + '@declaration.required-parameter-count': accessorCapture( + '@declaration.required-parameter-count', + acc, + arity, + ), + '@declaration.return-type': accessorCapture( + '@declaration.return-type', + acc, + acc.returnType, + ), + '@declaration.is-synthetic': accessorCapture('@declaration.is-synthetic', acc, 'true'), + }); + } + } + return captures; +} diff --git a/gitnexus/src/core/ingestion/languages/jvm/beanspec.ts b/gitnexus/src/core/ingestion/languages/jvm/beanspec.ts new file mode 100644 index 000000000..0a14209f5 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/jvm/beanspec.ts @@ -0,0 +1,49 @@ +/** + * Language-neutral JVM JavaBeans naming primitives. + * + * Language adapters choose whether to invent/preserve an `is` prefix and + * which single-character capitalization policy their compiler uses. + */ + +export function capitalizeBeanName(s: string): string { + if (s.length === 0) return s; + const first = s.charAt(0); + const upper = first.toUpperCase(); + // Java Character case conversion is one UTF-16 code unit. JavaScript + // full-case conversion may expand one unit (`ß` → `SS`), which would invent + // a method name no JVM compiler emits. + return (upper.length === 1 ? upper : first) + s.slice(1); +} + +/** + * Primitive-boolean / Kotlin `is`-prefix fields whose name already starts with + * `is` plus a non-lowercase character keep that name for the getter and drop + * the `is` prefix for the setter base (`isEnabled` → `isEnabled()` / + * `setEnabled(...)`, `is1` → `is1()` / `set1(...)`). Digits and punctuation + * count as non-lowercase, matching Lombok `!Character.isLowerCase` and kotlinc. + */ +export function booleanIsPrefixBase(fieldName: string, useIsPrefix: boolean): string | null { + if (!useIsPrefix || !fieldName.startsWith('is') || fieldName.length < 3) return null; + const third = fieldName.charAt(2); + return third === third.toUpperCase() ? fieldName.slice(2) : null; +} + +export function jvmGetterName( + fieldName: string, + useIsPrefix: boolean, + capitalize: (name: string) => string = capitalizeBeanName, +): string { + if (booleanIsPrefixBase(fieldName, useIsPrefix) !== null) return fieldName; + if (useIsPrefix) return `is${capitalize(fieldName)}`; + return `get${capitalize(fieldName)}`; +} + +export function jvmSetterName( + fieldName: string, + useIsPrefix: boolean, + capitalize: (name: string) => string = capitalizeBeanName, +): string { + const stripped = booleanIsPrefixBase(fieldName, useIsPrefix); + if (stripped !== null) return `set${stripped}`; + return `set${capitalize(fieldName)}`; +} diff --git a/gitnexus/src/core/ingestion/languages/kotlin.ts b/gitnexus/src/core/ingestion/languages/kotlin.ts index 18d4fd9c1..4107d5430 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin.ts @@ -41,6 +41,7 @@ import { kotlinMergeBindings, kotlinReceiverBinding, } from './kotlin/index.js'; +import { synthesizeLombokAccessors } from './kotlin/lombok-synthesizer.js'; /** Check if a Kotlin function_declaration capture is inside a class_body (i.e., a method). * Kotlin grammar uses function_declaration for both top-level functions and class methods. @@ -202,4 +203,5 @@ export const kotlinProvider = defineLanguage({ mergeBindings: (_scope, bindings) => kotlinMergeBindings(bindings), receiverBinding: kotlinReceiverBinding, arityCompatibility: kotlinArityCompatibility, + synthesizeStructureMembers: synthesizeLombokAccessors, }); diff --git a/gitnexus/src/core/ingestion/languages/kotlin/captures.ts b/gitnexus/src/core/ingestion/languages/kotlin/captures.ts index f3b00db03..6c8d6f6c3 100644 --- a/gitnexus/src/core/ingestion/languages/kotlin/captures.ts +++ b/gitnexus/src/core/ingestion/languages/kotlin/captures.ts @@ -28,6 +28,7 @@ import { } from './capture-side-channel.js'; import { captureKotlinPackageFact } from './package-facts.js'; import { synthesizeCallableFlowCaptures } from '../../utils/callable-flow-captures.js'; +import { synthesizeLombokAccessorCaptures } from './lombok-synthesizer.js'; import { captureKotlinSpringDiClassFact, type KotlinSpringDiClassFact } from './spring-di.js'; import type { SpringDynamicLookupFact } from '../../frameworks/spring/dynamic-lookups.js'; import { captureKotlinSpringDynamicLookupFact } from './spring-dynamic-lookup.js'; @@ -371,6 +372,7 @@ export function emitKotlinScopeCaptures( setKotlinSpringDiFacts(filePath, springDiFacts); setKotlinSpringDynamicLookupFacts(filePath, springDynamicLookupFacts); setKotlinSpringNonHttpHandlerFacts(filePath, springNonHttpHandlerFacts); + out.push(...synthesizeLombokAccessorCaptures(tree.rootNode)); out.push(...synthesizeCallableFlowCaptures(tree.rootNode, KOTLIN_CALLABLE_CAPTURE_OPTIONS)); return out; } diff --git a/gitnexus/src/core/ingestion/languages/kotlin/lombok-synthesizer.ts b/gitnexus/src/core/ingestion/languages/kotlin/lombok-synthesizer.ts new file mode 100644 index 000000000..bb0c3d5b7 --- /dev/null +++ b/gitnexus/src/core/ingestion/languages/kotlin/lombok-synthesizer.ts @@ -0,0 +1,512 @@ +/** + * Kotlin accessor synthesizer (same provider-hook role as Java Lombok). + * + * kotlinc emits JavaBeans getters/setters for `val`/`var` properties. Those + * methods are absent from the tree-sitter AST, so Java (and Kotlin) calls + * like `user.getName()` miss CALLS edges. Planning is Kotlin-specific; + * naming and Method emission share `jvm/beanspec` + `jvm/accessor-synthesis`. + * + * ## Supported subset (v1) + * - Class / data class / object / companion / interface `val`/`var` properties + * (interface accessors without a custom body are abstract JVM methods). + * - Primary-constructor `val`/`var` class parameters. + * - Names beginning with `is` + a non-lowercase character keep that getter name; all other + * properties, including `Boolean`, use `get`. + * - Custom `get()`/`set()` bodies still emit their JVM accessor Methods. + * - Explicit `fun getX` / `@JvmField` / `const` skip synthesis. + * - `@JvmName`-renamed accessors are suppressed until custom-name emission lands. + * Unsupported: `@JvmStatic` renaming, file-facade top-level properties. + */ +import type Parser from 'tree-sitter'; +import type { CaptureMatch } from 'gitnexus-shared'; +import { booleanIsPrefixBase, jvmGetterName, jvmSetterName } from '../jvm/beanspec.js'; +import { + createExistingMethodIndex, + createJvmAccessorSynthesis, + hasExistingMethod, + jvmTypeSimpleName, + rememberExistingMethod, + type ExistingMethodIndex, + type PlannedJvmAccessor, + type PlannedJvmAccessorOwner, + type SyntheticAccessorResult, + type SyntheticVisibility, +} from '../jvm/accessor-synthesis.js'; + +const KOTLIN_TYPE_DECLS = new Set(['class_declaration', 'object_declaration', 'companion_object']); + +function capitalizeAscii(name: string): string { + const first = name.charAt(0); + return first >= 'a' && first <= 'z' + ? String.fromCharCode(first.charCodeAt(0) - 32) + name.slice(1) + : name; +} + +export function kotlinGetterName(propertyName: string): string { + return jvmGetterName( + propertyName, + booleanIsPrefixBase(propertyName, true) !== null, + capitalizeAscii, + ); +} + +export function kotlinSetterName(propertyName: string): string { + return jvmSetterName(propertyName, true, capitalizeAscii); +} + +interface KtProperty { + name: string; + type: string; + isVar: boolean; + skipGetter: boolean; + skipSetter: boolean; + getterVisibility: SyntheticVisibility; + setterVisibility: SyntheticVisibility; + startLine: number; + endLine: number; + propertyNode: Parser.SyntaxNode; + declaratorNode: Parser.SyntaxNode; +} + +interface KtClass { + node: Parser.SyntaxNode; + name: string; + isStatic: boolean; + isInterface: boolean; + wasHoisted: boolean; + properties: KtProperty[]; + existingMethods: ExistingMethodIndex; +} + +interface KotlinImportIndex { + byLocalName: Map; + shadowedSimpleNames: Set; +} + +function collectKotlinImports(root: Parser.SyntaxNode): KotlinImportIndex { + const byLocalName = new Map(); + const shadowedSimpleNames = new Set(); + for (const child of root.children) { + if (child.type !== 'class_declaration') continue; + const name = jvmTypeSimpleName(child); + if (name) shadowedSimpleNames.add(name); + } + const importList = root.children.find((child) => child.type === 'import_list'); + for (const child of importList?.children ?? []) { + if (child.type !== 'import_header') continue; + const text = child.text + .replace(/^import\s+/, '') + .replace(/\/\*[\s\S]*?\*\//g, '') + .trim(); + const [pathText, aliasText] = text.split(/\s+as\s+/, 2); + const importPath = pathText?.replace(/\s+/g, ''); + if (!importPath || importPath.endsWith('.*')) continue; + const localName = aliasText?.trim() || importPath.split('.').pop(); + if (localName) byLocalName.set(localName, importPath); + } + return { byLocalName, shadowedSimpleNames }; +} + +function annotationUserTypeText(annotation: Parser.SyntaxNode): string { + const constructor = annotation.namedChildren.find((c) => c.type === 'constructor_invocation'); + const userType = + constructor?.namedChildren.find((c) => c.type === 'user_type') ?? + annotation.namedChildren.find((c) => c.type === 'user_type'); + return userType?.text ?? ''; +} + +function isKotlinJvmAnnotation( + annotation: Parser.SyntaxNode, + name: string, + imports: KotlinImportIndex, +): boolean { + const typeText = annotationUserTypeText(annotation); + const canonical = `kotlin.jvm.${name}`; + if (typeText.includes('.')) return typeText === canonical; + const imported = imports.byLocalName.get(typeText); + if (imported !== undefined) return imported === canonical; + if (imports.shadowedSimpleNames.has(typeText)) return false; + return typeText === name; +} + +function kotlinVisibility(modifiers: Parser.SyntaxNode | undefined): SyntheticVisibility { + if (!modifiers) return 'public'; + for (const child of modifiers.namedChildren) { + if (child.type !== 'visibility_modifier') continue; + if (child.text === 'private') return 'private'; + if (child.text === 'protected') return 'protected'; + if (child.text === 'internal') return 'package'; + } + return 'public'; +} + +function hasJvmField(node: Parser.SyntaxNode, imports: KotlinImportIndex): boolean { + const mods = node.children.find((c) => c.type === 'modifiers'); + return ( + mods?.namedChildren.some( + (child) => child.type === 'annotation' && isKotlinJvmAnnotation(child, 'JvmField', imports), + ) === true + ); +} + +function hasConst(node: Parser.SyntaxNode): boolean { + const mods = node.children.find((c) => c.type === 'modifiers'); + if ( + mods?.namedChildren.some( + (child) => child.type === 'property_modifier' && child.text === 'const', + ) + ) { + return true; + } + return node.namedChildren.some((child) => child.type === 'const'); +} + +function isVarBinding(node: Parser.SyntaxNode): boolean | null { + const kind = node.children.find((c) => c.type === 'binding_pattern_kind'); + const text = kind?.text; + if (text === 'var') return true; + if (text === 'val') return false; + return null; +} + +function inferredInitializerType(node: Parser.SyntaxNode): string | undefined { + switch (node.type) { + case 'string_literal': + case 'line_string_literal': + case 'multi_line_string_literal': + return 'String'; + case 'character_literal': + return 'Char'; + case 'boolean_literal': + case 'true': + case 'false': + return 'Boolean'; + case 'long_literal': + return 'Long'; + case 'unsigned_literal': + return /l$/i.test(node.text) ? 'ULong' : 'UInt'; + case 'integer_literal': + case 'decimal_integer_literal': + case 'hex_integer_literal': + case 'octal_integer_literal': + case 'binary_integer_literal': + return 'Int'; + case 'real_literal': + case 'decimal_floating_point_literal': + return /f$/i.test(node.text) ? 'Float' : 'Double'; + case 'prefix_expression': { + const operand = node.namedChildren.at(-1); + return operand ? inferredInitializerType(operand) : undefined; + } + case 'call_expression': { + const callee = node.namedChildren.find((child) => child.type === 'simple_identifier'); + if (!callee) return undefined; + const first = callee.text.charAt(0); + return first !== '' && first === first.toUpperCase() ? callee.text : undefined; + } + default: + return undefined; + } +} + +function propertyTypeText(node: Parser.SyntaxNode): string { + const declarator = + node.type === 'class_parameter' + ? node + : (node.children.find((c) => c.type === 'variable_declaration') ?? node); + const colon = declarator.children.find((c) => c.type === ':'); + let typeNode = colon?.nextNamedSibling ?? null; + while (typeNode?.type === 'type_modifiers') typeNode = typeNode.nextNamedSibling; + if (typeNode) return typeNode.text; + const initializer = node.namedChildren.find( + (child) => + child.id !== declarator.id && + child.type !== 'binding_pattern_kind' && + child.type !== 'modifiers', + ); + return initializer ? (inferredInitializerType(initializer) ?? 'unknown') : 'unknown'; +} + +function propertyNameNode(node: Parser.SyntaxNode): Parser.SyntaxNode | null { + if (node.type === 'class_parameter') { + return node.children.find((c) => c.type === 'simple_identifier') ?? null; + } + const decl = node.children.find((c) => c.type === 'variable_declaration'); + if (decl) { + return decl.children.find((c) => c.type === 'simple_identifier') ?? null; + } + return node.children.find((c) => c.type === 'simple_identifier') ?? null; +} + +function accessorMetadata( + prop: Parser.SyntaxNode, + propertyVisibility: SyntheticVisibility, + imports: KotlinImportIndex, +): { + getterVisibility: SyntheticVisibility; + setterVisibility: SyntheticVisibility; + skipGetter: boolean; + skipSetter: boolean; +} { + let getter = propertyVisibility; + let setter = propertyVisibility; + let skipGetter = false; + let skipSetter = false; + const propertyModifiers = prop.children.find((c) => c.type === 'modifiers'); + for (const annotation of propertyModifiers?.namedChildren ?? []) { + if ( + annotation.type !== 'annotation' || + !isKotlinJvmAnnotation(annotation, 'JvmName', imports) + ) { + continue; + } + const target = annotation.children.find((c) => c.type === 'use_site_target')?.text; + if (target === 'get:') skipGetter = true; + if (target === 'set:') skipSetter = true; + } + const apply = (node: Parser.SyntaxNode): void => { + const modifiers = node.children.find((c) => c.type === 'modifiers'); + if (!modifiers) return; + if (node.type === 'getter') getter = kotlinVisibility(modifiers); + if (node.type === 'setter') setter = kotlinVisibility(modifiers); + if ( + modifiers.namedChildren.some((annotation) => + isKotlinJvmAnnotation(annotation, 'JvmName', imports), + ) + ) { + if (node.type === 'getter') skipGetter = true; + if (node.type === 'setter') skipSetter = true; + } + }; + for (const child of prop.children) { + if (child.type === 'getter' || child.type === 'setter') apply(child); + } + let sib: Parser.SyntaxNode | null = prop.nextNamedSibling; + while (sib && (sib.type === 'getter' || sib.type === 'setter')) { + apply(sib); + sib = sib.nextNamedSibling; + } + return { + getterVisibility: getter, + setterVisibility: setter, + skipGetter, + skipSetter, + }; +} + +function hasKotlinAccessorBody(prop: Parser.SyntaxNode, kind: 'getter' | 'setter'): boolean { + const hasBody = (node: Parser.SyntaxNode): boolean => + node.type === kind && node.children.some((child) => child.type === 'function_body'); + if (prop.children.some(hasBody)) return true; + let sib: Parser.SyntaxNode | null = prop.nextNamedSibling; + while (sib && (sib.type === 'getter' || sib.type === 'setter')) { + if (hasBody(sib)) return true; + sib = sib.nextNamedSibling; + } + return false; +} + +function functionName(node: Parser.SyntaxNode): string | undefined { + return node.children.find((c) => c.type === 'simple_identifier')?.text; +} + +function functionArity(node: Parser.SyntaxNode): number { + const params = node.children.find((c) => c.type === 'function_value_parameters'); + let arity = + node.childForFieldName('receiver') !== null || + node.namedChildren.some((child) => child.type === 'receiver_type') + ? 1 + : 0; + const modifiers = node.children.find((child) => child.type === 'modifiers'); + if ( + modifiers?.namedChildren.some( + (child) => child.type === 'function_modifier' && child.text === 'suspend', + ) + ) { + arity += 1; + } + for (const child of params?.namedChildren ?? []) { + if (child.type === 'parameter' || child.type === 'parameter_with_optional_type') arity += 1; + } + return arity; +} + +function collectExistingMethods(...bodies: Array): ExistingMethodIndex { + const index = createExistingMethodIndex('exact'); + for (const body of bodies) { + if (!body) continue; + for (const child of body.children) { + if (child.type !== 'function_declaration') continue; + const name = functionName(child); + if (!name) continue; + rememberExistingMethod(index, name, functionArity(child)); + } + } + return index; +} + +function toKtProperty(child: Parser.SyntaxNode, imports: KotlinImportIndex): KtProperty | null { + const isVar = isVarBinding(child); + if (isVar === null) return null; + if (hasJvmField(child, imports) || hasConst(child)) return null; + const nameNode = propertyNameNode(child); + if (!nameNode) return null; + const mods = child.children.find((c) => c.type === 'modifiers'); + const visibility = kotlinVisibility(mods); + const accessor = accessorMetadata(child, visibility, imports); + return { + name: nameNode.text, + type: propertyTypeText(child), + isVar, + skipGetter: accessor.skipGetter, + skipSetter: accessor.skipSetter, + getterVisibility: accessor.getterVisibility, + setterVisibility: accessor.setterVisibility, + startLine: child.startPosition.row + 1, + endLine: child.endPosition.row + 1, + propertyNode: child, + declaratorNode: nameNode, + }; +} + +function collectTypedProperties( + parent: Parser.SyntaxNode | null, + type: 'class_parameter' | 'property_declaration', + imports: KotlinImportIndex, +): KtProperty[] { + if (!parent) return []; + const out: KtProperty[] = []; + for (const child of parent.namedChildren) { + if (child.type !== type) continue; + const prop = toKtProperty(child, imports); + if (prop) out.push(prop); + } + return out; +} + +function findKtClasses(root: Parser.SyntaxNode, imports: KotlinImportIndex): KtClass[] { + const classes: KtClass[] = []; + const graphOwnerNode = (node: Parser.SyntaxNode): Parser.SyntaxNode => { + if (node.type !== 'companion_object') return node; + if (jvmTypeSimpleName(node)) return node; + let current = node.parent; + while (current && !KOTLIN_TYPE_DECLS.has(current.type)) current = current.parent; + return current ?? node; + }; + const walk = (node: Parser.SyntaxNode): void => { + if (KOTLIN_TYPE_DECLS.has(node.type)) { + const ownerNode = graphOwnerNode(node); + const name = jvmTypeSimpleName(ownerNode) ?? ''; + const ctor = node.children.find((c) => c.type === 'primary_constructor') ?? null; + const body = node.children.find((c) => c.type === 'class_body') ?? null; + if (name) { + const properties = [ + ...collectTypedProperties(ctor, 'class_parameter', imports), + ...collectTypedProperties(body, 'property_declaration', imports), + ]; + if (properties.length > 0) { + const ownerBody = + ownerNode.id === node.id + ? null + : (ownerNode.children.find((child) => child.type === 'class_body') ?? null); + classes.push({ + node: ownerNode, + name, + isStatic: node.type === 'companion_object', + isInterface: node.children.some((child) => child.type === 'interface'), + wasHoisted: ownerNode.id !== node.id, + properties, + existingMethods: collectExistingMethods(body, ownerBody), + }); + } + } + if (body) { + for (const child of body.namedChildren) { + if (KOTLIN_TYPE_DECLS.has(child.type)) walk(child); + } + } + return; + } + for (const child of node.namedChildren) walk(child); + }; + walk(root); + return classes; +} + +function planAccessors(cls: KtClass): PlannedJvmAccessor[] { + const planned: PlannedJvmAccessor[] = []; + for (const prop of cls.properties) { + const gName = kotlinGetterName(prop.name); + if (!prop.skipGetter && !hasExistingMethod(cls.existingMethods, gName, 0)) { + planned.push({ + kind: 'getter', + name: gName, + returnType: prop.type, + parameterTypes: [], + visibility: prop.getterVisibility, + isStatic: cls.isStatic, + isAbstract: cls.isInterface && !hasKotlinAccessorBody(prop.propertyNode, 'getter'), + startLine: prop.startLine, + endLine: prop.endLine, + declaratorNode: prop.declaratorNode, + }); + } + if (prop.isVar && !prop.skipSetter) { + const sName = kotlinSetterName(prop.name); + if (!hasExistingMethod(cls.existingMethods, sName, 1)) { + planned.push({ + kind: 'setter', + name: sName, + returnType: 'void', + parameterTypes: [prop.type], + visibility: prop.setterVisibility, + isStatic: cls.isStatic, + isAbstract: cls.isInterface && !hasKotlinAccessorBody(prop.propertyNode, 'setter'), + startLine: prop.startLine, + endLine: prop.endLine, + declaratorNode: prop.declaratorNode, + }); + } + } + } + return planned; +} + +function planKotlinAccessorOwners(rootNode: Parser.SyntaxNode): PlannedJvmAccessorOwner[] { + const owners: PlannedJvmAccessorOwner[] = []; + const imports = collectKotlinImports(rootNode); + for (const cls of findKtClasses(rootNode, imports)) { + const accessors = planAccessors(cls); + const existingIndex = cls.wasHoisted + ? owners.findIndex((owner) => owner.node.id === cls.node.id) + : -1; + const existing = existingIndex >= 0 ? owners[existingIndex] : undefined; + if (existing) { + owners[existingIndex] = { + ...existing, + accessors: [...existing.accessors, ...accessors], + }; + } else { + owners.push({ node: cls.node, name: cls.name, accessors }); + } + } + return owners; +} + +const lombokAccessorSynthesis = createJvmAccessorSynthesis({ + language: 'kotlin', + synthetic: 'kotlin-jvm', + planOwners: planKotlinAccessorOwners, +}); + +export function synthesizeLombokAccessors( + tree: Parser.Tree, + filePath: string, + classOwnersById: ReadonlyMap, +): SyntheticAccessorResult { + return lombokAccessorSynthesis.synthesize(tree, filePath, classOwnersById); +} + +export function synthesizeLombokAccessorCaptures(rootNode: Parser.SyntaxNode): CaptureMatch[] { + return lombokAccessorSynthesis.captures(rootNode); +} diff --git a/gitnexus/src/core/ingestion/scope-extractor.ts b/gitnexus/src/core/ingestion/scope-extractor.ts index 39b8667ae..fde3c7b26 100644 --- a/gitnexus/src/core/ingestion/scope-extractor.ts +++ b/gitnexus/src/core/ingestion/scope-extractor.ts @@ -1775,6 +1775,7 @@ const KNOWN_SUB_TAGS: ReadonlySet = new Set([ '@scope.lexical-names', '@declaration.name', '@declaration.qualified_name', + '@declaration.is-synthetic', '@import.name', '@import.source', '@import.alias', diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index ab27cf7ac..3f3f70395 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -1578,6 +1578,11 @@ const processFileGroup = ( } const provider = getProvider(language); + // Owner map for provider.synthesizeStructureMembers: type-declaration AST + // node id → graph node id for classes THIS file's capture loop materialized. + // Keyed by in-memory AST identity (never persisted); filled below. + const classOwnersByNodeId = new Map(); + // #2687: ONE pass over `matches` yields both suppression sets — the // definition-name claims by rank (callable > Property > value), so the dedup // below cannot depend on tree-sitter's match order, and the concrete-typedef @@ -2987,6 +2992,17 @@ const processFileGroup = ( : {}), }); + // Class-like definitions register their AST node id → graph node id for + // provider.synthesizeStructureMembers. The definition node is the same + // type-declaration AST node that the provider-specific planner receives. + if ( + isClassLikeLabel && + definitionNode && + provider.classExtractor?.isTypeDeclaration(definitionNode) + ) { + classOwnersByNodeId.set(definitionNode.id, nodeId); + } + // Object-literal callables remain file definitions as well as members of // their exported binding. Class members still use HAS_METHOD alone. const isTopLevelObjectCallable = @@ -3092,6 +3108,19 @@ const processFileGroup = ( if (springTypes.length > 0) (result.springTypes ??= []).push(...springTypes); } + if (provider.synthesizeStructureMembers) { + const synthetic = provider.synthesizeStructureMembers(tree, file.path, classOwnersByNodeId); + for (const node of synthetic.nodes) { + result.nodes.push(node as ParsedNode); + } + for (const sym of synthetic.symbols) { + result.symbols.push(sym as ParsedSymbol); + } + for (const rel of synthetic.relationships) { + result.relationships.push(rel as ParsedRelationship); + } + } + // Vue: emit CALLS edges for components used in