GitNexus/gitnexus/test/unit/analyze-worker-core.test.ts
mengkaka 79543c8f83
feat(storage): add configurable index storage and content retention tiers (#3060)
* feat(storage): add configurable index storage and content retention tiers

Rebase #3060 onto current origin/main. Keep GITNEXUS_STORAGE_PATH,
GITNEXUS_STORAGE_ROOT, and GITNEXUS_CONTENT_RETENTION, and fold in
main's FTS skip, embed-session, and help-text updates.

Co-authored-by: Cursor <cursoragent@cursor.com>

* Address PR review feedback (#3060)

Keep legacy registry rows on the local storage fallback, resolve
symlinks before the destructive-path guard, and align hook lookup
with CLI branch slugs, branch-slot metadata, and longest-path match.

Co-authored-by: Cursor <cursoragent@cursor.com>

* Address PR review feedback (#3060)

Only list swept upload directories after a successful removal so
callers cannot treat a permission or transient rm failure as gone.

Co-authored-by: Cursor <cursoragent@cursor.com>

* Address PR review feedback (#3060)

Document that getStoragePath may consult registered storage while
this module still does not mutate the global registry.

Co-authored-by: Cursor <cursoragent@cursor.com>

* fix(storage): close review findings for external indexes and retention

Re-inspect ownership under the analyze lock, fail-closed when the
registry file is missing, and keep skip-git hook discovery plus
retention fields on HTTP/MCP list surfaces. /api/file stays 410
unless contentRetention is full.

Co-authored-by: Cursor <cursoragent@cursor.com>

* chore(autofix): apply prettier + eslint fixes via /autofix command

* Address PR review feedback (#3060)

Treat lock-only index dirs as empty, honor HTTP --force storage policy, and prefer registered plus branch-aware slots in hooks and augment.

Co-authored-by: Cursor <cursoragent@cursor.com>

* Address PR review feedback (#3060)

Keep hook fallbacks inside the current worktree, compare foreign-local slots canonically, and make storage fixtures survive ownership validation.

Co-authored-by: Cursor <cursoragent@cursor.com>

* Fix macOS hook test expecting realpath'd registry paths.

resolveHookRepo returns the written registry path, not a filesystem realpath, so the assertion must match that.

* Address gitnexus-check warnings on hook install docs and slot tests.

The Cursor troubleshooting list omitted registry-query.cjs, and the writable-slot test only checked that isDirectory exists instead of that the path is a directory.

* Align the HTTP catalog source-scan with skippable resolveRepo validation.

resolveRepo lists fresh repos with validate: options.validateStorage !== false so DELETE can skip prune; the test still required a literal validate: true.

* Harden storage path sinks so CodeQL path-injection and ReDoS alerts clear.

Contain every filesystem probe inside the resolved storage slot with the inline path.relative idiom, reject filesystem-root slots, and trim slot basenames in linear time.

* Settle bridge stamps before writing so CI size/mtime matches stay stable.

LadybugDB can still flush into bridge.lbug after close+rename; persist whole-millisecond mtimes and wait for consecutive stats to agree so a freshly written pair matches.

* Type the settled bridge stat as fs.Stats so tsc does not see bigint.

Awaited<ReturnType<typeof fsp.stat>> collapsed the bigint overload and broke prepare/typecheck on CI.

* Keep the bridge mtime stamp exact so same-size swaps still fail the pair check.

Co-authored-by: Cursor <cursoragent@cursor.com>

* Wrap the bridge stamp predicate so prettier --check stays green.

Co-authored-by: Cursor <cursoragent@cursor.com>

* Require a quiet interval before stamping a settled bridge file.

Co-authored-by: Cursor <cursoragent@cursor.com>

* Reuse shared storage and settle helpers instead of local copies.

Co-authored-by: Cursor <cursoragent@cursor.com>

---------

Co-authored-by: Gergo Magyar <gergomagyar0@gmail.com>
Co-authored-by: Cursor <cursoragent@cursor.com>
Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
2026-09-12 20:31:55 +00:00

210 lines
6.5 KiB
TypeScript

/**
* Unit tests for the analyze-worker core seam (#2264).
*
* P2: the worker must NOT report `complete` for a half-finalized repo (meta.json
* written but the global registry entry missing) — it must surface that as an
* error, mirroring the CLI's assertAnalysisFinalized guard.
*
* P3: a SIGTERM cancellation and a near-simultaneous completion must not both
* report a terminal outcome — the `claimTerminal` slot coordinates them.
*
* Driven via the side-effect-free `runWorkerAnalysis` seam with injected fakes, so
* no fork()/process.on side effects of the entry module are touched.
*/
import { describe, it, expect, vi } from 'vitest';
import {
runWorkerAnalysis,
createTerminalClaim,
type WorkerAnalysisDeps,
} from '../../src/server/analyze-worker-core.js';
import type { AnalyzeResult } from '../../src/core/run-analyze.js';
import type { WorkerMessage } from '../../src/server/analyze-worker.js';
import type { AnalyzerRunnerIdentity } from '../../src/storage/repo-manager.js';
import { IndexLockTimeoutError, type LockRecord } from '../../src/storage/index-lock.js';
const baseResult: AnalyzeResult = {
repoName: 'repo',
repoPath: '/repo',
storagePath: '/repo/.gitnexus',
stats: {},
alreadyUpToDate: false,
ftsRepairedOnly: false,
};
const okRun: WorkerAnalysisDeps['runFullAnalysis'] = vi.fn(async () => baseResult);
const okFinalize: WorkerAnalysisDeps['assertAnalysisFinalized'] = vi.fn(async () => undefined);
const alwaysClaim: WorkerAnalysisDeps['claimTerminal'] = () => true;
describe('runWorkerAnalysis — finalize guard (#2264 P2)', () => {
it('reports error (not complete) when finalization fails for an unregistered repo', async () => {
const send = vi.fn<(msg: WorkerMessage) => void>();
const assertAnalysisFinalized: WorkerAnalysisDeps['assertAnalysisFinalized'] = vi.fn(
async () => {
throw new Error('registry entry for /repo was not added');
},
);
await runWorkerAnalysis(
'/repo',
{},
{
runFullAnalysis: okRun,
assertAnalysisFinalized,
send,
claimTerminal: alwaysClaim,
},
);
expect(send).toHaveBeenCalledWith({
type: 'error',
message: 'registry entry for /repo was not added',
});
expect(send).not.toHaveBeenCalledWith(expect.objectContaining({ type: 'complete' }));
});
it('reports complete exactly once when finalization succeeds', async () => {
const send = vi.fn<(msg: WorkerMessage) => void>();
await runWorkerAnalysis(
'/repo',
{},
{
runFullAnalysis: okRun,
assertAnalysisFinalized: okFinalize,
send,
claimTerminal: alwaysClaim,
},
);
const completes = send.mock.calls.filter((c) => c[0].type === 'complete');
expect(completes).toHaveLength(1);
expect(okFinalize).toHaveBeenCalledWith('/repo', '/repo/.gitnexus');
});
it('threads the pre-import runner receipt into runFullAnalysis', async () => {
const send = vi.fn<(msg: WorkerMessage) => void>();
const run = vi.fn<WorkerAnalysisDeps['runFullAnalysis']>(async () => baseResult);
const receipt = { schemaVersion: 4 } as AnalyzerRunnerIdentity;
await runWorkerAnalysis(
'/repo',
{},
{
runFullAnalysis: run,
assertAnalysisFinalized: okFinalize,
send,
claimTerminal: alwaysClaim,
},
receipt,
);
expect(run.mock.calls[0]?.[3]).toBe(receipt);
});
it('reports error when finalization passes but the analysis itself throws', async () => {
const send = vi.fn<(msg: WorkerMessage) => void>();
const failingRun: WorkerAnalysisDeps['runFullAnalysis'] = vi.fn(async () => {
throw new Error('boom');
});
// Fresh local mock (not the shared okFinalize) so the "never called" assertion
// reflects only this test.
const finalize = vi.fn<WorkerAnalysisDeps['assertAnalysisFinalized']>(async () => undefined);
await runWorkerAnalysis(
'/repo',
{},
{
runFullAnalysis: failingRun,
assertAnalysisFinalized: finalize,
send,
claimTerminal: alwaysClaim,
},
);
expect(send).toHaveBeenCalledWith({ type: 'error', message: 'boom' });
expect(finalize).not.toHaveBeenCalled();
});
it.each([undefined, '/repo/.gitnexus/analyze.lock.guard'])(
'classifies lock timeout retryability for guard=%s',
async (guardPath) => {
const send = vi.fn<(msg: WorkerMessage) => void>();
const holder: LockRecord = {
v: 1,
pid: -1,
hostname: 'host',
startTime: null,
token: '',
invocationId: 'unknown',
acquiredAt: '',
};
const lockContended: WorkerAnalysisDeps['runFullAnalysis'] = vi.fn(async () => {
throw new IndexLockTimeoutError(holder, 600_000, false, guardPath);
});
await runWorkerAnalysis(
'/repo',
{},
{
runFullAnalysis: lockContended,
assertAnalysisFinalized: okFinalize,
send,
claimTerminal: alwaysClaim,
},
);
expect(send).toHaveBeenCalledWith(
expect.objectContaining({
type: 'error',
code: 'index-lock-timeout',
retryable: guardPath === undefined,
}),
);
if (guardPath) {
expect(send).toHaveBeenCalledWith(
expect.objectContaining({ message: expect.stringContaining('quiesced recovery') }),
);
}
},
);
});
describe('runWorkerAnalysis — terminal-claim coordination (#2264 P3)', () => {
it('sends NO terminal message when the slot is already claimed (cancellation won)', async () => {
const send = vi.fn<(msg: WorkerMessage) => void>();
const alreadyClaimed: WorkerAnalysisDeps['claimTerminal'] = () => false;
await runWorkerAnalysis(
'/repo',
{},
{
runFullAnalysis: okRun,
assertAnalysisFinalized: okFinalize,
send,
claimTerminal: alreadyClaimed,
},
);
const terminals = send.mock.calls.filter(
(c) => c[0].type === 'complete' || c[0].type === 'error',
);
expect(terminals).toHaveLength(0);
});
});
describe('createTerminalClaim (#2264 P3)', () => {
it('returns true for the first claim and false for every claim after', () => {
const claim = createTerminalClaim();
expect(claim()).toBe(true);
expect(claim()).toBe(false);
expect(claim()).toBe(false);
});
it('gives independent claims separate slots', () => {
const a = createTerminalClaim();
const b = createTerminalClaim();
expect(a()).toBe(true);
expect(b()).toBe(true);
expect(a()).toBe(false);
});
});