GitNexus/gitnexus/src/storage/repo-manager.ts
Gergo Magyar 6ad8d7d389 feat(storage): share the index store between clones automatically (#3352)
Clones of one repository now share like linked worktrees. A clone whose
normalized origin URL matches another registered, still-present clone
joins that clone's store, or founds one (keyed on its own path) that the
sibling joins on its next analyze. A lone clone keeps its own .gitnexus.
Graphs stay keyed by commit and feature key, so clones only ever share a
graph built from the same commit with the same settings.

`analyze --no-share` now records a lasting opt-out (`shareOptOut` on the
registry entry, preserved across re-registration); `--share-with` clears
it. The analyze worker reports the storage it wrote over IPC so the
server settles a clone's first shared slot.

A query-time base-plus-overlay graph stays out: LadybugDB reads one
database per query. Instead, private graph copies record whether the
filesystem cloned them copy-on-write (sharing unchanged pages on disk)
or made a full copy, and `status` reports it.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2026-09-24 21:43:48 +00:00

1945 lines
81 KiB
TypeScript

/**
* Repository Manager
*
* Manages GitNexus index storage:
* - Per-repo metadata file (gitnexus.json) under .gitnexus/, dual-written to a
* legacy meta.json mirror for backward compatibility (see MIGRATION.md)
* - .gitnexus/ directory for local metadata and caches (parse-cache, parsedfile-store)
* - Global registry at ~/.gitnexus/registry.json for MCP server discovery
*
* gitnexus.json is simply a filename distinct from the generic meta.json — it
* has no bearing on git worktree behavior. .gitnexus/ remains fully git-ignored
* in every case; each worktree already has its own independent .gitnexus/ by
* construction (getStoragePath is per-checkout), regardless of which filename
* the metadata inside it uses.
*/
import fs from 'fs/promises';
import { realpathSync } from 'fs';
import path from 'path';
import {
getGitRoot,
getInferredRepoName,
resolveRepoIdentityRoot,
stripUrlCredentials,
} from './git.js';
import { stripWindowsLongPathPrefix } from '../lib/utils.js';
import { writeFileAtomic } from './fs-atomic.js';
import { getGlobalDir } from './global-dir.js';
import { logger } from '../core/logger.js';
import { mapPool } from './map-pool.js';
import {
acquireIndexLock,
IndexLockTimeoutError,
requireExclusiveIndexLock,
type IndexLockHandle,
} from './index-lock.js';
import {
branchSlug,
BRANCHES_DIR,
resolveBranchPlacement,
type BranchSummary,
} from './branch-index.js';
import {
GITNEXUS_DIR,
INDEX_METADATA_FILE,
LEGACY_METADATA_FILE,
getStoragePath,
isMissingFilesystemError,
loadMeta,
type AnalyzerRunnerIdentity,
type RepoMeta,
} from './repo-meta.js';
import { LBUG_DIRECTORY } from './storage-constants.js';
import { resolveGraphPath } from './shared-store.js';
import {
defaultStoragePath,
ensureStoragePathWritable,
InvalidStoragePathError,
inspectResolvedStorage,
inspectRegisteredStorage,
inspectStoragePath,
isTransientStorageInspection,
LIST_STORAGE_REQUIREMENTS,
requireDeletableStoragePath,
StorageDeletionError,
validateConfiguredStoragePath,
} from './storage-resolver.js';
// Re-export the #2106 branch primitives (extracted to branch-index.ts, R10) so
// existing `repo-manager` import sites and tests keep working unchanged.
export { branchSlug, resolveBranchPlacement };
export type { BranchSummary };
// Re-export the metadata primitives (extracted to repo-meta.ts) for the same
// reason. They moved DOWN a layer so `branch-index.ts` can read the flat slot's
// metadata without importing back out of this module — see repo-meta.ts for the
// cycle that made the extraction necessary. `LEGACY_METADATA_FILE` stays
// module-private here, exactly as before.
export { getStoragePath, INDEX_METADATA_FILE, isMissingFilesystemError, loadMeta };
export { ensureStoragePathWritable, InvalidStoragePathError };
export { CONTENT_RETENTION_SCHEMA_VERSION } from './repo-meta.js';
export type { ContentRetention, FtsProfile } from './repo-meta.js';
export type { AnalyzerRunnerIdentity, RepoMeta };
export { getGlobalDir } from './global-dir.js';
/**
* Normalise a repo path for registry comparison across platforms
* (#664 review feedback from @evander-wang).
*
* Why this exists: `path.resolve` alone is NOT enough for
* cross-platform registry stability.
* - **macOS**: tmpdirs and `/var` are symlinks to `/private/var`.
* A child process that stored `/private/var/folders/.../repo` in
* the registry cannot later be matched by an outer caller that
* supplies the symlink form `/var/folders/.../repo`. `path.resolve`
* does not follow symlinks; `realpathSync.native` does.
* - **Windows**: GitHub runners surface tmpdirs in 8.3 short-name
* form (`RUNNERA~1\...`), but `process.cwd()` often returns the
* long form (`runneradmin\...`). `realpathSync.native` normalises
* both sides to the long-name canonical path.
* - **Windows, extended-length paths** (#2667): a caller can supply a
* `\\?\`-prefixed path — the usual MAX_PATH workaround — and
* `path.resolve` preserves the prefix, so the string compare below
* never matches the un-prefixed entry the registry stores. The
* realpath branch already dropped it (libuv strips the prefix inside
* `fs__realpath`), but the fallback branch did not, which is exactly
* the branch a missing path takes. `stripWindowsLongPathPrefix` is
* applied to both so the two branches agree.
*
* This normalisation is safe here precisely because the result is only ever
* compared, never opened: Node does NOT re-add `\\?\` for over-MAX_PATH
* paths, so an fs-facing path must keep whatever form the caller gave it.
* See the `registerRepo` comment on applying canonicalisation at COMPARE
* points only.
*
* Fallback behaviour: if the path does not exist on disk (e.g. a user
* passed `gitnexus remove some-alias` and the alias misses every
* registry entry, or the caller is resolving a path that was deleted
* after registration), we return `path.resolve(p)` rather than
* throwing. This preserves the idempotent-on-missing semantics of
* `resolveRegistryEntry` / `remove`.
*
* Backwards compatibility: this function is applied to BOTH the
* caller-supplied input AND each stored `entry.path` at compare time
* inside `resolveRegistryEntry`, so registries written by older
* versions still match correctly. Entries are NOT canonicalised at
* write time — `registerRepo` stores `path.resolve(repoPath)` — which
* is what makes the compare-only rule above hold.
*/
export const canonicalizePath = (p: string): string => {
const resolved = path.resolve(p);
try {
return stripWindowsLongPathPrefix(realpathSync.native(resolved));
} catch {
return stripWindowsLongPathPrefix(resolved);
}
};
/**
* Compare two already-canonicalised registry paths. Case-insensitive on Windows
* (its filesystem is), case-sensitive elsewhere. Both arguments must already be
* run through {@link canonicalizePath}; this is the single comparison the registry
* lookups/dedup/finalize checks all share so they answer identically.
*/
export const registryPathEquals = (a: string, b: string): boolean =>
process.platform === 'win32' ? a.toLowerCase() === b.toLowerCase() : a === b;
/**
* Does the clone dir derived from an entry's *name* actually belong to that
* entry? Registry names are not unique across storage locations: a cloned
* repo under `~/.gitnexus/repos/<name>` and a local repo registered under the
* same name share a `getCloneDir(entry.name)` result. The server's delete
* handler must therefore never remove the clone dir based on the name alone —
* only when the entry's own `path` resolves to that dir (mirroring its step-2b
* rule that cleanup is driven off `entry.path`, so a same-named sibling's
* clone is never removed). Both sides are canonicalised so symlinked or
* differently-spelled forms of the same dir still match.
*/
export const cloneDirBelongsToEntry = (cloneDir: string, entryPath: string): boolean =>
registryPathEquals(canonicalizePath(cloneDir), canonicalizePath(entryPath));
export interface IndexedRepo {
repoPath: string;
storagePath: string;
lbugPath: string;
metaPath: string;
meta: RepoMeta;
}
/**
* Shape of an entry in the global registry (~/.gitnexus/registry.json)
*/
export interface RegistryEntry {
name: string;
path: string;
storagePath: string;
indexedAt: string;
lastCommit: string;
/** See {@link RepoMeta.remoteUrl}. Mirrored from meta at register time. */
remoteUrl?: string;
stats?: RepoMeta['stats'];
/**
* Branch name owning the flat/primary index (#2106). Mirrors the flat
* `meta.branch`. Absent for legacy single-branch entries and non-git repos —
* additive and backward compatible.
*/
branch?: string;
/**
* Non-primary branch indexes for this same path (#2106). Absent when only the
* primary branch is indexed, preserving the one-entry-per-path model and the
* legacy registry shape.
*/
branches?: BranchSummary[];
/**
* The checkout left sharing with `analyze --no-share` (#3352), so it does
* not join a sibling clone's store automatically. Cleared by `--share-with`.
*/
shareOptOut?: true;
}
/** Path-only registry lookup. Canonicalizes `repoPath` once. Does not throw. */
export const findRegistryEntryByRepoPath = (
entries: readonly RegistryEntry[],
repoPath: string,
): RegistryEntry | undefined => {
const repoKey = canonicalizePath(repoPath);
return entries.find((entry) => registryPathEquals(canonicalizePath(entry.path), repoKey));
};
const GITNEXUS_EXCLUDE_ENTRY = `${GITNEXUS_DIR}/`;
// ─── Local Storage Helpers ─────────────────────────────────────────────
/**
* Get paths to key storage files.
*
* `storagePath` is ALWAYS the flat `<repo>/.gitnexus` — content-addressed
* caches (`parse-cache/`, `parsedfile-store/`) live there and are shared
* across branches (#2106 KTD7). When `branch` is provided, both `lbugPath`
* and `metaPath` are scoped under `branches/<slug>/`. For the flat call
* (no `branch`), `storagePath` and `lbugPath` remain byte-identical to the
* pre-multi-branch behavior (#2106), except that a shared-store checkout slot
* (#3352) returns the commit graph its metadata records (`resolveGraphPath`).
* `metaPath`'s FILENAME changed from `meta.json` to `gitnexus.json`
* (PR #2363) — `saveMeta` keeps a `meta.json` mirror in sync for consumers
* that still read the legacy name.
*
* Each branch slot has its own metadata file:
* - Primary/flat: <repo>/.gitnexus/gitnexus.json
* - Feature branches: <repo>/.gitnexus/branches/<slug>/gitnexus.json
*
* Callers should use `loadMeta(metaDir)` and `saveMeta(metaDir, meta)` where
* metaDir is the directory containing the metadata file — both handle the
* legacy mirror automatically.
*/
export const getStoragePaths = (
repoPath: string,
branch?: string,
resolvedStoragePath?: string,
) => {
const storagePath = resolvedStoragePath ?? getStoragePath(repoPath);
const baseDir = branch ? path.join(storagePath, BRANCHES_DIR, branchSlug(branch)) : storagePath;
return {
storagePath,
// Branch slots are always private; a flat shared-store slot may read a
// commit graph (#3352).
lbugPath: branch ? path.join(baseDir, LBUG_DIRECTORY) : resolveGraphPath(storagePath),
metaPath: path.join(baseDir, INDEX_METADATA_FILE), // Branch-specific metadata file
};
};
/**
* Check whether a KuzuDB index exists in the given storage path.
* Non-destructive — safe to call from status commands.
*/
export const hasKuzuIndex = async (storagePath: string): Promise<boolean> => {
try {
await fs.stat(path.join(storagePath, 'kuzu'));
return true;
} catch {
return false;
}
};
/**
* Clean up stale KuzuDB files after migration to LadybugDB.
*
* Returns:
* found — true if .gitnexus/kuzu existed and was deleted
* needsReindex — true if kuzu existed but lbug does not (re-analyze required)
*
* Callers own the user-facing messaging; this function only deletes files.
*/
export const cleanupOldKuzuFiles = async (
storagePath: string,
): Promise<{ found: boolean; needsReindex: boolean }> => {
const oldPath = path.join(storagePath, 'kuzu');
const newPath = path.join(storagePath, 'lbug');
try {
await fs.stat(oldPath);
// Old kuzu file/dir exists — determine if lbug is already present
let needsReindex = false;
try {
await fs.stat(newPath);
} catch {
needsReindex = true;
}
// Delete kuzu database file and its sidecars (.wal, .lock)
for (const suffix of ['', '.wal', '.lock']) {
try {
await fs.unlink(oldPath + suffix);
} catch {}
}
// Also handle the case where kuzu was stored as a directory
try {
await fs.rm(oldPath, { recursive: true, force: true });
} catch {}
return { found: true, needsReindex };
} catch {
// Old path doesn't exist — nothing to do
return { found: false, needsReindex: false };
}
};
/**
* Save metadata to the metadata file (gitnexus.json) in the given directory,
* dual-writing the legacy `meta.json` mirror for backward compatibility.
*
* Atomic via tmp-file + rename (matches `saveParseCache`'s pattern). The
* `incrementalInProgress` dirty flag travels through this file — a crash
* mid-write would leave a corrupt `gitnexus.json` that the next run's
* `loadMeta` would silently treat as "no prior index", losing the dirty
* flag and skipping the recovery full-rebuild. Write-and-rename rules
* that out: the rename is atomic on POSIX and on Windows (`fs.rename`
* on `node:fs/promises` uses `MoveFileEx(REPLACE_EXISTING)`), so either
* the old or the new file is observed at every moment.
*
* `gitnexus.json` is the primary write and must succeed. `meta.json` is a
* best-effort mirror kept for consumers that only know the legacy filename
* (see MIGRATION.md) — its write failure is logged, not thrown, so a
* mirror-write hiccup never fails the caller's analyze run.
*/
export const saveMeta = async (metaDir: string, meta: RepoMeta): Promise<void> => {
await fs.mkdir(metaDir, { recursive: true });
// Serialised once: `meta` carries a fileHashes entry per file, so on a large
// repo this string is megabytes and both writes want the identical bytes.
const json = JSON.stringify(meta, null, 2);
await writeFileAtomic(path.join(metaDir, INDEX_METADATA_FILE), json);
try {
await writeFileAtomic(path.join(metaDir, LEGACY_METADATA_FILE), json);
} catch (err) {
logger.warn({ err, metaDir }, 'Failed to write legacy meta.json mirror (non-critical)');
}
};
/** Check whether the resolved storage contains an owned, usable code index. */
export const hasIndex = async (repoPath: string): Promise<boolean> => {
const inspection = await inspectResolvedStorage(repoPath);
return inspection.state === 'owned' && inspection.hasCodeIndexDB;
};
/** Load an owned index from one already-determined repository path. */
export const loadRepo = async (repoPath: string): Promise<IndexedRepo | null> => {
const inspection = await inspectResolvedStorage(repoPath);
// Do not require LadybugDB here: clean needs to locate an owned
// metadata-only slot so it can remove interrupted or legacy remnants.
if (inspection.state !== 'owned') return null;
const paths = getStoragePaths(inspection.repoPath, undefined, inspection.storagePath);
const meta = await loadMeta(paths.storagePath);
if (!meta) return null;
return {
repoPath: inspection.repoPath,
...paths,
meta,
};
};
type ReconcileMetadataRead =
| { state: 'absent' }
| { state: 'invalid'; error: unknown }
| { state: 'valid'; meta: RepoMeta };
const readReconcileMetadata = async (
dir: string,
filename: typeof INDEX_METADATA_FILE | typeof LEGACY_METADATA_FILE,
): Promise<ReconcileMetadataRead> => {
let raw: string;
try {
raw = await fs.readFile(path.join(dir, filename), 'utf-8');
} catch (error) {
return isMissingFilesystemError(error) ? { state: 'absent' } : { state: 'invalid', error };
}
try {
return { state: 'valid', meta: JSON.parse(raw) as RepoMeta };
} catch (error) {
return { state: 'invalid', error };
}
};
/**
* Reconcile `gitnexus.json` and the legacy `meta.json` mirror in one directory.
* A valid primary is authoritative; legacy metadata is used only when the
* primary is provably absent. Never deletes anything.
* Returns true when a write occurred.
*/
const reconcileMetaDir = async (dir: string): Promise<boolean> => {
const primary = await readReconcileMetadata(dir, INDEX_METADATA_FILE);
if (primary.state === 'invalid') {
logger.warn(
{ dir, filename: INDEX_METADATA_FILE, err: primary.error },
'Primary metadata is unreadable/corrupt; leaving both metadata files unchanged',
);
return false;
}
if (primary.state === 'valid') {
const legacy = await readReconcileMetadata(dir, LEGACY_METADATA_FILE);
if (legacy.state === 'valid' && JSON.stringify(primary.meta) === JSON.stringify(legacy.meta)) {
return false;
}
await saveMeta(dir, primary.meta);
return true;
}
const legacy = await readReconcileMetadata(dir, LEGACY_METADATA_FILE);
if (legacy.state === 'valid') {
await saveMeta(dir, legacy.meta);
return true;
}
if (legacy.state === 'invalid') {
logger.warn(
{ dir, filename: LEGACY_METADATA_FILE, err: legacy.error },
'Legacy metadata is unreadable/corrupt; leaving it unchanged',
);
}
return false;
};
/**
* Reconcile the metadata files for a repo's flat slot and every
* `branches/<slug>/` slot. `resolvedStoragePath` is the ownership-validated
* write target supplied by analyze; callers that omit it retain the historic
* repository-path lookup for compatibility.
*
* This is a best-effort compatibility sync, NOT a one-way migration: the
* legacy `meta.json` mirror is kept in sync indefinitely (removal happens at
* a future major version — see MIGRATION.md), so older binaries, still-running
* MCP servers, and the shipped editor hooks keep working, and a rollback to a
* pre-rename version sees current metadata instead of "no prior index".
* Returns true when any file was written.
*/
export const reconcileMetadataFiles = async (
repoPath: string,
resolvedStoragePath?: string,
): Promise<boolean> => {
const storagePath = resolvedStoragePath ?? getStoragePath(repoPath);
let changed = await reconcileMetaDir(storagePath);
const branchesDir = path.join(storagePath, BRANCHES_DIR);
let branchDirs: string[];
try {
branchDirs = await fs.readdir(branchesDir);
} catch {
// branchesDir may not exist (not a multi-branch repo) — expected, silent.
return changed;
}
for (const branchDir of branchDirs) {
const branchPath = path.join(branchesDir, branchDir);
// Per-branch isolation: one bad branch dir (dangling symlink, EACCES)
// must not silently abort reconciliation for every branch after it —
// readdir order is stable, so an unguarded throw here would permanently
// starve the same trailing branches on every run.
try {
const stat = await fs.stat(branchPath);
if (!stat.isDirectory()) continue;
if (await reconcileMetaDir(branchPath)) changed = true;
} catch (err) {
logger.warn(
{ branchDir, err },
'Skipping branch directory during metadata reconciliation (non-critical)',
);
}
}
return changed;
};
/** Resolve the Git worktree first, then load only that repository's index. */
export const findRepo = async (startPath: string): Promise<IndexedRepo | null> => {
const resolved = path.resolve(startPath);
return loadRepo(getGitRoot(resolved) ?? resolved);
};
export function isReadOnlyFilesystemError(err: unknown): boolean {
const code = (err as NodeJS.ErrnoException)?.code;
return code === 'EROFS' || code === 'EACCES' || code === 'EPERM';
}
/**
* Keep .gitnexus/ ignored. It contains local index state and caches.
*/
export const ensureGitNexusIgnored = async (
repoPath: string,
resolvedStoragePath?: string,
): Promise<void> => {
const storagePath = resolvedStoragePath ?? getStoragePath(repoPath);
const gitignorePath = path.join(storagePath, '.gitignore');
const desired = '*\n';
// Idempotent fast path: skip the write entirely when the file already has
// the expected content. Lets this run cleanly on read-only mounts (e.g.
// the documented Docker workflow with WORKSPACE_DIR bound :ro) when an
// earlier `analyze` already created the file. See issue #1549.
try {
if ((await fs.readFile(gitignorePath, 'utf-8')) === desired) {
await ensureGitInfoExclude(repoPath);
return;
}
} catch (err: any) {
if (err?.code !== 'ENOENT') throw err;
}
try {
await fs.mkdir(path.dirname(gitignorePath), { recursive: true });
await fs.writeFile(gitignorePath, desired, 'utf-8');
} catch (err: any) {
if (isReadOnlyFilesystemError(err)) {
logger.warn(
{ path: gitignorePath, code: err.code },
'GitNexus storage filesystem is not writable; skipping .gitnexus/.gitignore. Cache files may appear as untracked in this repo locally.',
);
} else {
throw err;
}
}
await ensureGitInfoExclude(repoPath);
};
const ensureGitInfoExclude = async (repoPath: string): Promise<void> => {
const gitDirPath = path.join(path.resolve(repoPath), '.git');
const excludePath = path.join(gitDirPath, 'info', 'exclude');
try {
const gitDir = await fs.stat(gitDirPath);
if (!gitDir.isDirectory()) return;
} catch {
return;
}
let content = '';
try {
content = await fs.readFile(excludePath, 'utf-8');
} catch (err: any) {
if (err?.code !== 'ENOENT') throw err;
}
const excludes = content
.split(/\r?\n/)
.map((line) => line.trim())
.filter((line) => line && !line.startsWith('#'));
if (excludes.includes(GITNEXUS_DIR) || excludes.includes(GITNEXUS_EXCLUDE_ENTRY)) return;
const separator = content.length === 0 || content.endsWith('\n') ? '' : '\n';
try {
await fs.mkdir(path.dirname(excludePath), { recursive: true });
await fs.writeFile(excludePath, `${content}${separator}${GITNEXUS_EXCLUDE_ENTRY}\n`, 'utf-8');
} catch (err: any) {
if (isReadOnlyFilesystemError(err)) {
logger.warn(
{ path: excludePath, code: err.code },
'GitNexus storage filesystem is not writable; skipping .git/info/exclude update. .gitnexus/ cache directory may appear as untracked in `git status` locally.',
);
} else {
throw err;
}
}
};
// ─── Global Registry (~/.gitnexus/registry.json) ───────────────────────
/**
* Get the path to the global registry file
*/
export const getGlobalRegistryPath = (): string => {
return path.join(getGlobalDir(), 'registry.json');
};
/**
* Lock namespace for the global registry.
*
* Deliberately a dedicated sub-directory rather than {@link getGlobalDir}
* itself: an index slot's lock dir is always `<repo>/.gitnexus` (or
* `<repo>/.gitnexus/branches/<slug>`), so for a repository rooted at the
* user's home directory — dotfiles-at-`$HOME` is a real layout — the per-repo
* analyze lock and the global-dir lock would resolve to the SAME directory.
* `acquireIndexLock` is not reentrant, so `runFullAnalysis` (which holds the
* per-repo lock across its whole pipeline) would then self-deadlock the moment
* it reached `registerRepo`/`adoptFlatBranchLabel`. No repo's index slot can
* ever be named `registry-lock`, so this namespace cannot collide.
*/
const getRegistryLockDir = (): string => path.join(getGlobalDir(), 'registry-lock');
/**
* Wait ceiling for the registry lock. A registry transaction is a sub-second
* JSON read/merge/write, so it must NOT inherit the index lock's 10-minute
* default (sized for multi-minute analyze runs): `gitnexus augment` runs on
* every editor/agent tool call with a documented sub-500ms cold-start budget
* and reaches this lock via `listRegisteredRepos({ validate: true })`.
*/
const REGISTRY_LOCK_TIMEOUT_MS = 5_000;
/**
* Serialize global registry read/merge/write transactions across processes.
*
* The registry is shared by every indexed repository, so per-index locks do
* not protect this file. Reuse the cross-platform index lock primitive with a
* registry-private lock namespace; the handle is kernel-owned on supported
* platforms. The file fallback reclaims dead workload holders; an orphan
* acquisition guard requires quiesced recovery (RUNBOOK.md).
*
* On timeout the transaction fails closed: continuing unlocked would reintroduce
* the lost-update race this lock exists to prevent and can silently discard a
* concurrent registration.
*/
const withRegistryLock = async <T>(operation: () => Promise<T>): Promise<T> => {
let lock: IndexLockHandle | null = null;
try {
lock = await acquireIndexLock(getRegistryLockDir(), {
timeoutMs: REGISTRY_LOCK_TIMEOUT_MS,
// Registry contention was previously invisible: `acquireIndexLock`'s own
// `log` texts name an "analyze" holder, which misattributes a registry
// wait, so surface a registry-specific line instead (#2716 review).
onWaitStart: () =>
logger.info('Waiting for another GitNexus process to finish a registry update…'),
});
} catch (err) {
if (err instanceof IndexLockTimeoutError) {
logger.error(
{ timeoutMs: REGISTRY_LOCK_TIMEOUT_MS },
'Timed out waiting for the global registry lock; refusing an unlocked registry transaction.',
);
}
throw err;
}
try {
requireExclusiveIndexLock(
lock,
'Cannot acquire the global registry lock; refusing an unlocked registry transaction.',
);
return await operation();
} finally {
lock?.release();
}
};
/**
* Drop credentials from every entry's `remoteUrl` (#2914).
*
* Applied on BOTH registry edges. Capture-time stripping in `getRemoteUrl`
* only covers values this version writes; a `registry.json` (or a per-repo
* meta that a re-register copies forward) written by an older version still
* holds the credential. Reading through here keeps it out of every consumer —
* `listRegisteredRepos`, MCP `list_repos`, `gitnexus list`, group sync — and
* writing through here means the next registry write drops it at rest instead
* of round-tripping it back to disk.
*
* Sanitised values compare equal to a freshly captured `getRemoteUrl`, so
* sibling-clone matching (#2054) is unaffected: both sides lose the same span.
*/
const sanitizeEntries = (entries: RegistryEntry[]): RegistryEntry[] =>
entries.map((e) => {
if (!e.remoteUrl) return e;
const cleaned = stripUrlCredentials(e.remoteUrl);
return cleaned === e.remoteUrl ? e : { ...e, remoteUrl: cleaned };
});
/**
* A registry row we can actually resolve a repo from.
*
* `Array.isArray` is not enough on the strict path: `[{}]` is a JSON array, so
* a malformed registry passed the shape check, every configured repo failed to
* resolve, and — because none of them produced a load ERROR — the total-failure
* guard stayed off and a good contracts.json was replaced by an empty one. That
* is the same fail-open the strict mode exists to close, one level down from
* the file to the rows inside it.
*
* Only the three fields the resolution path actually depends on are required.
* `indexedAt` / `lastCommit` are deliberately NOT: callers already default them
* (`e?.indexedAt || ''`), so demanding them would reject a legacy row that
* resolves perfectly well — trading a fail-open for a fail-shut on real data.
*
* Two of the three must also be non-blank, because `typeof '' === 'string'`
* passes a row that cannot identify anything. `name` is what
* `defaultResolveHandle` matches a configured repo against, so a blank one
* matches nothing and puts every repo in `missingRepos` — the same fail-open,
* dressed as a clean answer. `storagePath` is what the resolved handle carries
* to `path.join(storagePath, 'lbug')`; blank, that joins to a relative `lbug`
* under the CWD, so the sync opens an index that is not the repo's.
*
* `path` stays at the bare string check, on the same reasoning that exempts
* `indexedAt` / `lastCommit`: require only what the resolution path depends on
* to IDENTIFY the repo. This check rejects the WHOLE registry, which is
* machine-wide, so a field tightened past what resolution needs would let one
* blank value in one row break every group sync on the machine — including
* groups whose repos all resolve.
*/
const isResolvableEntry = (value: unknown): value is RegistryEntry => {
if (!value || typeof value !== 'object' || Array.isArray(value)) return false;
const e = value as Record<string, unknown>;
const identifies = (v: unknown): boolean => typeof v === 'string' && v.trim() !== '';
return identifies(e.name) && identifies(e.storagePath) && typeof e.path === 'string';
};
/**
* Shared body for the two read modes below.
*
* `strict` distinguishes "the registry says nothing is registered" from "the
* registry could not be read". Lenient collapses both into `[]`.
*
* ENOENT is lenient in BOTH modes: no file genuinely means nothing has been
* registered yet, and every first-run path depends on that.
*/
const parseRegistryContents = async (raw: string, strict: boolean): Promise<RegistryEntry[]> => {
try {
// The parse gets its OWN guarded region, narrower than the checks below,
// and the parser's error is DISCARDED rather than rethrown.
//
// `JSON.parse`'s SyntaxError quotes a ten-character window of the source
// either side of the break — `Unexpected token 'L', ..."end.git"},<here>"...`.
// Registry rows carry remote URLs with their HTTPS userinfo verbatim, so a
// registry that breaks on one of those URLs puts the credential into that
// window, and the thrown message is not the only place it goes from there:
// `groupStatus` interpolates it into `unresolvableReason` for an MCP
// client, and `gitnexus group sync` prints it.
//
// Not logged and not attached as `cause` either, deliberately against this
// file's own convention of handing the `Error` object to the logger so it
// captures stack and cause: under MCP stdio the client writes those records
// to a log file on disk, so following the convention here would move the
// byte window from one channel to a more durable one. The parser's position
// offset is not worth a credential — the path and the failure class are
// what an operator acts on, and they are what the two errors below say too.
let data: unknown;
try {
data = JSON.parse(raw);
} catch {
throw new Error(`${getGlobalRegistryPath()} is not valid JSON (registry is corrupt)`);
}
if (!Array.isArray(data)) {
if (strict) {
throw new Error(`${getGlobalRegistryPath()} is not a JSON array (registry is corrupt)`);
}
return [];
}
// `storagePath` was not present in pre-external-storage registry files.
// Normalize only that legacy absence at the read boundary; malformed values
// remain visible to the strict destructive-operation safety checks below.
const entries = data.map((entry) =>
entry &&
typeof entry === 'object' &&
!Array.isArray(entry) &&
typeof (entry as Record<string, unknown>).path === 'string' &&
(entry as Record<string, unknown>).storagePath === undefined
? {
...(entry as Record<string, unknown>),
storagePath: defaultStoragePath((entry as Record<string, unknown>).path as string),
}
: entry,
) as RegistryEntry[];
if (strict) {
// Reject the WHOLE registry, never filter the bad rows out. Dropping them
// would report the repos they name as unregistered, which is precisely
// the unreadable-as-missing answer this mode refuses to give.
const bad = entries.findIndex((entry) => !isResolvableEntry(entry));
if (bad !== -1) {
throw new Error(
`${getGlobalRegistryPath()} entry ${bad} does not identify a repo — name and storagePath must be non-empty strings and path must be a string (registry is corrupt)`,
);
}
}
return sanitizeEntries(entries);
} catch (err) {
if (strict) throw err;
return [];
}
};
const readRegistryFile = async (strict: boolean): Promise<RegistryEntry[]> => {
let raw: string;
try {
raw = await fs.readFile(getGlobalRegistryPath(), 'utf-8');
} catch (err) {
if (strict && (err as NodeJS.ErrnoException).code !== 'ENOENT') throw err;
return [];
}
return parseRegistryContents(raw, strict);
};
/**
* Read the global registry. Returns empty array if not found — and, note, also
* when the file exists but cannot be read or parsed. That is fine for a
* read-only listing, where an unreadable registry and an empty one print the
* same nothing. It is not fine for a caller that ACTS on emptiness; see
* `readRegistryStrict`.
*/
export const readRegistry = async (): Promise<RegistryEntry[]> => readRegistryFile(false);
/**
* Read the global registry, refusing to report an unreadable one as empty.
*
* An EACCES after a `sudo gitnexus analyze`, a truncated registry.json, or an
* $HOME-on-NFS blip otherwise presents as "no repo is registered" — an
* unreadable condition reported as missing, which is exactly the conflation
* #3011 removes one frame further down. `syncGroup` is the caller that acts on
* that answer, by replacing a good contracts.json with an empty one.
*
* Deliberately a separate export rather than an option on `readRegistry`:
* leaving that signature untouched keeps every existing lenient call site
* provably unaffected, and the mode is legible at the call site.
*
* No count here on purpose. This comment carried one, it said nine, and the
* real figure was thirteen by the time anyone checked and fourteen shortly
* after — a number in prose beside code that moves is a claim that rots
* silently, which is the defect class this whole change set is about. The
* argument does not need the figure: it holds for one call site or fifty.
*/
export const readRegistryStrict = async (): Promise<RegistryEntry[]> => readRegistryFile(true);
/**
* Strict registry read that distinguishes "file is absent" from "file is
* empty or unreadable". ENOENT returns `undefined`; corrupt/unreadable
* files still throw. Callers that delete based on membership must not treat
* a missing file as an empty registry.
*/
export const readRegistryStrictIfPresent = async (): Promise<RegistryEntry[] | undefined> => {
let raw: string;
try {
raw = await fs.readFile(getGlobalRegistryPath(), 'utf-8');
} catch (err) {
if ((err as NodeJS.ErrnoException).code === 'ENOENT') return undefined;
throw err;
}
return parseRegistryContents(raw, true);
};
/**
* Write the global registry to disk.
*
* Atomic tmp+rename: a crash mid-write can never leave a truncated
* registry.json that the next load would treat as empty and silently drop
* every registered repo (#2106 R9). The tmp path must stay per-write — the
* registry is the one file every gitnexus process on the machine writes (#2888).
*/
const writeRegistry = async (entries: RegistryEntry[], attempts?: number): Promise<void> => {
const dir = getGlobalDir();
await fs.mkdir(dir, { recursive: true });
await writeFileAtomic(
getGlobalRegistryPath(),
JSON.stringify(sanitizeEntries(entries), null, 2),
attempts,
);
};
/**
* Options for {@link registerRepo}. All optional — callers without any
* disambiguation requirement can keep calling `registerRepo(path, meta)`
* unchanged.
*/
export interface RegisterRepoOptions {
/**
* User-provided alias from `analyze --name <alias>` (#829). Overrides
* the default basename-derived registry `name`. Persisted — subsequent
* re-analyses of the same path without `--name` preserve the alias.
*/
name?: string;
/**
* Best-effort notification after an explicit alias change has been committed
* to the registry. Invoked after the registry lock is released. Callback
* failures are ignored: reporting must not turn a successful registry write
* into an apparent transaction failure.
*/
onRename?: (previousName: string, nextName: string) => void | Promise<void>;
/**
* Allow two DIFFERENT repo paths to register under the same alias
* (#829). Mapped from the `--allow-duplicate-name` CLI flag.
*
* Scope: this flag governs cross-path alias sharing only — one repo
* path always has exactly one registry entry (and therefore exactly
* one alias). Re-analyzing the same path with `--name Y` overwrites
* a previous `--name X`; it does NOT create a second entry or a
* second alias for the same path (see the upsert-by-resolved-path
* logic in {@link registerRepo} and the
* `re-registerRepo with a different name overrides the previous
* alias` test in `test/unit/repo-manager.test.ts`).
*
* Distinct from `--force` (which only triggers pipeline re-index);
* a user accepting a duplicate alias should not be forced to also
* re-run the full pipeline.
*/
allowDuplicateName?: boolean;
/**
* Non-primary branch this run indexed (#2106). When set, the branch's
* summary is upserted into the entry's `branches[]` and the primary
* top-level fields are left untouched. When `undefined`, this is a
* primary/flat run that refreshes the top-level fields (and preserves any
* existing branch summaries).
*/
branch?: string;
/**
* The storage slot already selected and validated by the caller. Passing it
* prevents registry registration from re-resolving configuration after an
* analysis or index operation has begun.
*/
storagePath?: string;
/**
* Drop recorded `branches[]` summaries on a primary run. Set when the entry
* moves to a different storage location (a shared-store slot, #3352): the
* summaries name `branches/<slug>` sub-indexes the new location does not hold.
*/
dropBranches?: boolean;
}
/**
* Thrown by {@link registerRepo} when a requested name is already in
* use by a DIFFERENT path. The CLI layer surfaces this as an actionable
* error instead of relying on `.message` string-matching.
*
* The colliding alias is exposed as `err.registryName` (not `err.name`).
* `err.name` keeps its inherited `Error.prototype.name` semantics (the
* class name) so downstream code can do the usual `err.name ===
* 'RegistryNameCollisionError'` checks; use the `kind` discriminant or
* `instanceof RegistryNameCollisionError` for type-safe narrowing.
*/
export class RegistryNameCollisionError extends Error {
readonly kind = 'RegistryNameCollisionError' as const;
constructor(
public readonly registryName: string,
public readonly existingPath: string,
public readonly requestedPath: string,
) {
super(
`Registry name "${registryName}" is already used by "${existingPath}".\n` +
`Pass --name <alias> to register "${requestedPath}" under a different name, ` +
`or --allow-duplicate-name to allow both paths under the same name (leaves -r <name> ambiguous for these two).`,
);
this.name = 'RegistryNameCollisionError';
}
}
/** Returns true when a previously-registered entry's `name` differs from
* both `path.basename(entry.path)` and the git-remote-derived name —
* i.e. a user explicitly aliased it via `analyze --name <alias>` on a
* prior run. Used to preserve the alias across re-analyses that omit
* `--name`. The remote-derived name is treated as an inference, not a
* custom alias, so re-analyses keep tracking remote renames.
*
* `inferredName` is passed in (rather than re-derived) so callers can
* avoid a second `git config` subprocess invocation. */
const hasCustomAlias = (entry: RegistryEntry, inferredName: string | null): boolean => {
const resolved = path.resolve(entry.path);
if (entry.name === path.basename(resolved)) return false;
// Canonical-root-derived names are not user aliases either (#1259):
// a worktree registered under the canonical repo's basename
// (e.g. `{name: 'repo', path: '/repo/wt-feature'}`) must re-register
// cleanly without firing the duplicate-name collision guard. Without
// this check `entry.name = 'repo'` !== `path.basename('/repo/wt-feature') = 'wt-feature'`,
// so the prior check returns true → `isPreservedAlias = true` → guard
// throws `RegistryNameCollisionError` against the also-registered
// canonical checkout entry. The Claude-Code per-task worktree workflow
// — analyze canonical, then analyze worktree, then re-analyze worktree
// — would break on the third call.
if (entry.name === path.basename(resolveRepoIdentityRoot(resolved))) return false;
if (inferredName && entry.name === inferredName) return false;
return true;
};
type RegisterRepoUnlockedResult = {
name: string;
rename?: { previousName: string; nextName: string };
};
/**
* Register (add or update) a repo in the global registry.
* Called after `gitnexus analyze` completes.
*
* Name resolution precedence (#829, #979):
* 1. explicit `opts.name` (from `analyze --name <alias>`)
* 2. preserved alias on an existing entry for this path
* 3. `git config --get remote.origin.url` repo name (#979 — recovers
* a meaningful name for monorepo subprojects, git worktrees, and
* Gas-Town-style `<rig>/refinery/rig/` layouts where the basename
* is generic)
* 4. `path.basename(repoPath)` (the original default)
*
* Duplicate-name guard: if another path already uses the resolved
* `name`, throw {@link RegistryNameCollisionError} unless
* `opts.allowDuplicateName` is set. The guard ONLY fires when the user explicitly passed a
* `name`; un-aliased basename collisions continue to register silently
* so existing users who don't know about `--name` see no behaviour
* change.
*
* Returns the `name` that was actually written to the registry — the
* caller can re-use it to keep AGENTS.md / skill files aligned with the
* MCP-visible repo name (#979).
*/
const registerRepoUnlocked = async (
repoPath: string,
meta: RepoMeta,
opts?: RegisterRepoOptions,
): Promise<RegisterRepoUnlockedResult> => {
// Preserve the caller's chosen path form in the registry — don't
// canonicalise at write time. This matters for two reasons:
// 1. `list` and error messages show the path the user actually
// knows (e.g. the 8.3 short form they typed), not a runtime-
// resolved long form they've never seen.
// 2. Keeps pre-existing #829 test assertions that compare
// `err.existingPath` against `path.resolve(tmpPath)` stable.
// Canonicalisation is applied at COMPARE points only (see below),
// which is where the cross-platform divergence actually matters.
const resolved = path.resolve(repoPath);
const storagePath =
opts?.storagePath === undefined
? getStoragePaths(resolved).storagePath
: validateConfiguredStoragePath(opts.storagePath);
// Canonical form used strictly for comparison — `realpathSync.native`
// expands macOS /var → /private/var and Windows 8.3 → long-name,
// falling back to `path.resolve` when the path doesn't exist.
const canonicalInput = canonicalizePath(repoPath);
// Production write paths pass the storage slot they already validated. Do
// not let a changed environment/registry redirect their registry entry, and
// require the metadata receipt to describe that same repository and slot.
// The omitted-option path deliberately retains legacy direct-call behavior.
if (opts?.storagePath !== undefined) {
if (!registryPathEquals(canonicalizePath(meta.repoPath), canonicalInput)) {
throw new Error(
`Refusing to register ${resolved}: metadata belongs to ${meta.repoPath}, not this repository.`,
);
}
if (
meta.storagePath === undefined &&
!registryPathEquals(
canonicalizePath(storagePath),
canonicalizePath(defaultStoragePath(resolved)),
)
) {
throw new Error(
`Refusing to register ${resolved}: external storage metadata must bind storagePath to the selected directory.`,
);
}
if (
meta.storagePath !== undefined &&
!registryPathEquals(canonicalizePath(meta.storagePath), canonicalizePath(storagePath))
) {
throw new Error(
`Refusing to register ${resolved}: metadata storagePath does not match the selected storage directory.`,
);
}
}
// Mutating writes must not treat an unreadable/truncated registry as empty
// (#3094): lenient `readRegistry()` returns `[]` on parse failure and would
// replace the machine-wide file with only this entry. ENOENT stays empty.
const entries = await readRegistryStrict();
const existingIdx = entries.findIndex((e) => {
// Canonicalise the STORED entry too so pre-canonicalisation
// registries (written by older versions, or paths passed in a
// different form) still match correctly. `canonicalizePath` falls
// back to `path.resolve` when the path no longer exists on disk,
// so stale entries that have been rm'd externally still resolve
// to a stable key instead of throwing.
const a = canonicalizePath(e.path);
const b = canonicalInput;
return registryPathEquals(a, b);
});
const existing = existingIdx >= 0 ? entries[existingIdx] : null;
// Precedence: explicit --name > preserved alias > remote-inferred > basename.
// Skip the `git config` subprocess entirely when --name was passed —
// the remote isn't consulted in that case.
let name: string;
let isPreservedAlias = false;
if (opts?.name !== undefined) {
name = opts.name;
} else {
// Compute the remote-derived name at most once. It feeds both the
// alias-preservation check (`hasCustomAlias` needs it to distinguish
// a sticky user alias from a previously-stored remote inference) and
// the fallback name when neither --name nor a preserved alias apply.
const inferred = getInferredRepoName(resolved);
if (existing && hasCustomAlias(existing, inferred)) {
name = existing.name;
isPreservedAlias = true;
} else {
// Canonical-root fallback: when `resolved` is a worktree root,
// derive the registry name from the canonical repo's basename, not
// the worktree slug — see #1259. `resolveRepoIdentityRoot` confines
// the collapse to canonical checkouts and linked worktree roots only,
// so `--skip-git` subdirs of unrelated parent git repos keep using
// their own basename (preserves the #1232/#1233 fix's intent).
name = inferred ?? path.basename(resolveRepoIdentityRoot(resolved));
}
}
// Duplicate-name guard: only fire when the user EXPLICITLY asked for
// this name (via opts.name or a preserved alias). Unqualified basename
// and remote-inferred collisions are preserved for backward-compat —
// they still register, and the user sees the ambiguity at `-r` / `list`
// resolution time (which is already improved by the disambiguated error
// messages and list output #829 ships).
const explicitName = opts?.name !== undefined || isPreservedAlias;
if (explicitName && !opts?.allowDuplicateName) {
// Compare canonical-vs-canonical here too so `/var/foo` and
// `/private/var/foo` (same repo, different form) aren't treated as
// two colliding paths.
const collidingEntry = entries.find(
(e, i) =>
i !== existingIdx &&
e.name.toLowerCase() === name.toLowerCase() &&
canonicalizePath(e.path) !== canonicalInput,
);
if (collidingEntry) {
throw new RegistryNameCollisionError(name, collidingEntry.path, resolved);
}
}
// This run's branch summary (non-primary runs only); hoisted so the
// re-read-before-write merge below can re-apply it against a fresh snapshot.
const summary: BranchSummary | null = opts?.branch
? {
branch: opts.branch,
indexedAt: meta.indexedAt,
lastCommit: meta.lastCommit,
stats: meta.stats,
}
: null;
let entry: RegistryEntry;
if (summary) {
// Non-primary branch run (#2106): keep the primary's top-level fields and
// upsert this branch into branches[]. One entry per path is preserved.
// When the registry entry is missing (lost/rebuilt registry.json), rebuild
// the primary top-level from the FLAT meta.json rather than this branch's
// meta, so `--branch <primary>` can still resolve (#2106 review).
const flatMeta = existing ? null : await loadMeta(storagePath);
const base: RegistryEntry = existing ?? {
name,
path: resolved,
storagePath,
indexedAt: flatMeta?.indexedAt ?? meta.indexedAt,
lastCommit: flatMeta?.lastCommit ?? meta.lastCommit,
remoteUrl: flatMeta?.remoteUrl ?? meta.remoteUrl,
stats: flatMeta?.stats ?? meta.stats,
...(flatMeta?.branch ? { branch: flatMeta.branch } : {}),
};
const branches = (base.branches ?? []).filter((b) => b.branch !== summary.branch);
branches.push(summary);
entry = { ...base, name, branches };
} else {
// Primary/flat run: refresh top-level fields, preserve any branch summaries
// already recorded for this path so a primary re-analyze does not drop them.
entry = {
name,
path: resolved,
storagePath,
indexedAt: meta.indexedAt,
lastCommit: meta.lastCommit,
remoteUrl: meta.remoteUrl,
stats: meta.stats,
...(meta.branch ? { branch: meta.branch } : {}),
...(existing?.branches ? { branches: existing.branches } : {}),
};
}
// Re-read immediately before writing to narrow the lost-update window (#2106
// R9): re-derive THIS run's delta against the FRESHEST snapshot so a
// concurrent change to the OTHER axis (a branch upsert vs a primary refresh)
// survives instead of being clobbered by a stale entry-time view.
const fresh = await readRegistryStrict();
const freshIdx = fresh.findIndex((e) => {
const a = canonicalizePath(e.path);
return registryPathEquals(a, canonicalInput);
});
const freshExisting = freshIdx >= 0 ? fresh[freshIdx] : null;
let merged: RegistryEntry;
if (summary) {
// Branch run: keep the FRESH top-level + branches, just upsert our summary.
const base = freshExisting ?? entry;
const branches = (base.branches ?? []).filter((b) => b.branch !== summary.branch);
branches.push(summary);
merged = { ...base, name, branches };
} else {
// Primary run: apply our refreshed top-level, but defer to the FRESH
// branches[] (a concurrent branch upsert or `clean --branch` wins).
merged = { ...entry };
if (freshExisting?.branches && !opts?.dropBranches) merged.branches = freshExisting.branches;
else delete merged.branches;
if (freshExisting?.shareOptOut) merged.shareOptOut = true;
}
if (freshIdx >= 0) {
fresh[freshIdx] = merged;
} else {
fresh.push(merged);
}
await writeRegistry(fresh);
const rename =
opts?.name !== undefined && freshExisting && freshExisting.name !== name
? { previousName: freshExisting.name, nextName: name }
: undefined;
return { name, ...(rename ? { rename } : {}) };
};
export const registerRepo = async (
repoPath: string,
meta: RepoMeta,
opts?: RegisterRepoOptions,
): Promise<string> => {
const { name, rename } = await withRegistryLock(() => registerRepoUnlocked(repoPath, meta, opts));
if (rename) {
try {
await opts?.onRename?.(rename.previousName, rename.nextName);
} catch {
// The rename is already durable; observer failures cannot roll it back.
}
}
return name;
};
/**
* Remove a repo from the global registry.
* Called after `gitnexus clean`.
*/
const unregisterRepoUnlocked = async (repoPath: string): Promise<void> => {
// Canonicalise BOTH sides so an unregister call issued with the
// symlink form (`/var/folders/.../repo`) still matches an entry
// written with the realpath form (`/private/var/folders/.../repo`),
// and vice versa. Matches the semantics of `registerRepo` and
// `resolveRegistryEntry` post-#1003 review.
const resolved = canonicalizePath(repoPath);
// Same rule as `registerRepoUnlocked` (#3094): a mutating write must not
// treat an unreadable/truncated registry as empty. The lenient reader
// returned `[]` on any read error (EBUSY/EPERM racing another gitnexus
// process's atomic rename on Windows, EIO, a half-written file) and this
// function then wrote `[]` back — deregistering every repo on the machine
// to remove one. A missing file means nothing to remove.
const entries = await readRegistryStrictIfPresent();
if (entries === undefined) return;
const filtered = entries.filter((e) => !registryPathEquals(canonicalizePath(e.path), resolved));
if (filtered.length === entries.length) return;
await writeRegistry(filtered);
};
export const unregisterRepo = async (repoPath: string): Promise<void> =>
withRegistryLock(() => unregisterRepoUnlocked(repoPath));
/**
* Record (or clear) a checkout's opt-out from automatic clone sharing
* (#3352). A no-op when the checkout is not registered.
*/
export const setShareOptOut = async (repoPath: string, optOut: boolean): Promise<void> =>
withRegistryLock(async () => {
const entries = await readRegistryStrict();
const entry = findRegistryEntryByRepoPath(entries, repoPath);
if (!entry || !!entry.shareOptOut === optOut) return;
if (optOut) entry.shareOptOut = true;
else delete entry.shareOptOut;
await writeRegistry(entries);
});
/**
* Remove a single non-primary branch's summary from a repo's registry entry
* (#2106 R7). Called by `gitnexus clean --branch`. Returns `true` when a
* matching `branches[]` summary was found and removed; `false` otherwise (so
* the CLI can report "no such indexed branch" without crashing). The top-level
* primary entry is left intact; an empty `branches[]` is dropped to keep the
* registry shape legacy-clean.
*/
const removeBranchIndexUnlocked = async (repoPath: string, branch: string): Promise<boolean> => {
const resolved = canonicalizePath(repoPath);
const entries = await readRegistry();
const idx = entries.findIndex((e) => registryPathEquals(canonicalizePath(e.path), resolved));
if (idx < 0) return false;
const entry = entries[idx];
const before = entry.branches?.length ?? 0;
if (!entry.branches || before === 0) return false;
const remaining = entry.branches.filter((b) => b.branch !== branch);
if (remaining.length === before) return false; // branch not recorded
if (remaining.length > 0) entry.branches = remaining;
else delete entry.branches;
entries[idx] = entry;
await writeRegistry(entries);
return true;
};
export const removeBranchIndex = async (repoPath: string, branch: string): Promise<boolean> =>
withRegistryLock(() => removeBranchIndexUnlocked(repoPath, branch));
/**
* Record that the flat workspace slot now serves `branch` (#2354).
*
* The flat index follows the checked-out working tree, so when a plain
* analyze lands on a branch that also has a pinned `branches/<slug>/`
* sub-index, that sub-index becomes permanently shadowed — explicit
* `--branch` runs re-resolve to the flat slot and query-side branch scoping
* serves the flat handle first. Delete the shadowed directory and drop its
* registry summary in the same pass (leaving either half behind would strand
* un-cleanable disk bloat), and refresh the entry's top-level `branch` label
* so `list`/`list_repos`/branch-scoped queries stay coherent.
*
* Deliberately narrow for the analyze fast path: a missing registry entry is
* a no-op — including the sub-index deletion, which only runs for registered
* repos (never self-heals an unregistered repo, per #2264/#1169; the registry
* check precedes the rm per #2364 review F2) — and no subprocess is spawned.
*
* Only the closing re-read/mutate/write runs under the registry lock. The
* recursive `rm` stays outside it — mirroring `clean.ts`, which deletes the
* branch directory before calling the (locked) `removeBranchIndex` — so a slow
* delete (large sub-index, AV scan, network mount) never blocks every other
* registry operation on the machine.
*/
export const adoptFlatBranchLabel = async (
repoPath: string,
branch: string,
resolvedStoragePath?: string,
): Promise<void> => {
const canonicalInput = canonicalizePath(repoPath);
const isRegistered = (list: RegistryEntry[]): number =>
list.findIndex((e) => registryPathEquals(canonicalizePath(e.path), canonicalInput));
// Cheap membership gate only (#2364 review F2): never touch the disk for an
// unregistered repo. The mutate below re-reads its own fresh snapshot.
const initialEntries = await readRegistry();
const initialIdx = isRegistered(initialEntries);
if (initialIdx < 0) return; // no-op, disk included (no self-heal)
const resolved = path.resolve(repoPath);
const storagePath =
resolvedStoragePath === undefined
? getStoragePaths(resolved).storagePath
: validateConfiguredStoragePath(resolvedStoragePath);
const registeredStoragePath = initialEntries[initialIdx].storagePath;
if (!registryPathEquals(canonicalizePath(registeredStoragePath), canonicalizePath(storagePath))) {
throw new Error(
`Refusing to adopt branch metadata: the registry storage path does not match the selected storage directory.`,
);
}
if (resolvedStoragePath !== undefined) {
const inspection = await inspectStoragePath(storagePath, resolved);
if (inspection.state !== 'owned') {
throw new Error(
`Refusing to adopt branch metadata: storage is not owned by this repository (${inspection.state}).`,
);
}
}
// Remove a shadowed sub-index directory, mirroring `clean --branch`'s
// containment guard: the target MUST live under .gitnexus/branches/.
const branchesRoot = path.join(storagePath, BRANCHES_DIR);
const branchDir = path.join(branchesRoot, branchSlug(branch));
const branchRelativePath = path.relative(branchesRoot, branchDir);
const branchDirIsContained =
branchRelativePath !== '' &&
branchRelativePath !== '..' &&
!branchRelativePath.startsWith(`..${path.sep}`) &&
!path.isAbsolute(branchRelativePath);
let dirGone = false;
if (branchDirIsContained) {
let rmError: NodeJS.ErrnoException | undefined;
await fs.rm(branchDir, { recursive: true, force: true }).catch((err: unknown) => {
rmError = err as NodeJS.ErrnoException;
});
// The registry summary may be dropped only for a verifiably-gone
// directory: `clean --branch` resolves its target solely via the
// recorded summary, so dropping it while the dir survives (e.g. Windows
// EBUSY on an lbug held open by a live MCP server) would strand
// un-cleanable disk bloat (#2364 review F4). A resolved force:true rm
// proves absence; on failure, probe the disk and treat only
// provably-absent errno as gone — EACCES/EIO are "not provably absent",
// the same polarity as listRegisteredRepos({ validate: true }).
if (!rmError) {
dirGone = true;
} else {
const probeCode = await fs.access(branchDir).then(
() => null,
(e: unknown) => (e as NodeJS.ErrnoException)?.code ?? 'UNKNOWN',
);
dirGone = probeCode === 'ENOENT' || probeCode === 'ENOTDIR';
}
if (dirGone) {
// Non-recursive by design: only removes the parent when no other pinned
// sub-index remains, so an empty branches/ dir doesn't read as "pinned".
await fs.rmdir(path.join(storagePath, BRANCHES_DIR)).catch(() => {});
} else {
logger.warn(
{ path: branchDir, code: rmError?.code },
'Could not remove the shadowed branch sub-index; keeping its registry summary so `gitnexus clean --branch` can still target it.',
);
}
} else {
throw new Error('Refusing to adopt branch metadata: branch storage target escapes branches/.');
}
// Re-read AFTER the potentially slow recursive rm, and under the lock: the
// registry is a multi-writer whole-file overwrite, and writing a pre-rm
// snapshot would silently clobber concurrent registerRepo/removeBranchIndex
// writers — the #2106 R9 re-read-before-write discipline registerRepo follows.
await withRegistryLock(async () => {
const entries = await readRegistry();
const idx = isRegistered(entries);
if (idx < 0) return; // unregistered concurrently → still a no-op
const entry = entries[idx];
if (!registryPathEquals(canonicalizePath(entry.storagePath), canonicalizePath(storagePath))) {
return; // a concurrent registration selected a different slot
}
const remaining = dirGone ? entry.branches?.filter((b) => b.branch !== branch) : entry.branches;
const droppedSummary = (entry.branches?.length ?? 0) !== (remaining?.length ?? 0);
if (entry.branch === branch && !droppedSummary) return; // already coherent
entry.branch = branch;
if (remaining && remaining.length > 0) entry.branches = remaining;
else delete entry.branches;
entries[idx] = entry;
await writeRegistry(entries);
});
};
/**
* Thrown by {@link resolveRegistryEntry} when no registered repo matches
* the caller's target string (by alias, basename, remote-inferred name,
* or resolved path). CLI callers that want idempotent "remove" semantics
* should catch this and exit 0 with a warning; non-idempotent callers
* (e.g. MCP tools) can surface the error directly.
*/
export class RegistryNotFoundError extends Error {
readonly kind = 'RegistryNotFoundError' as const;
constructor(
public readonly target: string,
public readonly availableNames: string[],
) {
const hint =
availableNames.length > 0
? ` Available: ${availableNames.join(', ')}.`
: ' No repositories are currently registered.';
super(`No registered repo matches "${target}".${hint}`);
this.name = 'RegistryNotFoundError';
}
}
/**
* Thrown by {@link resolveRegistryEntry} when the target string matches
* the `name` of two or more entries — only possible when the user
* previously registered duplicates via `analyze --name X
* --allow-duplicate-name` (#829). The error carries enough information
* for the caller to render an actionable disambiguation hint without
* string-matching on `.message`.
*
* `kind` is a string literal discriminant (same pattern as
* {@link RegistryNameCollisionError}) so callers can narrow via
* `err.kind === 'RegistryAmbiguousTargetError'` without importing the
* class.
*/
export class RegistryAmbiguousTargetError extends Error {
readonly kind = 'RegistryAmbiguousTargetError' as const;
constructor(
public readonly target: string,
public readonly matches: RegistryEntry[],
) {
const listing = matches.map((m) => ` - ${m.name} (${m.path})`).join('\n');
super(
`Multiple registered repos match "${target}":\n${listing}\n` +
`Pass the absolute path instead to disambiguate.`,
);
this.name = 'RegistryAmbiguousTargetError';
}
}
/**
* Thrown by {@link assertAnalysisFinalized} when a successful `analyze`
* run did not actually persist the index metadata file or did not register
* the repo in `~/.gitnexus/registry.json` (#1169).
*
* Why this exists: on Windows, `gitnexus analyze` has been observed to
* exit cleanly (code 0) with `lbug.wal` written but no metadata file,
* leaving the repo invisible to `gitnexus list`/`status` and downstream
* MCP discovery. The only signal to the user was an empty banner —
* which is indistinguishable from a no-op early return. This invariant
* fails loudly with an actionable diagnostic so the silent-finalize bug
* surfaces with a non-zero exit code and a recoverable error message
* regardless of the upstream root cause (re-exec churn, native module
* side effects, antivirus, or future regressions).
*/
export class AnalysisNotFinalizedError extends Error {
readonly kind = 'AnalysisNotFinalizedError' as const;
constructor(
public readonly repoPath: string,
public readonly storagePath: string,
public readonly missing: 'meta' | 'registry-entry',
public readonly registryPath: string,
) {
const detail =
missing === 'meta'
? `${INDEX_METADATA_FILE} was not written to ${path.join(storagePath, INDEX_METADATA_FILE)}`
: `registry entry for ${repoPath} was not added to ${registryPath}`;
super(
`Analysis did not finalize for ${repoPath}: ${detail}. ` +
`The on-disk index is incomplete and was not registered. ` +
`Re-run "gitnexus analyze" — if the problem persists, inspect ` +
`${storagePath} for a stale lbug.wal that signals an aborted write.`,
);
this.name = 'AnalysisNotFinalizedError';
}
}
/**
* True when the global registry already contains an entry whose canonical path
* matches `repoPath`. Uses the same canonical, case-folded (Windows) comparison
* as {@link assertAnalysisFinalized} so "is it registered?" answers identically
* at the analyze fast-path gate and at the finalize assertion. Pure read.
*/
export const isRepoRegistered = async (repoPath: string): Promise<boolean> => {
const entries = await readRegistry();
const canonicalInput = canonicalizePath(path.resolve(repoPath));
return entries.some((e) => registryPathEquals(canonicalizePath(e.path), canonicalInput));
};
/**
* Verify that a successful `analyze` call actually produced an indexed,
* registered repo on disk. Two checks, both strictly required:
*
* 1. `gitnexus.json` must exist in the storage directory selected for this
* analysis (the caller may pass that exact directory).
* (the primary metadata file; the legacy `meta.json` mirror is not
* sufficient — a finalized analyze always writes the primary).
* 2. The global registry (`getGlobalRegistryPath()`) must contain an
* entry whose canonical path matches `repoPath`. The registry is read
* with {@link readRegistryStrict}: a corrupt or unreadable file throws
* rather than being treated as a missing registry-entry.
*
* Throws {@link AnalysisNotFinalizedError} on the first failure with the
* specific missing artifact. Pure read — does not mutate disk state.
*
* The optional `resolvedStoragePath` carries the exact slot used by the
* analysis. Omitting it preserves the legacy local-resolution behavior for
* direct callers.
*/
export const assertAnalysisFinalized = async (
repoPath: string,
resolvedStoragePath?: string,
): Promise<void> => {
const resolved = path.resolve(repoPath);
const storagePath =
resolvedStoragePath === undefined
? getStoragePaths(resolved).storagePath
: validateConfiguredStoragePath(resolvedStoragePath);
const metaPath = path.join(storagePath, INDEX_METADATA_FILE);
try {
await fs.access(metaPath);
} catch {
throw new AnalysisNotFinalizedError(resolved, storagePath, 'meta', getGlobalRegistryPath());
}
const canonicalRepoPath = canonicalizePath(resolved);
const canonicalStoragePath = canonicalizePath(storagePath);
const registeredAtStoragePath = (await readRegistryStrict()).some(
(entry) =>
registryPathEquals(canonicalizePath(entry.path), canonicalRepoPath) &&
registryPathEquals(canonicalizePath(entry.storagePath), canonicalStoragePath),
);
if (!registeredAtStoragePath) {
throw new AnalysisNotFinalizedError(
resolved,
storagePath,
'registry-entry',
getGlobalRegistryPath(),
);
}
};
/**
* Thrown by {@link assertSafeStoragePath} when {@link requireDeletableStoragePath}
* rejects the registry entry. Repository-local `.gitnexus` may still be
* removed when it is missing, empty, unowned, or foreign; an external slot
* is removable only when metadata binds both the repository and that exact
* path. CLI destructive commands (`remove`, `clean --all`) should catch this
* and exit non-zero without deleting anything.
*/
export class UnsafeStoragePathError extends Error {
readonly kind = 'UnsafeStoragePathError' as const;
constructor(
public readonly entry: RegistryEntry,
public readonly expectedStoragePath: string,
public readonly actualStoragePath: string,
) {
super(
`Refusing to remove storage path for safety: expected ` +
`"${expectedStoragePath}" under the repo's .gitnexus subfolder, ` +
`but the registry entry has "${actualStoragePath}". ` +
`This usually means the registry entry is corrupted or was ` +
`hand-edited. Delete the entry manually from ~/.gitnexus/registry.json ` +
`and re-run analyze.`,
);
this.name = 'UnsafeStoragePathError';
}
}
/**
* Guard rail for destructive CLI paths (`remove` #664,
* `clean --all` #258, future MCP `remove` tool): verify that a
* registry entry's `storagePath` names the registered index. Repository-local
* indexes are validated by their canonical `<repo>/.gitnexus` path; external
* slots must additionally prove their ownership through matching persisted
* metadata before a recursive deletion is allowed.
*
* Why this exists (#1003 review — @magyargergo):
* - `~/.gitnexus/registry.json` is a plain-text user-writable file.
* A corrupted, hand-edited, or downgrade/upgrade-racing entry
* could plausibly end up with `storagePath === ""` (resolves to
* cwd), `storagePath === path` (the repo root!), `storagePath`
* equal to a parent/sibling of the repo, or simply any arbitrary
* filesystem path.
* - `fs.rm(recursive: true, force: true)` on ANY of those would be
* a runtime disaster — at best delete the user's working tree, at
* worst nuke an unrelated directory tree they happen to own.
* - `clean` (default, cwd-scoped) is safe by construction — it
* re-derives storagePath from `findRepo(cwd)` and never trusts
* the registry field. But `clean --all` DOES iterate the registry
* and trust each entry's stored storagePath (same shape as
* `remove`), so this helper must be wired into that loop too.
* - An external slot is intentionally not constrained under a checkout. Its
* own metadata must bind both the source checkout path and the resolved
* storage path before it may be removed.
*
* The resolver preserves the legacy local-path allowance while also rejecting
* foreign or malformed metadata. External slots additionally require metadata
* ownership so a hand-edited registry cannot redirect a destructive command to
* an arbitrary directory.
*/
export const assertSafeStoragePath = async (entry: RegistryEntry): Promise<void> => {
try {
await requireDeletableStoragePath(entry);
} catch (error) {
if (error instanceof StorageDeletionError) {
throw new UnsafeStoragePathError(entry, error.expectedStoragePath, error.actualStoragePath);
}
throw error;
}
};
/**
* Resolve a user-supplied target string (from `gitnexus remove <target>`
* or equivalent MCP tool argument) to a single registry entry.
*
* Match precedence (first hit wins, subsequent tiers are only tried if
* the prior tier produces zero matches):
* 1. Exact resolved-path match (Windows: case-insensitive).
* Paths are unique by registry construction, so a path match can
* never be ambiguous.
* 2. Exact `name` match (case-insensitive). If ≥ 2 entries share the
* name — only possible via `--allow-duplicate-name` (#829) —
* throws {@link RegistryAmbiguousTargetError}.
*
* No fuzzy / partial matching — unambiguous, scriptable behaviour is
* more important than convenience for destructive commands.
*
* Throws {@link RegistryNotFoundError} if no entry matches.
*
* `entries` is passed in (rather than re-read) so callers that already
* hold the registry snapshot (e.g. to print a "before" state) can avoid
* a second disk read, and so tests can inject fixtures without touching
* `GITNEXUS_HOME`.
*/
export const resolveRegistryEntry = (entries: RegistryEntry[], target: string): RegistryEntry => {
// Tier 1: path match. Canonicalise BOTH sides so symlink and
// Windows-8.3 quirks don't cause a false miss — e.g. the caller
// passes `/var/folders/.../repo` while the registry has
// `/private/var/folders/.../repo` (both resolve to the same
// `realpath.native`). See `canonicalizePath` for the rationale.
//
// Canonicalising the STORED entry (not just the input) is what gives
// us backward-compat for registries written by versions that only
// ran `path.resolve` — both get canonicalised here at compare time.
const canonicalTarget = canonicalizePath(target);
const pathMatch = entries.find((e) => {
const a = canonicalizePath(e.path);
const b = canonicalTarget;
return registryPathEquals(a, b);
});
if (pathMatch) return pathMatch;
// Tier 2: name match. Case-insensitive on all platforms — registry
// name collisions are already filtered case-insensitively in
// `registerRepo`, so "APP" vs "app" are considered the same key.
const targetLower = target.toLowerCase();
const nameMatches = entries.filter((e) => e.name.toLowerCase() === targetLower);
if (nameMatches.length === 1) return nameMatches[0];
if (nameMatches.length > 1) {
throw new RegistryAmbiguousTargetError(target, nameMatches);
}
// Tier 3: miss. Build the available-names hint ONCE; resolveRepo-style
// disambiguated labels (`app (/path)`) are applied when the same name
// appears in multiple entries so the user sees the same hint shape as
// `-r <name>` errors.
const nameCounts = new Map<string, number>();
for (const e of entries) {
const key = e.name.toLowerCase();
nameCounts.set(key, (nameCounts.get(key) ?? 0) + 1);
}
const availableNames = entries.map((e) =>
(nameCounts.get(e.name.toLowerCase()) ?? 0) > 1 ? `${e.name} (${e.path})` : e.name,
);
throw new RegistryNotFoundError(target, availableNames);
};
/**
* Name-only registry match (the name tier of {@link resolveRegistryEntry},
* without path matching). Used by `group.yaml` member *values*, which are
* registry aliases, not filesystem paths.
*
* Zero matches → `undefined` (caller treats as missing). One match → that
* entry. Two or more → {@link RegistryAmbiguousTargetError}.
*/
export const findRegistryEntryByName = (
entries: RegistryEntry[],
name: string,
): RegistryEntry | undefined => {
const targetLower = name.toLowerCase();
const nameMatches = entries.filter((e) => e.name.toLowerCase() === targetLower);
if (nameMatches.length === 1) return nameMatches[0];
if (nameMatches.length > 1) {
throw new RegistryAmbiguousTargetError(name, nameMatches);
}
return undefined;
};
/**
* List all registered repos from the global registry.
*
* With `validate: true`, returns only entries whose storage is owned by the
* registered repository and contains a LadybugDB index. Entries whose storage
* path is provably gone, empty, or has no ownership metadata are pruned and
* persisted on a best-effort basis. A storage entry that cannot be inspected
* because of a transient filesystem error is kept in the returned view, so an
* I/O storm cannot make the registry disappear; it remains unconfirmed until a
* later validating read succeeds.
*/
export const listRegisteredRepos = async (opts?: {
validate?: boolean;
}): Promise<RegistryEntry[]> => {
const entries = await readRegistry();
if (!opts?.validate) return entries;
// Validate each entry through the shared storage resolver. The registry's
// storagePath is intentional here: GITNEXUS_STORAGE_PATH must not redirect
// validation of an explicitly registered repository to another slot.
const valid: RegistryEntry[] = [];
// Keep the exact inspected registry slot, not just the repository path.
// A concurrent analyze may re-register the same checkout into another
// external slot while this read-only validation walk is in flight.
const prunedSlots = new Set<string>();
const registrySlotKey = (entry: RegistryEntry): string =>
`${canonicalizePath(entry.path)}\0${canonicalizePath(entry.storagePath)}`;
const inspections = await mapPool(entries, inspectRegisteredStorage, 8);
for (const [entry, inspection] of entries.map(
(entry, i) => [entry, inspections[i]] as [RegistryEntry, (typeof inspections)[number]],
)) {
const meetsRequirements =
LIST_STORAGE_REQUIREMENTS.allowedStates.includes(inspection.state) &&
(!LIST_STORAGE_REQUIREMENTS.requireCodeIndexDB || inspection.hasCodeIndexDB);
if (meetsRequirements) {
valid.push(entry);
} else if (
inspection.state === 'missing' ||
inspection.state === 'empty' ||
(inspection.state === 'unowned' &&
inspection.reason === 'Storage directory contains data but no valid ownership metadata.')
) {
// A missing/empty directory or a registry slot with no ownership
// metadata is not a usable index. Removing only the registry row is
// safe; the storage directory itself is never deleted here.
prunedSlots.add(registrySlotKey(entry));
} else if (isTransientStorageInspection(inspection)) {
// Not provably absent or invalid. Keep the old safety behavior for
// EIO/EAGAIN/EBUSY/EACCES-style filesystem failures.
valid.push(entry);
} else {
logger.warn(
{
name: entry.name,
storagePath: entry.storagePath,
state: inspection.state,
hasCodeIndexDB: inspection.hasCodeIndexDB,
reason: inspection.reason,
},
'Skipping registry entry during validation because its storage is not a usable owned code index.',
);
}
}
// If we pruned any entries, save the cleaned registry — under the lock, and
// only then. The validation walk above is read-only and can touch several
// files per entry, so holding the global lock across it would serialize
// every `gitnexus augment` behind unrelated registry work for no benefit.
// Re-read inside the lock and drop only the same provably-absent storage
// slots from that fresh snapshot, so a concurrent re-registration of the
// same repository path into another slot survives.
if (prunedSlots.size > 0) {
try {
await withRegistryLock(async () => {
const fresh = await readRegistry();
await writeRegistry(
fresh.filter((entry) => !prunedSlots.has(registrySlotKey(entry))),
1,
);
});
} catch (err) {
// Best-effort housekeeping: callers consume the returned view, and the
// prune set is recomputed on the next validating read. It must not throw
// — this runs on MCP startup (LocalBackend.init → refreshRepos), where
// nothing catches and a rejection reads as "Server disconnected".
logger.warn(
{ err, prunedCount: prunedSlots.size },
'Could not persist the pruned global registry; continuing with the in-memory pruned view.',
);
}
}
return valid;
};
// ─── Global CLI Config (~/.gitnexus/config.json) ─────────────────────────
export interface CLIConfig {
apiKey?: string;
model?: string;
baseUrl?: string;
provider?:
| 'openai'
| 'openrouter'
| 'azure'
| 'custom'
| 'cursor'
| 'claude'
| 'codex'
| 'opencode'
| 'grok'
| 'minimax';
cursorModel?: string;
claudeModel?: string;
codexModel?: string;
opencodeModel?: string;
grokModel?: string;
/** Azure api-version query param (e.g. '2024-10-21'). Only used when provider is 'azure'. */
apiVersion?: string;
/** Set true when the deployment is a reasoning model (o1, o3, o4-mini). Auto-detected for OpenAI; must be set for Azure deployments. */
isReasoningModel?: boolean;
}
/**
* Get the path to the global CLI config file
*/
export const getGlobalConfigPath = (): string => {
return path.join(getGlobalDir(), 'config.json');
};
/**
* Load CLI config from ~/.gitnexus/config.json
*/
export const loadCLIConfig = async (): Promise<CLIConfig> => {
try {
const raw = await fs.readFile(getGlobalConfigPath(), 'utf-8');
return JSON.parse(raw) as CLIConfig;
} catch {
return {};
}
};
/**
* Save CLI config to ~/.gitnexus/config.json
*/
export const saveCLIConfig = async (config: CLIConfig): Promise<void> => {
const dir = getGlobalDir();
await fs.mkdir(dir, { recursive: true });
const configPath = getGlobalConfigPath();
await fs.writeFile(configPath, JSON.stringify(config, null, 2), 'utf-8');
// Restrict file permissions on Unix (config may contain API keys)
if (process.platform !== 'win32') {
try {
await fs.chmod(configPath, 0o600);
} catch {
/* best-effort */
}
}
};
// ─── Sibling-clone detection ─────────────────────────────────────────────
//
// A "sibling clone" is a different on-disk path that points at the same
// logical repository (same `origin` remote URL) as a registered index.
// This shows up in three operationally important shapes (see issue):
//
// 1. The same repo is checked out under multiple paths (worktrees,
// multi-agent workspaces). Only one is indexed; the others silently
// diverge from the graph.
// 2. The indexed clone is itself behind its own HEAD (the existing
// `checkStaleness` already handles this case).
// 3. A query is issued from a `cwd` that lives inside a sibling clone
// whose HEAD has drifted from the indexed `lastCommit`.
//
// Detection is intentionally remote-URL-based and does NOT walk the
// filesystem hunting for unregistered clones — only registered entries
// are considered. The `cwd`-driven branch ({@link checkSiblingDrift})
// also accepts an unregistered cwd, because the live caller's working
// directory is the one place we can cheaply learn about an
// unregistered clone.
/**
* Find other registered entries whose `remoteUrl` matches the given
* one, excluding `selfPath` (case-insensitive on Windows). Entries
* without a `remoteUrl` are ignored — we cannot prove sibling-ness
* without a fingerprint.
*/
export const findSiblingClones = async (
remoteUrl: string | undefined,
selfPath: string,
): Promise<RegistryEntry[]> => {
if (!remoteUrl) return [];
const entries = await readRegistry();
const isWin = process.platform === 'win32';
const norm = (p: string) => (isWin ? path.resolve(p).toLowerCase() : path.resolve(p));
const self = norm(selfPath);
return entries.filter((e) => e.remoteUrl === remoteUrl && norm(e.path) !== self);
};
/**
* Description of how a working directory relates to a registered index.
*
* `match` semantics:
* - `path` — `cwd` is inside the registered entry's path.
* - `sibling-by-remote` — `cwd` is in a different on-disk clone of the
* same repo (same `remoteUrl`).
* - `none` — no relationship found.
*/
export interface CwdMatch {
match: 'path' | 'sibling-by-remote' | 'none';
entry?: RegistryEntry;
/** The git toplevel of `cwd`, when `cwd` is inside a git work tree. */
cwdGitRoot?: string;
/** HEAD of the cwd's clone, when resolvable. */
cwdHead?: string;
/**
* Number of commits the registered `lastCommit` is behind the
* sibling-clone HEAD, when both refs are known to the cwd's clone.
* `undefined` when the comparison cannot be performed (e.g. the
* indexed commit isn't reachable from cwd).
*/
drift?: number;
/** Human-readable hint, set whenever the situation warrants warning. */
hint?: string;
}