mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-05 02:43:32 +00:00
Merge branch 'main' into codex/cpp-braced-init-list-overload
This commit is contained in:
commit
8abfb2f1d6
46 changed files with 4627 additions and 684 deletions
|
|
@ -17,3 +17,9 @@ WEB_HOST_PORT=4173
|
|||
# Optional read-only mount, exposed to the server as /workspace.
|
||||
# Override with the directory that contains the repos you want to index.
|
||||
WORKSPACE_DIR=./
|
||||
|
||||
# Azure DevOps Server Integration (passed to the server container)
|
||||
# Prefer https:// — the PAT rides in an Authorization header, so cleartext
|
||||
# http:// exposes it on the wire (still supported for internal-only instances).
|
||||
# AZURE_DEVOPS_URL=https://azuredevops.example.com
|
||||
# AZURE_DEVOPS_PAT=your-pat-here
|
||||
|
|
|
|||
18
.github/workflows/ci-tests.yml
vendored
18
.github/workflows/ci-tests.yml
vendored
|
|
@ -276,6 +276,24 @@ jobs:
|
|||
run: node --expose-gc --import tsx bench/cfg/measure.mjs --check
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Emit-persistence throughput / byte-identity guards (#2203)
|
||||
# Build-free: asserts streamAllCSVsToDisk output is byte-identical
|
||||
# (order-independent CSV-line fingerprint — the #2203 U2/U3 emit
|
||||
# optimisations must not change graph content) and that emit wall-time
|
||||
# stays linear in node+edge count. The LadybugDB COPY half needs a real
|
||||
# DB, so its timing lives in the runtime PROF_LBUG_LOAD breakdown.
|
||||
run: node --import tsx bench/emit-persistence/measure.mjs --check
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Streaming PDG-emit byte-identity / bounded-RSS guards (#2202)
|
||||
# Build-free: asserts the streaming PdgEmitSink emits a CSV row SET
|
||||
# byte-identical to the whole-graph streamAllCSVsToDisk emit, AND that
|
||||
# the in-memory graph retains zero BasicBlock nodes (the O(chunk) peak-RSS
|
||||
# bound that unblocks full-kernel-scale repos). Fails on fingerprint drift
|
||||
# or any resident BasicBlock.
|
||||
run: node --import tsx bench/emit-persistence/measure-streaming.mjs --check
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Cross-language pipeline benchmarks (GITNEXUS_BENCH, serial)
|
||||
env:
|
||||
GITNEXUS_BENCH: '1'
|
||||
|
|
|
|||
|
|
@ -3,10 +3,17 @@ title = "GitNexus"
|
|||
[extend]
|
||||
useDefault = true
|
||||
|
||||
# Fake embedding API keys in unit tests (current probe + historical placeholder).
|
||||
# Fake credentials in unit tests — none are real secrets:
|
||||
# - embedding API keys in the http-embedder tests (regexes below)
|
||||
# - synthetic GitHub PAT fixtures in the git-clone PAT-injection tests
|
||||
# (e.g. ghp_secret123, ghp_uniqueRawSecret_98765) — allowlisted by path
|
||||
# so the exception is bounded to that one test file.
|
||||
[allowlist]
|
||||
description = "fake embedding API keys in http-embedder unit tests"
|
||||
description = "fake credentials in unit tests (no real secrets)"
|
||||
regexes = [
|
||||
'''secret-key-12345''',
|
||||
'''test-api-key-redaction-check''',
|
||||
]
|
||||
paths = [
|
||||
'''gitnexus/test/unit/git-clone\.test\.ts''',
|
||||
]
|
||||
|
|
|
|||
|
|
@ -324,6 +324,7 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max
|
|||
| `GITNEXUS_VERBOSE` | unset | When `1`, enables verbose ingestion logs (skipped-file warnings, per-chunk throughput, parse-cache stats). Equivalent to `--verbose`. | Debugging an analyze that "completed" but seems to have missed files; tuning `--workers` / chunk concurrency against observable throughput. |
|
||||
| `GITNEXUS_PROFILE_DEFERRED` | unset | When `1`, emits `[deferred-profile]` timing/progress logs for the post-chunk deferred resolution band (imports → heritage → buildHeritageMap → legacy call resolution). Implied by `GITNEXUS_VERBOSE`. | Diagnosing analyze stalls in "Resolving calls (all chunks)" on large Java/Kotlin repos (issue #1741) without the full verbose ingestion noise. |
|
||||
| `GITNEXUS_PROFILE_DEFERRED_SLOW_MS` | `3000` (verbose) / `5000` | Per-file threshold in ms above which `processCallsFromExtracted` emits a `slow file …` log line. Parsed via `Number()`: accepts integers (`5000`), scientific notation (`2.5e3`), decimals (`.5`), and hex (`0x10`). Non-finite or non-positive values fall back to the default. | Hunting a few outlier files dominating the deferred call-resolution stage; lower to surface more, raise to focus only on the worst. |
|
||||
| `PROF_LBUG_LOAD` | unset | When `1`, emits one `[lbug-load prof]` summary line per `loadGraphToLbug` call breaking the graph-DB persistence wall into stages (`csv-emit` / `copy-nodes` / `copy-rels` / `fallback` / `total`) plus node & edge counts. Zero-cost when unset. | Attributing large-repo analyze wall time across CSV generation vs. LadybugDB `COPY` (issue #2203) — the analyze "emit" timing is the scope-resolution bucket, not this DB-write path. |
|
||||
| `GITNEXUS_MAX_FILE_SIZE` | `512` (KB) | Walker skip threshold in KB. Hard cap is `32768` (tree-sitter buffer ceiling). Equivalent to `--max-file-size <kb>`. | Indexing repos with intentionally-large source files (generated parsers, vendored bundles) that should still be parsed. |
|
||||
| `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS` | `30000` | Worker idle timeout in milliseconds before retry/fallback. Equivalent to `--worker-timeout <seconds>` × 1000. | Slow-parsing files (large minified JS, deeply-nested TS types) that legitimately need more than 30s. |
|
||||
| `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold in bytes. Equivalent to `--wal-checkpoint-threshold <bytes>`. `-1` keeps LadybugDB's stock threshold (~16 MiB). Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | You need a larger or smaller WAL auto-checkpoint threshold for your analyze workload. |
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@
|
|||
*
|
||||
* The "empty state" card rendered inside DropZone's Crossfade when the server
|
||||
* is connected but zero repos are indexed. Replaces the generic error message
|
||||
* with a first-class GitHub URL input flow.
|
||||
* with a first-class repository URL input flow.
|
||||
*
|
||||
* Rendering context:
|
||||
* DropZone (Crossfade, phase="analyze")
|
||||
|
|
@ -15,7 +15,7 @@
|
|||
* the app to the graph explorer.
|
||||
*/
|
||||
|
||||
import { Sparkles, Github } from '@/lib/lucide-icons';
|
||||
import { Sparkles, GitBranch } from '@/lib/lucide-icons';
|
||||
import { RepoAnalyzer } from './RepoAnalyzer';
|
||||
import { useTranslation } from 'react-i18next';
|
||||
|
||||
|
|
@ -46,7 +46,7 @@ export const AnalyzeOnboarding = ({ onComplete }: AnalyzeOnboardingProps) => {
|
|||
|
||||
{/* Icon */}
|
||||
<div className="mx-auto mb-4 flex h-14 w-14 items-center justify-center rounded-2xl border border-accent/30 bg-gradient-to-br from-accent/20 to-accent-dim/10 shadow-glow-soft">
|
||||
<Github className="h-7 w-7 text-accent" />
|
||||
<GitBranch className="h-7 w-7 text-accent" />
|
||||
</div>
|
||||
|
||||
<h2 className="text-lg leading-snug font-semibold text-text-primary">
|
||||
|
|
|
|||
|
|
@ -10,12 +10,14 @@ import { useState, useRef, useEffect, useId } from 'react';
|
|||
import {
|
||||
Github,
|
||||
Gitlab,
|
||||
AzureDevops,
|
||||
FolderOpen,
|
||||
Loader2,
|
||||
Check,
|
||||
ArrowRight,
|
||||
AlertCircle,
|
||||
Sparkles,
|
||||
Key,
|
||||
} from '@/lib/lucide-icons';
|
||||
import {
|
||||
startAnalyze,
|
||||
|
|
@ -30,10 +32,14 @@ import { useTranslation } from 'react-i18next';
|
|||
|
||||
// ── Helpers ──────────────────────────────────────────────────────────────────
|
||||
|
||||
type InputMode = 'github' | 'gitlab' | 'local';
|
||||
type InputMode = 'github' | 'gitlab' | 'azure' | 'local';
|
||||
|
||||
const GITHUB_RE = /^https?:\/\/(www\.)?github\.com\/[^/\s]+\/[^/\s]+/i;
|
||||
const GITLAB_RE = /^https?:\/\/[^/\s]+\/[^/\s]+\/[^/\s]+(\/.*)?$/i;
|
||||
// One-or-more path segments before `/_git/`, so the legacy single-project
|
||||
// cloud form (myorg.visualstudio.com/project/_git/repo) is accepted too —
|
||||
// the backend already supports it (isAzureDevOpsUrl / extractRepoName).
|
||||
const AZURE_RE = /^https?:\/\/[^/\s]+\/(?:[^/\s]+\/)+_git\/[^/\s]+/i;
|
||||
const IS_WINDOWS = navigator.userAgent.toLowerCase().includes('win');
|
||||
|
||||
function isValidGithubUrl(value: string): boolean {
|
||||
|
|
@ -44,6 +50,10 @@ function isValidGitlabUrl(value: string): boolean {
|
|||
return GITLAB_RE.test(value.trim());
|
||||
}
|
||||
|
||||
function isValidAzureUrl(value: string): boolean {
|
||||
return AZURE_RE.test(value.trim());
|
||||
}
|
||||
|
||||
// ── Mode tabs ────────────────────────────────────────────────────────────────
|
||||
|
||||
function ModeTabs({ mode, onChange }: { mode: InputMode; onChange: (m: InputMode) => void }) {
|
||||
|
|
@ -81,6 +91,19 @@ function ModeTabs({ mode, onChange }: { mode: InputMode; onChange: (m: InputMode
|
|||
<Gitlab className="h-3 w-3" />
|
||||
{t('repoAnalyzer.gitlabUrl')}
|
||||
</button>
|
||||
<button
|
||||
role="tab"
|
||||
aria-selected={mode === 'azure'}
|
||||
onClick={() => onChange('azure')}
|
||||
className={`flex flex-1 cursor-pointer items-center justify-center gap-1.5 rounded-md px-3 py-1.5 text-xs font-medium transition-all duration-150 ${
|
||||
mode === 'azure'
|
||||
? 'bg-accent text-white shadow-sm'
|
||||
: 'text-text-muted hover:text-text-secondary'
|
||||
} `}
|
||||
>
|
||||
<AzureDevops className="h-3 w-3" />
|
||||
{t('repoAnalyzer.azureDevOpsUrl')}
|
||||
</button>
|
||||
<button
|
||||
role="tab"
|
||||
aria-selected={mode === 'local'}
|
||||
|
|
@ -173,7 +196,9 @@ export const RepoAnalyzer = ({ variant, onComplete, onCancel }: RepoAnalyzerProp
|
|||
null,
|
||||
);
|
||||
const [githubUrl, setGithubUrl] = useState('');
|
||||
const [githubToken, setGithubToken] = useState('');
|
||||
const [gitlabUrl, setGitlabUrl] = useState('');
|
||||
const [azureUrl, setAzureUrl] = useState('');
|
||||
const [localPath, setLocalPath] = useState('');
|
||||
const [phase, setPhase] = useState<InternalPhase>('input');
|
||||
const [validationError, setValidationError] = useState<string | null>(null);
|
||||
|
|
@ -239,7 +264,9 @@ export const RepoAnalyzer = ({ variant, onComplete, onCancel }: RepoAnalyzerProp
|
|||
invalidateRequest();
|
||||
setMode(m);
|
||||
setGithubUrl('');
|
||||
setGithubToken('');
|
||||
setGitlabUrl('');
|
||||
setAzureUrl('');
|
||||
setLocalPath('');
|
||||
setValidationError(null);
|
||||
setUploadSummary(null);
|
||||
|
|
@ -259,7 +286,9 @@ export const RepoAnalyzer = ({ variant, onComplete, onCancel }: RepoAnalyzerProp
|
|||
? isValidGithubUrl(githubUrl) && (phase === 'input' || phase === 'error')
|
||||
: mode === 'gitlab'
|
||||
? isValidGitlabUrl(gitlabUrl) && (phase === 'input' || phase === 'error')
|
||||
: localPath.trim().length > 1 && (phase === 'input' || phase === 'error');
|
||||
: mode === 'azure'
|
||||
? isValidAzureUrl(azureUrl) && (phase === 'input' || phase === 'error')
|
||||
: localPath.trim().length > 1 && (phase === 'input' || phase === 'error');
|
||||
|
||||
const handleAnalyze = async () => {
|
||||
if (mode === 'github' && !isValidGithubUrl(githubUrl)) {
|
||||
|
|
@ -270,6 +299,10 @@ export const RepoAnalyzer = ({ variant, onComplete, onCancel }: RepoAnalyzerProp
|
|||
setValidationError('Please enter a valid GitLab repository URL.');
|
||||
return;
|
||||
}
|
||||
if (mode === 'azure' && !isValidAzureUrl(azureUrl)) {
|
||||
setValidationError(t('errors:invalidAzureDevOpsUrl'));
|
||||
return;
|
||||
}
|
||||
if (mode === 'local' && localPath.trim().length < 2) {
|
||||
setValidationError(t('errors:missingFolderPath'));
|
||||
return;
|
||||
|
|
@ -285,10 +318,15 @@ export const RepoAnalyzer = ({ variant, onComplete, onCancel }: RepoAnalyzerProp
|
|||
try {
|
||||
const request =
|
||||
mode === 'github'
|
||||
? { url: githubUrl.trim() }
|
||||
? {
|
||||
url: githubUrl.trim(),
|
||||
...(githubToken.trim() ? { token: githubToken.trim() } : {}),
|
||||
}
|
||||
: mode === 'gitlab'
|
||||
? { url: gitlabUrl.trim() }
|
||||
: { path: localPath.trim() };
|
||||
: mode === 'azure'
|
||||
? { url: azureUrl.trim() }
|
||||
: { path: localPath.trim() };
|
||||
const { jobId } = await startAnalyze(request);
|
||||
// Stale resolution: return without cancelling — URL jobIds may be
|
||||
// dedup-aliased to a job another session owns (see cancelStaleUploadJob).
|
||||
|
|
@ -299,7 +337,9 @@ export const RepoAnalyzer = ({ variant, onComplete, onCancel }: RepoAnalyzerProp
|
|||
? githubUrl.trim()
|
||||
: mode === 'gitlab'
|
||||
? gitlabUrl.trim()
|
||||
: localPath.trim();
|
||||
: mode === 'azure'
|
||||
? azureUrl.trim()
|
||||
: localPath.trim();
|
||||
trackJob(jobId, nameSource);
|
||||
} catch (err) {
|
||||
// Unmount aborts the controller, so this also covers the unmounted case.
|
||||
|
|
@ -327,6 +367,7 @@ export const RepoAnalyzer = ({ variant, onComplete, onCancel }: RepoAnalyzerProp
|
|||
: undefined) ??
|
||||
t('onboarding:repoAnalyzer.defaultRepoName');
|
||||
setCompletedRepoName(name);
|
||||
setGithubToken('');
|
||||
setPhase('done');
|
||||
sseControllerRef.current = null;
|
||||
completeTimerRef.current = setTimeout(() => {
|
||||
|
|
@ -396,6 +437,7 @@ export const RepoAnalyzer = ({ variant, onComplete, onCancel }: RepoAnalyzerProp
|
|||
} catch {}
|
||||
jobIdRef.current = null;
|
||||
}
|
||||
setGithubToken('');
|
||||
setPhase('input');
|
||||
setProgress({ phase: 'queued', percent: 0, message: t('common:analyzePhases.queued') });
|
||||
setUploading(false);
|
||||
|
|
@ -460,6 +502,39 @@ export const RepoAnalyzer = ({ variant, onComplete, onCancel }: RepoAnalyzerProp
|
|||
</div>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Optional GitHub Personal Access Token for private repos */}
|
||||
<div className="space-y-1.5 pt-1">
|
||||
<label
|
||||
htmlFor={`${inputId}-token`}
|
||||
className="block text-xs font-medium tracking-wider text-text-secondary uppercase"
|
||||
>
|
||||
{t('onboarding:repoAnalyzer.githubTokenLabel')}
|
||||
</label>
|
||||
<div className="flex items-center gap-3 rounded-xl border border-border-default bg-void px-4 py-3 transition-all duration-200 focus-within:border-accent/40">
|
||||
<Key className="h-4 w-4 shrink-0 text-text-muted" />
|
||||
<input
|
||||
id={`${inputId}-token`}
|
||||
type="password"
|
||||
value={githubToken}
|
||||
onChange={(e) => setGithubToken(e.target.value)}
|
||||
onKeyDown={(e) => {
|
||||
if (e.key === 'Enter' && canSubmit && !isLoading) {
|
||||
e.preventDefault();
|
||||
handleAnalyze();
|
||||
}
|
||||
}}
|
||||
disabled={isLoading}
|
||||
placeholder={t('onboarding:repoAnalyzer.githubTokenPlaceholder')}
|
||||
autoComplete="off"
|
||||
spellCheck={false}
|
||||
className="flex-1 border-none bg-transparent font-mono text-sm text-text-primary outline-none placeholder:text-text-muted disabled:opacity-50"
|
||||
/>
|
||||
</div>
|
||||
<p className="text-xs text-text-muted">
|
||||
{t('onboarding:repoAnalyzer.githubTokenHelp')}
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
|
|
@ -516,6 +591,61 @@ export const RepoAnalyzer = ({ variant, onComplete, onCancel }: RepoAnalyzerProp
|
|||
</div>
|
||||
)}
|
||||
|
||||
{/* Azure DevOps URL input */}
|
||||
{showInput && mode === 'azure' && (
|
||||
<div className="space-y-2">
|
||||
<label
|
||||
htmlFor={inputId}
|
||||
className="block text-xs font-medium tracking-wider text-text-secondary uppercase"
|
||||
>
|
||||
{t('onboarding:repoAnalyzer.azureDevOpsRepositoryUrl')}
|
||||
</label>
|
||||
<div
|
||||
className={`flex items-center gap-3 rounded-xl border bg-void px-4 py-3.5 transition-all duration-200 ${
|
||||
validationError && phase === 'error'
|
||||
? 'border-red-500/50'
|
||||
: isValidAzureUrl(azureUrl)
|
||||
? 'border-accent/50 shadow-[0_0_0_3px_rgba(124,58,237,0.08)]'
|
||||
: 'border-border-default focus-within:border-accent/40'
|
||||
} `}
|
||||
>
|
||||
<AzureDevops className="h-4 w-4 shrink-0 text-text-muted" />
|
||||
<input
|
||||
id={inputId}
|
||||
type="url"
|
||||
value={azureUrl}
|
||||
onChange={(e) => {
|
||||
setAzureUrl(e.target.value);
|
||||
if (validationError) setValidationError(null);
|
||||
}}
|
||||
onKeyDown={(e) => {
|
||||
if (e.key === 'Enter' && canSubmit && !isLoading) {
|
||||
e.preventDefault();
|
||||
handleAnalyze();
|
||||
}
|
||||
}}
|
||||
disabled={isLoading}
|
||||
placeholder="http://azuredevops.example.com/Collection/Project/_git/Repo"
|
||||
autoComplete="url"
|
||||
spellCheck={false}
|
||||
className="flex-1 border-none bg-transparent font-mono text-sm text-text-primary outline-none placeholder:text-text-muted disabled:opacity-50"
|
||||
/>
|
||||
{azureUrl.length > 10 && (
|
||||
<div className="shrink-0">
|
||||
{isValidAzureUrl(azureUrl) ? (
|
||||
<Check className="h-3.5 w-3.5 text-emerald-400" />
|
||||
) : (
|
||||
<AlertCircle className="h-3.5 w-3.5 text-text-muted" />
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
<p className="text-xs text-text-muted">
|
||||
{t('onboarding:repoAnalyzer.azureDevOpsSupported')}
|
||||
</p>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Local folder input */}
|
||||
{showInput && mode === 'local' && (
|
||||
<div className="space-y-2">
|
||||
|
|
|
|||
|
|
@ -185,6 +185,41 @@ export const Gitlab = forwardRef<SVGSVGElement, LucideProps>(function Gitlab(
|
|||
);
|
||||
});
|
||||
|
||||
/**
|
||||
* Azure DevOps mark — SVG path data from simple-icons (CC0-1.0).
|
||||
*
|
||||
* The Azure DevOps logo is a registered trademark of Microsoft Corporation.
|
||||
* We use it here only to indicate Azure DevOps source-repo integration.
|
||||
*
|
||||
* API-compatible with `lucide-react` icons (`LucideProps`).
|
||||
*/
|
||||
export const AzureDevops = forwardRef<SVGSVGElement, LucideProps>(function AzureDevops(
|
||||
{
|
||||
size = 24,
|
||||
color = 'currentColor',
|
||||
className,
|
||||
strokeWidth: _strokeWidth,
|
||||
absoluteStrokeWidth: _absoluteStrokeWidth,
|
||||
...rest
|
||||
},
|
||||
ref,
|
||||
) {
|
||||
return (
|
||||
<svg
|
||||
ref={ref}
|
||||
xmlns="http://www.w3.org/2000/svg"
|
||||
width={size}
|
||||
height={size}
|
||||
viewBox="0 0 24 24"
|
||||
fill={color}
|
||||
className={className}
|
||||
{...rest}
|
||||
>
|
||||
<path d="M0 8.877L2.247 5.91l8.405-3.416V.022l7.37 5.393L2.966 8.338v8.225L0 15.707zm24-4.45v14.651l-5.753 4.9-9.303-3.057v3.056l-5.978-7.416 15.057 1.798V5.415z" />
|
||||
</svg>
|
||||
);
|
||||
});
|
||||
|
||||
export const Github = forwardRef<SVGSVGElement, LucideProps>(function Github(
|
||||
{
|
||||
size = 24,
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@
|
|||
"analysisFailed": "Analysis failed. Check server logs.",
|
||||
"startAnalysisFailed": "Failed to start analysis",
|
||||
"invalidGithubUrl": "Please enter a valid GitHub repository URL.",
|
||||
"invalidAzureDevOpsUrl": "Please enter a valid Azure DevOps repository URL.",
|
||||
"missingFolderPath": "Please enter a folder path.",
|
||||
"backend": {
|
||||
"reconnecting": "Server connection lost. Reconnecting…",
|
||||
|
|
|
|||
|
|
@ -31,7 +31,7 @@
|
|||
},
|
||||
"analyzeFirst": {
|
||||
"title": "Analyze your first repository",
|
||||
"description": "Paste a GitHub URL and GitNexus will clone it, parse the code, and build a live knowledge graph — right in your browser.",
|
||||
"description": "Paste a repository URL and GitNexus will clone it, parse the code, and build a live knowledge graph — right in your browser.",
|
||||
"footer": "Public repos only · Cloned locally by the server · No data leaves your machine"
|
||||
},
|
||||
"landing": {
|
||||
|
|
@ -51,6 +51,7 @@
|
|||
"inputType": "Input type",
|
||||
"githubUrl": "GitHub URL",
|
||||
"gitlabUrl": "GitLab URL",
|
||||
"azureDevOpsUrl": "Azure DevOps",
|
||||
"localFolder": "Local Folder",
|
||||
"starting": "Starting analysis...",
|
||||
"analyzeRepository": "Analyze Repository",
|
||||
|
|
@ -58,8 +59,13 @@
|
|||
"loadingGraph": "Loading graph...",
|
||||
"defaultRepoName": "repository",
|
||||
"githubRepositoryUrl": "GitHub Repository URL",
|
||||
"githubTokenLabel": "Personal Access Token (optional)",
|
||||
"githubTokenPlaceholder": "ghp_… or github_pat_…",
|
||||
"githubTokenHelp": "Required for private repos. Needs the 'repo' (or fine-grained Contents:read) scope. Sent once, not stored.",
|
||||
"gitlabRepositoryUrl": "GitLab Repository URL",
|
||||
"gitlabSupported": "Supports GitLab.com and self-hosted GitLab instances.",
|
||||
"azureDevOpsRepositoryUrl": "Azure DevOps Repository URL",
|
||||
"azureDevOpsSupported": "Format: https://dev.azure.com/organization/project/_git/repository",
|
||||
"localFolderPath": "Local Folder Path",
|
||||
"hideBackground": "Hide (analysis continues in background)",
|
||||
"upload": {
|
||||
|
|
|
|||
|
|
@ -6,6 +6,7 @@
|
|||
"analysisFailed": "分析失败,请检查服务器日志。",
|
||||
"startAnalysisFailed": "启动分析失败",
|
||||
"invalidGithubUrl": "请输入有效的 GitHub 仓库 URL。",
|
||||
"invalidAzureDevOpsUrl": "请输入有效的 Azure DevOps 仓库 URL。",
|
||||
"missingFolderPath": "请输入文件夹路径。",
|
||||
"backend": {
|
||||
"reconnecting": "服务器连接已断开,正在重连…",
|
||||
|
|
|
|||
|
|
@ -31,7 +31,7 @@
|
|||
},
|
||||
"analyzeFirst": {
|
||||
"title": "分析你的第一个仓库",
|
||||
"description": "粘贴 GitHub URL,GitNexus 会克隆仓库、解析代码,并直接在浏览器中构建实时知识图谱。",
|
||||
"description": "粘贴仓库 URL,GitNexus 会克隆仓库、解析代码,并直接在浏览器中构建实时知识图谱。",
|
||||
"footer": "仅支持公开仓库 · 服务器本地克隆 · 数据不会离开你的机器"
|
||||
},
|
||||
"landing": {
|
||||
|
|
@ -51,6 +51,7 @@
|
|||
"inputType": "输入类型",
|
||||
"githubUrl": "GitHub URL",
|
||||
"gitlabUrl": "GitLab URL",
|
||||
"azureDevOpsUrl": "Azure DevOps",
|
||||
"localFolder": "本地文件夹",
|
||||
"starting": "正在启动分析...",
|
||||
"analyzeRepository": "分析仓库",
|
||||
|
|
@ -58,8 +59,13 @@
|
|||
"loadingGraph": "正在加载图数据...",
|
||||
"defaultRepoName": "仓库",
|
||||
"githubRepositoryUrl": "GitHub 仓库 URL",
|
||||
"githubTokenLabel": "个人访问令牌(可选)",
|
||||
"githubTokenPlaceholder": "ghp_… 或 github_pat_…",
|
||||
"githubTokenHelp": "私有仓库需要此项。需 'repo' 范围(或细粒度 Contents:read)。仅发送一次,不会保存。",
|
||||
"gitlabRepositoryUrl": "GitLab 仓库 URL",
|
||||
"gitlabSupported": "支持 GitLab.com 和自托管 GitLab 实例。",
|
||||
"azureDevOpsRepositoryUrl": "Azure DevOps 仓库 URL",
|
||||
"azureDevOpsSupported": "格式: http://server/Collection/Project/_git/Repository",
|
||||
"localFolderPath": "本地文件夹路径",
|
||||
"hideBackground": "隐藏(分析继续在后台进行)",
|
||||
"upload": {
|
||||
|
|
|
|||
|
|
@ -854,6 +854,7 @@ export const startAnalyze = async (request: {
|
|||
path?: string;
|
||||
force?: boolean;
|
||||
embeddings?: boolean;
|
||||
token?: string;
|
||||
}): Promise<{ jobId: string; status: string }> => {
|
||||
const response = await fetchWithTimeout(
|
||||
`${_backendUrl}/api/analyze`,
|
||||
|
|
|
|||
|
|
@ -10,3 +10,13 @@
|
|||
|
||||
# Works with Infinity, vLLM, TEI, llama.cpp, Ollama, LM Studio, or OpenAI.
|
||||
# See README for details.
|
||||
|
||||
# Azure DevOps Server (Self-Hosted) Integration
|
||||
# Base URL of your Azure DevOps Server instance. Prefer https:// — the PAT is
|
||||
# sent in an Authorization header, so cleartext http:// exposes it on the wire
|
||||
# (http:// is still supported for internal-only instances; the server warns).
|
||||
# AZURE_DEVOPS_URL=https://azuredevops.example.com
|
||||
|
||||
# Personal Access Token with Code (Read) scope for cloning private repos.
|
||||
# Used for both self-hosted and cloud (dev.azure.com) Azure DevOps.
|
||||
# AZURE_DEVOPS_PAT=your-pat-here
|
||||
|
|
|
|||
58
gitnexus/bench/emit-persistence/README.md
Normal file
58
gitnexus/bench/emit-persistence/README.md
Normal file
|
|
@ -0,0 +1,58 @@
|
|||
# Emit-persistence bench (#2203)
|
||||
|
||||
Build-free throughput + byte-identity guard for the **CSV-generation half** of
|
||||
the graph-DB persistence pipeline (`streamAllCSVsToDisk`), which dominates
|
||||
large-repo `analyze` wall time alongside parsing (issue #2203).
|
||||
|
||||
```bash
|
||||
# from gitnexus/
|
||||
node --import tsx bench/emit-persistence/measure.mjs # print one JSON line
|
||||
node --import tsx bench/emit-persistence/measure.mjs --check # gate vs baselines.json
|
||||
```
|
||||
|
||||
## What it measures
|
||||
|
||||
A synthetic `KnowledgeGraph` (files + functions + classes + 4 edge types across
|
||||
the `File→Function`, `File→Class`, `Function→Function` label pairs) at two
|
||||
scales:
|
||||
|
||||
- **`elapsed_ms_small` / `elapsed_ms_large`** — median wall-clock over `REPS`
|
||||
runs of `streamAllCSVsToDisk`.
|
||||
- **`scaling_ratio`** — `(t_large/t_small)/(LARGE/SMALL)`; ~1.0 is linear. The
|
||||
`--check` gate fails if it exceeds `scaling_budget` (catches an O(n²)
|
||||
re-regression in the emit/routing path).
|
||||
- **`fingerprint`** — order-independent sha256 over every emitted CSV line (node
|
||||
CSVs + per-FROM→TO-label-pair rel CSVs). This is the **byte-identity gate**:
|
||||
the U2 (direct per-pair routing) and U3 (per-row microtask elimination)
|
||||
optimisations must not change graph content, and any future change that does
|
||||
fails `--check`. Byte-identity holds for all quote-free ids; for an id
|
||||
containing a `"` the router intentionally diverges from — and is more correct
|
||||
than — the legacy regex oracle (see `src/core/lbug/rel-pair-routing.ts`).
|
||||
|
||||
## What it does NOT measure
|
||||
|
||||
- **The LadybugDB `COPY` half.** Bulk loading needs a live writable DB
|
||||
connection, so it can't run build-free. Its per-stage timing lives in the
|
||||
runtime `PROF_LBUG_LOAD=1` breakdown (`[lbug-load prof] csv-emit=… copy-nodes=…
|
||||
copy-rels=… fallback=… total=…`) and is exercised end-to-end by the
|
||||
integration round-trip tests (`test/integration/basicblock-roundtrip.test.ts`,
|
||||
`lbug-core-adapter.test.ts`).
|
||||
- **Content extraction.** Bench nodes have no backing source files, so the
|
||||
`content` column is empty — emit cost here reflects the CSV machinery
|
||||
(routing, escaping, buffering, disk writes), not file reads.
|
||||
- **At-scale absolute numbers.** The real postgres / kernel-`fs/` wall (issue
|
||||
#2203's table) is a maintainer-run measurement; this synthetic bench is the
|
||||
reproducible regression guard, not a substitute for those runs.
|
||||
|
||||
## Deferred follow-up
|
||||
|
||||
Parallelising the `COPY` loop (`PARALLEL=false` is load-bearing; LadybugDB is
|
||||
single-writer) is **out of scope** for #2203 pending empirical validation of
|
||||
concurrent-COPY support — the `PROF_LBUG_LOAD` breakdown is the prerequisite
|
||||
that shows whether COPY is the dominant cost worth that risk.
|
||||
|
||||
## Regenerating the baseline
|
||||
|
||||
```bash
|
||||
node --import tsx bench/emit-persistence/measure.mjs # copy fingerprint + ratio into baselines.json
|
||||
```
|
||||
4
gitnexus/bench/emit-persistence/baselines-streaming.json
Normal file
4
gitnexus/bench/emit-persistence/baselines-streaming.json
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
{
|
||||
"fingerprint": "386f432c74f4992455055d8891dbe6c873ea95afa60ef4be023a21aed7b4bcb1",
|
||||
"_note": "Byte-identity + bounded-retention gate for streaming/chunked PDG emit (#2202). fingerprint = sha256 of the sorted, header-stripped BasicBlock + PDG-edge data rows of the canonical synthetic set. --check also asserts the streamed PdgEmitSink output is byte-identical to the whole-graph streamAllCSVsToDisk emit (byte_identical_nodes/edges) and that the in-memory graph retains 0 BasicBlocks (resident_basic_blocks === 0, the O(chunk) RSS bound). Regenerate via `node --import tsx bench/emit-persistence/measure-streaming.mjs`."
|
||||
}
|
||||
6
gitnexus/bench/emit-persistence/baselines.json
Normal file
6
gitnexus/bench/emit-persistence/baselines.json
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"fingerprint": "1b9dd0b783899b47067c36511d241860f291ac736e57682b0ece14148e3958ff",
|
||||
"scaling_budget": 1.8,
|
||||
"max_ms_large": 1000,
|
||||
"_note": "fingerprint = sha256 over per-file digests (filename + sha256(file bytes)), entry list sorted — binds each emitted line to its file so a row routed to the WRONG pair file changes the hash, AND catches within-file row reordering (file bytes hashed as-written). Byte-identity gate for #2203 U2/U3. NOTE: a future change that legitimately reorders emit (without changing the node/edge SET) will trip --check; regenerate then. scaling_budget bounds (t_large/t_small)/(LARGE/SMALL): observed ~0.95-1.05 (linear); 1.8 tolerates disk-I/O timing noise on CI while still catching an O(n^2) re-regression (~4x). max_ms_large=1000ms is a coarse absolute backstop (observed ~200ms) that catches a gross uniform slowdown the ratio gate misses; generous so CI host noise won't flake it. Regenerate via `node --import tsx bench/emit-persistence/measure.mjs`."
|
||||
}
|
||||
199
gitnexus/bench/emit-persistence/measure-streaming.mjs
Normal file
199
gitnexus/bench/emit-persistence/measure-streaming.mjs
Normal file
|
|
@ -0,0 +1,199 @@
|
|||
/**
|
||||
* Build-free byte-identity + bounded-retention bench for streaming/chunked PDG
|
||||
* graph emit (issue #2202).
|
||||
*
|
||||
* Proves the two acceptance criteria at scale, without a DB connection:
|
||||
* 1. BYTE-IDENTITY (R2): emitting a BasicBlock + intra-file PDG-edge set via
|
||||
* the streaming `PdgEmitSink` produces the IDENTICAL CSV data-row set as
|
||||
* the whole-graph `streamAllCSVsToDisk` path. Compared per file
|
||||
* (basicblock.csv, rel_BasicBlock_BasicBlock.csv) over header-stripped,
|
||||
* sorted lines so it is a pure function of the emitted row SET.
|
||||
* 2. BOUNDED RETENTION (R1): with streaming on, the in-memory graph holds
|
||||
* ZERO BasicBlock nodes regardless of how many are emitted — the PDG layer
|
||||
* never accumulates in process memory (peak RSS O(chunk), not O(graph)).
|
||||
*
|
||||
* Build-free: imports the `.ts` hotpaths through tsx
|
||||
* (`node --import tsx bench/emit-persistence/measure-streaming.mjs`).
|
||||
*
|
||||
* Without args: prints one JSON object. With `--check`: asserts byte-identity,
|
||||
* retention, and fingerprint == the committed baseline; exits non-zero on any
|
||||
* failure.
|
||||
*/
|
||||
import fs from 'node:fs';
|
||||
import fsp from 'node:fs/promises';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
import crypto from 'node:crypto';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.ts';
|
||||
import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.ts';
|
||||
import { PdgEmitSink } from '../../src/core/lbug/pdg-emit-sink.ts';
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const BASELINE_PATH = path.resolve(__dirname, 'baselines-streaming.json');
|
||||
|
||||
// PDG edge types streamed per file (all intra-block BasicBlock→BasicBlock).
|
||||
const PDG_TYPES = ['CFG', 'REACHING_DEF', 'CDG', 'POST_DOMINATE', 'TAINTED', 'SANITIZES'];
|
||||
|
||||
const FUNCS = 1200; // functions
|
||||
const BLOCKS = 6; // basic blocks per function ⇒ FUNCS*BLOCKS BasicBlocks total
|
||||
const CHUNK_ROWS = 64; // tiny streamed buffer to exercise frequent flushing
|
||||
|
||||
/**
|
||||
* Build the canonical PDG node/edge SET: `FUNCS` functions each with `BLOCKS`
|
||||
* BasicBlocks and a chain of intra-function PDG edges. Returns the structural
|
||||
* nodes (File/Function) separately from the BasicBlock + PDG-edge layer so the
|
||||
* streamed path can route them to different sinks.
|
||||
*/
|
||||
function buildSet() {
|
||||
const structuralNodes = [];
|
||||
const structuralRels = [];
|
||||
const bbNodes = [];
|
||||
const pdgEdges = [];
|
||||
for (let f = 0; f < FUNCS; f++) {
|
||||
const fp = `src/m${f % 50}.ts`;
|
||||
const fnId = `Function:${fp}:fn${f}:1`;
|
||||
structuralNodes.push({
|
||||
id: fnId,
|
||||
label: 'Function',
|
||||
properties: { name: `fn${f}`, filePath: fp, startLine: 1, endLine: 99 },
|
||||
});
|
||||
for (let b = 0; b < BLOCKS; b++) {
|
||||
bbNodes.push({
|
||||
id: `BasicBlock:${fp}:1:0:${f}_${b}`,
|
||||
label: 'BasicBlock',
|
||||
properties: {
|
||||
name: '',
|
||||
filePath: fp,
|
||||
startLine: b * 3,
|
||||
endLine: b * 3 + 2,
|
||||
text: `f${f}b${b}`,
|
||||
},
|
||||
});
|
||||
}
|
||||
for (let b = 0; b < BLOCKS - 1; b++) {
|
||||
const from = `BasicBlock:${fp}:1:0:${f}_${b}`;
|
||||
const to = `BasicBlock:${fp}:1:0:${f}_${b + 1}`;
|
||||
for (const type of PDG_TYPES) {
|
||||
pdgEdges.push({
|
||||
id: `${type}:${f}:${b}`,
|
||||
sourceId: from,
|
||||
targetId: to,
|
||||
type,
|
||||
confidence: 1,
|
||||
reason: type === 'REACHING_DEF' ? `v${b}` : type === 'CDG' ? 'T' : '',
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
// A few File nodes so the structural emit produces a realistic multi-table mix.
|
||||
for (let m = 0; m < 50; m++) {
|
||||
structuralNodes.push({
|
||||
id: `File:src/m${m}.ts`,
|
||||
label: 'File',
|
||||
properties: { name: `m${m}.ts`, filePath: `src/m${m}.ts` },
|
||||
});
|
||||
}
|
||||
return { structuralNodes, structuralRels, bbNodes, pdgEdges };
|
||||
}
|
||||
|
||||
/** Header-stripped, sorted, non-empty data rows of one CSV file (or [] if absent). */
|
||||
async function dataRows(csvPath) {
|
||||
let text;
|
||||
try {
|
||||
text = await fsp.readFile(csvPath, 'utf8');
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
const lines = text.split('\n').filter((l) => l.length > 0);
|
||||
return lines.slice(1).sort(); // drop the header line
|
||||
}
|
||||
|
||||
const sha = (rows) => crypto.createHash('sha256').update(rows.join('\n')).digest('hex');
|
||||
|
||||
async function measure() {
|
||||
// mkdtemp (unpredictable, unique) rather than a predictable pid-based tmp path.
|
||||
const tmpRoot = await fsp.mkdtemp(path.join(os.tmpdir(), 'gitnexus-stream-bench-'));
|
||||
try {
|
||||
const { structuralNodes, structuralRels, bbNodes, pdgEdges } = buildSet();
|
||||
|
||||
// ── whole-graph path ─────────────────────────────────────────────────
|
||||
const wholeGraph = createKnowledgeGraph();
|
||||
for (const n of structuralNodes) wholeGraph.addNode(n);
|
||||
for (const n of bbNodes) wholeGraph.addNode(n);
|
||||
for (const r of structuralRels) wholeGraph.addRelationship(r);
|
||||
for (const e of pdgEdges) wholeGraph.addRelationship(e);
|
||||
const wholeDir = path.join(tmpRoot, 'whole');
|
||||
await streamAllCSVsToDisk(wholeGraph, path.join(tmpRoot, 'no-repo'), wholeDir);
|
||||
|
||||
// ── streamed path ────────────────────────────────────────────────────
|
||||
const realGraph = createKnowledgeGraph();
|
||||
const sink = new PdgEmitSink(realGraph, path.join(tmpRoot, 'pdg-csv'), CHUNK_ROWS);
|
||||
for (const n of structuralNodes) realGraph.addNode(n); // structural → real graph
|
||||
for (const r of structuralRels) realGraph.addRelationship(r);
|
||||
for (const n of bbNodes) sink.addNode(n); // BasicBlock layer → sink (CSV)
|
||||
for (const e of pdgEdges) sink.addRelationship(e);
|
||||
sink.finalize();
|
||||
const streamedCsvDir = path.join(tmpRoot, 'streamed');
|
||||
await streamAllCSVsToDisk(realGraph, path.join(tmpRoot, 'no-repo'), streamedCsvDir);
|
||||
|
||||
// ── retention (R1): the real graph holds ZERO BasicBlocks ────────────
|
||||
let residentBasicBlocks = 0;
|
||||
for (const n of realGraph.iterNodes()) if (n.label === 'BasicBlock') residentBasicBlocks++;
|
||||
|
||||
// ── byte-identity (R2): per-file data-row set equality ───────────────
|
||||
const wholeBb = await dataRows(path.join(wholeDir, 'basicblock.csv'));
|
||||
const streamedBb = await dataRows(path.join(tmpRoot, 'pdg-csv', 'basicblock.csv'));
|
||||
const wholeRel = await dataRows(path.join(wholeDir, 'rel_BasicBlock_BasicBlock.csv'));
|
||||
const streamedRel = await dataRows(
|
||||
path.join(tmpRoot, 'pdg-csv', 'rel_BasicBlock_BasicBlock.csv'),
|
||||
);
|
||||
|
||||
const bbIdentical = sha(wholeBb) === sha(streamedBb);
|
||||
const relIdentical = sha(wholeRel) === sha(streamedRel);
|
||||
// Fingerprint over the canonical PDG data-row set (drift gate).
|
||||
const fingerprint = sha([...wholeBb, ...wholeRel].sort());
|
||||
|
||||
return {
|
||||
scenario: 'streamingPdgEmit',
|
||||
basic_blocks: bbNodes.length,
|
||||
pdg_edges: pdgEdges.length,
|
||||
chunk_rows: CHUNK_ROWS,
|
||||
resident_basic_blocks: residentBasicBlocks,
|
||||
byte_identical_nodes: bbIdentical,
|
||||
byte_identical_edges: relIdentical,
|
||||
fingerprint,
|
||||
};
|
||||
} finally {
|
||||
await fsp.rm(tmpRoot, { recursive: true, force: true }).catch(() => {});
|
||||
}
|
||||
}
|
||||
|
||||
const CHECK = process.argv.includes('--check');
|
||||
const result = await measure();
|
||||
|
||||
if (!CHECK) {
|
||||
process.stdout.write(JSON.stringify(result) + '\n');
|
||||
} else {
|
||||
const base = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8'));
|
||||
const failures = [];
|
||||
if (!result.byte_identical_nodes)
|
||||
failures.push('streamed BasicBlock rows differ from whole-graph emit');
|
||||
if (!result.byte_identical_edges)
|
||||
failures.push('streamed PDG-edge rows differ from whole-graph emit');
|
||||
if (result.resident_basic_blocks !== 0) {
|
||||
failures.push(
|
||||
`RSS bound violated: ${result.resident_basic_blocks} BasicBlock node(s) retained in the in-memory graph (expected 0)`,
|
||||
);
|
||||
}
|
||||
if (result.fingerprint !== base.fingerprint) {
|
||||
failures.push(`fingerprint drift (got ${result.fingerprint}, expected ${base.fingerprint})`);
|
||||
}
|
||||
process.stdout.write(JSON.stringify(result) + '\n');
|
||||
if (failures.length > 0) {
|
||||
for (const f of failures) process.stderr.write(`[stream-pdg-emit --check] FAIL: ${f}\n`);
|
||||
process.exit(1);
|
||||
}
|
||||
process.stderr.write('[stream-pdg-emit --check] PASS\n');
|
||||
}
|
||||
216
gitnexus/bench/emit-persistence/measure.mjs
Normal file
216
gitnexus/bench/emit-persistence/measure.mjs
Normal file
|
|
@ -0,0 +1,216 @@
|
|||
/**
|
||||
* Build-free emit-path throughput + byte-identity bench for the graph-DB
|
||||
* persistence pipeline (issue #2203).
|
||||
*
|
||||
* Measures `streamAllCSVsToDisk` — the CSV-generation half of the persistence
|
||||
* path that U2 (direct per-pair relationship routing) and U3 (per-row
|
||||
* microtask elimination) optimised. The LadybugDB `COPY` half needs a real DB
|
||||
* connection, so its timing lives in the runtime `PROF_LBUG_LOAD` breakdown +
|
||||
* the integration round-trip tests, NOT here (see README.md).
|
||||
*
|
||||
* For a synthetic KnowledgeGraph at two scales it reports:
|
||||
* - elapsed_ms_small / elapsed_ms_large (median over REPS) + a scaling ratio
|
||||
* `(t_large/t_small)/(LARGE/SMALL)`: ~1.0 linear, ~3.x quadratic;
|
||||
* - an order-independent sha256 fingerprint over every emitted CSV line
|
||||
* (node CSVs + per-FROM→TO-label-pair rel CSVs), as the byte-identity gate
|
||||
* guarding the issue's "byte-identical graph content" requirement.
|
||||
*
|
||||
* Build-free: imports the `.ts` hotpaths through tsx
|
||||
* (`node --import tsx bench/emit-persistence/measure.mjs`). Static `.ts`
|
||||
* imports work; a top-level `await import()` breaks tsx's lexer.
|
||||
*
|
||||
* Without args: prints one JSON object per scenario.
|
||||
* With `--check`: asserts the fingerprint == the committed baseline AND the
|
||||
* scaling ratio < the recorded budget; exits non-zero on drift/regression.
|
||||
*/
|
||||
import fs from 'node:fs';
|
||||
import fsp from 'node:fs/promises';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
import crypto from 'node:crypto';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.ts';
|
||||
import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.ts';
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const BASELINE_PATH = path.resolve(__dirname, 'baselines.json');
|
||||
|
||||
// ---- synthetic graph generation (deterministic — no randomness) ----
|
||||
|
||||
/**
|
||||
* Build a graph of `entityCount` files, each with 2 functions + 1 class and 4
|
||||
* relationships. ids carry valid table-label prefixes (File:/Function:/Class:)
|
||||
* so edges route to real pairs (File→Function, File→Class, Function→Function)
|
||||
* — exercising the U2 router across multiple label pairs. Node `content` is
|
||||
* never populated (no backing files), so emit cost reflects the CSV machinery
|
||||
* (routing, escaping, buffering, disk writes), not content extraction.
|
||||
*/
|
||||
function generateGraph(entityCount) {
|
||||
const graph = createKnowledgeGraph();
|
||||
for (let i = 0; i < entityCount; i++) {
|
||||
const fp = `src/e${i}.ts`;
|
||||
const fileId = `File:${fp}`;
|
||||
const fnA = `Function:${fp}:fnA:1`;
|
||||
const fnB = `Function:${fp}:fnB:10`;
|
||||
const cls = `Class:${fp}:C:20`;
|
||||
graph.addNode({ id: fileId, label: 'File', properties: { name: `e${i}.ts`, filePath: fp } });
|
||||
graph.addNode({
|
||||
id: fnA,
|
||||
label: 'Function',
|
||||
properties: { name: 'fnA', filePath: fp, startLine: 1, endLine: 5, isExported: true },
|
||||
});
|
||||
graph.addNode({
|
||||
id: fnB,
|
||||
label: 'Function',
|
||||
properties: { name: 'fnB', filePath: fp, startLine: 10, endLine: 15, isExported: false },
|
||||
});
|
||||
graph.addNode({
|
||||
id: cls,
|
||||
label: 'Class',
|
||||
properties: { name: 'C', filePath: fp, startLine: 20, endLine: 30, isExported: true },
|
||||
});
|
||||
graph.addRelationship({
|
||||
id: `${fileId}->${fnA}`,
|
||||
sourceId: fileId,
|
||||
targetId: fnA,
|
||||
type: 'CONTAINS',
|
||||
confidence: 1,
|
||||
reason: '',
|
||||
});
|
||||
graph.addRelationship({
|
||||
id: `${fileId}->${fnB}`,
|
||||
sourceId: fileId,
|
||||
targetId: fnB,
|
||||
type: 'CONTAINS',
|
||||
confidence: 1,
|
||||
reason: '',
|
||||
});
|
||||
graph.addRelationship({
|
||||
id: `${fileId}->${cls}`,
|
||||
sourceId: fileId,
|
||||
targetId: cls,
|
||||
type: 'CONTAINS',
|
||||
confidence: 1,
|
||||
reason: '',
|
||||
});
|
||||
graph.addRelationship({
|
||||
id: `${fnA}->${fnB}`,
|
||||
sourceId: fnA,
|
||||
targetId: fnB,
|
||||
type: 'CALLS',
|
||||
confidence: 1,
|
||||
reason: '',
|
||||
});
|
||||
}
|
||||
return graph;
|
||||
}
|
||||
|
||||
// ---- byte-identity fingerprint (order-independent) ----
|
||||
|
||||
/** sha256 over every non-empty line of every emitted CSV file, sorted so the
|
||||
* digest is a pure function of the emitted line SET (insertion-order agnostic). */
|
||||
async function fingerprintEmit(graph, dir) {
|
||||
await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir);
|
||||
// Per-file digest bound to the filename: a row routed to the WRONG pair file
|
||||
// (or a header written to the wrong file) changes the fingerprint — a global
|
||||
// line-flatten could not catch that. File bytes are hashed as-written (so it
|
||||
// also catches within-file row reordering); the entry list is sorted so
|
||||
// readdir order doesn't matter.
|
||||
const entries = [];
|
||||
for (const name of fs.readdirSync(dir)) {
|
||||
if (!name.endsWith('.csv')) continue;
|
||||
const bytes = await fsp.readFile(path.join(dir, name));
|
||||
entries.push(`${name}\n${crypto.createHash('sha256').update(bytes).digest('hex')}`);
|
||||
}
|
||||
return crypto.createHash('sha256').update(entries.sort().join('\n')).digest('hex');
|
||||
}
|
||||
|
||||
// ---- timing ----
|
||||
|
||||
function median(xs) {
|
||||
const s = [...xs].sort((a, b) => a - b);
|
||||
const m = Math.floor(s.length / 2);
|
||||
return s.length % 2 ? s[m] : (s[m - 1] + s[m]) / 2;
|
||||
}
|
||||
|
||||
async function timeEmit(graph, dir, reps) {
|
||||
await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir); // warmup (not counted)
|
||||
const samples = [];
|
||||
for (let i = 0; i < reps; i++) {
|
||||
const start = process.hrtime.bigint();
|
||||
await streamAllCSVsToDisk(graph, path.join(dir, 'no-such-repo'), dir);
|
||||
samples.push(Number(process.hrtime.bigint() - start) / 1e6);
|
||||
}
|
||||
return median(samples);
|
||||
}
|
||||
|
||||
const SMALL = 600;
|
||||
const LARGE = 2400;
|
||||
const REPS = 5;
|
||||
|
||||
async function measure() {
|
||||
const tmpRoot = path.join(os.tmpdir(), `gitnexus-emit-bench-${process.pid}`);
|
||||
await fsp.mkdir(tmpRoot, { recursive: true });
|
||||
try {
|
||||
const smallGraph = generateGraph(SMALL);
|
||||
const largeGraph = generateGraph(LARGE);
|
||||
|
||||
const fingerprint = await fingerprintEmit(largeGraph, path.join(tmpRoot, 'fp'));
|
||||
const small = await timeEmit(smallGraph, path.join(tmpRoot, 'small'), REPS);
|
||||
const large = await timeEmit(largeGraph, path.join(tmpRoot, 'large'), REPS);
|
||||
const scalingRatio = small > 0 ? large / small / (LARGE / SMALL) : 0;
|
||||
|
||||
return {
|
||||
scenario: 'streamAllCSVsToDisk',
|
||||
entities_small: SMALL,
|
||||
entities_large: LARGE,
|
||||
nodes_large: LARGE * 4,
|
||||
rels_large: LARGE * 4,
|
||||
elapsed_ms_small: Number(small.toFixed(2)),
|
||||
elapsed_ms_large: Number(large.toFixed(2)),
|
||||
scaling_ratio: Number(scalingRatio.toFixed(3)),
|
||||
fingerprint,
|
||||
};
|
||||
} finally {
|
||||
await fsp.rm(tmpRoot, { recursive: true, force: true }).catch(() => {});
|
||||
}
|
||||
}
|
||||
|
||||
// ---- run ----
|
||||
|
||||
const CHECK = process.argv.includes('--check');
|
||||
const result = await measure();
|
||||
|
||||
if (!CHECK) {
|
||||
process.stdout.write(JSON.stringify(result) + '\n');
|
||||
} else {
|
||||
const base = JSON.parse(fs.readFileSync(BASELINE_PATH, 'utf8'));
|
||||
const failures = [];
|
||||
if (result.fingerprint !== base.fingerprint) {
|
||||
failures.push(
|
||||
`byte-identity fingerprint drift (got ${result.fingerprint}, expected ${base.fingerprint})`,
|
||||
);
|
||||
}
|
||||
if (result.scaling_ratio >= base.scaling_budget) {
|
||||
failures.push(
|
||||
`scaling ratio ${result.scaling_ratio} >= budget ${base.scaling_budget} ` +
|
||||
`(${SMALL}->${LARGE} entities, ms ${result.elapsed_ms_small}->${result.elapsed_ms_large})`,
|
||||
);
|
||||
}
|
||||
// Absolute backstop: the scaling ratio alone passes a uniform Nx slowdown (it
|
||||
// only compares large/small). A generous, host-noise-tolerant ceiling catches
|
||||
// a gross absolute regression. Opt-in (only enforced when max_ms_large is set).
|
||||
if (base.max_ms_large !== undefined && result.elapsed_ms_large >= base.max_ms_large) {
|
||||
failures.push(
|
||||
`absolute wall-time regression: elapsed_ms_large ${result.elapsed_ms_large}ms >= budget ` +
|
||||
`${base.max_ms_large}ms (coarse backstop, not a tight SLA)`,
|
||||
);
|
||||
}
|
||||
process.stdout.write(JSON.stringify(result) + '\n');
|
||||
if (failures.length > 0) {
|
||||
for (const f of failures) process.stderr.write(`[emit-persistence --check] FAIL: ${f}\n`);
|
||||
process.exit(1);
|
||||
}
|
||||
process.stderr.write('[emit-persistence --check] PASS\n');
|
||||
}
|
||||
46
gitnexus/package-lock.json
generated
46
gitnexus/package-lock.json
generated
|
|
@ -1340,19 +1340,18 @@
|
|||
"license": "BSD-3-Clause"
|
||||
},
|
||||
"node_modules/@protobufjs/eventemitter": {
|
||||
"version": "1.1.0",
|
||||
"resolved": "https://registry.npmjs.org/@protobufjs/eventemitter/-/eventemitter-1.1.0.tgz",
|
||||
"integrity": "sha512-j9ednRT81vYJ9OfVuXG6ERSTdEL1xVsNgqpkxMsbIabzSo3goCjDIveeGv5d03om39ML71RdmrGNjG5SReBP/Q==",
|
||||
"version": "1.1.1",
|
||||
"resolved": "https://registry.npmjs.org/@protobufjs/eventemitter/-/eventemitter-1.1.1.tgz",
|
||||
"integrity": "sha512-vW1GmwMZNnL+gMRaovlh9yZX74kc+TTU3FObkkurpMaRtBfLP3ldjS9KQWlwZgraRE0+dheEEoAxdzcJQ8eXZg==",
|
||||
"license": "BSD-3-Clause"
|
||||
},
|
||||
"node_modules/@protobufjs/fetch": {
|
||||
"version": "1.1.0",
|
||||
"resolved": "https://registry.npmjs.org/@protobufjs/fetch/-/fetch-1.1.0.tgz",
|
||||
"integrity": "sha512-lljVXpqXebpsijW71PZaCYeIcE5on1w5DlQy5WH6GLbFryLUrBD4932W/E2BSpfRJWseIL4v/KPgBFxDOIdKpQ==",
|
||||
"version": "1.1.1",
|
||||
"resolved": "https://registry.npmjs.org/@protobufjs/fetch/-/fetch-1.1.1.tgz",
|
||||
"integrity": "sha512-GpptLrs57adMSuHi3VNj0mAF8dwh36LMaYF6XyJ6JMWlVsc+t42tm1HSEDmOs3A8fC9yyeisgLhsTVQokOZ0zw==",
|
||||
"license": "BSD-3-Clause",
|
||||
"dependencies": {
|
||||
"@protobufjs/aspromise": "^1.1.1",
|
||||
"@protobufjs/inquire": "^1.1.0"
|
||||
"@protobufjs/aspromise": "^1.1.1"
|
||||
}
|
||||
},
|
||||
"node_modules/@protobufjs/float": {
|
||||
|
|
@ -1361,12 +1360,6 @@
|
|||
"integrity": "sha512-Ddb+kVXlXst9d+R9PfTIxh1EdNkgoRe5tOX6t01f1lYWOvJnSPDBlG241QLzcyPdoNTsblLUdujGSE4RzrTZGQ==",
|
||||
"license": "BSD-3-Clause"
|
||||
},
|
||||
"node_modules/@protobufjs/inquire": {
|
||||
"version": "1.1.1",
|
||||
"resolved": "https://registry.npmjs.org/@protobufjs/inquire/-/inquire-1.1.1.tgz",
|
||||
"integrity": "sha512-mnzgDV26ueAvk7rsbt9L7bE0SuAoqyuys/sMMrmVcN5x9VsxpcG3rqAUSgDyLp0UZlmNfIbQ4fHfCtreVBk8Ew==",
|
||||
"license": "BSD-3-Clause"
|
||||
},
|
||||
"node_modules/@protobufjs/path": {
|
||||
"version": "1.1.2",
|
||||
"resolved": "https://registry.npmjs.org/@protobufjs/path/-/path-1.1.2.tgz",
|
||||
|
|
@ -1822,9 +1815,9 @@
|
|||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@types/node": {
|
||||
"version": "25.9.2",
|
||||
"resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.2.tgz",
|
||||
"integrity": "sha512-G05zqtJhcDLb8uslf5EjCxXg9G1KQxiV8OS0R26IC//Eoyitzqe8z37I7cqvnZlrlSfgocQRfSn/AHBZJJFyGw==",
|
||||
"version": "25.9.3",
|
||||
"resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.3.tgz",
|
||||
"integrity": "sha512-603BddQMv3pUcr4U2dhujk83N2tTDVr/34wII2B6bJy6g+8WD6yUb11jszNs0gdi4PesVWl7ABt8nYMVpnLUcg==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"undici-types": ">=7.24.0 <7.24.7"
|
||||
|
|
@ -4351,24 +4344,23 @@
|
|||
"license": "MIT"
|
||||
},
|
||||
"node_modules/protobufjs": {
|
||||
"version": "7.5.8",
|
||||
"resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.5.8.tgz",
|
||||
"integrity": "sha512-dvpCIeLPbXZS/Ete7yLaO7RenOdken2NHKykBXbsaGxZT0UTltcarBciw+A78SRQs9iMAAVpsYA+l8b1hTePIA==",
|
||||
"version": "7.6.4",
|
||||
"resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.6.4.tgz",
|
||||
"integrity": "sha512-RJJPTTpvFfHcWLkIa2JFWK4XvtSzS0yEWDmunqHXli1h3JlkbcQZXDZdcWxv+JK3Xsl5/UFDPZ0iGm7DAengYw==",
|
||||
"hasInstallScript": true,
|
||||
"license": "BSD-3-Clause",
|
||||
"dependencies": {
|
||||
"@protobufjs/aspromise": "^1.1.2",
|
||||
"@protobufjs/base64": "^1.1.2",
|
||||
"@protobufjs/codegen": "^2.0.5",
|
||||
"@protobufjs/eventemitter": "^1.1.0",
|
||||
"@protobufjs/fetch": "^1.1.0",
|
||||
"@protobufjs/eventemitter": "^1.1.1",
|
||||
"@protobufjs/fetch": "^1.1.1",
|
||||
"@protobufjs/float": "^1.0.2",
|
||||
"@protobufjs/inquire": "^1.1.1",
|
||||
"@protobufjs/path": "^1.1.2",
|
||||
"@protobufjs/pool": "^1.1.0",
|
||||
"@protobufjs/utf8": "^1.1.1",
|
||||
"@types/node": ">=13.7.0",
|
||||
"long": "^5.0.0"
|
||||
"long": "^5.3.2"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=12.0.0"
|
||||
|
|
@ -4917,9 +4909,9 @@
|
|||
}
|
||||
},
|
||||
"node_modules/tar": {
|
||||
"version": "7.5.13",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.13.tgz",
|
||||
"integrity": "sha512-tOG/7GyXpFevhXVh8jOPJrmtRpOTsYqUIkVdVooZYJS/z8WhfQUX8RJILmeuJNinGAMSu1veBr4asSHFt5/hng==",
|
||||
"version": "7.5.16",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.16.tgz",
|
||||
"integrity": "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w==",
|
||||
"license": "BlueOak-1.0.0",
|
||||
"dependencies": {
|
||||
"@isaacs/fs-minipass": "^4.0.0",
|
||||
|
|
|
|||
|
|
@ -121,6 +121,23 @@ export interface PipelineOptions {
|
|||
/** Per-run `TAINT_PATH` edge cap (#2084 review P1-3). `undefined` ⇒
|
||||
* `DEFAULT_PDG_MAX_INTERPROC_EDGES` (1000); `0` ⇒ no cap. */
|
||||
pdgMaxInterprocEdges?: number;
|
||||
/**
|
||||
* Streaming/chunked PDG graph emit (#2202). When true, the BasicBlock +
|
||||
* intra-file PDG-edge layer (CFG / REACHING_DEF / CDG / POST_DOMINATE /
|
||||
* TAINTED / SANITIZES) is streamed to CSV-on-disk during the scope-resolution
|
||||
* emit loop instead of being materialized in the in-memory graph, bounding
|
||||
* peak RSS to O(chunk) rather than O(graph) at full-kernel scale. Already
|
||||
* gated by the caller to full-rebuild runs only (the incremental writeback
|
||||
* reads BasicBlocks back from the in-memory graph). Memory-only — produces a
|
||||
* byte-identical persisted graph and is NOT part of `RepoMeta.pdg`, so
|
||||
* toggling it never trips `pdgModeMismatch`. Default/false ⇒ today's
|
||||
* whole-graph emit.
|
||||
*/
|
||||
streamPdgEmit?: boolean;
|
||||
/** Streamed PDG-emit write buffer (rows) when `streamPdgEmit` is on (#2202).
|
||||
* `undefined` ⇒ `DEFAULT_PDG_EMIT_CHUNK_ROWS`. Memory-only; does not affect
|
||||
* emitted bytes. */
|
||||
pdgEmitChunkSize?: number;
|
||||
/**
|
||||
* Request parsing with the worker pool disabled. The sequential parser was
|
||||
* removed — the worker pool is the sole parse path — so setting this now
|
||||
|
|
@ -287,10 +304,10 @@ export const runPipelineFromRepo = async (
|
|||
|
||||
let communityResult: CommunitiesOutput['communityResult'] | undefined;
|
||||
let processResult: ProcessesOutput['processResult'] | undefined;
|
||||
const resolutionOutcomes = getPhaseOutput<ScopeResolutionOutput>(
|
||||
results,
|
||||
'scopeResolution',
|
||||
).resolutionOutcomes;
|
||||
const scopeResolutionOutput = getPhaseOutput<ScopeResolutionOutput>(results, 'scopeResolution');
|
||||
const resolutionOutcomes = scopeResolutionOutput.resolutionOutcomes;
|
||||
// Streamed PDG-emit manifest (#2202): present only when streaming was on.
|
||||
const pdgEmitManifest = scopeResolutionOutput.pdgEmitManifest;
|
||||
|
||||
if (!options?.skipGraphPhases) {
|
||||
communityResult = getPhaseOutput<CommunitiesOutput>(results, 'communities').communityResult;
|
||||
|
|
@ -319,5 +336,6 @@ export const runPipelineFromRepo = async (
|
|||
processResult,
|
||||
resolutionOutcomes,
|
||||
usedWorkerPool,
|
||||
pdgEmitManifest,
|
||||
};
|
||||
};
|
||||
|
|
|
|||
|
|
@ -44,6 +44,8 @@ import {
|
|||
import type { ResolutionOutcome } from '../resolution-outcome.js';
|
||||
import type { FunctionSummary } from '../../taint/summary-model.js';
|
||||
import { buildFunctionNodeIndex } from '../../taint/summary-harvest-driver.js';
|
||||
import { PdgEmitSink, type PdgEmitManifest } from '../../../lbug/pdg-emit-sink.js';
|
||||
import { resolveNativeSafeStorageDir } from '../../../lbug/lbug-config.js';
|
||||
|
||||
import { logger } from '../../../logger.js';
|
||||
export interface ScopeResolutionOutput {
|
||||
|
|
@ -72,6 +74,14 @@ export interface ScopeResolutionOutput {
|
|||
* The `taintSummaries` phase composes these over the `CALLS` graph.
|
||||
*/
|
||||
readonly functionSummaries: readonly FunctionSummary[];
|
||||
/**
|
||||
* Streamed PDG-emit COPY manifest (#2202). Present only when streaming was on
|
||||
* (full rebuild + `--pdg` + enabled): the BasicBlock node CSV + per-pair PDG
|
||||
* edge CSVs that were flushed to disk during the emit loop, for the persistence
|
||||
* step to COPY alongside the structural CSVs. Absent ⇒ the PDG layer (if any)
|
||||
* is in the in-memory graph and persists via the normal whole-graph emit.
|
||||
*/
|
||||
readonly pdgEmitManifest?: PdgEmitManifest;
|
||||
}
|
||||
|
||||
const NOOP_OUTPUT: ScopeResolutionOutput = Object.freeze({
|
||||
|
|
@ -242,236 +252,291 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
? buildFunctionNodeIndex(ctx.graph)
|
||||
: undefined;
|
||||
|
||||
for (const [lang, provider] of SCOPE_RESOLVERS) {
|
||||
// Standalone providers (COBOL, JCL) don't emit graph edges yet
|
||||
// through the scope-resolution path. This is the canonical guard:
|
||||
// runScopeResolution is never called for standalone providers, which
|
||||
// keeps cobolPhase as the sole IMPORTS edge producer. Keep this guard
|
||||
// in sync with any additional standalone providers added to
|
||||
// SCOPE_RESOLVERS.
|
||||
if (provider.languageProvider.parseStrategy === 'standalone') continue;
|
||||
|
||||
const primaryLangFiles = filesByLang.get(lang) ?? [];
|
||||
if (primaryLangFiles.length === 0) continue;
|
||||
const primaryFilePaths = primaryLangFiles.map((f) => f.path);
|
||||
|
||||
// Load per-language import-resolution config (tsconfig paths,
|
||||
// composer.json autoload, go.mod, ...). One I/O round trip per
|
||||
// workspace pass — cached implicitly by the result handed to
|
||||
// every `resolveImportTarget` call below.
|
||||
const resolutionConfig =
|
||||
provider.loadResolutionConfig !== undefined
|
||||
? await provider.loadResolutionConfig(ctx.repoPath)
|
||||
: undefined;
|
||||
|
||||
// Some languages (e.g. Vue) expand their file universe beyond the
|
||||
// primary-language files via the `collectScopeContextPaths` hook.
|
||||
// The hook receives raw source contents of the primary files so it
|
||||
// can trace import closures without a second tree-sitter parse.
|
||||
//
|
||||
// To avoid reading primary files twice (once for the hook, once for
|
||||
// the resolution pass), we read them upfront and merge with the
|
||||
// extra context paths the hook may add.
|
||||
// Stream this language's pre-built ParsedFiles in from the disk store
|
||||
// FIRST (huge-repo path). Doing it before reading source lets us skip
|
||||
// loading content for files the store already covers — for a provider
|
||||
// with no content-consuming hook that source is pure dead weight once
|
||||
// extraction is served from the store (~1.5 GB on the kernel's C pass).
|
||||
// Merged into `preExtractedByPath`; the per-language release block below
|
||||
// evicts these again before the next language, so only one language's
|
||||
// ParsedFiles are resident at a time.
|
||||
const loadStoreFor = async (paths: ReadonlySet<string>): Promise<void> => {
|
||||
if (!parsedFileStorePath) return;
|
||||
const fromDisk = await loadParsedFilesForPaths(parsedFileStorePath, paths);
|
||||
for (const [fp, pf] of fromDisk) preExtractedByPath.set(fp, pf);
|
||||
};
|
||||
|
||||
// A provider that feeds source text into a post-extract hook
|
||||
// (populateWorkspaceOwners / populateNamespaceSiblings /
|
||||
// populateRangeBindings / emitPostResolutionEdges) needs content for ALL
|
||||
// its files; one without those hooks only needs content for files the
|
||||
// store does NOT cover (fresh-extract fallback). Keep this in sync with
|
||||
// the getFileContents() call-sites in run.ts.
|
||||
const providerNeedsAllContent =
|
||||
provider.populateWorkspaceOwners !== undefined ||
|
||||
provider.populateNamespaceSiblings !== undefined ||
|
||||
provider.populateRangeBindings !== undefined ||
|
||||
provider.emitPostResolutionEdges !== undefined;
|
||||
|
||||
let scopeFilePaths: Set<string>;
|
||||
let contents: Map<string, string>;
|
||||
if (provider.collectScopeContextPaths !== undefined) {
|
||||
// Context-expanding providers (e.g. Vue) need every primary file's
|
||||
// source up front for the closure hook, so load it all.
|
||||
const entryFileContents = await readFileContents(ctx.repoPath, primaryFilePaths);
|
||||
scopeFilePaths = provider.collectScopeContextPaths({
|
||||
primaryFilePaths,
|
||||
preExtractedByPath,
|
||||
entryFileContents,
|
||||
allScannedPaths,
|
||||
resolutionConfig,
|
||||
});
|
||||
// Read only the extra context files (TS/JS etc.) not already loaded.
|
||||
const extraPaths = [...scopeFilePaths].filter((p) => !entryFileContents.has(p));
|
||||
const extraContents = await readFileContents(ctx.repoPath, extraPaths);
|
||||
contents = new Map([...entryFileContents, ...extraContents]);
|
||||
await loadStoreFor(scopeFilePaths);
|
||||
// Streaming/chunked PDG emit (#2202): when enabled (the caller has already
|
||||
// gated this to full-rebuild + `--pdg`), route the BasicBlock + intra-file
|
||||
// PDG-edge layer to CSV-on-disk through one sink shared across every
|
||||
// language pass, so it never accumulates in `ctx.graph` (peak RSS O(chunk)).
|
||||
// Needs the storage dir (the parse-cache store path, the same `.gitnexus`
|
||||
// dir loadGraphToLbug COPYs from); if that is somehow absent we skip
|
||||
// streaming and fall back to the in-memory whole-graph emit.
|
||||
let pdgEmitSink: PdgEmitSink | undefined;
|
||||
if (ctx.options?.streamPdgEmit === true && totalScopeFiles > 0) {
|
||||
if (parsedFileStorePath) {
|
||||
pdgEmitSink = new PdgEmitSink(
|
||||
ctx.graph,
|
||||
// Same ASCII-safe relocation the structural CSVs get (#2202 review #2):
|
||||
// on Windows non-ASCII storage paths the COPY can't open files under
|
||||
// the native path, so the dir is relocated to a hashed os.tmpdir().
|
||||
resolveNativeSafeStorageDir(parsedFileStorePath, 'pdg-csv'),
|
||||
ctx.options?.pdgEmitChunkSize,
|
||||
);
|
||||
} else {
|
||||
scopeFilePaths = new Set(primaryFilePaths);
|
||||
await loadStoreFor(scopeFilePaths);
|
||||
const pathsToRead = providerNeedsAllContent
|
||||
? primaryFilePaths
|
||||
: primaryFilePaths.filter((p) => !preExtractedByPath.has(p));
|
||||
contents = await readFileContents(ctx.repoPath, pathsToRead);
|
||||
}
|
||||
const filePaths = [...scopeFilePaths];
|
||||
const files: { path: string; content: string }[] = [];
|
||||
for (const fp of filePaths) {
|
||||
const content = contents.get(fp);
|
||||
if (content !== undefined) {
|
||||
files.push({ path: fp, content });
|
||||
} else if (preExtractedByPath.has(fp)) {
|
||||
// Store covers extraction for this file and we deliberately skipped
|
||||
// reading its source; the empty string is never consumed (the
|
||||
// extract loop uses the pre-extracted ParsedFile and this provider
|
||||
// has no content hook).
|
||||
files.push({ path: fp, content: '' });
|
||||
}
|
||||
// else: uncovered AND unreadable → skip (unchanged from prior behavior).
|
||||
}
|
||||
|
||||
const langFileCount = files.length;
|
||||
logHeapProbe(
|
||||
'scope-lang-start',
|
||||
`lang=${lang} files=${langFileCount} contentsLoaded=${contents.size}`,
|
||||
);
|
||||
const langLabel = lang.charAt(0).toUpperCase() + lang.slice(1);
|
||||
currentLangIdx++;
|
||||
const langTag =
|
||||
totalScopeLangs > 1 ? `${langLabel} [${currentLangIdx}/${totalScopeLangs}]` : langLabel;
|
||||
|
||||
if (totalScopeFiles > 0) {
|
||||
const pct =
|
||||
SCOPE_PCT_START + Math.round((processedScopeFiles / totalScopeFiles) * SCOPE_PCT_RANGE);
|
||||
ctx.onProgress({
|
||||
phase: 'scopeResolution',
|
||||
percent: pct,
|
||||
message: 'Resolving types',
|
||||
detail: `${langTag}, ${langFileCount.toLocaleString()} files`,
|
||||
});
|
||||
}
|
||||
|
||||
const stats = runScopeResolution(
|
||||
{
|
||||
graph: ctx.graph,
|
||||
model,
|
||||
files,
|
||||
resolutionConfig,
|
||||
prebuiltNodeLookup: sharedNodeLookup,
|
||||
prebuiltFunctionNodeIndex: sharedFnNodeIndex,
|
||||
preExtractedParsedFiles: preExtractedByPath,
|
||||
scopeIndexStorePath: parsedFileStorePath,
|
||||
// CFG/PDG emission (#2081 M1) — opt-in; off ⇒ byte-identical graph.
|
||||
pdg: ctx.options?.pdg === true,
|
||||
pdgMaxEdgesPerFunction: ctx.options?.pdgMaxEdgesPerFunction,
|
||||
pdgMaxReachingDefEdgesPerFunction: ctx.options?.pdgMaxReachingDefEdgesPerFunction,
|
||||
pdgMaxCdgEdgesPerFunction: ctx.options?.pdgMaxCdgEdgesPerFunction,
|
||||
pdgMaxTaintFindingsPerFunction: ctx.options?.pdgMaxTaintFindingsPerFunction,
|
||||
pdgMaxTaintHops: ctx.options?.pdgMaxTaintHops,
|
||||
recordResolutionOutcome: (outcome) => {
|
||||
resolutionOutcomes.push(outcome);
|
||||
},
|
||||
onWarn: (msg) => {
|
||||
if (isSemanticModelValidatorEnabled()) {
|
||||
logger.warn(`[scope-resolution:${lang}] ${msg}`);
|
||||
}
|
||||
},
|
||||
onProgress:
|
||||
totalScopeFiles > 0
|
||||
? (subPhase: ScopeResolutionSubPhase, current, total) => {
|
||||
let langRatio: number;
|
||||
switch (subPhase) {
|
||||
case 'extracting':
|
||||
langRatio = total > 0 ? (current / total) * 0.5 : 0;
|
||||
break;
|
||||
case 'analyzing types':
|
||||
langRatio = 0.5;
|
||||
break;
|
||||
case 'resolving references':
|
||||
langRatio = 0.7;
|
||||
break;
|
||||
case 'linking symbols':
|
||||
langRatio = 0.85;
|
||||
break;
|
||||
default: {
|
||||
const _exhaustive: never = subPhase;
|
||||
langRatio = 0.85;
|
||||
}
|
||||
}
|
||||
const overallRatio = Math.min(
|
||||
1,
|
||||
(processedScopeFiles + langRatio * langFileCount) / totalScopeFiles,
|
||||
);
|
||||
const pct = SCOPE_PCT_START + Math.round(overallRatio * SCOPE_PCT_RANGE);
|
||||
ctx.onProgress({
|
||||
phase: 'scopeResolution',
|
||||
percent: pct,
|
||||
message: 'Resolving types',
|
||||
detail:
|
||||
subPhase === 'extracting'
|
||||
? `${langTag} — extracting ${current.toLocaleString()}/${total.toLocaleString()} files`
|
||||
: `${langTag} — ${subPhase}`,
|
||||
});
|
||||
}
|
||||
: undefined,
|
||||
},
|
||||
provider,
|
||||
);
|
||||
|
||||
// Release file contents and pre-extracted entries after each language
|
||||
// to reduce memory pressure. For large codebases (16K+ PHP files),
|
||||
// holding all source code simultaneously with scope trees causes OOM.
|
||||
// See: https://github.com/abhigyanpatwari/GitNexus/issues/1741
|
||||
//
|
||||
// Use `filePaths` (not `primaryFilePaths`) so that any context files
|
||||
// added by `collectScopeContextPaths` (e.g. TS/JS files pulled in for
|
||||
// Vue cross-file resolution) are also evicted and not held until GC.
|
||||
files.length = 0;
|
||||
contents.clear();
|
||||
for (const fp of filePaths) {
|
||||
preExtractedByPath.delete(fp);
|
||||
}
|
||||
// This language's ParsedFiles are now unreachable (runScopeResolution has
|
||||
// returned and the Map entries are deleted). Force a GC HERE so a heavy
|
||||
// language's ~17-20GB set (e.g. C/C++ on the Linux kernel) is reclaimed
|
||||
// BEFORE the next language's store-load — instead of leaving V8 to collect
|
||||
// it lazily under the next pass's allocation pressure (which, at a cap >=
|
||||
// RAM, degrades into swap-thrash). Collects only dead objects: the live
|
||||
// cross-file index of the next pass is untouched. The pre/post probe
|
||||
// confirms whether old-space fragmentation defeats the reclaim.
|
||||
logHeapProbe('lang-release-pre-gc', `lang=${lang}`);
|
||||
forceGc();
|
||||
logHeapProbe('lang-release-post-gc', `lang=${lang}`);
|
||||
logHeapProbe('scope-lang-end', `lang=${lang} filesProcessed=${stats.filesProcessed}`);
|
||||
|
||||
processedScopeFiles += langFileCount;
|
||||
anyRan = true;
|
||||
functionSummaries.push(...stats.functionSummaries);
|
||||
totalFiles += stats.filesProcessed;
|
||||
totalImports += stats.importsEmitted;
|
||||
totalRefs += stats.referenceEdgesEmitted;
|
||||
perLanguage.set(lang, {
|
||||
filesProcessed: stats.filesProcessed,
|
||||
importsEmitted: stats.importsEmitted,
|
||||
referenceEdgesEmitted: stats.referenceEdgesEmitted,
|
||||
});
|
||||
|
||||
if (isDev) {
|
||||
logger.info(
|
||||
`[scope-resolution:${lang}] ${stats.filesProcessed} files → ${stats.importsEmitted} IMPORTS + ${stats.referenceEdgesEmitted} reference edges (${stats.resolve.unresolved} unresolved sites, ${stats.referenceSkipped} skipped)`,
|
||||
logger.warn(
|
||||
'[scope-resolution] streaming PDG emit requested but no storage path is ' +
|
||||
'available; falling back to in-memory whole-graph emit',
|
||||
);
|
||||
}
|
||||
}
|
||||
// Cross-pass per-file dedup set for the streaming sink (#2202): one set
|
||||
// shared across every language pass so a file emitted in two passes (e.g. a
|
||||
// `.ts` module pulled into the Vue context pass) streams its PDG layer once.
|
||||
// Only created when streaming — the in-memory-graph path dedups via its Map.
|
||||
const pdgEmittedFiles = pdgEmitSink !== undefined ? new Set<string>() : undefined;
|
||||
|
||||
// Stream the PDG layer with guaranteed writer cleanup: a throw escaping the
|
||||
// per-language loop (outside run.ts's per-file try/catch — e.g. from
|
||||
// finalize/propagate/a provider hook) must still release the sink's file
|
||||
// descriptors. finalize() runs on the success path; the finally closes the
|
||||
// sink only when finalize did not (idempotent via the sink's `finalized`).
|
||||
let pdgEmitManifest: PdgEmitManifest | undefined;
|
||||
let pdgSinkSettled = false;
|
||||
try {
|
||||
for (const [lang, provider] of SCOPE_RESOLVERS) {
|
||||
// Standalone providers (COBOL, JCL) don't emit graph edges yet
|
||||
// through the scope-resolution path. This is the canonical guard:
|
||||
// runScopeResolution is never called for standalone providers, which
|
||||
// keeps cobolPhase as the sole IMPORTS edge producer. Keep this guard
|
||||
// in sync with any additional standalone providers added to
|
||||
// SCOPE_RESOLVERS.
|
||||
if (provider.languageProvider.parseStrategy === 'standalone') continue;
|
||||
|
||||
const primaryLangFiles = filesByLang.get(lang) ?? [];
|
||||
if (primaryLangFiles.length === 0) continue;
|
||||
const primaryFilePaths = primaryLangFiles.map((f) => f.path);
|
||||
|
||||
// Load per-language import-resolution config (tsconfig paths,
|
||||
// composer.json autoload, go.mod, ...). One I/O round trip per
|
||||
// workspace pass — cached implicitly by the result handed to
|
||||
// every `resolveImportTarget` call below.
|
||||
const resolutionConfig =
|
||||
provider.loadResolutionConfig !== undefined
|
||||
? await provider.loadResolutionConfig(ctx.repoPath)
|
||||
: undefined;
|
||||
|
||||
// Some languages (e.g. Vue) expand their file universe beyond the
|
||||
// primary-language files via the `collectScopeContextPaths` hook.
|
||||
// The hook receives raw source contents of the primary files so it
|
||||
// can trace import closures without a second tree-sitter parse.
|
||||
//
|
||||
// To avoid reading primary files twice (once for the hook, once for
|
||||
// the resolution pass), we read them upfront and merge with the
|
||||
// extra context paths the hook may add.
|
||||
// Stream this language's pre-built ParsedFiles in from the disk store
|
||||
// FIRST (huge-repo path). Doing it before reading source lets us skip
|
||||
// loading content for files the store already covers — for a provider
|
||||
// with no content-consuming hook that source is pure dead weight once
|
||||
// extraction is served from the store (~1.5 GB on the kernel's C pass).
|
||||
// Merged into `preExtractedByPath`; the per-language release block below
|
||||
// evicts these again before the next language, so only one language's
|
||||
// ParsedFiles are resident at a time.
|
||||
const loadStoreFor = async (paths: ReadonlySet<string>): Promise<void> => {
|
||||
if (!parsedFileStorePath) return;
|
||||
const fromDisk = await loadParsedFilesForPaths(parsedFileStorePath, paths);
|
||||
for (const [fp, pf] of fromDisk) preExtractedByPath.set(fp, pf);
|
||||
};
|
||||
|
||||
// A provider that feeds source text into a post-extract hook
|
||||
// (populateWorkspaceOwners / populateNamespaceSiblings /
|
||||
// populateRangeBindings / emitPostResolutionEdges) needs content for ALL
|
||||
// its files; one without those hooks only needs content for files the
|
||||
// store does NOT cover (fresh-extract fallback). Keep this in sync with
|
||||
// the getFileContents() call-sites in run.ts.
|
||||
const providerNeedsAllContent =
|
||||
provider.populateWorkspaceOwners !== undefined ||
|
||||
provider.populateNamespaceSiblings !== undefined ||
|
||||
provider.populateRangeBindings !== undefined ||
|
||||
provider.emitPostResolutionEdges !== undefined;
|
||||
|
||||
let scopeFilePaths: Set<string>;
|
||||
let contents: Map<string, string>;
|
||||
if (provider.collectScopeContextPaths !== undefined) {
|
||||
// Context-expanding providers (e.g. Vue) need every primary file's
|
||||
// source up front for the closure hook, so load it all.
|
||||
const entryFileContents = await readFileContents(ctx.repoPath, primaryFilePaths);
|
||||
scopeFilePaths = provider.collectScopeContextPaths({
|
||||
primaryFilePaths,
|
||||
preExtractedByPath,
|
||||
entryFileContents,
|
||||
allScannedPaths,
|
||||
resolutionConfig,
|
||||
});
|
||||
// Read only the extra context files (TS/JS etc.) not already loaded.
|
||||
const extraPaths = [...scopeFilePaths].filter((p) => !entryFileContents.has(p));
|
||||
const extraContents = await readFileContents(ctx.repoPath, extraPaths);
|
||||
contents = new Map([...entryFileContents, ...extraContents]);
|
||||
await loadStoreFor(scopeFilePaths);
|
||||
} else {
|
||||
scopeFilePaths = new Set(primaryFilePaths);
|
||||
await loadStoreFor(scopeFilePaths);
|
||||
const pathsToRead = providerNeedsAllContent
|
||||
? primaryFilePaths
|
||||
: primaryFilePaths.filter((p) => !preExtractedByPath.has(p));
|
||||
contents = await readFileContents(ctx.repoPath, pathsToRead);
|
||||
}
|
||||
const filePaths = [...scopeFilePaths];
|
||||
const files: { path: string; content: string }[] = [];
|
||||
for (const fp of filePaths) {
|
||||
const content = contents.get(fp);
|
||||
if (content !== undefined) {
|
||||
files.push({ path: fp, content });
|
||||
} else if (preExtractedByPath.has(fp)) {
|
||||
// Store covers extraction for this file and we deliberately skipped
|
||||
// reading its source; the empty string is never consumed (the
|
||||
// extract loop uses the pre-extracted ParsedFile and this provider
|
||||
// has no content hook).
|
||||
files.push({ path: fp, content: '' });
|
||||
}
|
||||
// else: uncovered AND unreadable → skip (unchanged from prior behavior).
|
||||
}
|
||||
|
||||
const langFileCount = files.length;
|
||||
logHeapProbe(
|
||||
'scope-lang-start',
|
||||
`lang=${lang} files=${langFileCount} contentsLoaded=${contents.size}`,
|
||||
);
|
||||
const langLabel = lang.charAt(0).toUpperCase() + lang.slice(1);
|
||||
currentLangIdx++;
|
||||
const langTag =
|
||||
totalScopeLangs > 1 ? `${langLabel} [${currentLangIdx}/${totalScopeLangs}]` : langLabel;
|
||||
|
||||
if (totalScopeFiles > 0) {
|
||||
const pct =
|
||||
SCOPE_PCT_START + Math.round((processedScopeFiles / totalScopeFiles) * SCOPE_PCT_RANGE);
|
||||
ctx.onProgress({
|
||||
phase: 'scopeResolution',
|
||||
percent: pct,
|
||||
message: 'Resolving types',
|
||||
detail: `${langTag}, ${langFileCount.toLocaleString()} files`,
|
||||
});
|
||||
}
|
||||
|
||||
const stats = runScopeResolution(
|
||||
{
|
||||
graph: ctx.graph,
|
||||
model,
|
||||
files,
|
||||
resolutionConfig,
|
||||
prebuiltNodeLookup: sharedNodeLookup,
|
||||
prebuiltFunctionNodeIndex: sharedFnNodeIndex,
|
||||
preExtractedParsedFiles: preExtractedByPath,
|
||||
scopeIndexStorePath: parsedFileStorePath,
|
||||
// CFG/PDG emission (#2081 M1) — opt-in; off ⇒ byte-identical graph.
|
||||
pdg: ctx.options?.pdg === true,
|
||||
pdgMaxEdgesPerFunction: ctx.options?.pdgMaxEdgesPerFunction,
|
||||
pdgMaxReachingDefEdgesPerFunction: ctx.options?.pdgMaxReachingDefEdgesPerFunction,
|
||||
pdgMaxCdgEdgesPerFunction: ctx.options?.pdgMaxCdgEdgesPerFunction,
|
||||
pdgMaxTaintFindingsPerFunction: ctx.options?.pdgMaxTaintFindingsPerFunction,
|
||||
pdgMaxTaintHops: ctx.options?.pdgMaxTaintHops,
|
||||
// Streaming PDG-emit sink (#2202) — undefined ⇒ emit to the in-memory graph.
|
||||
pdgEmitSink,
|
||||
// Cross-pass per-file dedup set (#2202) — undefined when not streaming.
|
||||
pdgEmittedFiles,
|
||||
recordResolutionOutcome: (outcome) => {
|
||||
resolutionOutcomes.push(outcome);
|
||||
},
|
||||
onWarn: (msg) => {
|
||||
if (isSemanticModelValidatorEnabled()) {
|
||||
logger.warn(`[scope-resolution:${lang}] ${msg}`);
|
||||
}
|
||||
},
|
||||
onProgress:
|
||||
totalScopeFiles > 0
|
||||
? (subPhase: ScopeResolutionSubPhase, current, total) => {
|
||||
let langRatio: number;
|
||||
switch (subPhase) {
|
||||
case 'extracting':
|
||||
langRatio = total > 0 ? (current / total) * 0.5 : 0;
|
||||
break;
|
||||
case 'analyzing types':
|
||||
langRatio = 0.5;
|
||||
break;
|
||||
case 'resolving references':
|
||||
langRatio = 0.7;
|
||||
break;
|
||||
case 'linking symbols':
|
||||
langRatio = 0.85;
|
||||
break;
|
||||
default: {
|
||||
const _exhaustive: never = subPhase;
|
||||
langRatio = 0.85;
|
||||
}
|
||||
}
|
||||
const overallRatio = Math.min(
|
||||
1,
|
||||
(processedScopeFiles + langRatio * langFileCount) / totalScopeFiles,
|
||||
);
|
||||
const pct = SCOPE_PCT_START + Math.round(overallRatio * SCOPE_PCT_RANGE);
|
||||
ctx.onProgress({
|
||||
phase: 'scopeResolution',
|
||||
percent: pct,
|
||||
message: 'Resolving types',
|
||||
detail:
|
||||
subPhase === 'extracting'
|
||||
? `${langTag} — extracting ${current.toLocaleString()}/${total.toLocaleString()} files`
|
||||
: `${langTag} — ${subPhase}`,
|
||||
});
|
||||
}
|
||||
: undefined,
|
||||
},
|
||||
provider,
|
||||
);
|
||||
|
||||
// Release file contents and pre-extracted entries after each language
|
||||
// to reduce memory pressure. For large codebases (16K+ PHP files),
|
||||
// holding all source code simultaneously with scope trees causes OOM.
|
||||
// See: https://github.com/abhigyanpatwari/GitNexus/issues/1741
|
||||
//
|
||||
// Use `filePaths` (not `primaryFilePaths`) so that any context files
|
||||
// added by `collectScopeContextPaths` (e.g. TS/JS files pulled in for
|
||||
// Vue cross-file resolution) are also evicted and not held until GC.
|
||||
files.length = 0;
|
||||
contents.clear();
|
||||
for (const fp of filePaths) {
|
||||
preExtractedByPath.delete(fp);
|
||||
}
|
||||
// This language's ParsedFiles are now unreachable (runScopeResolution has
|
||||
// returned and the Map entries are deleted). Force a GC HERE so a heavy
|
||||
// language's ~17-20GB set (e.g. C/C++ on the Linux kernel) is reclaimed
|
||||
// BEFORE the next language's store-load — instead of leaving V8 to collect
|
||||
// it lazily under the next pass's allocation pressure (which, at a cap >=
|
||||
// RAM, degrades into swap-thrash). Collects only dead objects: the live
|
||||
// cross-file index of the next pass is untouched. The pre/post probe
|
||||
// confirms whether old-space fragmentation defeats the reclaim.
|
||||
logHeapProbe('lang-release-pre-gc', `lang=${lang}`);
|
||||
forceGc();
|
||||
logHeapProbe('lang-release-post-gc', `lang=${lang}`);
|
||||
logHeapProbe('scope-lang-end', `lang=${lang} filesProcessed=${stats.filesProcessed}`);
|
||||
|
||||
processedScopeFiles += langFileCount;
|
||||
anyRan = true;
|
||||
functionSummaries.push(...stats.functionSummaries);
|
||||
totalFiles += stats.filesProcessed;
|
||||
totalImports += stats.importsEmitted;
|
||||
totalRefs += stats.referenceEdgesEmitted;
|
||||
perLanguage.set(lang, {
|
||||
filesProcessed: stats.filesProcessed,
|
||||
importsEmitted: stats.importsEmitted,
|
||||
referenceEdgesEmitted: stats.referenceEdgesEmitted,
|
||||
});
|
||||
|
||||
if (isDev) {
|
||||
logger.info(
|
||||
`[scope-resolution:${lang}] ${stats.filesProcessed} files → ${stats.importsEmitted} IMPORTS + ${stats.referenceEdgesEmitted} reference edges (${stats.resolve.unresolved} unresolved sites, ${stats.referenceSkipped} skipped)`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Finalize the streaming PDG sink (#2202) once after the last language:
|
||||
// flush + close its CSV writers and capture the COPY manifest. forceGc at
|
||||
// the boundary reclaims transient write buffers (mirrors the per-language
|
||||
// release below).
|
||||
pdgEmitManifest = pdgEmitSink?.finalize();
|
||||
pdgSinkSettled = true;
|
||||
if (pdgEmitSink !== undefined) forceGc();
|
||||
} finally {
|
||||
// Release fds if a throw skipped finalize (idempotent with finalize()).
|
||||
if (pdgEmitSink !== undefined && !pdgSinkSettled) pdgEmitSink.close();
|
||||
}
|
||||
|
||||
if (totalScopeFiles > 0 && anyRan) {
|
||||
ctx.onProgress({
|
||||
|
|
@ -494,7 +559,10 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
}
|
||||
}
|
||||
|
||||
if (!anyRan) return NOOP_OUTPUT;
|
||||
// Even when no language ran, surface a finalized manifest (its CSVs are on
|
||||
// disk) so loadGraphToLbug COPYs them rather than orphaning them — empty in
|
||||
// the no-files case, harmless.
|
||||
if (!anyRan) return pdgEmitManifest ? { ...NOOP_OUTPUT, pdgEmitManifest } : NOOP_OUTPUT;
|
||||
|
||||
return {
|
||||
ran: true,
|
||||
|
|
@ -504,6 +572,7 @@ export const scopeResolutionPhase: PipelinePhase<ScopeResolutionOutput> = {
|
|||
resolutionOutcomes,
|
||||
perLanguage,
|
||||
functionSummaries,
|
||||
pdgEmitManifest,
|
||||
};
|
||||
},
|
||||
};
|
||||
|
|
|
|||
|
|
@ -302,6 +302,30 @@ interface RunScopeResolutionInput {
|
|||
* `reason`; consumed by the U4 taint emit step). `undefined` ⇒
|
||||
* `DEFAULT_PDG_MAX_TAINT_HOPS` (32); `0` ⇒ no cap. */
|
||||
readonly pdgMaxTaintHops?: number;
|
||||
/**
|
||||
* Streaming PDG-emit sink (#2202). When present (streaming on, full rebuild),
|
||||
* the `--pdg` emit routes BasicBlock nodes + intra-file PDG edges to THIS
|
||||
* graph-shaped target instead of the in-memory `graph`, so the bulky PDG
|
||||
* layer never accumulates in memory (peak RSS O(chunk)). Typed as a plain
|
||||
* `KnowledgeGraph` so this module stays decoupled from the persistence layer;
|
||||
* the caller (the scope-resolution phase) owns its lifecycle and finalizes it
|
||||
* after the last language. Absent ⇒ the emit writes to `graph` as before
|
||||
* (byte-identical default).
|
||||
*/
|
||||
readonly pdgEmitSink?: KnowledgeGraph;
|
||||
/**
|
||||
* Cross-pass per-file dedup set for streaming PDG emit (#2202). Shared across
|
||||
* every language pass (owned by the scope-resolution phase). A file imported
|
||||
* by more than one language (e.g. a `.ts` module pulled into the Vue context
|
||||
* pass) is PDG-emitted in each pass over the same `cfgSideChannel`, producing
|
||||
* identical ids; the in-memory graph dedups that by id, but the streaming sink
|
||||
* is dedup-free (to stay O(write buffer), not O(total ids)). So when present
|
||||
* (streaming on), the emit loop skips a file whose PDG already streamed and
|
||||
* records the rest — keeping the streamed set byte-identical to the
|
||||
* Map-deduped whole-graph emit, for any language-pass order. Absent ⇒ no skip
|
||||
* (the graph Map dedups), so the default path is unchanged.
|
||||
*/
|
||||
readonly pdgEmittedFiles?: Set<string>;
|
||||
/**
|
||||
* Optional graph-node lookup built ONCE by the caller and shared across
|
||||
* every language pass. `buildGraphNodeLookup` scans the whole graph and is
|
||||
|
|
@ -769,6 +793,11 @@ export function runScopeResolution(
|
|||
// can bracket it. Printed as the PROF `taint=` segment.
|
||||
let taintMs = 0;
|
||||
if (input.pdg === true) {
|
||||
// Streaming target (#2202): when a sink is provided, BasicBlock nodes +
|
||||
// intra-file PDG edges are routed to CSV-on-disk through it instead of
|
||||
// accumulating in `graph`. The function-node index below is still built
|
||||
// from the real `graph` (Function/Method nodes live there, never the sink).
|
||||
const pdgTarget: KnowledgeGraph = input.pdgEmitSink ?? graph;
|
||||
let cfgBlocks = 0;
|
||||
let cfgEdges = 0;
|
||||
let cfgDroppedEdges = 0;
|
||||
|
|
@ -835,6 +864,15 @@ export function runScopeResolution(
|
|||
// shard that slipped the version gate) must skip emission, not throw a
|
||||
// TypeError mid-graph-build and abort scope-resolution for the language.
|
||||
if (!Array.isArray(cfgs) || cfgs.length === 0) continue;
|
||||
// Cross-pass per-file dedup (#2202): when streaming, a file whose PDG
|
||||
// already streamed in a prior language pass (e.g. a `.ts` module pulled
|
||||
// into the Vue context pass) would re-emit identical ids from the same
|
||||
// cfgSideChannel — the dedup-free streaming sink would double the rows.
|
||||
// Skip it here; the in-memory-graph path needs no skip (its Map dedups).
|
||||
if (input.pdgEmittedFiles !== undefined) {
|
||||
if (input.pdgEmittedFiles.has(pf.filePath)) continue;
|
||||
input.pdgEmittedFiles.add(pf.filePath);
|
||||
}
|
||||
try {
|
||||
// Per-element emit-safety filter (mirrors the parsedfile-store
|
||||
// reviver's POLICY: valid elements in a mixed array still emit; junk
|
||||
|
|
@ -854,7 +892,7 @@ export function runScopeResolution(
|
|||
}
|
||||
if (wellFormed.length === 0) continue;
|
||||
const emitted = emitFileCfgs(
|
||||
graph,
|
||||
pdgTarget,
|
||||
wellFormed,
|
||||
input.pdgMaxEdgesPerFunction ?? DEFAULT_MAX_CFG_EDGES_PER_FUNCTION,
|
||||
// Log cap-overflow drops UNCONDITIONALLY (not via input.onWarn, which is
|
||||
|
|
@ -873,7 +911,7 @@ export function runScopeResolution(
|
|||
// PROF-gated like every other checkpoint here (zero cost when off).
|
||||
const t0 = PROF ? performance.now() : 0;
|
||||
const rd = emitFileReachingDefs(
|
||||
graph,
|
||||
pdgTarget,
|
||||
wellFormed,
|
||||
input.pdgMaxReachingDefEdgesPerFunction ??
|
||||
DEFAULT_PDG_MAX_REACHING_DEF_EDGES_PER_FUNCTION,
|
||||
|
|
@ -892,7 +930,7 @@ export function runScopeResolution(
|
|||
// persisted and its time folds into the `pdg=` PROF segment next to RD.
|
||||
const tCdg = PROF ? performance.now() : 0;
|
||||
const cdg = emitFileCdg(
|
||||
graph,
|
||||
pdgTarget,
|
||||
wellFormed,
|
||||
input.pdgMaxCdgEdgesPerFunction ?? DEFAULT_PDG_MAX_CDG_EDGES_PER_FUNCTION,
|
||||
(message) => logger.warn(message), // unconditional — R6, no silent truncation
|
||||
|
|
@ -909,7 +947,7 @@ export function runScopeResolution(
|
|||
if (taintSpec !== undefined) {
|
||||
const t1 = PROF ? performance.now() : 0;
|
||||
const taint = emitFileTaint(
|
||||
graph,
|
||||
pdgTarget,
|
||||
wellFormed,
|
||||
pf.parsedImports,
|
||||
taintSpec,
|
||||
|
|
|
|||
|
|
@ -28,6 +28,18 @@ export const parseTruthyEnv = (raw: string | undefined): boolean => {
|
|||
return value === '1' || value === 'true' || value === 'yes';
|
||||
};
|
||||
|
||||
/**
|
||||
* Parse a positive-integer env-var value. Returns the integer when `raw` is a
|
||||
* finite integer `> 0`; otherwise (`undefined`, empty, non-numeric, `0`, or
|
||||
* negative) returns `undefined` so the caller falls back to its default. Used
|
||||
* for numeric tuning knobs like `GITNEXUS_PDG_EMIT_CHUNK_SIZE` (#2202).
|
||||
*/
|
||||
export const parsePositiveIntEnv = (raw: string | undefined): number | undefined => {
|
||||
if (raw === undefined) return undefined;
|
||||
const n = Number(raw.trim());
|
||||
return Number.isInteger(n) && n > 0 ? n : undefined;
|
||||
};
|
||||
|
||||
/**
|
||||
* Whether scope-resolution dev validators (e.g. `validateBindingsImmutability`)
|
||||
* should run AND emit warnings. Off by default in CLI runs to avoid silent
|
||||
|
|
|
|||
|
|
@ -17,7 +17,8 @@ import { createWriteStream, WriteStream } from 'fs';
|
|||
import path from 'path';
|
||||
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
|
||||
import { KnowledgeGraph } from '../graph/types.js';
|
||||
import { NodeTableName } from './schema.js';
|
||||
import { NodeTableName, NODE_TABLES } from './schema.js';
|
||||
import { RelPairRouter } from './rel-pair-routing.js';
|
||||
import { parseTruthyEnv } from '../ingestion/utils/env.js';
|
||||
|
||||
/**
|
||||
|
|
@ -182,13 +183,20 @@ class BufferedCSVWriter {
|
|||
this.buffer.push(header);
|
||||
}
|
||||
|
||||
addRow(row: string) {
|
||||
/**
|
||||
* Buffer a row. Returns a promise ONLY when the buffer crossed FLUSH_EVERY
|
||||
* and a disk write was issued; otherwise returns `undefined` so the caller
|
||||
* can skip awaiting (#2203 U3) — avoiding a microtask tick on every buffered
|
||||
* row (millions at scale). The flush promise still resolves on drain, so
|
||||
* backpressure is preserved on the rows that actually write.
|
||||
*/
|
||||
addRow(row: string): Promise<void> | undefined {
|
||||
this.buffer.push(row);
|
||||
this.rows++;
|
||||
if (this.buffer.length >= FLUSH_EVERY) {
|
||||
return this.flush();
|
||||
}
|
||||
return Promise.resolve();
|
||||
return undefined;
|
||||
}
|
||||
|
||||
flush(): Promise<void> {
|
||||
|
|
@ -223,10 +231,51 @@ class BufferedCSVWriter {
|
|||
// STREAMING CSV GENERATION — SINGLE PASS
|
||||
// ============================================================================
|
||||
|
||||
/** Canonical relationship CSV header — shared by the emit pass and the
|
||||
* `splitRelCsvByLabelPair` differential oracle. */
|
||||
export const REL_CSV_HEADER = 'from,to,type,confidence,reason,step';
|
||||
|
||||
/** Build the escaped CSV row (no trailing newline) for one relationship.
|
||||
* Single source of the relationship row bytes — used by the emit pass and by
|
||||
* the byte-identity differential test that feeds the legacy split oracle. */
|
||||
export const buildRelRow = (rel: GraphRelationship): string =>
|
||||
[
|
||||
escapeCSVField(rel.sourceId),
|
||||
escapeCSVField(rel.targetId),
|
||||
escapeCSVField(rel.type),
|
||||
escapeCSVNumber(rel.confidence, 1.0),
|
||||
escapeCSVField(rel.reason),
|
||||
escapeCSVNumber(rel.step, 0),
|
||||
].join(',');
|
||||
|
||||
/** Canonical BasicBlock node CSV header — taint/PDG substrate (issue #2080).
|
||||
* No `name` column; blocks are identified by id + source span. Shared by the
|
||||
* whole-graph emit pass and the streaming PDG emit sink (issue #2202) so the
|
||||
* two paths produce byte-identical BasicBlock rows by construction. */
|
||||
export const BASICBLOCK_CSV_HEADER = 'id,filePath,startLine,endLine,text';
|
||||
|
||||
/** Build the escaped CSV row (no trailing newline) for one BasicBlock node.
|
||||
* Single source of the BasicBlock row bytes — used by `streamAllCSVsToDisk`
|
||||
* and by the streaming `PdgEmitSink` (issue #2202). */
|
||||
export const buildBasicBlockRow = (node: GraphNode): string =>
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
escapeCSVField(node.properties.text || ''),
|
||||
].join(',');
|
||||
|
||||
export interface StreamedCSVResult {
|
||||
nodeFiles: Map<NodeTableName, { csvPath: string; rows: number }>;
|
||||
relCsvPath: string;
|
||||
relRows: number;
|
||||
/** pairKey (`From|To`) → per-FROM→TO-label-pair CSV file. */
|
||||
relsByPair: Map<string, { csvPath: string; rows: number }>;
|
||||
/** Header line shared by every per-pair file. */
|
||||
relHeader: string;
|
||||
/** Edges skipped because an endpoint label is not a valid node table. */
|
||||
skippedRels: number;
|
||||
/** Edges routed to a per-pair file. */
|
||||
totalValidRels: number;
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -253,255 +302,185 @@ export const streamAllCSVsToDisk = async (
|
|||
const prevMax = process.getMaxListeners();
|
||||
process.setMaxListeners(prevMax + 40);
|
||||
|
||||
const contentCache = new FileContentCache(repoPath);
|
||||
// try/finally so the listener bump is ALWAYS restored — including the
|
||||
// rel-routing throw path (#2203 U2) and any node-writer finish() rejection,
|
||||
// not just the success path (avoids leaking +40 listeners across failed runs
|
||||
// in long-lived hosts / the test suite).
|
||||
try {
|
||||
const contentCache = new FileContentCache(repoPath);
|
||||
|
||||
// Create writers for every node type up-front
|
||||
const fileWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'file.csv'),
|
||||
'id,name,filePath,content',
|
||||
);
|
||||
const folderWriter = new BufferedCSVWriter(path.join(csvDir, 'folder.csv'), 'id,name,filePath');
|
||||
const codeElementHeader = 'id,name,filePath,startLine,endLine,isExported,content,description';
|
||||
const functionWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'function.csv'),
|
||||
codeElementHeader,
|
||||
);
|
||||
const classWriter = new BufferedCSVWriter(path.join(csvDir, 'class.csv'), codeElementHeader);
|
||||
const interfaceWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'interface.csv'),
|
||||
codeElementHeader,
|
||||
);
|
||||
const methodHeader =
|
||||
'id,name,filePath,startLine,endLine,isExported,content,description,parameterCount,returnType';
|
||||
const methodWriter = new BufferedCSVWriter(path.join(csvDir, 'method.csv'), methodHeader);
|
||||
const codeElemWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'codeelement.csv'),
|
||||
codeElementHeader,
|
||||
);
|
||||
const communityWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'community.csv'),
|
||||
'id,label,heuristicLabel,keywords,description,enrichedBy,cohesion,symbolCount',
|
||||
);
|
||||
const processWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'process.csv'),
|
||||
'id,label,heuristicLabel,processType,stepCount,communities,entryPointId,terminalId',
|
||||
);
|
||||
|
||||
// Section nodes have an extra 'level' column
|
||||
const sectionWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'section.csv'),
|
||||
'id,name,filePath,startLine,endLine,level,content,description',
|
||||
);
|
||||
|
||||
// Route nodes for API endpoint mapping
|
||||
const routeWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'route.csv'),
|
||||
'id,name,filePath,responseKeys,errorKeys,middleware',
|
||||
);
|
||||
|
||||
// Tool nodes for MCP tool definitions
|
||||
const toolWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'tool.csv'),
|
||||
'id,name,filePath,description',
|
||||
);
|
||||
|
||||
// BasicBlock nodes — taint/PDG substrate (issue #2080). No `name` column;
|
||||
// blocks are identified by id + source span. Emitted by no phase yet.
|
||||
const basicBlockWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'basicblock.csv'),
|
||||
'id,filePath,startLine,endLine,text',
|
||||
);
|
||||
|
||||
// Multi-language node types share the same CSV shape (no isExported column)
|
||||
const multiLangHeader = 'id,name,filePath,startLine,endLine,content,description';
|
||||
const MULTI_LANG_TYPES = [
|
||||
'Struct',
|
||||
'Enum',
|
||||
'Macro',
|
||||
'Typedef',
|
||||
'Union',
|
||||
'Namespace',
|
||||
'Trait',
|
||||
'Impl',
|
||||
'TypeAlias',
|
||||
'Const',
|
||||
'Static',
|
||||
'Variable',
|
||||
'Property',
|
||||
'Record',
|
||||
'Delegate',
|
||||
'Annotation',
|
||||
'Constructor',
|
||||
'Template',
|
||||
'Module',
|
||||
] as const;
|
||||
const propertyHeader = 'id,name,filePath,startLine,endLine,content,description,declaredType';
|
||||
const multiLangWriters = new Map<string, BufferedCSVWriter>();
|
||||
for (const t of MULTI_LANG_TYPES) {
|
||||
multiLangWriters.set(
|
||||
t,
|
||||
new BufferedCSVWriter(
|
||||
path.join(csvDir, `${t.toLowerCase()}.csv`),
|
||||
t === 'Property' ? propertyHeader : multiLangHeader,
|
||||
),
|
||||
// Create writers for every node type up-front
|
||||
const fileWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'file.csv'),
|
||||
'id,name,filePath,content',
|
||||
);
|
||||
const folderWriter = new BufferedCSVWriter(path.join(csvDir, 'folder.csv'), 'id,name,filePath');
|
||||
const codeElementHeader = 'id,name,filePath,startLine,endLine,isExported,content,description';
|
||||
const functionWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'function.csv'),
|
||||
codeElementHeader,
|
||||
);
|
||||
const classWriter = new BufferedCSVWriter(path.join(csvDir, 'class.csv'), codeElementHeader);
|
||||
const interfaceWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'interface.csv'),
|
||||
codeElementHeader,
|
||||
);
|
||||
const methodHeader =
|
||||
'id,name,filePath,startLine,endLine,isExported,content,description,parameterCount,returnType';
|
||||
const methodWriter = new BufferedCSVWriter(path.join(csvDir, 'method.csv'), methodHeader);
|
||||
const codeElemWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'codeelement.csv'),
|
||||
codeElementHeader,
|
||||
);
|
||||
const communityWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'community.csv'),
|
||||
'id,label,heuristicLabel,keywords,description,enrichedBy,cohesion,symbolCount',
|
||||
);
|
||||
const processWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'process.csv'),
|
||||
'id,label,heuristicLabel,processType,stepCount,communities,entryPointId,terminalId',
|
||||
);
|
||||
}
|
||||
|
||||
const codeWriterMap: Record<string, BufferedCSVWriter> = {
|
||||
Function: functionWriter,
|
||||
Class: classWriter,
|
||||
Interface: interfaceWriter,
|
||||
CodeElement: codeElemWriter,
|
||||
};
|
||||
// Section nodes have an extra 'level' column
|
||||
const sectionWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'section.csv'),
|
||||
'id,name,filePath,startLine,endLine,level,content,description',
|
||||
);
|
||||
|
||||
// Deduplicate all node types — the pipeline can produce duplicate IDs across
|
||||
// all symbol types (Class, Method, Function, etc.), not just File nodes.
|
||||
// A single Set covering every label prevents PK violations on COPY.
|
||||
const seenNodeIds = new Set<string>();
|
||||
// Route nodes for API endpoint mapping
|
||||
const routeWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'route.csv'),
|
||||
'id,name,filePath,responseKeys,errorKeys,middleware',
|
||||
);
|
||||
|
||||
// --- SINGLE PASS over all nodes ---
|
||||
for (const node of orderedNodes(graph, sortOutput)) {
|
||||
if (seenNodeIds.has(node.id)) continue;
|
||||
seenNodeIds.add(node.id);
|
||||
// Tool nodes for MCP tool definitions
|
||||
const toolWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'tool.csv'),
|
||||
'id,name,filePath,description',
|
||||
);
|
||||
|
||||
switch (node.label) {
|
||||
case 'File': {
|
||||
const content = await extractContent(node, contentCache);
|
||||
await fileWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVField(content),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Folder':
|
||||
await folderWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
case 'Community': {
|
||||
const keywords = node.properties.keywords || [];
|
||||
const keywordsStr = `[${keywords.map((k: string) => `'${k.replace(/\\/g, '\\\\').replace(/'/g, "''").replace(/,/g, '\\,')}'`).join(',')}]`;
|
||||
await communityWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.heuristicLabel || ''),
|
||||
keywordsStr,
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
escapeCSVField(node.properties.enrichedBy || 'heuristic'),
|
||||
escapeCSVNumber(node.properties.cohesion, 0),
|
||||
escapeCSVNumber(node.properties.symbolCount, 0),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Process': {
|
||||
const communities = node.properties.communities || [];
|
||||
const communitiesStr = `[${communities.map((c: string) => `'${c.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
await processWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.heuristicLabel || ''),
|
||||
escapeCSVField(node.properties.processType || ''),
|
||||
escapeCSVNumber(node.properties.stepCount, 0),
|
||||
escapeCSVField(communitiesStr),
|
||||
escapeCSVField(node.properties.entryPointId || ''),
|
||||
escapeCSVField(node.properties.terminalId || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Method': {
|
||||
const content = await extractContent(node, contentCache);
|
||||
await methodWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
node.properties.isExported ? 'true' : 'false',
|
||||
escapeCSVField(content),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
escapeCSVNumber(node.properties.parameterCount, 0),
|
||||
escapeCSVField(node.properties.returnType || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Section': {
|
||||
const content = await extractContent(node, contentCache);
|
||||
await sectionWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
escapeCSVNumber(node.properties.level, 1),
|
||||
escapeCSVField(content),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Route': {
|
||||
const responseKeys = node.properties.responseKeys || [];
|
||||
// LadybugDB array literal inside a quoted CSV field: escapeCSVField wraps in "..."
|
||||
// and the array uses single-quoted elements
|
||||
const keysStr = `[${responseKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
const errorKeys = node.properties.errorKeys || [];
|
||||
const errorKeysStr = `[${errorKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
const middleware = node.properties.middleware || [];
|
||||
const middlewareStr = `[${middleware.map((m: string) => `'${m.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
await routeWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVField(keysStr),
|
||||
escapeCSVField(errorKeysStr),
|
||||
escapeCSVField(middlewareStr),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Tool':
|
||||
await toolWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
case 'BasicBlock':
|
||||
await basicBlockWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
escapeCSVField(node.properties.text || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
default: {
|
||||
// Code element nodes (Function, Class, Interface, CodeElement)
|
||||
const writer = codeWriterMap[node.label];
|
||||
if (writer) {
|
||||
// BasicBlock nodes — taint/PDG substrate (issue #2080). No `name` column;
|
||||
// blocks are identified by id + source span. Emitted by no phase yet.
|
||||
const basicBlockWriter = new BufferedCSVWriter(
|
||||
path.join(csvDir, 'basicblock.csv'),
|
||||
BASICBLOCK_CSV_HEADER,
|
||||
);
|
||||
|
||||
// Multi-language node types share the same CSV shape (no isExported column)
|
||||
const multiLangHeader = 'id,name,filePath,startLine,endLine,content,description';
|
||||
const MULTI_LANG_TYPES = [
|
||||
'Struct',
|
||||
'Enum',
|
||||
'Macro',
|
||||
'Typedef',
|
||||
'Union',
|
||||
'Namespace',
|
||||
'Trait',
|
||||
'Impl',
|
||||
'TypeAlias',
|
||||
'Const',
|
||||
'Static',
|
||||
'Variable',
|
||||
'Property',
|
||||
'Record',
|
||||
'Delegate',
|
||||
'Annotation',
|
||||
'Constructor',
|
||||
'Template',
|
||||
'Module',
|
||||
] as const;
|
||||
const propertyHeader = 'id,name,filePath,startLine,endLine,content,description,declaredType';
|
||||
const multiLangWriters = new Map<string, BufferedCSVWriter>();
|
||||
for (const t of MULTI_LANG_TYPES) {
|
||||
multiLangWriters.set(
|
||||
t,
|
||||
new BufferedCSVWriter(
|
||||
path.join(csvDir, `${t.toLowerCase()}.csv`),
|
||||
t === 'Property' ? propertyHeader : multiLangHeader,
|
||||
),
|
||||
);
|
||||
}
|
||||
|
||||
const codeWriterMap: Record<string, BufferedCSVWriter> = {
|
||||
Function: functionWriter,
|
||||
Class: classWriter,
|
||||
Interface: interfaceWriter,
|
||||
CodeElement: codeElemWriter,
|
||||
};
|
||||
|
||||
// Deduplicate all node types — the pipeline can produce duplicate IDs across
|
||||
// all symbol types (Class, Method, Function, etc.), not just File nodes.
|
||||
// A single Set covering every label prevents PK violations on COPY.
|
||||
const seenNodeIds = new Set<string>();
|
||||
|
||||
// --- SINGLE PASS over all nodes ---
|
||||
for (const node of orderedNodes(graph, sortOutput)) {
|
||||
if (seenNodeIds.has(node.id)) continue;
|
||||
seenNodeIds.add(node.id);
|
||||
|
||||
// addRow returns a promise only when it flushes; awaiting it once after the
|
||||
// switch (instead of `await`-ing every addRow) skips a per-row microtask
|
||||
// tick on the ~FLUSH_EVERY-1 buffered rows between flushes (#2203 U3).
|
||||
let pending: Promise<void> | undefined;
|
||||
switch (node.label) {
|
||||
case 'File': {
|
||||
const content = await extractContent(node, contentCache);
|
||||
await writer.addRow(
|
||||
pending = fileWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVField(content),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Folder':
|
||||
pending = folderWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
case 'Community': {
|
||||
const keywords = node.properties.keywords || [];
|
||||
const keywordsStr = `[${keywords.map((k: string) => `'${k.replace(/\\/g, '\\\\').replace(/'/g, "''").replace(/,/g, '\\,')}'`).join(',')}]`;
|
||||
pending = communityWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.heuristicLabel || ''),
|
||||
keywordsStr,
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
escapeCSVField(node.properties.enrichedBy || 'heuristic'),
|
||||
escapeCSVNumber(node.properties.cohesion, 0),
|
||||
escapeCSVNumber(node.properties.symbolCount, 0),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Process': {
|
||||
const communities = node.properties.communities || [];
|
||||
const communitiesStr = `[${communities.map((c: string) => `'${c.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
pending = processWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.heuristicLabel || ''),
|
||||
escapeCSVField(node.properties.processType || ''),
|
||||
escapeCSVNumber(node.properties.stepCount, 0),
|
||||
escapeCSVField(communitiesStr),
|
||||
escapeCSVField(node.properties.entryPointId || ''),
|
||||
escapeCSVField(node.properties.terminalId || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Method': {
|
||||
const content = await extractContent(node, contentCache);
|
||||
pending = methodWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
|
|
@ -511,101 +490,191 @@ export const streamAllCSVsToDisk = async (
|
|||
node.properties.isExported ? 'true' : 'false',
|
||||
escapeCSVField(content),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
escapeCSVNumber(node.properties.parameterCount, 0),
|
||||
escapeCSVField(node.properties.returnType || ''),
|
||||
].join(','),
|
||||
);
|
||||
} else {
|
||||
// Multi-language node types (Struct, Impl, Trait, Macro, etc.)
|
||||
const mlWriter = multiLangWriters.get(node.label);
|
||||
if (mlWriter) {
|
||||
break;
|
||||
}
|
||||
case 'Section': {
|
||||
const content = await extractContent(node, contentCache);
|
||||
pending = sectionWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
escapeCSVNumber(node.properties.level, 1),
|
||||
escapeCSVField(content),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Route': {
|
||||
const responseKeys = node.properties.responseKeys || [];
|
||||
// LadybugDB array literal inside a quoted CSV field: escapeCSVField wraps in "..."
|
||||
// and the array uses single-quoted elements
|
||||
const keysStr = `[${responseKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
const errorKeys = node.properties.errorKeys || [];
|
||||
const errorKeysStr = `[${errorKeys.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
const middleware = node.properties.middleware || [];
|
||||
const middlewareStr = `[${middleware.map((m: string) => `'${m.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
pending = routeWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVField(keysStr),
|
||||
escapeCSVField(errorKeysStr),
|
||||
escapeCSVField(middlewareStr),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
}
|
||||
case 'Tool':
|
||||
pending = toolWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
].join(','),
|
||||
);
|
||||
break;
|
||||
case 'BasicBlock':
|
||||
pending = basicBlockWriter.addRow(buildBasicBlockRow(node));
|
||||
break;
|
||||
default: {
|
||||
// Code element nodes (Function, Class, Interface, CodeElement)
|
||||
const writer = codeWriterMap[node.label];
|
||||
if (writer) {
|
||||
const content = await extractContent(node, contentCache);
|
||||
await mlWriter.addRow(
|
||||
pending = writer.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
node.properties.isExported ? 'true' : 'false',
|
||||
escapeCSVField(content),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
...(node.label === 'Property'
|
||||
? [escapeCSVField(node.properties.declaredType || '')]
|
||||
: []),
|
||||
].join(','),
|
||||
);
|
||||
} else {
|
||||
// Multi-language node types (Struct, Impl, Trait, Macro, etc.)
|
||||
const mlWriter = multiLangWriters.get(node.label);
|
||||
if (mlWriter) {
|
||||
const content = await extractContent(node, contentCache);
|
||||
pending = mlWriter.addRow(
|
||||
[
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
escapeCSVField(content),
|
||||
escapeCSVField(node.properties.description || ''),
|
||||
...(node.label === 'Property'
|
||||
? [escapeCSVField(node.properties.declaredType || '')]
|
||||
: []),
|
||||
].join(','),
|
||||
);
|
||||
} else {
|
||||
// Unknown label: not in codeWriterMap or multiLangWriters, so there
|
||||
// is no CSV table for it and it is intentionally NOT persisted —
|
||||
// `pending` stays undefined, so the loop awaits nothing. Made
|
||||
// explicit so a future node type isn't silently dropped here: wire
|
||||
// it into one of the writer maps above (or this branch).
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (pending) await pending;
|
||||
}
|
||||
|
||||
// Finish all node writers
|
||||
const allWriters = [
|
||||
fileWriter,
|
||||
folderWriter,
|
||||
functionWriter,
|
||||
classWriter,
|
||||
interfaceWriter,
|
||||
methodWriter,
|
||||
codeElemWriter,
|
||||
communityWriter,
|
||||
processWriter,
|
||||
sectionWriter,
|
||||
routeWriter,
|
||||
toolWriter,
|
||||
basicBlockWriter,
|
||||
...multiLangWriters.values(),
|
||||
];
|
||||
await Promise.all(allWriters.map((w) => w.finish()));
|
||||
|
||||
// --- Stream relationships directly to per-FROM→TO-label-pair files ---
|
||||
// (#2203 U2) Route every edge to its pair file in this single pass. The old
|
||||
// monolithic relations.csv — and its line-by-line re-read + per-edge regex
|
||||
// re-split in loadGraphToLbug — are gone, so the ~1M-edge set is written and
|
||||
// read once instead of twice. The router applies the SAME label-derivation +
|
||||
// validTables filter as the legacy splitRelCsvByLabelPair, so the per-pair
|
||||
// files are byte-identical (asserted by the differential test).
|
||||
const relRouter = new RelPairRouter(csvDir, REL_CSV_HEADER, new Set<string>(NODE_TABLES));
|
||||
try {
|
||||
for (const rel of orderedRelationships(graph, sortOutput)) {
|
||||
const pending = relRouter.route(rel.sourceId, rel.targetId, buildRelRow(rel));
|
||||
if (pending) await pending;
|
||||
}
|
||||
await relRouter.close();
|
||||
} catch (err) {
|
||||
relRouter.destroy();
|
||||
// Rethrow the real stream error (EMFILE / disk-full) rather than the generic
|
||||
// AbortError a pending drain-await rejects with — mirrors the retained
|
||||
// splitRelCsvByLabelPair's `throw streamError ?? err`.
|
||||
throw relRouter.lastError ?? err;
|
||||
}
|
||||
|
||||
// Build result map — only include tables that have rows
|
||||
const nodeFiles = new Map<NodeTableName, { csvPath: string; rows: number }>();
|
||||
const tableMap: [NodeTableName, BufferedCSVWriter][] = [
|
||||
['File', fileWriter],
|
||||
['Folder', folderWriter],
|
||||
['Function', functionWriter],
|
||||
['Class', classWriter],
|
||||
['Interface', interfaceWriter],
|
||||
['Method', methodWriter],
|
||||
['CodeElement', codeElemWriter],
|
||||
['Community', communityWriter],
|
||||
['Process', processWriter],
|
||||
['Section' as NodeTableName, sectionWriter],
|
||||
['Route' as NodeTableName, routeWriter],
|
||||
['Tool' as NodeTableName, toolWriter],
|
||||
['BasicBlock' as NodeTableName, basicBlockWriter],
|
||||
...Array.from(multiLangWriters.entries()).map(
|
||||
([name, w]) => [name as NodeTableName, w] as [NodeTableName, BufferedCSVWriter],
|
||||
),
|
||||
];
|
||||
for (const [name, writer] of tableMap) {
|
||||
if (writer.rows > 0) {
|
||||
nodeFiles.set(name, {
|
||||
csvPath: path.join(csvDir, `${name.toLowerCase()}.csv`),
|
||||
rows: writer.rows,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
nodeFiles,
|
||||
relsByPair: relRouter.byPair,
|
||||
relHeader: REL_CSV_HEADER,
|
||||
skippedRels: relRouter.skipped,
|
||||
totalValidRels: relRouter.total,
|
||||
};
|
||||
} finally {
|
||||
// Restore original process listener limit on every path (success or throw).
|
||||
process.setMaxListeners(prevMax);
|
||||
}
|
||||
|
||||
// Finish all node writers
|
||||
const allWriters = [
|
||||
fileWriter,
|
||||
folderWriter,
|
||||
functionWriter,
|
||||
classWriter,
|
||||
interfaceWriter,
|
||||
methodWriter,
|
||||
codeElemWriter,
|
||||
communityWriter,
|
||||
processWriter,
|
||||
sectionWriter,
|
||||
routeWriter,
|
||||
toolWriter,
|
||||
basicBlockWriter,
|
||||
...multiLangWriters.values(),
|
||||
];
|
||||
await Promise.all(allWriters.map((w) => w.finish()));
|
||||
|
||||
// --- Stream relationship CSV ---
|
||||
const relCsvPath = path.join(csvDir, 'relations.csv');
|
||||
const relWriter = new BufferedCSVWriter(relCsvPath, 'from,to,type,confidence,reason,step');
|
||||
for (const rel of orderedRelationships(graph, sortOutput)) {
|
||||
await relWriter.addRow(
|
||||
[
|
||||
escapeCSVField(rel.sourceId),
|
||||
escapeCSVField(rel.targetId),
|
||||
escapeCSVField(rel.type),
|
||||
escapeCSVNumber(rel.confidence, 1.0),
|
||||
escapeCSVField(rel.reason),
|
||||
escapeCSVNumber((rel as any).step, 0),
|
||||
].join(','),
|
||||
);
|
||||
}
|
||||
await relWriter.finish();
|
||||
|
||||
// Build result map — only include tables that have rows
|
||||
const nodeFiles = new Map<NodeTableName, { csvPath: string; rows: number }>();
|
||||
const tableMap: [NodeTableName, BufferedCSVWriter][] = [
|
||||
['File', fileWriter],
|
||||
['Folder', folderWriter],
|
||||
['Function', functionWriter],
|
||||
['Class', classWriter],
|
||||
['Interface', interfaceWriter],
|
||||
['Method', methodWriter],
|
||||
['CodeElement', codeElemWriter],
|
||||
['Community', communityWriter],
|
||||
['Process', processWriter],
|
||||
['Section' as NodeTableName, sectionWriter],
|
||||
['Route' as NodeTableName, routeWriter],
|
||||
['Tool' as NodeTableName, toolWriter],
|
||||
['BasicBlock' as NodeTableName, basicBlockWriter],
|
||||
...Array.from(multiLangWriters.entries()).map(
|
||||
([name, w]) => [name as NodeTableName, w] as [NodeTableName, BufferedCSVWriter],
|
||||
),
|
||||
];
|
||||
for (const [name, writer] of tableMap) {
|
||||
if (writer.rows > 0) {
|
||||
nodeFiles.set(name, {
|
||||
csvPath: path.join(csvDir, `${name.toLowerCase()}.csv`),
|
||||
rows: writer.rows,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Restore original process listener limit
|
||||
process.setMaxListeners(prevMax);
|
||||
|
||||
return { nodeFiles, relCsvPath, relRows: relWriter.rows };
|
||||
};
|
||||
|
|
|
|||
|
|
@ -4,8 +4,6 @@ import { createInterface } from 'readline';
|
|||
import { once } from 'events';
|
||||
import { finished } from 'stream/promises';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
import crypto from 'crypto';
|
||||
import lbug from '@ladybugdb/core';
|
||||
import { closeQueryResults } from './query-result-utils.js';
|
||||
import { KnowledgeGraph } from '../graph/types.js';
|
||||
|
|
@ -19,6 +17,8 @@ import {
|
|||
NodeTableName,
|
||||
} from './schema.js';
|
||||
import { streamAllCSVsToDisk } from './csv-generator.js';
|
||||
import type { PdgEmitManifest } from './pdg-emit-sink.js';
|
||||
import { getNodeLabel as deriveNodeLabel, type WriteStreamFactory } from './rel-pair-routing.js';
|
||||
import type { CachedEmbedding } from '../embeddings/types.js';
|
||||
import { extensionManager, type ExtensionEnsureOptions } from './extension-loader.js';
|
||||
import {
|
||||
|
|
@ -28,6 +28,7 @@ import {
|
|||
isWalCorruptionError,
|
||||
openLbugConnection,
|
||||
toNativeSafePath,
|
||||
resolveNativeSafeStorageDir,
|
||||
WAL_RECOVERY_SUGGESTION,
|
||||
waitForWindowsHandleRelease,
|
||||
type LbugConnectionHandle,
|
||||
|
|
@ -48,9 +49,9 @@ import { logger } from '../logger.js';
|
|||
// ---------------------------------------------------------------------------
|
||||
// Relationship CSV splitting — extracted for testability (PR #818)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Factory for creating WriteStreams — injectable for testing. */
|
||||
export type WriteStreamFactory = (filePath: string) => import('fs').WriteStream;
|
||||
// WriteStreamFactory is imported above from rel-pair-routing.ts (its canonical
|
||||
// home) for splitRelCsvByLabelPair's signature; no external code imports it from
|
||||
// here, so it is not re-exported.
|
||||
|
||||
/** Result of splitting the relationship CSV into per-label-pair files. */
|
||||
export interface RelCsvSplitResult {
|
||||
|
|
@ -64,6 +65,15 @@ export interface RelCsvSplitResult {
|
|||
/**
|
||||
* Split a relationship CSV into per-label-pair files on disk.
|
||||
*
|
||||
* @internal RETAINED AS A DIFFERENTIAL ORACLE. As of #2203 U2, production emit
|
||||
* routes relationships to per-pair files directly during the single pass (see
|
||||
* RelPairRouter in `rel-pair-routing.ts`), so this function has NO production
|
||||
* callers — it is kept ONLY so the byte-identity test in
|
||||
* `test/integration/csv-pipeline.test.ts` ("direct per-pair emit matches the
|
||||
* split oracle") can diff the direct-emit output against this proven path. Do
|
||||
* NOT delete it as dead code without also removing that test and accepting the
|
||||
* loss of the byte-identity guard (and likewise `test/unit/rel-csv-split.test.ts`).
|
||||
*
|
||||
* Streams the CSV line-by-line, routing each relationship to a file named
|
||||
* `rel_{fromLabel}_{toLabel}.csv`. Handles backpressure correctly: only one
|
||||
* drain listener per stream at a time, and readline resumes only when ALL
|
||||
|
|
@ -871,6 +881,15 @@ export const loadGraphToLbug = async (
|
|||
repoPath: string,
|
||||
storagePath: string,
|
||||
onProgress?: LbugProgressCallback,
|
||||
/**
|
||||
* Streamed PDG-emit manifest (#2202). When present (streaming was on, full
|
||||
* rebuild), the BasicBlock node CSV + per-pair PDG-edge CSVs it points at
|
||||
* were already flushed to disk during the emit loop; they are merged into the
|
||||
* COPY plan below so they load alongside the structural CSVs. When streaming
|
||||
* was on the in-memory `graph` holds zero BasicBlocks, so `streamAllCSVsToDisk`
|
||||
* emits none — the manifest is the sole source and there is no double-COPY.
|
||||
*/
|
||||
pdgEmitManifest?: PdgEmitManifest,
|
||||
) => {
|
||||
if (!conn) {
|
||||
throw new Error('LadybugDB not initialized. Call initLbug first.');
|
||||
|
|
@ -878,23 +897,57 @@ export const loadGraphToLbug = async (
|
|||
|
||||
const log = onProgress || (() => {});
|
||||
|
||||
let csvDir: string;
|
||||
if (process.platform === 'win32' && /[^\x00-\x7F]/.test(storagePath)) {
|
||||
const hash = crypto.createHash('sha256').update(storagePath).digest('hex').slice(0, 16);
|
||||
csvDir = toNativeSafePath(path.join(os.tmpdir(), `gitnexus-csv-${hash}`));
|
||||
} else {
|
||||
csvDir = path.join(storagePath, 'csv');
|
||||
}
|
||||
// ── #2203 persistence-path profiling ──────────────────────────────────
|
||||
// Mirrors the PROF_SCOPE_RESOLUTION pattern (scope-resolution/pipeline/
|
||||
// run.ts): zero-cost when off — process.hrtime.bigint() is only read under
|
||||
// PROF_LBUG_LOAD=1, and the summary is logged behind the same gate. Fills
|
||||
// the gap that the DB-persistence path is un-timed today (the analyze
|
||||
// "emit" number is the scope-resolution emit bucket, not this COPY path).
|
||||
const PROF = process.env.PROF_LBUG_LOAD === '1';
|
||||
const mark = (): bigint => (PROF ? process.hrtime.bigint() : 0n);
|
||||
const span = (a: bigint, b: bigint): string => (Number(b - a) / 1e6).toFixed(1);
|
||||
const tStart = mark();
|
||||
|
||||
const csvDir = resolveNativeSafeStorageDir(storagePath, 'csv');
|
||||
|
||||
log('Streaming CSVs to disk...');
|
||||
const csvResult = await streamAllCSVsToDisk(graph, repoPath, csvDir);
|
||||
|
||||
// Merge the streamed PDG-emit CSVs (#2202) into the COPY plan so the
|
||||
// BasicBlock node table + per-pair PDG edges (CFG / REACHING_DEF / CDG /
|
||||
// POST_DOMINATE / TAINTED / SANITIZES) load through the SAME node + per-pair
|
||||
// COPY loops as the structural CSVs. The graph held zero BasicBlocks when
|
||||
// streaming, so `streamAllCSVsToDisk` produced none of these — the manifest
|
||||
// is the sole source and there is no double-COPY. Absent ⇒ no-op.
|
||||
if (pdgEmitManifest) {
|
||||
for (const [table, meta] of pdgEmitManifest.nodeFiles) {
|
||||
// A collision means a BasicBlock leaked into the in-memory graph during a
|
||||
// streamed run (streamAllCSVsToDisk then emitted a structural basicblock.csv).
|
||||
// That is a streaming-invariant violation — fail loudly rather than
|
||||
// silently overwrite one CSV with the other and drop its rows (#2202 review #3).
|
||||
if (csvResult.nodeFiles.has(table)) {
|
||||
throw new Error(
|
||||
`Streaming PDG manifest collides with a structural node CSV for "${table}" — ` +
|
||||
`the in-memory graph should hold zero ${table} nodes when streaming. ` +
|
||||
`A ${table} node leaked into the graph during a streamed emit.`,
|
||||
);
|
||||
}
|
||||
csvResult.nodeFiles.set(table, meta);
|
||||
}
|
||||
for (const [pairKey, meta] of pdgEmitManifest.relsByPair) {
|
||||
if (csvResult.relsByPair.has(pairKey)) {
|
||||
throw new Error(
|
||||
`Streaming PDG manifest collides with a structural relationship CSV for pair ` +
|
||||
`"${pairKey}" — a PDG edge leaked into the in-memory graph during a streamed emit.`,
|
||||
);
|
||||
}
|
||||
csvResult.relsByPair.set(pairKey, meta);
|
||||
csvResult.totalValidRels += meta.rows;
|
||||
}
|
||||
}
|
||||
const tCsv = mark();
|
||||
|
||||
const validTables = new Set<string>(NODE_TABLES as readonly string[]);
|
||||
const getNodeLabel = (nodeId: string): string => {
|
||||
if (nodeId.startsWith('comm_')) return 'Community';
|
||||
if (nodeId.startsWith('proc_')) return 'Process';
|
||||
return nodeId.split(':')[0];
|
||||
};
|
||||
|
||||
// Bulk COPY all node CSVs (sequential — LadybugDB allows only one write txn at a time)
|
||||
const nodeFiles = [...csvResult.nodeFiles.entries()];
|
||||
|
|
@ -924,37 +977,32 @@ export const loadGraphToLbug = async (
|
|||
}
|
||||
}
|
||||
|
||||
// Bulk COPY relationships — split by FROM→TO label pair (LadybugDB requires it)
|
||||
const { relHeader, relsByPairMeta, pairWriteStreams, skippedRels, totalValidRels } =
|
||||
await splitRelCsvByLabelPair(csvResult.relCsvPath, csvDir, validTables, getNodeLabel);
|
||||
const tCopyNodes = mark();
|
||||
|
||||
// Close all per-pair write streams before COPY. `stream/promises.finished`
|
||||
// resolves on the stream's 'finish' event and rejects on 'error' — replaces
|
||||
// a hand-rolled promisification with the stdlib primitive.
|
||||
await Promise.all(
|
||||
Array.from(pairWriteStreams.values()).map(async (ws) => {
|
||||
ws.end();
|
||||
await finished(ws);
|
||||
}),
|
||||
);
|
||||
// Bulk COPY relationships. They were already routed to per-FROM→TO-label-pair
|
||||
// files during the emit pass (#2203 U2) — there is no monolithic relations.csv
|
||||
// to re-read/re-split here; we COPY each pair file directly.
|
||||
const { relsByPair, relHeader, skippedRels, totalValidRels } = csvResult;
|
||||
let tCopyRels = tCopyNodes;
|
||||
let tFallback = tCopyNodes;
|
||||
|
||||
const insertedRels = totalValidRels;
|
||||
const warnings: string[] = [];
|
||||
if (insertedRels > 0) {
|
||||
log(`Loading edges: ${insertedRels.toLocaleString()} across ${relsByPairMeta.size} types`);
|
||||
log(`Loading edges: ${insertedRels.toLocaleString()} across ${relsByPair.size} types`);
|
||||
|
||||
let pairIdx = 0;
|
||||
let failedPairEdges = 0;
|
||||
const failedPairCsvPaths = new Set<string>();
|
||||
|
||||
for (const [pairKey, { csvPath: pairCsvPath, rows }] of relsByPairMeta) {
|
||||
for (const [pairKey, { csvPath: pairCsvPath, rows }] of relsByPair) {
|
||||
pairIdx++;
|
||||
const [fromLabel, toLabel] = pairKey.split('|');
|
||||
const normalizedPath = normalizeCopyPath(pairCsvPath);
|
||||
const copyQuery = `COPY ${REL_TABLE_NAME} FROM "${normalizedPath}" (from="${fromLabel}", to="${toLabel}", HEADER=true, ESCAPE='"', DELIM=',', QUOTE='"', PARALLEL=false, auto_detect=false)`;
|
||||
|
||||
if (pairIdx % 5 === 0 || rows > 1000) {
|
||||
log(`Loading edges: ${pairIdx}/${relsByPairMeta.size} types (${fromLabel} -> ${toLabel})`);
|
||||
log(`Loading edges: ${pairIdx}/${relsByPair.size} types (${fromLabel} -> ${toLabel})`);
|
||||
}
|
||||
|
||||
try {
|
||||
|
|
@ -980,6 +1028,7 @@ export const loadGraphToLbug = async (
|
|||
} catch {}
|
||||
}
|
||||
}
|
||||
tCopyRels = mark();
|
||||
|
||||
if (failedPairCsvPaths.size > 0) {
|
||||
log(`Inserting ${failedPairEdges} edges individually (missing schema pairs)`);
|
||||
|
|
@ -999,15 +1048,14 @@ export const loadGraphToLbug = async (
|
|||
} catch {}
|
||||
}
|
||||
if (allLines.length > 1) {
|
||||
await fallbackRelationshipInserts(allLines, validTables, getNodeLabel);
|
||||
await fallbackRelationshipInserts(allLines, validTables, deriveNodeLabel);
|
||||
}
|
||||
}
|
||||
tFallback = mark();
|
||||
}
|
||||
|
||||
// Cleanup all CSVs
|
||||
try {
|
||||
await fs.unlink(csvResult.relCsvPath);
|
||||
} catch {}
|
||||
// Cleanup all CSVs (per-pair rel files are unlinked in the COPY loop above;
|
||||
// the remaining sweep below catches node CSVs + any leftover pair files).
|
||||
for (const [, { csvPath }] of csvResult.nodeFiles) {
|
||||
try {
|
||||
await fs.unlink(csvPath);
|
||||
|
|
@ -1025,6 +1073,18 @@ export const loadGraphToLbug = async (
|
|||
await fs.rmdir(csvDir);
|
||||
} catch {}
|
||||
|
||||
if (PROF) {
|
||||
const tEnd = mark();
|
||||
let totalNodeRows = 0;
|
||||
for (const [, { rows }] of csvResult.nodeFiles) totalNodeRows += rows;
|
||||
logger.warn(
|
||||
`[lbug-load prof] csv-emit=${span(tStart, tCsv)}ms ` +
|
||||
`copy-nodes=${span(tCsv, tCopyNodes)}ms copy-rels=${span(tCopyNodes, tCopyRels)}ms ` +
|
||||
`fallback=${span(tCopyRels, tFallback)}ms total=${span(tStart, tEnd)}ms ` +
|
||||
`(${totalNodeRows} nodes, ${insertedRels} rels)`,
|
||||
);
|
||||
}
|
||||
|
||||
return { success: true, insertedRels, skippedRels, warnings };
|
||||
};
|
||||
|
||||
|
|
|
|||
|
|
@ -189,6 +189,36 @@ export function toNativeSafePath(p: string): string {
|
|||
return p;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the on-disk CSV staging dir for `<storagePath>/<subdir>`, applying the
|
||||
* same ASCII-safe relocation `toNativeSafePath` enables: on Windows with a
|
||||
* non-ASCII storage path, LadybugDB's bulk COPY cannot open files under that
|
||||
* path, so the dir is relocated under `os.tmpdir()`. Shared by the structural
|
||||
* `csv/` dir and the streaming `pdg-csv/` dir (#2202) so the two can never
|
||||
* diverge on platform handling; the `gitnexus-<subdir>-` prefix keeps their tmp
|
||||
* locations distinct and recognizable.
|
||||
*
|
||||
* The relocated dir is created with `fs.mkdtempSync` (a unique, mode-0700,
|
||||
* guaranteed-not-pre-existing suffix) rather than a deterministic
|
||||
* `gitnexus-<subdir>-<hash>` name. A predictable name in the world-readable OS
|
||||
* temp dir is information-disclosure-prone and pre-plantable
|
||||
* (CWE-377/378 / CodeQL `js/insecure-temporary-file`); mkdtemp's random suffix
|
||||
* is the documented mitigation and is what reaches the streaming sink's
|
||||
* `fs.openSync`. The non-Windows / ASCII path stays a pure `path.join` (no dir
|
||||
* created) and is byte-identical to before.
|
||||
*/
|
||||
export function resolveNativeSafeStorageDir(storagePath: string, subdir: string): string {
|
||||
if (process.platform === 'win32' && NON_ASCII_RE.test(storagePath)) {
|
||||
// 8.3-shorten the tmpdir base first (a non-ASCII Windows *profile* path can
|
||||
// make os.tmpdir() itself non-ASCII), THEN mkdtemp so the returned path —
|
||||
// the one that flows into fs.openSync — is provably mkdtemp-sourced (random,
|
||||
// exclusive) and clears the insecure-temp-file dataflow.
|
||||
const base = toNativeSafePath(os.tmpdir());
|
||||
return fsSync.mkdtempSync(path.join(base, `gitnexus-${subdir}-`));
|
||||
}
|
||||
return path.join(storagePath, subdir);
|
||||
}
|
||||
|
||||
/**
|
||||
* Shared configuration for `@ladybugdb/core` `Database` construction.
|
||||
*
|
||||
|
|
|
|||
395
gitnexus/src/core/lbug/pdg-emit-sink.ts
Normal file
395
gitnexus/src/core/lbug/pdg-emit-sink.ts
Normal file
|
|
@ -0,0 +1,395 @@
|
|||
/**
|
||||
* Streaming PDG graph-emit sink (issue #2202).
|
||||
*
|
||||
* The PDG emit loop (`scope-resolution/pipeline/run.ts`, the `--pdg` block)
|
||||
* materializes BasicBlock nodes + intra-file PDG edges (CFG / REACHING_DEF /
|
||||
* CDG / POST_DOMINATE / TAINTED / SANITIZES) into the in-memory
|
||||
* `KnowledgeGraph`. At full-kernel scale that layer dominates peak RSS
|
||||
* (~7 GB at 511K BasicBlocks; ~100 GB extrapolated to the full kernel → OOM).
|
||||
*
|
||||
* `PdgEmitSink` is a write-routing façade over the real graph: the emit
|
||||
* functions are write-only and compute every edge endpoint by deterministic id
|
||||
* (audited — no read-back), so the sink can route BasicBlock node rows and PDG
|
||||
* edge rows straight to bounded CSV-on-disk writers and **never store them**.
|
||||
* The graph's resident size stops growing with the PDG layer → peak RSS becomes
|
||||
* O(chunk buffer), not O(graph). Everything else (structural nodes/edges, the
|
||||
* whole-program M4 TAINT_PATH edges) is delegated to the real graph unchanged.
|
||||
*
|
||||
* Why synchronous writers? The whole PDG emit (`runScopeResolution` and its
|
||||
* per-file loop) is synchronous — there is no `await` point to drain an async
|
||||
* stream, so a `BufferedCSVWriter` (Node `WriteStream`) would accumulate
|
||||
* unwritten chunks in process memory across millions of rows, defeating the RSS
|
||||
* bound. `fs.writeSync` goes straight to the OS; resident memory is bounded to
|
||||
* one `chunkRows` buffer. This mirrors the sync-shard pattern in
|
||||
* `storage/parsedfile-store.ts`.
|
||||
*
|
||||
* Byte-identity (issue acceptance): the sink reuses the SAME shared row
|
||||
* builders (`buildBasicBlockRow`, `buildRelRow`) and label derivation
|
||||
* (`getNodeLabel`) as `streamAllCSVsToDisk`, so the streamed CSV line SET is
|
||||
* identical to the whole-graph emit's, and the bulk COPY loads the same rows →
|
||||
* the persisted graph is SET-identical and DB-identical. The guarantee is
|
||||
* set-level, not byte-level on the CSV file: the sink streams rows in emit
|
||||
* order and does NOT re-sort them under `GITNEXUS_SORT_GRAPH_OUTPUT`, so a
|
||||
* streamed CSV file is not necessarily byte-for-byte equal to the sorted
|
||||
* whole-graph CSV — but the row set, and therefore the DB outcome, is. (The
|
||||
* streamed CSVs are deleted right after the COPY, so their on-disk byte order
|
||||
* is never observed.) Cross-pass dedup is done upstream, per FILE, in the emit
|
||||
* loop (`run.ts` skips a file whose PDG already streamed) rather than in the
|
||||
* sink, because a file can be PDG-emitted in more than one language pass (a
|
||||
* `.ts` module imported by a `.vue` SFC is emitted in both the TypeScript and
|
||||
* Vue context passes over the same worker-built CFG) and a sink-level per-id
|
||||
* dedup set would retain every id → O(total ids) memory, defeating the
|
||||
* O(chunk) RSS bound (#2202 review #1). The differential fingerprint test
|
||||
* (issue #2202 U6) and the Vue+TS cross-pass integration test guard the set.
|
||||
*/
|
||||
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import type { GraphNode, GraphRelationship, RelationshipType } from 'gitnexus-shared';
|
||||
import type { KnowledgeGraph } from '../graph/types.js';
|
||||
import {
|
||||
BASICBLOCK_CSV_HEADER,
|
||||
REL_CSV_HEADER,
|
||||
buildBasicBlockRow,
|
||||
buildRelRow,
|
||||
} from './csv-generator.js';
|
||||
import { getNodeLabel } from './rel-pair-routing.js';
|
||||
import { NODE_TABLES, type NodeTableName } from './schema.js';
|
||||
|
||||
/**
|
||||
* PDG edge types streamed per-file (all intra-block BasicBlock→BasicBlock).
|
||||
* `TAINT_PATH` is intentionally excluded — it is the whole-program M4 edge
|
||||
* (Function→Function), computed in a separate post-resolution phase over the
|
||||
* complete CALLS graph, and stays in the in-memory graph (it is small and is
|
||||
* persisted by the normal whole-graph emit).
|
||||
*/
|
||||
const PDG_EDGE_TYPES: ReadonlySet<RelationshipType> = new Set<RelationshipType>([
|
||||
'CFG',
|
||||
'REACHING_DEF',
|
||||
'CDG',
|
||||
'POST_DOMINATE',
|
||||
'TAINTED',
|
||||
'SANITIZES',
|
||||
]);
|
||||
|
||||
/** Default streamed-write buffer (rows). Matches the whole-graph emit's
|
||||
* `FLUSH_EVERY` order of magnitude; overridable via `GITNEXUS_PDG_EMIT_CHUNK_SIZE`. */
|
||||
export const DEFAULT_PDG_EMIT_CHUNK_ROWS = 500;
|
||||
|
||||
/**
|
||||
* Synchronous buffered CSV writer. Buffers up to `chunkRows` rows, then issues
|
||||
* one `fs.writeSync` straight to the OS (no in-process stream buffer). Header
|
||||
* is written into the buffer at construction and is NOT counted in `rows`
|
||||
* (matching `BufferedCSVWriter` semantics, so manifest row counts line up).
|
||||
*/
|
||||
class SyncCsvWriter {
|
||||
private fd: number;
|
||||
private buf: string[] = [];
|
||||
private readonly chunkRows: number;
|
||||
rows = 0;
|
||||
/**
|
||||
* First IO error this writer hit (a `fs.writeSync` short-write loop throwing
|
||||
* on e.g. disk-full). Once poisoned the writer refuses further rows and
|
||||
* skips its final flush; the sink surfaces it from {@link PdgEmitSink.finalize}
|
||||
* so a truncated CSV is never handed to the bulk COPY (#2202 review #4). A
|
||||
* streamed-write failure is an IO fault, not the CFG-logic error that the
|
||||
* emit loop's per-file try/catch is built to swallow — poisoning routes it
|
||||
* past that catch to a loud failure.
|
||||
*/
|
||||
poison: unknown | undefined = undefined;
|
||||
|
||||
constructor(
|
||||
readonly csvPath: string,
|
||||
header: string,
|
||||
chunkRows: number,
|
||||
) {
|
||||
// Guard a 0/negative buffer: the flush modulo would never fire and `buf`
|
||||
// would grow unbounded, defeating the whole point of streaming.
|
||||
this.chunkRows = Math.max(1, chunkRows);
|
||||
// Exclusive create (O_EXCL): the streamed-CSV dir is wiped + recreated fresh
|
||||
// by the PdgEmitSink constructor before any writer opens a file, so the path
|
||||
// never pre-exists — 'wx' both matches that invariant and refuses to follow
|
||||
// a pre-planted symlink at the path (CWE-377 / CodeQL js/insecure-temporary-file).
|
||||
this.fd = fs.openSync(csvPath, 'wx');
|
||||
this.buf.push(header);
|
||||
}
|
||||
|
||||
addRow(row: string): void {
|
||||
// A poisoned writer is dead — stop buffering so memory can't grow on a
|
||||
// writer whose fd is already in a bad state; finalize will report the fault.
|
||||
if (this.poison !== undefined) return;
|
||||
this.buf.push(row);
|
||||
this.rows++;
|
||||
// Flush on DATA-row count, not buffer length: the header occupies buf[0]
|
||||
// until the first flush, so a `buf.length >= chunkRows` test would fire one
|
||||
// row early on the first chunk. Counting rows makes every flush exactly
|
||||
// `chunkRows` rows.
|
||||
if (this.rows % this.chunkRows === 0) this.flushOrPoison();
|
||||
}
|
||||
|
||||
/** Flush, recording (and re-throwing) any IO error as poison. Re-throwing
|
||||
* lets the immediate caller log the per-file failure; the persisted `poison`
|
||||
* is the backstop that makes finalize fail loudly even when that throw is
|
||||
* swallowed by the emit loop's CFG try/catch. */
|
||||
private flushOrPoison(): void {
|
||||
try {
|
||||
this.flush();
|
||||
} catch (e) {
|
||||
this.poison ??= e;
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
private flush(): void {
|
||||
if (this.buf.length === 0) return;
|
||||
const data = Buffer.from(this.buf.join('\n') + '\n', 'utf8');
|
||||
// fs.writeSync can return a short byte count; loop until the whole buffer
|
||||
// lands so a partial write never truncates a CSV row mid-field.
|
||||
let offset = 0;
|
||||
while (offset < data.length) {
|
||||
offset += fs.writeSync(this.fd, data, offset, data.length - offset);
|
||||
}
|
||||
this.buf.length = 0;
|
||||
}
|
||||
|
||||
/** Flush remaining rows (unless already poisoned) and close the fd. Never
|
||||
* throws: a final-flush IO error is recorded as poison and the fd is still
|
||||
* closed, so a write error neither leaks an fd nor escapes here — the sink
|
||||
* reads {@link poison} after closing every writer and fails loudly then. */
|
||||
close(): void {
|
||||
try {
|
||||
if (this.poison === undefined) this.flush();
|
||||
} catch (e) {
|
||||
this.poison ??= e;
|
||||
} finally {
|
||||
try {
|
||||
fs.closeSync(this.fd);
|
||||
} catch {
|
||||
/* fd may already be invalid after an IO fault — nothing to recover */
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* COPY manifest produced by {@link PdgEmitSink.finalize}. Shaped to merge
|
||||
* directly into `StreamedCSVResult` so `loadGraphToLbug` COPYs the streamed
|
||||
* PDG CSVs through the same per-table / per-pair loops as the structural CSVs.
|
||||
* Paths are absolute, so persistence needs no dir recomputation.
|
||||
*/
|
||||
export interface PdgEmitManifest {
|
||||
/** Node-table CSVs (only `BasicBlock` today). */
|
||||
readonly nodeFiles: Map<NodeTableName, { csvPath: string; rows: number }>;
|
||||
/** pairKey (`From|To`) → per-pair edge CSV. */
|
||||
readonly relsByPair: Map<string, { csvPath: string; rows: number }>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Write-routing graph façade. Construct one per analyze run, thread it into the
|
||||
* per-language `runScopeResolution` calls in place of the real graph during the
|
||||
* `--pdg` emit, then {@link finalize} once after the last language.
|
||||
*/
|
||||
export class PdgEmitSink implements KnowledgeGraph {
|
||||
private readonly validTables: Set<string>;
|
||||
private bbWriter: SyncCsvWriter | undefined;
|
||||
/** pairKey (`From|To`) → writer. PDG edges are all `BasicBlock|BasicBlock`,
|
||||
* but the map keeps the sink general and the manifest pair-keyed. */
|
||||
private readonly relWriters = new Map<string, SyncCsvWriter>();
|
||||
private finalized = false;
|
||||
/**
|
||||
* First writer-construction failure (a `fs.openSync` throwing on e.g. EMFILE
|
||||
* — out of file descriptors). The failure happens inside the `SyncCsvWriter`
|
||||
* constructor before a writer object exists to carry poison, so it is held
|
||||
* here at the sink level and folded into the {@link finalize} error check.
|
||||
* Like an in-flight write fault, an open failure mid-emit would otherwise be
|
||||
* swallowed by the emit loop's per-file try/catch and silently drop the rest
|
||||
* of that file's rows (#2202 review #4/#6).
|
||||
*/
|
||||
private openFailure: unknown | undefined = undefined;
|
||||
// NOTE on dedup: the same file can be PDG-emitted in more than one language
|
||||
// pass (e.g. a `.ts` module imported by a `.vue` SFC is emitted in both the
|
||||
// TypeScript pass and the Vue context pass over the same worker-built
|
||||
// `cfgSideChannel`). The in-memory graph dedups that by id (first-writer-wins);
|
||||
// this sink does NOT — to keep peak memory O(write buffer) rather than
|
||||
// O(total ids), cross-pass dedup is done upstream, per FILE, in the emit loop
|
||||
// (`run.ts` skips a file whose PDG already streamed via `pdgEmittedFiles`).
|
||||
// The sink therefore receives each id exactly once and is a faithful
|
||||
// pass-through; it must not be fed duplicate ids.
|
||||
|
||||
constructor(
|
||||
private readonly real: KnowledgeGraph,
|
||||
private readonly pdgCsvDir: string,
|
||||
private readonly chunkRows: number = DEFAULT_PDG_EMIT_CHUNK_ROWS,
|
||||
) {
|
||||
this.validTables = new Set<string>(NODE_TABLES as readonly string[]);
|
||||
// Clear any streamed CSVs left by a previous (possibly crashed) run so a
|
||||
// later COPY never picks up stale rows.
|
||||
fs.rmSync(pdgCsvDir, { recursive: true, force: true });
|
||||
fs.mkdirSync(pdgCsvDir, { recursive: true });
|
||||
}
|
||||
|
||||
// ── routed writes ──────────────────────────────────────────────────────────
|
||||
|
||||
addNode(node: GraphNode): void {
|
||||
if (node.label === 'BasicBlock') {
|
||||
if (this.bbWriter === undefined) {
|
||||
try {
|
||||
this.bbWriter = new SyncCsvWriter(
|
||||
path.join(this.pdgCsvDir, 'basicblock.csv'),
|
||||
BASICBLOCK_CSV_HEADER,
|
||||
this.chunkRows,
|
||||
);
|
||||
} catch (e) {
|
||||
this.openFailure ??= e;
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
this.bbWriter.addRow(buildBasicBlockRow(node));
|
||||
return;
|
||||
}
|
||||
this.real.addNode(node);
|
||||
}
|
||||
|
||||
addRelationship(relationship: GraphRelationship): void {
|
||||
if (PDG_EDGE_TYPES.has(relationship.type)) {
|
||||
const fromLabel = getNodeLabel(relationship.sourceId);
|
||||
const toLabel = getNodeLabel(relationship.targetId);
|
||||
// Skip edges whose endpoint labels are not valid node tables — mirrors
|
||||
// `RelPairRouter` exactly so the streamed set matches the whole-graph set.
|
||||
if (!this.validTables.has(fromLabel) || !this.validTables.has(toLabel)) return;
|
||||
const pairKey = `${fromLabel}|${toLabel}`;
|
||||
let writer = this.relWriters.get(pairKey);
|
||||
if (writer === undefined) {
|
||||
try {
|
||||
writer = new SyncCsvWriter(
|
||||
path.join(this.pdgCsvDir, `rel_${fromLabel}_${toLabel}.csv`),
|
||||
REL_CSV_HEADER,
|
||||
this.chunkRows,
|
||||
);
|
||||
} catch (e) {
|
||||
this.openFailure ??= e;
|
||||
throw e;
|
||||
}
|
||||
this.relWriters.set(pairKey, writer);
|
||||
}
|
||||
writer.addRow(buildRelRow(relationship));
|
||||
return;
|
||||
}
|
||||
this.real.addRelationship(relationship);
|
||||
}
|
||||
|
||||
/** Flush + close every streamed writer and return the COPY manifest. Every
|
||||
* fd is closed even when a writer is poisoned (its `close` never throws); any
|
||||
* IO fault — an in-flight write that poisoned a writer, a final-flush failure,
|
||||
* or a writer-open failure (EMFILE) — is surfaced loudly here so a disk-full
|
||||
* / out-of-fds run never hands a truncated CSV to the bulk COPY (#2202 review
|
||||
* #4). The emit loop's per-file try/catch swallows the synchronous throw, so
|
||||
* this poison check is the backstop that turns a silent partial manifest into
|
||||
* a hard failure. */
|
||||
finalize(): PdgEmitManifest {
|
||||
if (this.finalized) throw new Error('PdgEmitSink.finalize() called twice');
|
||||
this.finalized = true;
|
||||
|
||||
const errors: unknown[] = [];
|
||||
if (this.openFailure !== undefined) errors.push(this.openFailure);
|
||||
|
||||
const nodeFiles = new Map<NodeTableName, { csvPath: string; rows: number }>();
|
||||
if (this.bbWriter !== undefined) {
|
||||
this.bbWriter.close();
|
||||
if (this.bbWriter.poison !== undefined) errors.push(this.bbWriter.poison);
|
||||
nodeFiles.set('BasicBlock' as NodeTableName, {
|
||||
csvPath: this.bbWriter.csvPath,
|
||||
rows: this.bbWriter.rows,
|
||||
});
|
||||
}
|
||||
|
||||
const relsByPair = new Map<string, { csvPath: string; rows: number }>();
|
||||
for (const [pairKey, writer] of this.relWriters) {
|
||||
writer.close();
|
||||
if (writer.poison !== undefined) errors.push(writer.poison);
|
||||
relsByPair.set(pairKey, { csvPath: writer.csvPath, rows: writer.rows });
|
||||
}
|
||||
|
||||
if (errors.length > 0) {
|
||||
const first = errors[0];
|
||||
throw new Error(
|
||||
`PdgEmitSink: ${errors.length} streamed CSV writer(s) hit an IO error ` +
|
||||
`(disk-full / out-of-fds) during the emit — the persisted graph would ` +
|
||||
`be truncated, so the run is failed rather than COPYing a partial CSV: ${
|
||||
first instanceof Error ? first.message : String(first)
|
||||
}`,
|
||||
);
|
||||
}
|
||||
|
||||
return { nodeFiles, relsByPair };
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort fd release for the error path — when a language pass throws
|
||||
* before {@link finalize} runs, the caller's `finally` calls this so the
|
||||
* BasicBlock + per-pair fds never leak. Idempotent with finalize via the
|
||||
* `finalized` flag; close errors are swallowed because the run is already
|
||||
* failing.
|
||||
*/
|
||||
close(): void {
|
||||
if (this.finalized) return;
|
||||
this.finalized = true;
|
||||
try {
|
||||
this.bbWriter?.close();
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
for (const writer of this.relWriters.values()) {
|
||||
try {
|
||||
writer.close();
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── delegated reads / non-PDG mutations ─────────────────────────────────────
|
||||
// The PDG emit functions never call these on the routed graph, but the
|
||||
// façade implements the full KnowledgeGraph surface so it is a drop-in for
|
||||
// the emit target and any non-PDG write transparently reaches the real graph.
|
||||
|
||||
get nodes(): GraphNode[] {
|
||||
return this.real.nodes;
|
||||
}
|
||||
get relationships(): GraphRelationship[] {
|
||||
return this.real.relationships;
|
||||
}
|
||||
iterNodes(): IterableIterator<GraphNode> {
|
||||
return this.real.iterNodes();
|
||||
}
|
||||
iterRelationships(): IterableIterator<GraphRelationship> {
|
||||
return this.real.iterRelationships();
|
||||
}
|
||||
iterRelationshipsByType(type: RelationshipType): IterableIterator<GraphRelationship> {
|
||||
return this.real.iterRelationshipsByType(type);
|
||||
}
|
||||
forEachNode(fn: (node: GraphNode) => void): void {
|
||||
this.real.forEachNode(fn);
|
||||
}
|
||||
forEachRelationship(fn: (rel: GraphRelationship) => void): void {
|
||||
this.real.forEachRelationship(fn);
|
||||
}
|
||||
getNode(id: string): GraphNode | undefined {
|
||||
return this.real.getNode(id);
|
||||
}
|
||||
get nodeCount(): number {
|
||||
return this.real.nodeCount;
|
||||
}
|
||||
get relationshipCount(): number {
|
||||
return this.real.relationshipCount;
|
||||
}
|
||||
removeNode(nodeId: string): boolean {
|
||||
return this.real.removeNode(nodeId);
|
||||
}
|
||||
removeNodesByFile(filePath: string): number {
|
||||
return this.real.removeNodesByFile(filePath);
|
||||
}
|
||||
removeRelationship(relationshipId: string): boolean {
|
||||
return this.real.removeRelationship(relationshipId);
|
||||
}
|
||||
}
|
||||
159
gitnexus/src/core/lbug/rel-pair-routing.ts
Normal file
159
gitnexus/src/core/lbug/rel-pair-routing.ts
Normal file
|
|
@ -0,0 +1,159 @@
|
|||
/**
|
||||
* Relationship per-label-pair routing (#2203 U2).
|
||||
*
|
||||
* LadybugDB's bulk `COPY` into the single `CodeRelation` rel table requires a
|
||||
* separate CSV per FROM→TO node-label pair (the `from=`/`to=` COPY params).
|
||||
* Historically the emit pass wrote one monolithic `relations.csv`, which
|
||||
* `loadGraphToLbug` then RE-READ line-by-line (regex per edge) and re-split
|
||||
* into per-pair files — writing and reading the entire ~1M-edge set twice.
|
||||
*
|
||||
* This router lets the single emit pass route each edge to its per-pair file
|
||||
* directly, so the monolithic write + re-read + per-edge regex are all gone.
|
||||
* The label-derivation + validTables filtering + per-pair-file format here match
|
||||
* the legacy `splitRelCsvByLabelPair`, so the per-pair files are byte-identical
|
||||
* for all quote-free ids — see the differential test in
|
||||
* `test/integration/csv-pipeline.test.ts`. ONE intentional divergence: this
|
||||
* router derives the label from the RAW id, while the oracle re-derives it via a
|
||||
* regex over the ESCAPED row — so for an id containing a `"` the router is the
|
||||
* more-correct path (it routes the edge to the right pair; the oracle's regex
|
||||
* mis-buckets or drops it). `splitRelCsvByLabelPair` is retained as the
|
||||
* differential oracle (the quote-in-id divergence is asserted explicitly).
|
||||
*
|
||||
* Backpressure: at most one stream is awaited at a time (the caller routes
|
||||
* edges sequentially and awaits the returned drain promise before the next),
|
||||
* mirroring the legacy split's `for await` invariant. The hot path (existing
|
||||
* pair, no backpressure) returns `void` — no microtask per edge.
|
||||
*/
|
||||
import path from 'path';
|
||||
import { createWriteStream, type WriteStream } from 'fs';
|
||||
import { once } from 'events';
|
||||
import { finished } from 'stream/promises';
|
||||
|
||||
/** Injectable for tests (backpressure/error simulation), mirroring split. */
|
||||
export type WriteStreamFactory = (filePath: string) => WriteStream;
|
||||
|
||||
/**
|
||||
* Derive a node's table label from its graph id. Matches the legacy
|
||||
* `getNodeLabel` that lived inline in `loadGraphToLbug`:
|
||||
* - `comm_*` → Community
|
||||
* - `proc_*` → Process
|
||||
* - otherwise the prefix before the first `:` (e.g. `Function:…` → Function)
|
||||
*/
|
||||
export const getNodeLabel = (nodeId: string): string => {
|
||||
if (nodeId.startsWith('comm_')) return 'Community';
|
||||
if (nodeId.startsWith('proc_')) return 'Process';
|
||||
return nodeId.split(':')[0];
|
||||
};
|
||||
|
||||
export interface RelPairMeta {
|
||||
csvPath: string;
|
||||
rows: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Routes already-escaped relationship CSV rows to per-FROM→TO-label-pair
|
||||
* files. Filters edges whose endpoint labels are not valid node tables
|
||||
* (counted as `skipped`), exactly as the legacy split did.
|
||||
*/
|
||||
export class RelPairRouter {
|
||||
/** pairKey (`From|To`) → { csvPath, rows } */
|
||||
readonly byPair = new Map<string, RelPairMeta>();
|
||||
private readonly streams = new Map<string, WriteStream>();
|
||||
skipped = 0;
|
||||
total = 0;
|
||||
|
||||
private streamError: Error | null = null;
|
||||
private readonly abort = new AbortController();
|
||||
|
||||
constructor(
|
||||
private readonly csvDir: string,
|
||||
private readonly header: string,
|
||||
private readonly validTables: Set<string>,
|
||||
private readonly wsFactory: WriteStreamFactory = (p) => createWriteStream(p, 'utf-8'),
|
||||
) {}
|
||||
|
||||
private markError = (err: Error): void => {
|
||||
this.streamError ??= err;
|
||||
this.abort.abort(err);
|
||||
};
|
||||
|
||||
/**
|
||||
* The first stream error observed, if any. Lets the emit caller rethrow the
|
||||
* real error (EMFILE / disk-full) instead of the generic `AbortError` that a
|
||||
* pending `once(ws,'drain',{signal})` rejects with when the abort fires —
|
||||
* mirroring the retained `splitRelCsvByLabelPair`'s `throw streamError ?? err`.
|
||||
*/
|
||||
get lastError(): Error | null {
|
||||
return this.streamError;
|
||||
}
|
||||
|
||||
/**
|
||||
* Route one already-escaped CSV row (no trailing newline) to its pair file.
|
||||
* Returns `void` on the synchronous hot path; a `Promise<void>` only when a
|
||||
* stream signals backpressure (or a new pair's header does) — the caller
|
||||
* awaits the promise before routing the next edge.
|
||||
*/
|
||||
route(fromId: string, toId: string, row: string): void | Promise<void> {
|
||||
if (this.streamError) throw this.streamError;
|
||||
|
||||
const fromLabel = getNodeLabel(fromId);
|
||||
const toLabel = getNodeLabel(toId);
|
||||
if (!this.validTables.has(fromLabel) || !this.validTables.has(toLabel)) {
|
||||
this.skipped++;
|
||||
return;
|
||||
}
|
||||
|
||||
const pairKey = `${fromLabel}|${toLabel}`;
|
||||
const ws = this.streams.get(pairKey);
|
||||
if (ws === undefined) {
|
||||
// First edge for this pair: open the stream, write header + row.
|
||||
return this.openAndWrite(pairKey, fromLabel, toLabel, row);
|
||||
}
|
||||
|
||||
this.byPair.get(pairKey)!.rows++;
|
||||
this.total++;
|
||||
if (!ws.write(row + '\n')) {
|
||||
return once(ws, 'drain', { signal: this.abort.signal }).then(() => undefined);
|
||||
}
|
||||
}
|
||||
|
||||
private async openAndWrite(
|
||||
pairKey: string,
|
||||
fromLabel: string,
|
||||
toLabel: string,
|
||||
row: string,
|
||||
): Promise<void> {
|
||||
const csvPath = path.join(this.csvDir, `rel_${fromLabel}_${toLabel}.csv`);
|
||||
const ws = this.wsFactory(csvPath);
|
||||
ws.on('error', this.markError);
|
||||
this.streams.set(pairKey, ws);
|
||||
this.byPair.set(pairKey, { csvPath, rows: 1 });
|
||||
this.total++;
|
||||
if (!ws.write(this.header + '\n')) {
|
||||
await once(ws, 'drain', { signal: this.abort.signal });
|
||||
}
|
||||
if (!ws.write(row + '\n')) {
|
||||
await once(ws, 'drain', { signal: this.abort.signal });
|
||||
}
|
||||
}
|
||||
|
||||
/** Flush + close every pair stream. Rejects if any stream errored. */
|
||||
async close(): Promise<void> {
|
||||
if (this.streamError) {
|
||||
this.destroy();
|
||||
throw this.streamError;
|
||||
}
|
||||
await Promise.all(
|
||||
Array.from(this.streams.values()).map(async (ws) => {
|
||||
ws.end();
|
||||
await finished(ws);
|
||||
}),
|
||||
);
|
||||
if (this.streamError) throw this.streamError;
|
||||
}
|
||||
|
||||
/** Tear down all streams (no flush) — used on the error path. */
|
||||
destroy(): void {
|
||||
for (const ws of this.streams.values()) ws.destroy();
|
||||
}
|
||||
}
|
||||
|
|
@ -61,6 +61,7 @@ import {
|
|||
} from './ingestion/taint/interproc-solver.js';
|
||||
import { DEFAULT_PDG_MAX_INTERPROC_EDGES } from './ingestion/taint/interproc-emit.js';
|
||||
import { taintModelVersion } from './ingestion/taint/typescript-model.js';
|
||||
import { parseTruthyEnv, parsePositiveIntEnv } from './ingestion/utils/env.js';
|
||||
import { computeFileHashes, diffFileHashes } from '../storage/file-hash.js';
|
||||
import {
|
||||
extractChangedSubgraph,
|
||||
|
|
@ -170,6 +171,19 @@ export interface AnalyzeOptions {
|
|||
pdgMaxInterprocFindings?: number;
|
||||
pdgMaxInterprocHops?: number;
|
||||
pdgMaxInterprocEdges?: number;
|
||||
/**
|
||||
* Stream the BasicBlock + intra-file PDG-edge layer to CSV-on-disk during the
|
||||
* emit loop instead of materializing it in the in-memory graph, bounding peak
|
||||
* RSS to O(chunk) for full-kernel-scale repos (#2202). Only engages on a full
|
||||
* rebuild — `resolveStreamPdgEmit` additionally requires `force === true`
|
||||
* (the pre-pipeline guarantee of a full rebuild). May also be enabled via
|
||||
* `GITNEXUS_STREAM_PDG_EMIT`. Memory-only; byte-identical output; not stamped
|
||||
* into `RepoMeta.pdg`. */
|
||||
streamPdgEmit?: boolean;
|
||||
/** Streamed PDG-emit write buffer (rows). `undefined` ⇒
|
||||
* `DEFAULT_PDG_EMIT_CHUNK_ROWS`. May also be set via
|
||||
* `GITNEXUS_PDG_EMIT_CHUNK_SIZE`. Memory-only (#2202). */
|
||||
pdgEmitChunkSize?: number;
|
||||
/**
|
||||
* Default branch threaded into generated AGENTS.md / CLAUDE.md so the
|
||||
* regression-compare example uses the configured branch instead of a
|
||||
|
|
@ -426,6 +440,48 @@ export const resolvePdgConfig = (options: PdgOptions): RepoMeta['pdg'] =>
|
|||
}
|
||||
: undefined;
|
||||
|
||||
/**
|
||||
* Whether streaming/chunked PDG graph emit (#2202) engages this run.
|
||||
*
|
||||
* Streaming flushes the BasicBlock + intra-file PDG-edge layer to CSV-on-disk
|
||||
* during the emit loop and never lands it in the in-memory graph, bounding peak
|
||||
* RSS to O(chunk). It is sound ONLY on a full rebuild: the incremental
|
||||
* writeback (`extractChangedSubgraph`) reads BasicBlock nodes back out of the
|
||||
* in-memory graph, which streaming has already offloaded. `force === true` is
|
||||
* the pre-pipeline guarantee of a full rebuild — `isIncremental` has
|
||||
* `!force` as a necessary condition — so gating on it avoids the deliberately
|
||||
* absent pre-pipeline incremental prediction (see the `isIncremental` note).
|
||||
*
|
||||
* Requires `pdg === true` (nothing to stream otherwise). Enabled by either the
|
||||
* explicit `streamPdgEmit` option or the `GITNEXUS_STREAM_PDG_EMIT` env toggle.
|
||||
* Memory-only — NOT part of {@link resolvePdgConfig}, so toggling it never
|
||||
* trips `pdgModeMismatch`. Read every call (not memoized) so `vi.stubEnv`
|
||||
* works in tests. Pure + exported for testing.
|
||||
*/
|
||||
export const resolveStreamPdgEmit = (options: {
|
||||
pdg?: boolean;
|
||||
force?: boolean;
|
||||
streamPdgEmit?: boolean;
|
||||
}): boolean =>
|
||||
options.pdg === true &&
|
||||
options.force === true &&
|
||||
(options.streamPdgEmit === true || parseTruthyEnv(process.env.GITNEXUS_STREAM_PDG_EMIT));
|
||||
|
||||
/**
|
||||
* Resolve the streamed PDG-emit write-buffer size (#2202). Explicit option wins
|
||||
* over `GITNEXUS_PDG_EMIT_CHUNK_SIZE`; `undefined` ⇒ the sink's
|
||||
* `DEFAULT_PDG_EMIT_CHUNK_ROWS`. Memory-only; does not affect emitted bytes.
|
||||
*/
|
||||
export const resolvePdgEmitChunkSize = (options: {
|
||||
pdgEmitChunkSize?: number;
|
||||
}): number | undefined => {
|
||||
// Only honor a positive-integer explicit option; `0`/negative is NOT nullish
|
||||
// so `?? env` would pass it through and make the sink flush every row.
|
||||
const opt = options.pdgEmitChunkSize;
|
||||
if (opt !== undefined && Number.isInteger(opt) && opt > 0) return opt;
|
||||
return parsePositiveIntEnv(process.env.GITNEXUS_PDG_EMIT_CHUNK_SIZE);
|
||||
};
|
||||
|
||||
/**
|
||||
* Whether the requested `--pdg` configuration differs from the one the
|
||||
* existing index's DB rows were built under (#2099 F1). An absent recorded
|
||||
|
|
@ -828,6 +884,11 @@ export async function runFullAnalysis(
|
|||
pdgMaxInterprocFindings: options.pdgMaxInterprocFindings,
|
||||
pdgMaxInterprocHops: options.pdgMaxInterprocHops,
|
||||
pdgMaxInterprocEdges: options.pdgMaxInterprocEdges,
|
||||
// Streaming/chunked PDG emit (#2202) — gated to full-rebuild runs
|
||||
// (force === true) so the incremental writeback never reads back an
|
||||
// offloaded BasicBlock layer. Memory-only; byte-identical output.
|
||||
streamPdgEmit: resolveStreamPdgEmit(options),
|
||||
pdgEmitChunkSize: resolvePdgEmitChunkSize(options),
|
||||
fetchWrappers: options.fetchWrappers,
|
||||
},
|
||||
);
|
||||
|
|
@ -1069,11 +1130,21 @@ export async function runFullAnalysis(
|
|||
});
|
||||
} else {
|
||||
// ── Full rebuild ───────────────────────────────────────────────
|
||||
await loadGraphToLbug(pipelineResult.graph, pipelineResult.repoPath, storagePath, (msg) => {
|
||||
lbugMsgCount++;
|
||||
const pct = Math.min(84, 60 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 24));
|
||||
progress('lbug', pct, msg);
|
||||
});
|
||||
// Pass the streamed PDG-emit manifest (#2202) so the BasicBlock layer that
|
||||
// was flushed to CSV during the emit loop is COPY'd alongside the
|
||||
// structural CSVs. Only ever set on a full rebuild (streaming is
|
||||
// force-gated), so the incremental branch above never carries it.
|
||||
await loadGraphToLbug(
|
||||
pipelineResult.graph,
|
||||
pipelineResult.repoPath,
|
||||
storagePath,
|
||||
(msg) => {
|
||||
lbugMsgCount++;
|
||||
const pct = Math.min(84, 60 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 24));
|
||||
progress('lbug', pct, msg);
|
||||
},
|
||||
pipelineResult.pdgEmitManifest,
|
||||
);
|
||||
}
|
||||
|
||||
// ── Phase 3: FTS (85–90%) ─────────────────────────────────────────
|
||||
|
|
|
|||
|
|
@ -33,7 +33,13 @@ import { mountMCPEndpoints } from './mcp-http.js';
|
|||
import { fileURLToPath } from 'url';
|
||||
import { JobManager } from './analyze-job.js';
|
||||
import { assertString, escapeRegExp, BadRequestError, createRouteLimiter } from './validation.js';
|
||||
import { extractRepoName, getCloneDir, cloneOrPull } from './git-clone.js';
|
||||
import {
|
||||
extractRepoName,
|
||||
getCloneDir,
|
||||
cloneOrPull,
|
||||
warnIfInsecureAzureConfig,
|
||||
GITHUB_TOKEN_HOSTS,
|
||||
} from './git-clone.js';
|
||||
import { createAnalyzeUploadHandler } from './analyze-upload.js';
|
||||
import { createLocalhostOriginGuard, normalizeBoundHost } from './middleware.js';
|
||||
import { createLaunchAnalysisWorker } from './analyze-launch.js';
|
||||
|
|
@ -673,7 +679,47 @@ export const handleQueryRequest = async (
|
|||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Validate the optional `token` field of POST /api/analyze. Returns an
|
||||
* { status, error } to send, or null when the token is absent or valid.
|
||||
*
|
||||
* The token is a GitHub PAT: charset-restricted (blocks CRLF header
|
||||
* smuggling), length-bounded (1–256), and bound to github.com using the SAME
|
||||
* GITHUB_TOKEN_HOSTS allowlist + hostname parse as resolveGitCredential, so a
|
||||
* token the API accepts is exactly the one buildGitEnv will inject — and one
|
||||
* it rejects is never sent off github.com.
|
||||
*
|
||||
* Exported for unit tests (the route validation is otherwise only reachable
|
||||
* by booting the server).
|
||||
*/
|
||||
export function validateAnalyzeToken(
|
||||
repoToken: unknown,
|
||||
repoUrl: unknown,
|
||||
): { status: number; error: string } | null {
|
||||
if (repoToken === undefined) return null;
|
||||
if (typeof repoToken !== 'string') return { status: 400, error: '"token" must be a string' };
|
||||
if (repoToken.length === 0 || repoToken.length > 256)
|
||||
return { status: 400, error: '"token" length must be between 1 and 256' };
|
||||
if (!/^[A-Za-z0-9._~+/=-]+$/.test(repoToken))
|
||||
return { status: 400, error: '"token" contains invalid characters' };
|
||||
if (!repoUrl || typeof repoUrl !== 'string')
|
||||
return { status: 400, error: '"token" requires "url"' };
|
||||
let tokenHost: string;
|
||||
try {
|
||||
tokenHost = new URL(repoUrl).hostname.toLowerCase();
|
||||
} catch {
|
||||
return { status: 400, error: '"url" must be a valid URL when "token" is provided' };
|
||||
}
|
||||
if (!GITHUB_TOKEN_HOSTS.has(tokenHost))
|
||||
return { status: 400, error: '"token" is only supported for github.com URLs' };
|
||||
return null;
|
||||
}
|
||||
|
||||
export const createServer = async (port: number, host: string = '127.0.0.1') => {
|
||||
// Surface a cleartext Azure DevOps PAT config at boot (operators rarely
|
||||
// read per-request logs). Warn-only — http:// self-hosted stays supported.
|
||||
warnIfInsecureAzureConfig();
|
||||
|
||||
const app = express();
|
||||
app.disable('x-powered-by');
|
||||
|
||||
|
|
@ -1461,7 +1507,14 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
|
|||
requireLocalhostOrigin,
|
||||
async (req, res) => {
|
||||
try {
|
||||
const { url: repoUrl, path: repoLocalPath, force, embeddings, dropEmbeddings } = req.body;
|
||||
const {
|
||||
url: repoUrl,
|
||||
path: repoLocalPath,
|
||||
force,
|
||||
embeddings,
|
||||
dropEmbeddings,
|
||||
token: repoToken,
|
||||
} = req.body;
|
||||
|
||||
// Input type validation
|
||||
if (repoUrl !== undefined && typeof repoUrl !== 'string') {
|
||||
|
|
@ -1478,6 +1531,14 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
|
|||
return;
|
||||
}
|
||||
|
||||
// Token: optional, restricted charset to prevent header smuggling
|
||||
// (CRLF), bound length, and bound to github.com (see validateAnalyzeToken).
|
||||
const tokenError = validateAnalyzeToken(repoToken, repoUrl);
|
||||
if (tokenError) {
|
||||
res.status(tokenError.status).json({ error: tokenError.error });
|
||||
return;
|
||||
}
|
||||
|
||||
// Path validation. The previous `normalize !== resolve` guard was inert
|
||||
// (both collapse `..` identically) and only false-rejected trailing
|
||||
// slashes, so it is dropped. Analyzing a local path the operator names
|
||||
|
|
@ -1496,9 +1557,19 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
|
|||
|
||||
const job = jobManager.createJob({ repoUrl, repoPath: repoLocalPath });
|
||||
|
||||
// If job was already running (dedup), just return its id
|
||||
// If job was already running (dedup), just return its id. The token is
|
||||
// not part of the dedup identity and is never stored on the job, so a
|
||||
// token on THIS request had no effect — the existing job already
|
||||
// cloned (or is cloning) with whatever credentials its originating
|
||||
// request supplied. Surface `tokenIgnored` so an authenticated caller
|
||||
// isn't misled into thinking their PAT took effect on a reused job.
|
||||
if (job.status !== 'queued') {
|
||||
res.status(202).json({ jobId: job.id, status: job.status });
|
||||
const body: { jobId: string; status: string; tokenIgnored?: boolean } = {
|
||||
jobId: job.id,
|
||||
status: job.status,
|
||||
};
|
||||
if (repoToken !== undefined) body.tokenIgnored = true;
|
||||
res.status(202).json(body);
|
||||
return;
|
||||
}
|
||||
|
||||
|
|
@ -1520,11 +1591,16 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
|
|||
progress: { phase: 'cloning', percent: 0, message: `Cloning ${repoUrl}...` },
|
||||
});
|
||||
|
||||
await cloneOrPull(repoUrl, targetPath, (progress) => {
|
||||
jobManager.updateJob(job.id, {
|
||||
progress: { phase: progress.phase, percent: 5, message: progress.message },
|
||||
});
|
||||
});
|
||||
await cloneOrPull(
|
||||
repoUrl,
|
||||
targetPath,
|
||||
(progress) => {
|
||||
jobManager.updateJob(job.id, {
|
||||
progress: { phase: progress.phase, percent: 5, message: progress.message },
|
||||
});
|
||||
},
|
||||
repoToken ? { token: repoToken } : undefined,
|
||||
);
|
||||
}
|
||||
|
||||
if (!targetPath) {
|
||||
|
|
|
|||
|
|
@ -236,6 +236,61 @@ export interface CloneProgress {
|
|||
*
|
||||
* Exported so the separator placement is testable without mocking spawn.
|
||||
*/
|
||||
/**
|
||||
* Detect Azure DevOps URLs — both self-hosted (via AZURE_DEVOPS_URL env)
|
||||
* and cloud (dev.azure.com / *.visualstudio.com).
|
||||
*
|
||||
* Self-hosted Azure DevOps Server instances use arbitrary hostnames
|
||||
* (e.g. `http://tfs.corp.example/Collection/Project/_git/Repo`), so the
|
||||
* function checks `AZURE_DEVOPS_URL` first. Cloud addresses are a
|
||||
* hardcoded fallback so PAT injection works out-of-the-box for
|
||||
* dev.azure.com without extra configuration.
|
||||
*/
|
||||
export function isAzureDevOpsUrl(url: string): boolean {
|
||||
try {
|
||||
// Strip a single trailing dot: `dev.azure.com.` is a valid absolute FQDN
|
||||
// that resolves to the same host, so it must match too.
|
||||
const host = new URL(url).hostname.toLowerCase().replace(/\.$/, '');
|
||||
|
||||
// Self-hosted: match against the configured base URL.
|
||||
const configuredBase = process.env.AZURE_DEVOPS_URL;
|
||||
if (configuredBase) {
|
||||
try {
|
||||
const baseHost = new URL(configuredBase).hostname.toLowerCase().replace(/\.$/, '');
|
||||
if (host === baseHost) return true;
|
||||
} catch {
|
||||
/* invalid AZURE_DEVOPS_URL — fall through to cloud check */
|
||||
}
|
||||
}
|
||||
|
||||
// Cloud fallback.
|
||||
return host === 'dev.azure.com' || host.endsWith('.visualstudio.com');
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* One-time startup warning when AZURE_DEVOPS_URL is configured over cleartext
|
||||
* http:// — the Azure DevOps PAT would then be sent unencrypted on every
|
||||
* clone. Self-hosted instances that only serve http are still supported (we
|
||||
* do not refuse), but operators rarely read request-time logs, so surface it
|
||||
* at boot too. Call once from server startup.
|
||||
*/
|
||||
export function warnIfInsecureAzureConfig(): void {
|
||||
const base = process.env.AZURE_DEVOPS_URL;
|
||||
if (!base) return;
|
||||
try {
|
||||
if (new URL(base).protocol === 'http:') {
|
||||
logger.warn(
|
||||
'AZURE_DEVOPS_URL is configured over cleartext http:// — the Azure DevOps PAT will be sent unencrypted. Prefer https:// where your instance supports it.',
|
||||
);
|
||||
}
|
||||
} catch {
|
||||
/* invalid AZURE_DEVOPS_URL — isAzureDevOpsUrl already tolerates this */
|
||||
}
|
||||
}
|
||||
|
||||
export function buildCloneArgs(url: string, targetDir: string): string[] {
|
||||
return ['clone', '--depth', '1', '--', url, targetDir];
|
||||
}
|
||||
|
|
@ -379,6 +434,7 @@ export async function cloneOrPull(
|
|||
url: string,
|
||||
targetDir: string,
|
||||
onProgress?: (progress: CloneProgress) => void,
|
||||
options?: { token?: string },
|
||||
): Promise<string> {
|
||||
// Containment barrier — inline with the canonical path.relative idiom so
|
||||
// CodeQL recognizes the sanitizer at every following filesystem and
|
||||
|
|
@ -413,29 +469,176 @@ export async function cloneOrPull(
|
|||
// whatever remote the dir was originally cloned from.
|
||||
await assertRemoteMatchesRequestedUrl(safeTarget, url);
|
||||
onProgress?.({ phase: 'pulling', message: 'Pulling latest changes...' });
|
||||
await runGit(['pull', '--ff-only'], safeTarget);
|
||||
await runGit(['pull', '--ff-only'], safeTarget, { token: options?.token, url });
|
||||
} else {
|
||||
await fs.mkdir(path.dirname(safeTarget), { recursive: true });
|
||||
onProgress?.({ phase: 'cloning', message: `Cloning ${url}...` });
|
||||
await runGit(buildCloneArgs(url, safeTarget));
|
||||
await runGit(buildCloneArgs(url, safeTarget), undefined, { token: options?.token, url });
|
||||
}
|
||||
|
||||
return safeTarget;
|
||||
}
|
||||
|
||||
function runGit(args: string[], cwd?: string): Promise<void> {
|
||||
/**
|
||||
* Hosts the per-request GitHub PAT may be sent to. Exported so the
|
||||
* /api/analyze boundary check and this injection-site check share one
|
||||
* allowlist (they must agree, or a token accepted by the API could be
|
||||
* silently dropped — or worse — at injection).
|
||||
*/
|
||||
export const GITHUB_TOKEN_HOSTS: ReadonlySet<string> = new Set(['github.com', 'www.github.com']);
|
||||
|
||||
/**
|
||||
* Resolve at most ONE git credential for a clone/pull, by server-side policy
|
||||
* keyed on the clone host against a fixed allowlist (never a free-form user
|
||||
* toggle):
|
||||
* 1. a per-request GitHub PAT — only for hosts in GITHUB_TOKEN_HOSTS;
|
||||
* 2. else the server's AZURE_DEVOPS_PAT — only for Azure DevOps hosts;
|
||||
* 3. else none.
|
||||
* The two host sets are disjoint, so at most one credential ever applies; the
|
||||
* GitHub token taking precedence is deterministic for the pathological case
|
||||
* where AZURE_DEVOPS_URL is itself configured to a github.com host. Returns
|
||||
* the base64 of the Basic-auth `user:secret` pair, or undefined.
|
||||
*
|
||||
* Security note (re CodeQL js/user-controlled-bypass): the clone URL is
|
||||
* user-controlled and selects WHICH credential applies, but it cannot
|
||||
* redirect a credential to an arbitrary host — the host is matched against
|
||||
* fixed server-side allowlists (GITHUB_TOKEN_HOSTS, isAzureDevOpsUrl's
|
||||
* dev.azure.com/*.visualstudio.com/configured AZURE_DEVOPS_URL), and the
|
||||
* emitted header is host-scoped (buildExtraHeaderKey). A URL outside the
|
||||
* allowlists yields no credential. The selection is therefore server-policy,
|
||||
* not a bypass the user can steer.
|
||||
*/
|
||||
function resolveGitCredential(options?: { token?: string; url?: string }): string | undefined {
|
||||
const url = options?.url;
|
||||
if (!url) return undefined;
|
||||
|
||||
let host: string;
|
||||
try {
|
||||
host = new URL(url).hostname.toLowerCase();
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
// 1. Per-request GitHub PAT — github.com only (mirrors the /api/analyze
|
||||
// host-bind so the user's token is never sent off github.com).
|
||||
if (options.token && GITHUB_TOKEN_HOSTS.has(host)) {
|
||||
return Buffer.from(`x-access-token:${options.token}`).toString('base64');
|
||||
}
|
||||
|
||||
// 2. Server-configured Azure DevOps PAT — Azure hosts only.
|
||||
const azurePat = process.env.AZURE_DEVOPS_PAT;
|
||||
if (azurePat && isAzureDevOpsUrl(url)) {
|
||||
return Buffer.from(`:${azurePat}`).toString('base64');
|
||||
}
|
||||
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the host-scoped git config key `http.<origin+path>.extraHeader` from
|
||||
* the raw clone URL, so the Authorization header is attached only to the
|
||||
* intended origin (and its clone sub-requests like /info/refs), never a
|
||||
* redirect target. Derived from the SAME raw URL git clones from — not the
|
||||
* normalize-for-compare form, which strips `.git` and would desync the key
|
||||
* from the wire URL and silently disable the header. Userinfo/query/fragment
|
||||
* are dropped (not part of git's URL match) and control characters stripped
|
||||
* (git rejects a newline in a config key outright).
|
||||
*/
|
||||
function buildExtraHeaderKey(url: string): string | undefined {
|
||||
let scoped: string;
|
||||
try {
|
||||
const u = new URL(url);
|
||||
u.username = '';
|
||||
u.password = '';
|
||||
u.search = '';
|
||||
u.hash = '';
|
||||
scoped = `${u.protocol}//${u.host}${u.pathname}`;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
scoped = scoped.replace(/[\r\n\0]/g, '');
|
||||
return `http.${scoped}.extraHeader`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Warn (do not block) when a credential is about to be sent over cleartext
|
||||
* http://. Base64 is encoding, not encryption, so an on-path observer can
|
||||
* read the PAT. We keep http:// working for self-hosted Azure DevOps Server.
|
||||
*/
|
||||
function warnIfCleartextCredential(url?: string): void {
|
||||
if (!url) return;
|
||||
try {
|
||||
const u = new URL(url);
|
||||
if (u.protocol === 'http:') {
|
||||
logger.warn(
|
||||
`Sending a git credential over cleartext http:// (${u.host}) — base64 is not encryption. Prefer https:// where the host supports it.`,
|
||||
);
|
||||
}
|
||||
} catch {
|
||||
/* resolver already validated the URL */
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the spawn env for `git`. Suppresses credential prompts and, when a
|
||||
* credential resolves (see resolveGitCredential), injects a single
|
||||
* host-scoped Authorization header via the `GIT_CONFIG_*` env protocol
|
||||
* (git ≥2.31) so credentials never appear in argv or the URL. Appends after
|
||||
* any existing `GIT_CONFIG_COUNT` rather than overwriting it. Exported for
|
||||
* unit tests.
|
||||
*/
|
||||
export function buildGitEnv(
|
||||
baseEnv: NodeJS.ProcessEnv,
|
||||
options?: { token?: string; url?: string },
|
||||
): NodeJS.ProcessEnv {
|
||||
const env: NodeJS.ProcessEnv = {
|
||||
...baseEnv,
|
||||
// Prevent git from prompting for credentials (hangs the process)
|
||||
GIT_TERMINAL_PROMPT: '0',
|
||||
// Ensure no credential helper tries to open a GUI prompt
|
||||
GIT_ASKPASS: process.platform === 'win32' ? 'echo' : '/bin/true',
|
||||
// Scrub git's HTTP/transport trace vars: if inherited from the parent
|
||||
// process they dump every request header — including the injected
|
||||
// Authorization header — to stderr, which runGit captures and logs.
|
||||
// `undefined` makes child_process omit the key from the child env.
|
||||
GIT_TRACE: undefined,
|
||||
GIT_TRACE_CURL: undefined,
|
||||
GIT_TRACE_PACKET: undefined,
|
||||
GIT_CURL_VERBOSE: undefined,
|
||||
};
|
||||
|
||||
const credential = resolveGitCredential(options);
|
||||
const key = options?.url ? buildExtraHeaderKey(options.url) : undefined;
|
||||
if (credential && key) {
|
||||
// Append after any GIT_CONFIG_* the operator already set, so we never
|
||||
// clobber their git config (e.g. an enforced http.sslVerify).
|
||||
const existing = Number.parseInt(env.GIT_CONFIG_COUNT ?? '', 10);
|
||||
const base = Number.isInteger(existing) && existing > 0 ? existing : 0;
|
||||
env.GIT_CONFIG_COUNT = String(base + 1);
|
||||
env[`GIT_CONFIG_KEY_${base}`] = key;
|
||||
env[`GIT_CONFIG_VALUE_${base}`] = `Authorization: Basic ${credential}`;
|
||||
warnIfCleartextCredential(options?.url);
|
||||
}
|
||||
|
||||
return env;
|
||||
}
|
||||
|
||||
// `options` carries the inputs the credential resolver needs: a per-request
|
||||
// GitHub `token` and the clone `url`. buildGitEnv injects at most ONE
|
||||
// host-scoped Authorization header (GitHub PAT for github.com, else the
|
||||
// server's AZURE_DEVOPS_PAT for Azure hosts) via the GIT_CONFIG_* protocol —
|
||||
// never in argv. See resolveGitCredential / buildExtraHeaderKey.
|
||||
function runGit(
|
||||
args: string[],
|
||||
cwd?: string,
|
||||
options?: { token?: string; url?: string },
|
||||
): Promise<void> {
|
||||
return new Promise((resolve, reject) => {
|
||||
const proc = spawn('git', args, {
|
||||
cwd,
|
||||
stdio: ['ignore', 'pipe', 'pipe'],
|
||||
windowsHide: true,
|
||||
env: {
|
||||
...process.env,
|
||||
// Prevent git from prompting for credentials (hangs the process)
|
||||
GIT_TERMINAL_PROMPT: '0',
|
||||
// Ensure no credential helper tries to open a GUI prompt
|
||||
GIT_ASKPASS: process.platform === 'win32' ? 'echo' : '/bin/true',
|
||||
},
|
||||
env: buildGitEnv(process.env, options),
|
||||
});
|
||||
|
||||
let stderr = '';
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@ import type { KnowledgeGraph } from '../core/graph/types.js';
|
|||
import { CommunityDetectionResult } from '../core/ingestion/community-processor.js';
|
||||
import { ProcessDetectionResult } from '../core/ingestion/process-processor.js';
|
||||
import type { ResolutionOutcome } from '../core/ingestion/scope-resolution/resolution-outcome.js';
|
||||
import type { PdgEmitManifest } from '../core/lbug/pdg-emit-sink.js';
|
||||
|
||||
// CLI-specific: in-memory result with graph + detection results
|
||||
export interface PipelineResult {
|
||||
|
|
@ -27,4 +28,12 @@ export interface PipelineResult {
|
|||
* affordance so regression suites can prove the pool engaged.
|
||||
*/
|
||||
usedWorkerPool: boolean;
|
||||
/**
|
||||
* Streamed PDG-emit COPY manifest (#2202). Present only when streaming/chunked
|
||||
* PDG emit was active (full rebuild + `--pdg` + enabled): the BasicBlock node
|
||||
* CSV + per-pair PDG-edge CSVs flushed to disk during the emit loop, for
|
||||
* `loadGraphToLbug` to COPY alongside the structural CSVs. Absent ⇒ the PDG
|
||||
* layer (if any) is resident in `graph` and persists via the whole-graph emit.
|
||||
*/
|
||||
pdgEmitManifest?: PdgEmitManifest;
|
||||
}
|
||||
|
|
|
|||
28
gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/app.vue
Normal file
28
gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/app.vue
Normal file
|
|
@ -0,0 +1,28 @@
|
|||
<!--
|
||||
Vue SFC that imports a sibling TypeScript module (`./shared`). The Vue
|
||||
provider's `collectScopeContextPaths` follows that import and adds shared.ts
|
||||
to the Vue resolution pass, so shared.ts is PDG-emitted in BOTH the
|
||||
TypeScript pass and the Vue context pass — the cross-pass double-emit the
|
||||
#2202 streaming per-file dedup must collapse (review #8a). The <script>'s own
|
||||
functions add a second, Vue-side CFG so the graph holds blocks from both
|
||||
files.
|
||||
-->
|
||||
<template>
|
||||
<div class="panel" @click="onClick">{{ label }}</div>
|
||||
</template>
|
||||
|
||||
<script setup lang="ts">
|
||||
import { ref } from 'vue';
|
||||
import { classify, accumulate, guard } from './shared';
|
||||
|
||||
const count = ref(0);
|
||||
const label = classify(guard(accumulate(10)));
|
||||
|
||||
function onClick(): void {
|
||||
if (count.value > 0) {
|
||||
count.value = count.value - 1;
|
||||
} else {
|
||||
count.value = 0;
|
||||
}
|
||||
}
|
||||
</script>
|
||||
44
gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/shared.ts
Normal file
44
gitnexus/test/integration/cfg/fixtures/vue-ts-pdg/shared.ts
Normal file
|
|
@ -0,0 +1,44 @@
|
|||
// Shared TypeScript module imported by app.vue. Because app.vue's
|
||||
// `collectScopeContextPaths` does a transitive import closure, THIS file is
|
||||
// pulled into the Vue scope-resolution pass IN ADDITION to the primary
|
||||
// TypeScript pass — so its worker-built CFG (the functions below) is
|
||||
// PDG-emitted in BOTH passes over the same `cfgSideChannel`, producing
|
||||
// identical BasicBlock + PDG-edge ids. The in-memory graph dedups those by id
|
||||
// (first-writer-wins Map); the streaming sink relies on per-file dedup in
|
||||
// run.ts (`pdgEmittedFiles`). This is the real cross-pass double-emit the
|
||||
// #2202 streaming dedup must collapse (review #8a).
|
||||
|
||||
export function classify(x: number): string {
|
||||
let label: string;
|
||||
if (x > 0) {
|
||||
label = 'positive';
|
||||
} else if (x < 0) {
|
||||
label = 'negative';
|
||||
} else {
|
||||
label = 'zero';
|
||||
}
|
||||
return label;
|
||||
}
|
||||
|
||||
export function accumulate(n: number): number {
|
||||
let sum = 0;
|
||||
for (let i = 0; i < n; i++) {
|
||||
if (i % 2 === 0) {
|
||||
sum += i;
|
||||
} else {
|
||||
sum -= 1;
|
||||
}
|
||||
}
|
||||
return sum;
|
||||
}
|
||||
|
||||
export function guard(value: number): number {
|
||||
if (value > 100) {
|
||||
return clamp(value);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
function clamp(value: number): number {
|
||||
return value > 100 ? 100 : value;
|
||||
}
|
||||
167
gitnexus/test/integration/cfg/pipeline-pdg-streaming.test.ts
Normal file
167
gitnexus/test/integration/cfg/pipeline-pdg-streaming.test.ts
Normal file
|
|
@ -0,0 +1,167 @@
|
|||
/**
|
||||
* End-to-end proof of streaming/chunked PDG graph emit (issue #2202).
|
||||
*
|
||||
* Runs the real pipeline (workers + scope-resolution) on the pdg-repo fixture
|
||||
* TWICE — a non-streamed baseline and a streamed run — and asserts:
|
||||
* - the streamed run's in-memory graph holds ZERO BasicBlock nodes and ZERO
|
||||
* intra-file PDG edges (the bulky layer was flushed to CSV, never resident
|
||||
* — the O(chunk) RSS bound, R1);
|
||||
* - the streamed run produces a `pdgEmitManifest` whose BasicBlock + PDG-edge
|
||||
* row counts EQUAL the baseline's resident counts (same emitted SET, R2);
|
||||
* - the baseline (streaming off) still emits the PDG layer into the graph,
|
||||
* i.e. the default path is unchanged (R3).
|
||||
*
|
||||
* Both runs use a durable parse cache (streaming requires `parseCache.storagePath`
|
||||
* for its CSV dir), differing ONLY in `streamPdgEmit`, so streaming is the only
|
||||
* variable.
|
||||
*/
|
||||
import { describe, it, expect, afterAll } from 'vitest';
|
||||
import fs from 'fs';
|
||||
import os from 'os';
|
||||
import path from 'path';
|
||||
import { runPipelineFromRepo } from '../../../src/core/ingestion/pipeline.js';
|
||||
import { loadParseCache } from '../../../src/storage/parse-cache.js';
|
||||
import type { PipelineResult } from '../../../src/types/pipeline.js';
|
||||
|
||||
const FIXTURE = path.join(__dirname, 'fixtures', 'pdg-repo');
|
||||
// A `.vue` SFC importing a `.ts` module: the TS module is PDG-emitted in BOTH
|
||||
// the TypeScript pass and the Vue context pass (review #8a, cross-pass dedup).
|
||||
const VUE_TS_FIXTURE = path.join(__dirname, 'fixtures', 'vue-ts-pdg');
|
||||
const PDG_EDGE_TYPES = new Set([
|
||||
'CFG',
|
||||
'REACHING_DEF',
|
||||
'CDG',
|
||||
'POST_DOMINATE',
|
||||
'TAINTED',
|
||||
'SANITIZES',
|
||||
]);
|
||||
|
||||
const tmpDirs: string[] = [];
|
||||
function freshRepo(fixture: string = FIXTURE): string {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-pdg-stream-'));
|
||||
fs.cpSync(fixture, dir, { recursive: true });
|
||||
tmpDirs.push(dir);
|
||||
return dir;
|
||||
}
|
||||
function freshStorage(): string {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-pdg-store-'));
|
||||
tmpDirs.push(dir);
|
||||
return dir;
|
||||
}
|
||||
|
||||
function pdgCounts(result: PipelineResult): { basicBlocks: number; pdgEdges: number } {
|
||||
let basicBlocks = 0;
|
||||
result.graph.forEachNode((n) => {
|
||||
if (n.label === 'BasicBlock') basicBlocks++;
|
||||
});
|
||||
let pdgEdges = 0;
|
||||
for (const rel of result.graph.iterRelationships()) {
|
||||
if (PDG_EDGE_TYPES.has(rel.type)) pdgEdges++;
|
||||
}
|
||||
return { basicBlocks, pdgEdges };
|
||||
}
|
||||
|
||||
describe('#2202 — streaming PDG emit end-to-end', () => {
|
||||
afterAll(() => {
|
||||
for (const d of tmpDirs) fs.rmSync(d, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
it('streams the PDG layer out of the graph while preserving the emitted set', async () => {
|
||||
// ── Baseline: --pdg on, streaming OFF (durable cache, same as streamed) ──
|
||||
const baseStorage = freshStorage();
|
||||
const baseline = await runPipelineFromRepo(freshRepo(), () => {}, {
|
||||
pdg: true,
|
||||
parseCache: await loadParseCache(baseStorage),
|
||||
});
|
||||
const base = pdgCounts(baseline);
|
||||
// R3 / sanity: the default path still materializes the PDG layer in-graph.
|
||||
expect(base.basicBlocks).toBeGreaterThan(0);
|
||||
expect(base.pdgEdges).toBeGreaterThan(0);
|
||||
expect(baseline.pdgEmitManifest).toBeUndefined();
|
||||
|
||||
// ── Streamed: --pdg on, streaming ON ─────────────────────────────────
|
||||
const streamStorage = freshStorage();
|
||||
const streamed = await runPipelineFromRepo(freshRepo(), () => {}, {
|
||||
pdg: true,
|
||||
streamPdgEmit: true,
|
||||
parseCache: await loadParseCache(streamStorage),
|
||||
});
|
||||
const streamedCounts = pdgCounts(streamed);
|
||||
|
||||
// R1: the bulky PDG layer never accumulated in the in-memory graph.
|
||||
expect(streamedCounts.basicBlocks).toBe(0);
|
||||
expect(streamedCounts.pdgEdges).toBe(0);
|
||||
|
||||
// R2: the streamed manifest carries the SAME emitted set as the baseline.
|
||||
const manifest = streamed.pdgEmitManifest;
|
||||
expect(manifest).toBeDefined();
|
||||
const bbRows = manifest!.nodeFiles.get('BasicBlock')?.rows ?? 0;
|
||||
expect(bbRows).toBe(base.basicBlocks);
|
||||
let manifestEdgeRows = 0;
|
||||
for (const [, meta] of manifest!.relsByPair) manifestEdgeRows += meta.rows;
|
||||
expect(manifestEdgeRows).toBe(base.pdgEdges);
|
||||
|
||||
// The streamed BasicBlock CSV exists on disk under the storage dir.
|
||||
const bbCsv = manifest!.nodeFiles.get('BasicBlock')?.csvPath;
|
||||
expect(bbCsv).toBeDefined();
|
||||
expect(fs.existsSync(bbCsv!)).toBe(true);
|
||||
});
|
||||
|
||||
it('collapses the real Vue+TS cross-pass double-emit to one streamed copy (review #8a)', async () => {
|
||||
// A `.ts` module imported by a `.vue` SFC is PDG-emitted in BOTH the
|
||||
// TypeScript pass and the Vue context pass (the Vue provider's
|
||||
// `collectScopeContextPaths` follows the import) over the same worker-built
|
||||
// `cfgSideChannel` → identical ids. The in-memory graph dedups those by id
|
||||
// (Map first-writer-wins); the streaming sink is dedup-free, so the emit
|
||||
// loop dedups per FILE via `pdgEmittedFiles`. Without that dedup the streamed
|
||||
// manifest would carry shared.ts's blocks TWICE (verified out-of-band: 33 →
|
||||
// 61 BasicBlock rows). This asserts the streamed SET equals the Map-deduped
|
||||
// baseline — the load-bearing dedup regression guard.
|
||||
const baseStorage = freshStorage();
|
||||
const baseline = await runPipelineFromRepo(freshRepo(VUE_TS_FIXTURE), () => {}, {
|
||||
pdg: true,
|
||||
parseCache: await loadParseCache(baseStorage),
|
||||
});
|
||||
const base = pdgCounts(baseline);
|
||||
// Both files contribute blocks — the cross-pass case is actually present.
|
||||
expect(base.basicBlocks).toBeGreaterThan(0);
|
||||
expect(base.pdgEdges).toBeGreaterThan(0);
|
||||
|
||||
const streamStorage = freshStorage();
|
||||
const streamed = await runPipelineFromRepo(freshRepo(VUE_TS_FIXTURE), () => {}, {
|
||||
pdg: true,
|
||||
streamPdgEmit: true,
|
||||
parseCache: await loadParseCache(streamStorage),
|
||||
});
|
||||
// Streamed graph holds none of the PDG layer.
|
||||
const streamedCounts = pdgCounts(streamed);
|
||||
expect(streamedCounts.basicBlocks).toBe(0);
|
||||
expect(streamedCounts.pdgEdges).toBe(0);
|
||||
|
||||
// The streamed manifest carries each file's PDG layer EXACTLY ONCE — equal
|
||||
// to the Map-deduped baseline. A broken per-file dedup would double the
|
||||
// shared module's rows and fail here.
|
||||
const manifest = streamed.pdgEmitManifest;
|
||||
expect(manifest).toBeDefined();
|
||||
expect(manifest!.nodeFiles.get('BasicBlock')?.rows ?? 0).toBe(base.basicBlocks);
|
||||
let manifestEdgeRows = 0;
|
||||
for (const [, meta] of manifest!.relsByPair) manifestEdgeRows += meta.rows;
|
||||
expect(manifestEdgeRows).toBe(base.pdgEdges);
|
||||
});
|
||||
|
||||
it('falls back to in-memory emit when streaming is on but no storage path exists', async () => {
|
||||
// `streamPdgEmit: true` with NO parse cache → `parsedFileStorePath` is
|
||||
// undefined, so phase.ts cannot place the streamed CSV dir and falls back to
|
||||
// the in-memory whole-graph emit (the `else` branch that warns). The PDG
|
||||
// layer must still land in the graph and NO manifest is produced.
|
||||
const fellBack = await runPipelineFromRepo(freshRepo(VUE_TS_FIXTURE), () => {}, {
|
||||
pdg: true,
|
||||
streamPdgEmit: true,
|
||||
// intentionally no parseCache → no storagePath
|
||||
});
|
||||
const counts = pdgCounts(fellBack);
|
||||
expect(counts.basicBlocks).toBeGreaterThan(0); // emitted in-memory, not streamed
|
||||
expect(counts.pdgEdges).toBeGreaterThan(0);
|
||||
expect(fellBack.pdgEmitManifest).toBeUndefined(); // no streaming happened
|
||||
});
|
||||
});
|
||||
|
|
@ -4,17 +4,45 @@
|
|||
* Tests: streamAllCSVsToDisk with real graph data.
|
||||
* Covers hardening fixes: LRU cache (#24), BufferedCSVWriter flush
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
|
||||
import { describe, it, expect, beforeAll, beforeEach, afterAll } from 'vitest';
|
||||
import fs from 'fs/promises';
|
||||
import { finished } from 'stream/promises';
|
||||
import path from 'path';
|
||||
import { createTempDir, type TestDBHandle } from '../helpers/test-db.js';
|
||||
import { buildTestGraph, type TestNodeInput, type TestRelInput } from '../helpers/test-graph.js';
|
||||
import { streamAllCSVsToDisk } from '../../src/core/lbug/csv-generator.js';
|
||||
import {
|
||||
streamAllCSVsToDisk,
|
||||
buildRelRow,
|
||||
REL_CSV_HEADER,
|
||||
} from '../../src/core/lbug/csv-generator.js';
|
||||
import { splitRelCsvByLabelPair } from '../../src/core/lbug/lbug-adapter.js';
|
||||
import { getNodeLabel } from '../../src/core/lbug/rel-pair-routing.js';
|
||||
import { NODE_TABLES } from '../../src/core/lbug/schema.js';
|
||||
|
||||
let tmpHandle: TestDBHandle;
|
||||
let csvDir: string;
|
||||
let repoDir: string;
|
||||
|
||||
/** Data rows (header dropped) of one CSV file's text. */
|
||||
const dataRowsOf = (csv: string): string[] =>
|
||||
csv
|
||||
.trim()
|
||||
.split('\n')
|
||||
.slice(1)
|
||||
.filter((l) => l.length > 0);
|
||||
|
||||
/** Concatenate data rows from every per-pair rel file (#2203 U2), pair keys
|
||||
* sorted so the concatenation order is deterministic regardless of map order. */
|
||||
const readAllRelRows = async (
|
||||
relsByPair: Map<string, { csvPath: string; rows: number }>,
|
||||
): Promise<string[]> => {
|
||||
const rows: string[] = [];
|
||||
for (const key of [...relsByPair.keys()].sort()) {
|
||||
rows.push(...dataRowsOf(await fs.readFile(relsByPair.get(key)!.csvPath, 'utf-8')));
|
||||
}
|
||||
return rows;
|
||||
};
|
||||
|
||||
beforeAll(async () => {
|
||||
tmpHandle = await createTempDir('csv-pipeline-test-');
|
||||
csvDir = path.join(tmpHandle.dbPath, 'csv');
|
||||
|
|
@ -76,9 +104,9 @@ describe('streamAllCSVsToDisk', () => {
|
|||
{ id: 'folder:src', label: 'Folder', name: 'src', filePath: 'src' },
|
||||
],
|
||||
[
|
||||
{ sourceId: 'func:main', targetId: 'func:helper', type: 'CALLS' },
|
||||
{ sourceId: 'file:src/index.ts', targetId: 'func:main', type: 'CONTAINS' },
|
||||
{ sourceId: 'file:src/utils.ts', targetId: 'func:helper', type: 'CONTAINS' },
|
||||
{ sourceId: 'Function:main', targetId: 'Function:helper', type: 'CALLS' },
|
||||
{ sourceId: 'File:src/index.ts', targetId: 'Function:main', type: 'CONTAINS' },
|
||||
{ sourceId: 'File:src/utils.ts', targetId: 'Function:helper', type: 'CONTAINS' },
|
||||
],
|
||||
);
|
||||
|
||||
|
|
@ -86,7 +114,8 @@ describe('streamAllCSVsToDisk', () => {
|
|||
|
||||
// Check that CSV files were created
|
||||
expect(result.nodeFiles.size).toBeGreaterThan(0);
|
||||
expect(result.relRows).toBe(3);
|
||||
expect(result.totalValidRels).toBe(3);
|
||||
expect(result.skippedRels).toBe(0);
|
||||
|
||||
// Verify File CSV
|
||||
const fileCsv = result.nodeFiles.get('File');
|
||||
|
|
@ -108,10 +137,12 @@ describe('streamAllCSVsToDisk', () => {
|
|||
expect(folderCsv).toBeDefined();
|
||||
expect(folderCsv!.rows).toBe(1);
|
||||
|
||||
// Verify relations CSV exists
|
||||
const relContent = await fs.readFile(result.relCsvPath, 'utf-8');
|
||||
const relLines = relContent.trim().split('\n');
|
||||
expect(relLines.length).toBe(4); // header + 3 relationships
|
||||
// Relationships are routed to per-FROM→TO-label-pair files (#2203 U2):
|
||||
// Function→Function (CALLS) + File→Function (2× CONTAINS).
|
||||
expect(result.relsByPair.has('Function|Function')).toBe(true);
|
||||
expect(result.relsByPair.has('File|Function')).toBe(true);
|
||||
expect(result.relsByPair.get('File|Function')!.rows).toBe(2);
|
||||
expect(await readAllRelRows(result.relsByPair)).toHaveLength(3);
|
||||
});
|
||||
|
||||
it('CSV content is properly escaped', async () => {
|
||||
|
|
@ -210,7 +241,8 @@ describe('streamAllCSVsToDisk', () => {
|
|||
const graph = buildTestGraph([], []);
|
||||
const result = await streamAllCSVsToDisk(graph, repoDir, csvDir);
|
||||
expect(result.nodeFiles.size).toBe(0);
|
||||
expect(result.relRows).toBe(0);
|
||||
expect(result.totalValidRels).toBe(0);
|
||||
expect(result.relsByPair.size).toBe(0);
|
||||
});
|
||||
|
||||
it('handles node with empty string properties', async () => {
|
||||
|
|
@ -221,6 +253,27 @@ describe('streamAllCSVsToDisk', () => {
|
|||
expect(fileCsv).toBeDefined();
|
||||
expect(fileCsv!.rows).toBe(1);
|
||||
});
|
||||
|
||||
it('crosses the BufferedCSVWriter FLUSH_EVERY boundary without losing rows', async () => {
|
||||
// FLUSH_EVERY=500; a >500-node graph forces ≥1 mid-stream flush, exercising
|
||||
// addRow's flush-promise return + the loop's `if (pending) await pending`
|
||||
// path that the small fixtures above never reach (only the bench did).
|
||||
const N = 600;
|
||||
const nodes = Array.from({ length: N }, (_, i) => ({
|
||||
id: `File:src/f${i}.ts`,
|
||||
label: 'File' as const,
|
||||
name: `f${i}.ts`,
|
||||
filePath: `src/f${i}.ts`,
|
||||
}));
|
||||
const result = await streamAllCSVsToDisk(buildTestGraph(nodes), repoDir, csvDir);
|
||||
|
||||
const fileCsv = result.nodeFiles.get('File');
|
||||
expect(fileCsv).toBeDefined();
|
||||
expect(fileCsv!.rows).toBe(N); // no rows dropped/duplicated at the flush boundary
|
||||
const dataRows = dataRowsOf(await fs.readFile(fileCsv!.csvPath, 'utf-8'));
|
||||
expect(dataRows).toHaveLength(N);
|
||||
expect(new Set(dataRows).size).toBe(N); // all distinct — no flush-boundary corruption
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
|
|
@ -234,15 +287,18 @@ describe('streamAllCSVsToDisk — deterministic output ordering', () => {
|
|||
// Folder nodes: single-line CSV rows (no multi-line `content` column), so the
|
||||
// id is the first comma-separated field and split('\n') is safe. ids are
|
||||
// deliberately NOT in insertion order (c, a, b).
|
||||
// ids use the `Folder:` prefix so getNodeLabel derives the valid `Folder`
|
||||
// table — edges route to rel_Folder_Folder.csv (#2203 U2). Deliberately NOT
|
||||
// in insertion order (c, a, b).
|
||||
const NODES: TestNodeInput[] = [
|
||||
{ id: 'folder:c', label: 'Folder', name: 'c', filePath: 'c' },
|
||||
{ id: 'folder:a', label: 'Folder', name: 'a', filePath: 'a' },
|
||||
{ id: 'folder:b', label: 'Folder', name: 'b', filePath: 'b' },
|
||||
{ id: 'Folder:c', label: 'Folder', name: 'c', filePath: 'c' },
|
||||
{ id: 'Folder:a', label: 'Folder', name: 'a', filePath: 'a' },
|
||||
{ id: 'Folder:b', label: 'Folder', name: 'b', filePath: 'b' },
|
||||
];
|
||||
const RELS: TestRelInput[] = [
|
||||
{ sourceId: 'folder:c', targetId: 'folder:a', type: 'CONTAINS' },
|
||||
{ sourceId: 'folder:a', targetId: 'folder:b', type: 'CONTAINS' },
|
||||
{ sourceId: 'folder:b', targetId: 'folder:c', type: 'CONTAINS' },
|
||||
{ sourceId: 'Folder:c', targetId: 'Folder:a', type: 'CONTAINS' },
|
||||
{ sourceId: 'Folder:a', targetId: 'Folder:b', type: 'CONTAINS' },
|
||||
{ sourceId: 'Folder:b', targetId: 'Folder:c', type: 'CONTAINS' },
|
||||
];
|
||||
const dataRows = (csv: string): string[] =>
|
||||
csv
|
||||
|
|
@ -270,7 +326,7 @@ describe('streamAllCSVsToDisk — deterministic output ordering', () => {
|
|||
const folderIds = folderCsv
|
||||
? dataRows(await fs.readFile(folderCsv.csvPath, 'utf-8')).map(firstCol)
|
||||
: [];
|
||||
const relRows = dataRows(await fs.readFile(result.relCsvPath, 'utf-8'));
|
||||
const relRows = await readAllRelRows(result.relsByPair);
|
||||
return { folderIds, relRows };
|
||||
} finally {
|
||||
delete process.env.GITNEXUS_SORT_GRAPH_OUTPUT;
|
||||
|
|
@ -307,3 +363,205 @@ describe('streamAllCSVsToDisk — deterministic output ordering', () => {
|
|||
expect([...onFwd.relRows].sort()).toEqual([...offFwd.relRows].sort());
|
||||
});
|
||||
});
|
||||
|
||||
/**
|
||||
* #2203 U2 byte-identity: for all quote-free ids the direct per-pair emit must
|
||||
* produce per-pair files byte-for-byte identical to the legacy
|
||||
* splitRelCsvByLabelPair oracle run over an equivalent monolithic relations.csv
|
||||
* from the same graph. This is the load-bearing guard for "byte-identical graph
|
||||
* content" (issue acceptance). The ONE intentional divergence — ids containing a
|
||||
* double-quote, where the router (raw-id label) is more correct than the oracle
|
||||
* (regex over the escaped row) — is asserted explicitly in its own test below.
|
||||
*/
|
||||
describe('streamAllCSVsToDisk — direct per-pair emit matches the split oracle', () => {
|
||||
// The oracle always emits in graph.iterRelationships() (unsorted) order; the
|
||||
// production path honours GITNEXUS_SORT_GRAPH_OUTPUT. Clear it so a value
|
||||
// leaked from a prior test can't desync the two and produce a spurious diff.
|
||||
beforeEach(() => {
|
||||
delete process.env.GITNEXUS_SORT_GRAPH_OUTPUT;
|
||||
});
|
||||
|
||||
it('produces byte-identical per-pair files + identical skip/total accounting', async () => {
|
||||
// Multiple valid pairs, getNodeLabel special prefixes (comm_ AND proc_), and
|
||||
// one invalid-label edge that BOTH paths must skip identically.
|
||||
const graph = buildTestGraph(
|
||||
[
|
||||
{ id: 'File:a.ts', label: 'File', name: 'a.ts', filePath: 'a.ts' },
|
||||
{ id: 'Function:a.ts:f:1', label: 'Function', name: 'f', filePath: 'a.ts' },
|
||||
{ id: 'Function:a.ts:g:5', label: 'Function', name: 'g', filePath: 'a.ts' },
|
||||
{ id: 'comm_1', label: 'Community' as never, name: 'c1', filePath: '' },
|
||||
{ id: 'comm_2', label: 'Community' as never, name: 'c2', filePath: '' },
|
||||
{ id: 'proc_1', label: 'Process' as never, name: 'p1', filePath: '' },
|
||||
{ id: 'proc_2', label: 'Process' as never, name: 'p2', filePath: '' },
|
||||
],
|
||||
[
|
||||
{ sourceId: 'File:a.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' },
|
||||
{ sourceId: 'File:a.ts', targetId: 'Function:a.ts:g:5', type: 'CONTAINS' },
|
||||
{ sourceId: 'Function:a.ts:f:1', targetId: 'Function:a.ts:g:5', type: 'CALLS' },
|
||||
{ sourceId: 'comm_1', targetId: 'comm_2', type: 'CONTAINS' },
|
||||
// proc_ prefix → Process label (getNodeLabel special case).
|
||||
{ sourceId: 'proc_1', targetId: 'proc_2', type: 'CONTAINS' },
|
||||
// Invalid FROM label ('Bogus' ∉ NODE_TABLES) — skipped by both paths.
|
||||
{ sourceId: 'Bogus:x', targetId: 'File:a.ts', type: 'CONTAINS' },
|
||||
// Invalid TO label — exercises the OTHER branch of the skip condition.
|
||||
{ sourceId: 'File:a.ts', targetId: 'Bogus:y', type: 'CONTAINS' },
|
||||
],
|
||||
);
|
||||
|
||||
const directDir = path.join(csvDir, 'diff-direct');
|
||||
const oracleDir = path.join(csvDir, 'diff-oracle');
|
||||
await fs.mkdir(oracleDir, { recursive: true });
|
||||
|
||||
// Direct emit (production path).
|
||||
const direct = await streamAllCSVsToDisk(graph, repoDir, directDir);
|
||||
|
||||
// Oracle: build the monolithic relations.csv this graph would have produced
|
||||
// (same insertion order, same row bytes via buildRelRow), then split it.
|
||||
const relCsv = path.join(oracleDir, 'relations.csv');
|
||||
const lines = [REL_CSV_HEADER];
|
||||
for (const rel of graph.iterRelationships()) lines.push(buildRelRow(rel));
|
||||
await fs.writeFile(relCsv, lines.join('\n') + '\n', 'utf-8');
|
||||
|
||||
const split = await splitRelCsvByLabelPair(
|
||||
relCsv,
|
||||
oracleDir,
|
||||
new Set<string>(NODE_TABLES),
|
||||
getNodeLabel,
|
||||
);
|
||||
await Promise.all(
|
||||
Array.from(split.pairWriteStreams.values()).map(async (ws) => {
|
||||
ws.end();
|
||||
await finished(ws);
|
||||
}),
|
||||
);
|
||||
|
||||
// Identical accounting.
|
||||
expect(direct.totalValidRels).toBe(split.totalValidRels);
|
||||
expect(direct.totalValidRels).toBe(5);
|
||||
expect(direct.skippedRels).toBe(split.skippedRels);
|
||||
expect(direct.skippedRels).toBe(2); // invalid-FROM + invalid-TO, both skipped
|
||||
expect(direct.relHeader).toBe(split.relHeader);
|
||||
|
||||
// Identical pair set.
|
||||
expect([...direct.relsByPair.keys()].sort()).toEqual([...split.relsByPairMeta.keys()].sort());
|
||||
|
||||
// Byte-identical per-pair file contents.
|
||||
for (const key of direct.relsByPair.keys()) {
|
||||
const directContent = await fs.readFile(direct.relsByPair.get(key)!.csvPath, 'utf-8');
|
||||
const oracleContent = await fs.readFile(split.relsByPairMeta.get(key)!.csvPath, 'utf-8');
|
||||
expect(directContent, `pair ${key}`).toBe(oracleContent);
|
||||
}
|
||||
});
|
||||
|
||||
it('quote-in-id edge: router routes it (raw-id label) while the oracle drops it — intended divergence', async () => {
|
||||
// A node id with an embedded double-quote (legal in a POSIX filePath). The
|
||||
// router derives the label from the RAW id (`File`), so it routes the edge;
|
||||
// the oracle re-derives the label via /"([^"]*)","([^"]*)"/ over the ESCAPED
|
||||
// row (`"File:a""b.ts",...`), mis-reads the field, and drops it. This locks
|
||||
// the intended divergence so a future change can't silently revert the
|
||||
// router to the buggy regex semantics.
|
||||
const graph = buildTestGraph(
|
||||
[
|
||||
{ id: 'File:clean.ts', label: 'File', name: 'clean.ts', filePath: 'clean.ts' },
|
||||
{ id: 'File:a"b.ts', label: 'File', name: 'a"b.ts', filePath: 'a"b.ts' },
|
||||
{ id: 'Function:a.ts:f:1', label: 'Function', name: 'f', filePath: 'a.ts' },
|
||||
],
|
||||
[
|
||||
{ sourceId: 'File:clean.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' },
|
||||
{ sourceId: 'File:a"b.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' },
|
||||
],
|
||||
);
|
||||
|
||||
const directDir = path.join(csvDir, 'qd-direct');
|
||||
const oracleDir = path.join(csvDir, 'qd-oracle');
|
||||
await fs.mkdir(oracleDir, { recursive: true });
|
||||
|
||||
const direct = await streamAllCSVsToDisk(graph, repoDir, directDir);
|
||||
|
||||
const relCsv = path.join(oracleDir, 'relations.csv');
|
||||
const lines = [REL_CSV_HEADER];
|
||||
for (const rel of graph.iterRelationships()) lines.push(buildRelRow(rel));
|
||||
await fs.writeFile(relCsv, lines.join('\n') + '\n', 'utf-8');
|
||||
const split = await splitRelCsvByLabelPair(
|
||||
relCsv,
|
||||
oracleDir,
|
||||
new Set<string>(NODE_TABLES),
|
||||
getNodeLabel,
|
||||
);
|
||||
await Promise.all(
|
||||
Array.from(split.pairWriteStreams.values()).map(async (ws) => {
|
||||
ws.end();
|
||||
await finished(ws);
|
||||
}),
|
||||
);
|
||||
|
||||
// Router routes BOTH edges — the raw-id label `File` is valid for both.
|
||||
expect(direct.totalValidRels).toBe(2);
|
||||
expect(direct.skippedRels).toBe(0);
|
||||
expect(direct.relsByPair.get('File|Function')!.rows).toBe(2);
|
||||
|
||||
// Oracle DIVERGES: its regex mis-reads the quote-in-id row and drops that
|
||||
// edge, so it routes strictly fewer edges. Asserted robustly — we do NOT
|
||||
// pin the oracle's exact mis-derived label.
|
||||
expect(split.totalValidRels).toBeLessThan(direct.totalValidRels);
|
||||
expect(split.skippedRels).toBeGreaterThan(direct.skippedRels);
|
||||
});
|
||||
|
||||
it('sorted path (GITNEXUS_SORT_GRAPH_OUTPUT=1): per-pair files byte-identical to the oracle', async () => {
|
||||
// The earlier differential test covers the default (insertion-order) path.
|
||||
// Here the sorted emit path must also match the oracle — fed the SAME
|
||||
// id-sorted order orderedRelationships() uses (sort by rel.id).
|
||||
process.env.GITNEXUS_SORT_GRAPH_OUTPUT = '1';
|
||||
try {
|
||||
const graph = buildTestGraph(
|
||||
[
|
||||
{ id: 'File:a.ts', label: 'File', name: 'a.ts', filePath: 'a.ts' },
|
||||
{ id: 'Function:a.ts:f:1', label: 'Function', name: 'f', filePath: 'a.ts' },
|
||||
{ id: 'Function:a.ts:g:5', label: 'Function', name: 'g', filePath: 'a.ts' },
|
||||
],
|
||||
// Deliberately NOT in id-sorted order so the sort actually reorders rows.
|
||||
[
|
||||
{ sourceId: 'Function:a.ts:f:1', targetId: 'Function:a.ts:g:5', type: 'CALLS' },
|
||||
{ sourceId: 'File:a.ts', targetId: 'Function:a.ts:g:5', type: 'CONTAINS' },
|
||||
{ sourceId: 'File:a.ts', targetId: 'Function:a.ts:f:1', type: 'CONTAINS' },
|
||||
],
|
||||
);
|
||||
|
||||
const directDir = path.join(csvDir, 'sorted-direct');
|
||||
const oracleDir = path.join(csvDir, 'sorted-oracle');
|
||||
await fs.mkdir(oracleDir, { recursive: true });
|
||||
|
||||
const direct = await streamAllCSVsToDisk(graph, repoDir, directDir);
|
||||
|
||||
// Oracle fed the same id-sorted order the sorted emit produces.
|
||||
const sortedRels = [...graph.iterRelationships()].sort((a, b) =>
|
||||
a.id < b.id ? -1 : a.id > b.id ? 1 : 0,
|
||||
);
|
||||
const relCsv = path.join(oracleDir, 'relations.csv');
|
||||
const lines = [REL_CSV_HEADER];
|
||||
for (const rel of sortedRels) lines.push(buildRelRow(rel));
|
||||
await fs.writeFile(relCsv, lines.join('\n') + '\n', 'utf-8');
|
||||
const split = await splitRelCsvByLabelPair(
|
||||
relCsv,
|
||||
oracleDir,
|
||||
new Set<string>(NODE_TABLES),
|
||||
getNodeLabel,
|
||||
);
|
||||
await Promise.all(
|
||||
Array.from(split.pairWriteStreams.values()).map(async (ws) => {
|
||||
ws.end();
|
||||
await finished(ws);
|
||||
}),
|
||||
);
|
||||
|
||||
expect([...direct.relsByPair.keys()].sort()).toEqual([...split.relsByPairMeta.keys()].sort());
|
||||
for (const key of direct.relsByPair.keys()) {
|
||||
const directContent = await fs.readFile(direct.relsByPair.get(key)!.csvPath, 'utf-8');
|
||||
const oracleContent = await fs.readFile(split.relsByPairMeta.get(key)!.csvPath, 'utf-8');
|
||||
expect(directContent, `pair ${key} (sorted)`).toBe(oracleContent);
|
||||
}
|
||||
} finally {
|
||||
delete process.env.GITNEXUS_SORT_GRAPH_OUTPUT;
|
||||
}
|
||||
});
|
||||
});
|
||||
|
|
|
|||
141
gitnexus/test/integration/lbug-load-prof.test.ts
Normal file
141
gitnexus/test/integration/lbug-load-prof.test.ts
Normal file
|
|
@ -0,0 +1,141 @@
|
|||
/**
|
||||
* Integration test: PROF_LBUG_LOAD persistence-path profiling (#2203 U1).
|
||||
*
|
||||
* loadGraphToLbug is un-timed in production today; the analyze "emit" number
|
||||
* is the scope-resolution emit bucket, not this CSV→COPY persistence path.
|
||||
* U1 adds a zero-cost-when-off per-stage breakdown gated by PROF_LBUG_LOAD=1,
|
||||
* mirroring the PROF_SCOPE_RESOLUTION pattern. These tests assert the gate:
|
||||
* - flag off → no `[lbug-load prof]` line is logged, behaviour unchanged
|
||||
* - flag on → exactly one summary line with every stage key + node/rel counts
|
||||
*
|
||||
* Needs a real LadybugDB connection (initLbug), so it lives under integration.
|
||||
* Logger assertions use `_captureLogger()` — the exported `logger` is a Proxy
|
||||
* over a lazily-built pino instance and is not directly spy-able.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, beforeEach, afterAll, afterEach } from 'vitest';
|
||||
import fs from 'fs/promises';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
import { buildTestGraph } from '../helpers/test-graph.js';
|
||||
import { _captureLogger, type LoggerCapture } from '../../src/core/logger.js';
|
||||
|
||||
let tmpBase: string;
|
||||
let storagePath: string;
|
||||
let dbPath: string;
|
||||
let cap: LoggerCapture;
|
||||
|
||||
const PROF_LINE = '[lbug-load prof]';
|
||||
|
||||
const profLines = (): string[] =>
|
||||
cap
|
||||
.records()
|
||||
.map((r) => (typeof r.msg === 'string' ? r.msg : ''))
|
||||
.filter((msg) => msg.includes(PROF_LINE));
|
||||
|
||||
beforeAll(async () => {
|
||||
tmpBase = path.join(os.tmpdir(), `gitnexus-lbug-prof-${Date.now()}-${process.pid}`);
|
||||
storagePath = path.join(tmpBase, '.gitnexus');
|
||||
dbPath = path.join(storagePath, 'lbug');
|
||||
await fs.mkdir(dbPath, { recursive: true });
|
||||
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
await adapter.initLbug(dbPath);
|
||||
});
|
||||
|
||||
beforeEach(() => {
|
||||
cap = _captureLogger();
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
cap.restore();
|
||||
delete process.env.PROF_LBUG_LOAD;
|
||||
});
|
||||
|
||||
afterAll(async () => {
|
||||
try {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
await adapter.closeLbug();
|
||||
} catch {
|
||||
/* may not have opened */
|
||||
}
|
||||
try {
|
||||
await fs.rm(tmpBase, { recursive: true, force: true });
|
||||
} catch {
|
||||
/* best-effort */
|
||||
}
|
||||
});
|
||||
|
||||
describe('PROF_LBUG_LOAD persistence-path profiling (#2203 U1)', () => {
|
||||
it('does NOT log a prof summary when the flag is unset', async () => {
|
||||
delete process.env.PROF_LBUG_LOAD;
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
|
||||
const graph = buildTestGraph(
|
||||
[
|
||||
{ id: 'File:src/off.ts', label: 'File', name: 'off.ts', filePath: 'src/off.ts' },
|
||||
{
|
||||
id: 'Function:src/off.ts:offFn:1',
|
||||
label: 'Function',
|
||||
name: 'offFn',
|
||||
filePath: 'src/off.ts',
|
||||
startLine: 1,
|
||||
endLine: 2,
|
||||
},
|
||||
],
|
||||
[{ sourceId: 'File:src/off.ts', targetId: 'Function:src/off.ts:offFn:1', type: 'DEFINES' }],
|
||||
);
|
||||
|
||||
const result = await adapter.loadGraphToLbug(graph, tmpBase, storagePath);
|
||||
|
||||
expect(result.success).toBe(true);
|
||||
expect(profLines()).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('logs exactly one summary line with all stage keys + counts when the flag is set', async () => {
|
||||
process.env.PROF_LBUG_LOAD = '1';
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
|
||||
// Distinct ids from the flag-off graph so the COPY does not hit a
|
||||
// PK-dup IGNORE_ERRORS retry on the shared singleton connection.
|
||||
const graph = buildTestGraph(
|
||||
[
|
||||
{ id: 'File:src/on.ts', label: 'File', name: 'on.ts', filePath: 'src/on.ts' },
|
||||
{
|
||||
id: 'Function:src/on.ts:onFn:1',
|
||||
label: 'Function',
|
||||
name: 'onFn',
|
||||
filePath: 'src/on.ts',
|
||||
startLine: 1,
|
||||
endLine: 2,
|
||||
},
|
||||
{
|
||||
id: 'Class:src/on.ts:OnClass:5',
|
||||
label: 'Class',
|
||||
name: 'OnClass',
|
||||
filePath: 'src/on.ts',
|
||||
startLine: 5,
|
||||
endLine: 8,
|
||||
},
|
||||
],
|
||||
[
|
||||
{ sourceId: 'File:src/on.ts', targetId: 'Function:src/on.ts:onFn:1', type: 'DEFINES' },
|
||||
{ sourceId: 'File:src/on.ts', targetId: 'Class:src/on.ts:OnClass:5', type: 'DEFINES' },
|
||||
],
|
||||
);
|
||||
|
||||
const result = await adapter.loadGraphToLbug(graph, tmpBase, storagePath);
|
||||
expect(result.success).toBe(true);
|
||||
|
||||
const lines = profLines();
|
||||
expect(lines).toHaveLength(1);
|
||||
|
||||
const line = lines[0];
|
||||
// Relationships are routed to per-pair files during csv-emit (#2203 U2),
|
||||
// so there is no separate rel-split stage.
|
||||
for (const key of ['csv-emit=', 'copy-nodes=', 'copy-rels=', 'fallback=', 'total=']) {
|
||||
expect(line).toContain(key);
|
||||
}
|
||||
// 3 node rows (File, Function, Class), 2 valid rels emitted.
|
||||
expect(line).toContain('(3 nodes, 2 rels)');
|
||||
});
|
||||
});
|
||||
181
gitnexus/test/integration/pdg-emit-streaming-roundtrip.test.ts
Normal file
181
gitnexus/test/integration/pdg-emit-streaming-roundtrip.test.ts
Normal file
|
|
@ -0,0 +1,181 @@
|
|||
/**
|
||||
* Integration test: streamed PDG-emit manifest round-trips the bulk-COPY load
|
||||
* path (issue #2202 U5).
|
||||
*
|
||||
* Simulates the streaming case end-to-end at the persistence boundary: the
|
||||
* BasicBlock + intra-file PDG-edge layer is flushed to CSV by a real
|
||||
* `PdgEmitSink` (so the in-memory graph holds ZERO BasicBlocks, exactly as in a
|
||||
* streamed run), and `loadGraphToLbug` is handed the resulting manifest. Asserts
|
||||
* the BasicBlock nodes + every PDG edge type land in the DB via the manifest,
|
||||
* alongside the structural graph — and that there is no double-COPY.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
|
||||
import fs from 'fs/promises';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph.js';
|
||||
import { PdgEmitSink } from '../../src/core/lbug/pdg-emit-sink.js';
|
||||
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
|
||||
|
||||
let tmpBase: string;
|
||||
let storagePath: string;
|
||||
|
||||
const FILE_ID = 'File:src/a.ts';
|
||||
const BB = (i: number) => `BasicBlock:src/a.ts:1:0:${i}`;
|
||||
const PDG_TYPES = ['CFG', 'REACHING_DEF', 'CDG', 'POST_DOMINATE', 'TAINTED', 'SANITIZES'] as const;
|
||||
|
||||
beforeAll(async () => {
|
||||
// mkdtemp (unpredictable, unique) — not a predictable os-temp path.
|
||||
tmpBase = await fs.mkdtemp(path.join(os.tmpdir(), 'gitnexus-pdg-stream-rt-'));
|
||||
storagePath = path.join(tmpBase, '.gitnexus');
|
||||
await fs.mkdir(path.join(storagePath, 'lbug'), { recursive: true });
|
||||
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
await adapter.initLbug(path.join(storagePath, 'lbug'));
|
||||
|
||||
// Structural graph — NO BasicBlock nodes (they were "streamed out").
|
||||
const graph = createKnowledgeGraph();
|
||||
graph.addNode({
|
||||
id: FILE_ID,
|
||||
label: 'File',
|
||||
properties: { name: 'a.ts', filePath: 'src/a.ts' },
|
||||
});
|
||||
|
||||
// Real sink → real manifest: route 3 BasicBlocks + one edge of each PDG type.
|
||||
const sink = new PdgEmitSink(graph, path.join(storagePath, 'pdg-csv'));
|
||||
for (let i = 0; i < 3; i++) {
|
||||
const node: GraphNode = {
|
||||
id: BB(i),
|
||||
label: 'BasicBlock',
|
||||
properties: {
|
||||
name: '',
|
||||
filePath: 'src/a.ts',
|
||||
startLine: i + 1,
|
||||
endLine: i + 2,
|
||||
text: `b${i}`,
|
||||
},
|
||||
};
|
||||
sink.addNode(node);
|
||||
}
|
||||
for (const type of PDG_TYPES) {
|
||||
const rel: GraphRelationship = {
|
||||
id: `${type}:0->1`,
|
||||
sourceId: BB(0),
|
||||
targetId: BB(1),
|
||||
type,
|
||||
confidence: 1,
|
||||
reason: type === 'REACHING_DEF' ? 'x' : type === 'CDG' ? 'T' : `${type}-edge`,
|
||||
};
|
||||
sink.addRelationship(rel);
|
||||
}
|
||||
const manifest = sink.finalize();
|
||||
|
||||
// The sink offloaded the whole PDG layer — the graph has only the File node.
|
||||
expect(graph.nodeCount).toBe(1);
|
||||
|
||||
await adapter.loadGraphToLbug(graph, tmpBase, storagePath, undefined, manifest);
|
||||
});
|
||||
|
||||
afterAll(async () => {
|
||||
try {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
await adapter.closeLbug();
|
||||
} catch {
|
||||
/* may not have opened */
|
||||
}
|
||||
if (tmpBase) {
|
||||
for (let attempt = 0; attempt < 5; attempt++) {
|
||||
try {
|
||||
await fs.rm(tmpBase, { recursive: true, force: true });
|
||||
return;
|
||||
} catch {
|
||||
if (attempt < 4) await new Promise((r) => setTimeout(r, 200 * (attempt + 1)));
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
describe('streamed PDG manifest → bulk COPY (#2202 U5)', () => {
|
||||
it('BasicBlock nodes from the manifest land in the DB with span + text', async () => {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
const rows = await adapter.executeQuery(
|
||||
'MATCH (n:BasicBlock) RETURN n.id AS id, n.text AS text, n.startLine AS startLine ORDER BY n.id',
|
||||
);
|
||||
expect(rows).toHaveLength(3);
|
||||
expect(rows[0].id).toBe(BB(0));
|
||||
expect(rows[0].text).toBe('b0');
|
||||
expect(Number(rows[0].startLine)).toBe(1);
|
||||
});
|
||||
|
||||
it('the structural graph (File node) loaded alongside the manifest', async () => {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
const rows = await adapter.executeQuery(
|
||||
`MATCH (f:File {id: '${FILE_ID}'}) RETURN count(f) AS c`,
|
||||
);
|
||||
expect(Number(rows[0].c)).toBe(1);
|
||||
});
|
||||
|
||||
it('every PDG edge type round-trips via the manifest (no double-COPY)', async () => {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
for (const type of PDG_TYPES) {
|
||||
const rows = await adapter.executeQuery(
|
||||
`MATCH (:BasicBlock)-[r:CodeRelation {type: '${type}'}]->(:BasicBlock) RETURN count(r) AS c`,
|
||||
);
|
||||
// Exactly one — not two (double-COPY would double these).
|
||||
expect(Number(rows[0].c), `${type} should round-trip exactly once`).toBe(1);
|
||||
}
|
||||
});
|
||||
|
||||
it('REACHING_DEF carries its variable in reason (manifest path)', async () => {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
const rows = await adapter.executeQuery(
|
||||
"MATCH (a:BasicBlock)-[r:CodeRelation {type: 'REACHING_DEF', reason: 'x'}]->(b:BasicBlock) RETURN a.id AS from, b.id AS to",
|
||||
);
|
||||
expect(rows).toHaveLength(1);
|
||||
expect(rows[0].from).toBe(BB(0));
|
||||
expect(rows[0].to).toBe(BB(1));
|
||||
});
|
||||
});
|
||||
|
||||
describe('streamed PDG manifest → disjoint-key merge guard (#2202 review #3)', () => {
|
||||
// The merge in loadGraphToLbug assumes the streamed manifest and the
|
||||
// structural csvResult are disjoint: when streaming is on the in-memory graph
|
||||
// holds ZERO BasicBlocks, so streamAllCSVsToDisk emits no basicblock.csv and
|
||||
// the manifest is the sole source. A future BasicBlock-leak-into-graph would
|
||||
// make both sides carry a "BasicBlock" entry; silently overwriting one CSV
|
||||
// with the other would drop its rows. The guard fails loudly instead.
|
||||
|
||||
it('throws when the manifest collides with a structural node CSV', async () => {
|
||||
const adapter = await import('../../src/core/lbug/lbug-adapter.js');
|
||||
|
||||
// A graph that DOES contain a BasicBlock → streamAllCSVsToDisk emits a
|
||||
// structural basicblock.csv (the invariant-violation scenario).
|
||||
const leakyGraph = createKnowledgeGraph();
|
||||
leakyGraph.addNode({
|
||||
id: BB(0),
|
||||
label: 'BasicBlock',
|
||||
properties: { name: '', filePath: 'src/a.ts', startLine: 1, endLine: 2, text: 'leak' },
|
||||
});
|
||||
|
||||
// A manifest that ALSO declares a BasicBlock node CSV → disjoint-key clash.
|
||||
const collideBase = await fs.mkdtemp(path.join(os.tmpdir(), 'gitnexus-pdg-collide-'));
|
||||
const collideStorage = path.join(collideBase, '.gitnexus');
|
||||
await fs.mkdir(collideStorage, { recursive: true });
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), path.join(collideStorage, 'pdg-csv'));
|
||||
sink.addNode({
|
||||
id: BB(1),
|
||||
label: 'BasicBlock',
|
||||
properties: { name: '', filePath: 'src/a.ts', startLine: 3, endLine: 4, text: 'm' },
|
||||
});
|
||||
const manifest = sink.finalize();
|
||||
|
||||
try {
|
||||
await expect(
|
||||
adapter.loadGraphToLbug(leakyGraph, collideBase, collideStorage, undefined, manifest),
|
||||
).rejects.toThrow(/collides with a structural node CSV for "BasicBlock"/);
|
||||
} finally {
|
||||
await fs.rm(collideBase, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
});
|
||||
|
|
@ -0,0 +1,197 @@
|
|||
/**
|
||||
* End-to-end HTTP test of POST /api/analyze token validation.
|
||||
*
|
||||
* The unit tests (api-analyze-token.test.ts) cover validateAnalyzeToken in
|
||||
* isolation; this proves the REAL production route actually wires it in —
|
||||
* express.json body parsing, the requireLocalhostOrigin guard, the route
|
||||
* handler invoking the validator, and the 400 status/error shape on the wire.
|
||||
* Closes the gap the PR #2223 tri-review noted: "the route validation is
|
||||
* otherwise only reachable by booting the server."
|
||||
*
|
||||
* Only rejection paths are asserted: each returns 400 BEFORE any clone, so the
|
||||
* test is hermetic (no network, no background git, no real repo). The accepted
|
||||
* path would spawn a background clone and is left to the unit coverage.
|
||||
*
|
||||
* Mirrors the spawn+health-poll harness in server-http-startup.test.ts; the
|
||||
* integration suite always builds dist first (pretest:integration).
|
||||
*/
|
||||
import { spawn, type ChildProcessWithoutNullStreams } from 'node:child_process';
|
||||
import fs from 'node:fs';
|
||||
import http from 'node:http';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
import { afterAll, beforeAll, describe, expect, it } from 'vitest';
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const REPO_ROOT = path.resolve(__dirname, '..', '..');
|
||||
const DIST_CLI = path.join(REPO_ROOT, 'dist', 'cli', 'index.js');
|
||||
const STARTUP_BUDGET_MS = process.env.CI ? 30_000 : 15_000;
|
||||
|
||||
const allocateFreePort = (): Promise<number> =>
|
||||
new Promise((resolve, reject) => {
|
||||
const probe = http.createServer();
|
||||
probe.once('error', reject);
|
||||
probe.listen(0, '127.0.0.1', () => {
|
||||
const addr = probe.address();
|
||||
if (typeof addr !== 'object' || !addr) {
|
||||
probe.close();
|
||||
reject(new Error('could not allocate ephemeral port'));
|
||||
return;
|
||||
}
|
||||
const port = addr.port;
|
||||
probe.close((err) => (err ? reject(err) : resolve(port)));
|
||||
});
|
||||
});
|
||||
|
||||
const httpJson = (
|
||||
port: number,
|
||||
method: string,
|
||||
reqPath: string,
|
||||
body?: unknown,
|
||||
): Promise<{ status: number; body: string }> =>
|
||||
new Promise((resolve, reject) => {
|
||||
const payload = body === undefined ? undefined : JSON.stringify(body);
|
||||
const req = http.request(
|
||||
{
|
||||
host: '127.0.0.1',
|
||||
port,
|
||||
path: reqPath,
|
||||
method,
|
||||
headers: payload
|
||||
? { 'content-type': 'application/json', 'content-length': Buffer.byteLength(payload) }
|
||||
: {},
|
||||
},
|
||||
(res) => {
|
||||
const chunks: Buffer[] = [];
|
||||
res.on('data', (c) => chunks.push(c));
|
||||
res.on('end', () =>
|
||||
resolve({ status: res.statusCode ?? 0, body: Buffer.concat(chunks).toString('utf8') }),
|
||||
);
|
||||
},
|
||||
);
|
||||
req.on('error', reject);
|
||||
req.setTimeout(5_000, () => {
|
||||
req.destroy();
|
||||
reject(new Error(`${method} ${reqPath} timed out`));
|
||||
});
|
||||
if (payload) req.write(payload);
|
||||
req.end();
|
||||
});
|
||||
|
||||
const postAnalyze = (port: number, body: unknown) => httpJson(port, 'POST', '/api/analyze', body);
|
||||
|
||||
// Spawned `serve` on Windows can report ready before the socket is reachable
|
||||
// from the parent (see server-http-startup.test.ts); the validateAnalyzeToken
|
||||
// unit tests cover the validation logic on every platform.
|
||||
const describeBlock = process.platform === 'win32' ? describe.skip : describe;
|
||||
|
||||
describeBlock('POST /api/analyze token validation (real server)', () => {
|
||||
let proc: ChildProcessWithoutNullStreams | undefined;
|
||||
let homeDir: string | undefined;
|
||||
let port = 0;
|
||||
|
||||
beforeAll(async () => {
|
||||
if (!fs.existsSync(DIST_CLI)) {
|
||||
throw new Error(`Missing ${DIST_CLI} — run npm run build before integration tests`);
|
||||
}
|
||||
|
||||
port = await allocateFreePort();
|
||||
homeDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-analyze-token-'));
|
||||
|
||||
proc = spawn(
|
||||
process.execPath,
|
||||
[DIST_CLI, 'serve', '--port', String(port), '--host', '127.0.0.1'],
|
||||
{
|
||||
cwd: REPO_ROOT,
|
||||
env: { ...process.env, GITNEXUS_HOME: homeDir, NODE_OPTIONS: '' },
|
||||
stdio: ['ignore', 'pipe', 'pipe'],
|
||||
},
|
||||
);
|
||||
|
||||
let stderr = '';
|
||||
proc.stderr.on('data', (buf) => {
|
||||
stderr += buf.toString();
|
||||
});
|
||||
|
||||
const startedAt = Date.now();
|
||||
while (Date.now() - startedAt < STARTUP_BUDGET_MS) {
|
||||
if (proc.exitCode !== null) {
|
||||
throw new Error(`serve exited ${proc.exitCode} before ready.\nstderr:\n${stderr}`);
|
||||
}
|
||||
try {
|
||||
const { status } = await httpJson(port, 'GET', '/api/health');
|
||||
if (status === 200) return;
|
||||
} catch {
|
||||
// Server still starting — retry until budget expires.
|
||||
}
|
||||
await new Promise((r) => setTimeout(r, 100));
|
||||
}
|
||||
throw new Error(
|
||||
`serve did not become ready within ${STARTUP_BUDGET_MS}ms.\nstderr:\n${stderr}`,
|
||||
);
|
||||
}, 60_000);
|
||||
|
||||
afterAll(async () => {
|
||||
if (proc && !proc.killed) {
|
||||
proc.kill('SIGTERM');
|
||||
await new Promise<void>((resolve) => {
|
||||
const timer = setTimeout(() => {
|
||||
proc?.kill('SIGKILL');
|
||||
resolve();
|
||||
}, 3_000);
|
||||
proc?.on('exit', () => {
|
||||
clearTimeout(timer);
|
||||
resolve();
|
||||
});
|
||||
});
|
||||
}
|
||||
proc = undefined;
|
||||
if (homeDir) {
|
||||
fs.rmSync(homeDir, { recursive: true, force: true });
|
||||
homeDir = undefined;
|
||||
}
|
||||
});
|
||||
|
||||
it('rejects a GitHub token paired with a non-github host (finding 6)', async () => {
|
||||
const { status, body } = await postAnalyze(port, {
|
||||
url: 'https://gitlab.com/owner/repo',
|
||||
token: 'ghp_validformat123',
|
||||
});
|
||||
expect(status).toBe(400);
|
||||
expect(JSON.parse(body).error).toContain('only supported for github.com');
|
||||
});
|
||||
|
||||
it('rejects a GitHub token paired with an Azure DevOps URL (cross-credential trigger)', async () => {
|
||||
const { status, body } = await postAnalyze(port, {
|
||||
url: 'https://dev.azure.com/org/proj/_git/repo',
|
||||
token: 'ghp_validformat123',
|
||||
});
|
||||
expect(status).toBe(400);
|
||||
expect(JSON.parse(body).error).toContain('only supported for github.com');
|
||||
});
|
||||
|
||||
it('rejects a token whose characters could smuggle a header (CRLF/space)', async () => {
|
||||
const { status, body } = await postAnalyze(port, {
|
||||
url: 'https://github.com/owner/repo',
|
||||
token: 'bad token',
|
||||
});
|
||||
expect(status).toBe(400);
|
||||
expect(JSON.parse(body).error).toContain('invalid characters');
|
||||
});
|
||||
|
||||
it('rejects a token with no url (routed via a path so it reaches token validation)', async () => {
|
||||
const { status, body } = await postAnalyze(port, {
|
||||
path: '/tmp/gitnexus-nonexistent-abs-path',
|
||||
token: 'ghp_validformat123',
|
||||
});
|
||||
expect(status).toBe(400);
|
||||
expect(JSON.parse(body).error).toContain('requires "url"');
|
||||
});
|
||||
|
||||
it('still rejects a request with neither url nor path (route reachable, json parsed)', async () => {
|
||||
const { status, body } = await postAnalyze(port, {});
|
||||
expect(status).toBe(400);
|
||||
expect(JSON.parse(body).error).toContain('Provide');
|
||||
});
|
||||
});
|
||||
75
gitnexus/test/unit/api-analyze-token.test.ts
Normal file
75
gitnexus/test/unit/api-analyze-token.test.ts
Normal file
|
|
@ -0,0 +1,75 @@
|
|||
/**
|
||||
* Unit tests for validateAnalyzeToken — the optional `token` validation on
|
||||
* POST /api/analyze. Tested directly (not via a booted server) since the
|
||||
* route validation is otherwise only reachable through createServer.
|
||||
*
|
||||
* Closes the test gap (finding 12) and locks the github.com host-bind
|
||||
* (finding 6) and CRLF/charset guards from the PR #2223 tri-review.
|
||||
*/
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import { validateAnalyzeToken } from '../../src/server/api.js';
|
||||
|
||||
const GH = 'https://github.com/owner/repo';
|
||||
|
||||
describe('validateAnalyzeToken', () => {
|
||||
it('returns null when no token is provided', () => {
|
||||
expect(validateAnalyzeToken(undefined, GH)).toBeNull();
|
||||
expect(validateAnalyzeToken(undefined, undefined)).toBeNull();
|
||||
});
|
||||
|
||||
it('accepts a well-formed token for a github.com URL', () => {
|
||||
expect(validateAnalyzeToken('ghp_abc123', GH)).toBeNull();
|
||||
expect(validateAnalyzeToken('ghp_abc123', 'https://www.github.com/o/r')).toBeNull();
|
||||
});
|
||||
|
||||
it('rejects a non-string token', () => {
|
||||
expect(validateAnalyzeToken(123 as unknown, GH)).toEqual({
|
||||
status: 400,
|
||||
error: '"token" must be a string',
|
||||
});
|
||||
});
|
||||
|
||||
it('rejects an empty or over-long token', () => {
|
||||
expect(validateAnalyzeToken('', GH)?.error).toBe('"token" length must be between 1 and 256');
|
||||
expect(validateAnalyzeToken('a'.repeat(257), GH)?.error).toBe(
|
||||
'"token" length must be between 1 and 256',
|
||||
);
|
||||
});
|
||||
|
||||
it('rejects a token with characters that could smuggle a header (CRLF/space/colon)', () => {
|
||||
for (const bad of ['abc def', 'abc\r\nHost: x', 'x-access-token:abc', 'abc<script>']) {
|
||||
expect(validateAnalyzeToken(bad, GH)?.error).toBe('"token" contains invalid characters');
|
||||
}
|
||||
});
|
||||
|
||||
it('rejects a token without a url', () => {
|
||||
expect(validateAnalyzeToken('ghp_abc123', undefined)?.error).toBe('"token" requires "url"');
|
||||
expect(validateAnalyzeToken('ghp_abc123', '')?.error).toBe('"token" requires "url"');
|
||||
});
|
||||
|
||||
it('rejects a token for a non-github host (host-bind)', () => {
|
||||
for (const url of [
|
||||
'https://gitlab.com/o/r',
|
||||
'https://dev.azure.com/o/p/_git/r',
|
||||
'https://github.com.evil.com/o/r',
|
||||
'https://api.github.com/o/r',
|
||||
]) {
|
||||
expect(validateAnalyzeToken('ghp_abc123', url)?.error).toBe(
|
||||
'"token" is only supported for github.com URLs',
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
it('treats github.com@evil.com as the evil host (userinfo, not host)', () => {
|
||||
// new URL(...).hostname is evil.com here — must be rejected.
|
||||
expect(validateAnalyzeToken('ghp_abc123', 'https://github.com@evil.com/o/r')?.error).toBe(
|
||||
'"token" is only supported for github.com URLs',
|
||||
);
|
||||
});
|
||||
|
||||
it('rejects a token when the url is unparseable', () => {
|
||||
expect(validateAnalyzeToken('ghp_abc123', 'not-a-url')?.error).toBe(
|
||||
'"url" must be a valid URL when "token" is provided',
|
||||
);
|
||||
});
|
||||
});
|
||||
|
|
@ -1,12 +1,30 @@
|
|||
import { afterAll, beforeAll, describe, it, expect } from 'vitest';
|
||||
import { afterAll, beforeAll, describe, it, expect, vi } from 'vitest';
|
||||
|
||||
// The logger is a Proxy with no `set` trap, so vi.spyOn can't patch it.
|
||||
// Mock the module and expose `warn` as a countable spy (other levels no-op).
|
||||
const warnSpy = vi.fn();
|
||||
vi.mock('../../src/core/logger.js', () => ({
|
||||
logger: {
|
||||
debug: () => {},
|
||||
info: () => {},
|
||||
warn: (...args: unknown[]) => warnSpy(...args),
|
||||
error: () => {},
|
||||
trace: () => {},
|
||||
fatal: () => {},
|
||||
},
|
||||
}));
|
||||
|
||||
import {
|
||||
extractRepoName,
|
||||
getCloneDir,
|
||||
validateGitUrl,
|
||||
cloneOrPull,
|
||||
buildCloneArgs,
|
||||
buildGitEnv,
|
||||
normalizeGitUrlForCompare,
|
||||
assertRemoteMatchesRequestedUrl,
|
||||
isAzureDevOpsUrl,
|
||||
warnIfInsecureAzureConfig,
|
||||
} from '../../src/server/git-clone.js';
|
||||
import path from 'node:path';
|
||||
import os from 'node:os';
|
||||
|
|
@ -311,6 +329,142 @@ describe('git-clone', () => {
|
|||
// --depth must be before the `--` separator (it's an option, not a positional).
|
||||
expect(depthIdx).toBeLessThan(args.indexOf('--'));
|
||||
});
|
||||
|
||||
it('never embeds a token in argv: credentials travel through env, not URL', () => {
|
||||
// buildCloneArgs is URL-only; the credential must travel through env
|
||||
// (buildGitEnv) so it cannot appear in `ps auxww` or in command logs.
|
||||
const args = buildCloneArgs('https://github.com/owner/repo.git', '/safe/target');
|
||||
// No credential material in argv — assert on the credential markers
|
||||
// directly rather than substring-matching the host (which CodeQL flags
|
||||
// as incomplete URL sanitization, js/incomplete-url-substring).
|
||||
expect(args.some((a) => a.includes('ghp_'))).toBe(false);
|
||||
expect(args.some((a) => a.toLowerCase().includes('authorization'))).toBe(false);
|
||||
expect(args.some((a) => a.includes('extraHeader'))).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('buildGitEnv — token injection', () => {
|
||||
// The token MUST travel via GIT_CONFIG_* env vars (git ≥2.31), not via
|
||||
// argv or URL. This keeps it out of `ps`, shell history, and stderr.
|
||||
|
||||
it('passes through base env and sets prompt-suppression env vars', () => {
|
||||
const env = buildGitEnv({ FOO: 'bar' });
|
||||
expect(env.FOO).toBe('bar');
|
||||
expect(env.GIT_TERMINAL_PROMPT).toBe('0');
|
||||
expect(env.GIT_ASKPASS).toBeDefined();
|
||||
});
|
||||
|
||||
it('scrubs inherited git trace vars that would log the Authorization header', () => {
|
||||
const env = buildGitEnv({
|
||||
GIT_TRACE: '1',
|
||||
GIT_TRACE_CURL: '1',
|
||||
GIT_TRACE_PACKET: '1',
|
||||
GIT_CURL_VERBOSE: '1',
|
||||
});
|
||||
expect(env.GIT_TRACE).toBeUndefined();
|
||||
expect(env.GIT_TRACE_CURL).toBeUndefined();
|
||||
expect(env.GIT_TRACE_PACKET).toBeUndefined();
|
||||
expect(env.GIT_CURL_VERBOSE).toBeUndefined();
|
||||
});
|
||||
|
||||
it('does not set GIT_CONFIG_* env vars when no token is provided', () => {
|
||||
const env = buildGitEnv({});
|
||||
expect(env.GIT_CONFIG_COUNT).toBeUndefined();
|
||||
expect(env.GIT_CONFIG_KEY_0).toBeUndefined();
|
||||
expect(env.GIT_CONFIG_VALUE_0).toBeUndefined();
|
||||
});
|
||||
|
||||
it('also leaves GIT_CONFIG_* unset when token is empty string', () => {
|
||||
const env = buildGitEnv({}, { token: '' });
|
||||
expect(env.GIT_CONFIG_COUNT).toBeUndefined();
|
||||
expect(env.GIT_CONFIG_KEY_0).toBeUndefined();
|
||||
expect(env.GIT_CONFIG_VALUE_0).toBeUndefined();
|
||||
});
|
||||
|
||||
it('injects a host-scoped Basic-auth header when a github.com token is provided', () => {
|
||||
const env = buildGitEnv({}, { token: 'ghp_secret123', url: 'https://github.com/owner/repo' });
|
||||
expect(env.GIT_CONFIG_COUNT).toBe('1');
|
||||
// Host-scoped key: the header attaches only to this origin's requests.
|
||||
expect(env.GIT_CONFIG_KEY_0).toBe('http.https://github.com/owner/repo.extraHeader');
|
||||
const expected =
|
||||
'Authorization: Basic ' + Buffer.from('x-access-token:ghp_secret123').toString('base64');
|
||||
expect(env.GIT_CONFIG_VALUE_0).toBe(expected);
|
||||
});
|
||||
|
||||
it('does not inject a token for a non-github host (defense-in-depth host bind)', () => {
|
||||
const env = buildGitEnv({}, { token: 'ghp_secret123', url: 'https://gitlab.com/owner/repo' });
|
||||
expect(env.GIT_CONFIG_COUNT).toBeUndefined();
|
||||
});
|
||||
|
||||
it('never includes the raw token value in any env entry', () => {
|
||||
// Defence-in-depth: token must only appear inside the base64 of the
|
||||
// Authorization header, never as a plain substring of any env var.
|
||||
const token = 'ghp_uniqueRawSecret_98765';
|
||||
const env = buildGitEnv({ EXISTING: 'value' }, { token, url: 'https://github.com/o/r' });
|
||||
for (const [key, value] of Object.entries(env)) {
|
||||
if (key === 'GIT_CONFIG_VALUE_0') continue;
|
||||
expect(String(value)).not.toContain(token);
|
||||
}
|
||||
});
|
||||
|
||||
it('injects the server Azure PAT (host-scoped) for an Azure URL', () => {
|
||||
const prev = process.env.AZURE_DEVOPS_PAT;
|
||||
process.env.AZURE_DEVOPS_PAT = 'azure-pat-xyz';
|
||||
try {
|
||||
const env = buildGitEnv({}, { url: 'https://dev.azure.com/org/proj/_git/repo' });
|
||||
expect(env.GIT_CONFIG_COUNT).toBe('1');
|
||||
expect(env.GIT_CONFIG_KEY_0).toBe(
|
||||
'http.https://dev.azure.com/org/proj/_git/repo.extraHeader',
|
||||
);
|
||||
const expected = 'Authorization: Basic ' + Buffer.from(':azure-pat-xyz').toString('base64');
|
||||
expect(env.GIT_CONFIG_VALUE_0).toBe(expected);
|
||||
} finally {
|
||||
if (prev === undefined) delete process.env.AZURE_DEVOPS_PAT;
|
||||
else process.env.AZURE_DEVOPS_PAT = prev;
|
||||
}
|
||||
});
|
||||
|
||||
it('emits EXACTLY ONE header when a github token and an Azure PAT could both apply', () => {
|
||||
// A token only injects for github.com (where isAzureDevOpsUrl is false),
|
||||
// so the two never collide — guard the resolver directly anyway.
|
||||
const prev = process.env.AZURE_DEVOPS_PAT;
|
||||
process.env.AZURE_DEVOPS_PAT = 'azure-pat-xyz';
|
||||
try {
|
||||
const env = buildGitEnv({}, { token: 'ghp_secret123', url: 'https://github.com/o/r' });
|
||||
expect(env.GIT_CONFIG_COUNT).toBe('1');
|
||||
expect(env.GIT_CONFIG_VALUE_1).toBeUndefined();
|
||||
for (const value of Object.values(env)) {
|
||||
expect(String(value)).not.toContain('azure-pat-xyz');
|
||||
}
|
||||
} finally {
|
||||
if (prev === undefined) delete process.env.AZURE_DEVOPS_PAT;
|
||||
else process.env.AZURE_DEVOPS_PAT = prev;
|
||||
}
|
||||
});
|
||||
|
||||
it('appends after an existing GIT_CONFIG_COUNT instead of overwriting it', () => {
|
||||
const env = buildGitEnv(
|
||||
{ GIT_CONFIG_COUNT: '1', GIT_CONFIG_KEY_0: 'http.sslVerify', GIT_CONFIG_VALUE_0: 'true' },
|
||||
{ token: 'ghp_secret123', url: 'https://github.com/o/r' },
|
||||
);
|
||||
expect(env.GIT_CONFIG_COUNT).toBe('2');
|
||||
// Operator's pre-existing config is preserved at index 0.
|
||||
expect(env.GIT_CONFIG_KEY_0).toBe('http.sslVerify');
|
||||
expect(env.GIT_CONFIG_VALUE_0).toBe('true');
|
||||
// Our credential is appended at index 1.
|
||||
expect(env.GIT_CONFIG_KEY_1).toBe('http.https://github.com/o/r.extraHeader');
|
||||
expect(env.GIT_CONFIG_VALUE_1).toContain('Authorization: Basic ');
|
||||
});
|
||||
|
||||
it('strips control characters from the config key (no key injection)', () => {
|
||||
const env = buildGitEnv(
|
||||
{},
|
||||
{ token: 'ghp_secret123', url: 'https://github.com/o/r%0Anewline' },
|
||||
);
|
||||
const key = env.GIT_CONFIG_KEY_0 ?? '';
|
||||
expect(key).not.toContain('\n');
|
||||
expect(key).not.toContain('\r');
|
||||
});
|
||||
});
|
||||
|
||||
describe('cloneOrPull — containment barrier', () => {
|
||||
|
|
@ -374,6 +528,128 @@ describe('git-clone', () => {
|
|||
});
|
||||
});
|
||||
|
||||
describe('isAzureDevOpsUrl', () => {
|
||||
it('recognizes dev.azure.com (cloud)', () => {
|
||||
expect(isAzureDevOpsUrl('https://dev.azure.com/org/project/_git/repo')).toBe(true);
|
||||
});
|
||||
|
||||
it('recognizes *.visualstudio.com (cloud legacy)', () => {
|
||||
expect(isAzureDevOpsUrl('https://myorg.visualstudio.com/project/_git/repo')).toBe(true);
|
||||
});
|
||||
|
||||
it('returns false for github.com', () => {
|
||||
expect(isAzureDevOpsUrl('https://github.com/user/repo')).toBe(false);
|
||||
});
|
||||
|
||||
it('returns false for gitlab.com', () => {
|
||||
expect(isAzureDevOpsUrl('https://gitlab.com/user/repo')).toBe(false);
|
||||
});
|
||||
|
||||
it('returns false for invalid URL', () => {
|
||||
expect(isAzureDevOpsUrl('not-a-url')).toBe(false);
|
||||
});
|
||||
|
||||
it('normalizes a trailing-dot FQDN (dev.azure.com.)', () => {
|
||||
expect(isAzureDevOpsUrl('https://dev.azure.com./org/proj/_git/repo')).toBe(true);
|
||||
expect(isAzureDevOpsUrl('https://myorg.visualstudio.com./project/_git/repo')).toBe(true);
|
||||
});
|
||||
|
||||
it('does not over-match a lookalike host with a trailing label', () => {
|
||||
expect(isAzureDevOpsUrl('https://dev.azure.com.evil.com/org/proj/_git/repo')).toBe(false);
|
||||
expect(isAzureDevOpsUrl('https://evilvisualstudio.com/project/_git/repo')).toBe(false);
|
||||
});
|
||||
|
||||
it('recognizes a self-hosted host configured via AZURE_DEVOPS_URL', () => {
|
||||
const prev = process.env.AZURE_DEVOPS_URL;
|
||||
try {
|
||||
process.env.AZURE_DEVOPS_URL = 'http://tfs.corp.example';
|
||||
expect(isAzureDevOpsUrl('http://tfs.corp.example/Coll/Proj/_git/Repo')).toBe(true);
|
||||
expect(isAzureDevOpsUrl('https://other.host.example/Coll/Proj/_git/Repo')).toBe(false);
|
||||
} finally {
|
||||
if (prev === undefined) delete process.env.AZURE_DEVOPS_URL;
|
||||
else process.env.AZURE_DEVOPS_URL = prev;
|
||||
}
|
||||
});
|
||||
|
||||
it('falls through to the cloud check when AZURE_DEVOPS_URL is invalid', () => {
|
||||
const prev = process.env.AZURE_DEVOPS_URL;
|
||||
try {
|
||||
process.env.AZURE_DEVOPS_URL = 'not-a-url';
|
||||
expect(isAzureDevOpsUrl('https://dev.azure.com/org/proj/_git/repo')).toBe(true);
|
||||
} finally {
|
||||
if (prev === undefined) delete process.env.AZURE_DEVOPS_URL;
|
||||
else process.env.AZURE_DEVOPS_URL = prev;
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('cleartext-credential warnings', () => {
|
||||
it('warns when injecting a credential over cleartext http://', () => {
|
||||
warnSpy.mockClear();
|
||||
buildGitEnv({}, { token: 'ghp_x', url: 'http://github.com/o/r' });
|
||||
expect(warnSpy).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('does not warn when injecting over https://', () => {
|
||||
warnSpy.mockClear();
|
||||
buildGitEnv({}, { token: 'ghp_x', url: 'https://github.com/o/r' });
|
||||
expect(warnSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('does not warn over http:// when no credential is injected', () => {
|
||||
warnSpy.mockClear();
|
||||
// gitlab host is not in the token allowlist and no Azure PAT is set,
|
||||
// so nothing is injected — and nothing is warned about.
|
||||
buildGitEnv({}, { token: 'ghp_x', url: 'http://gitlab.com/o/r' });
|
||||
expect(warnSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('warnIfInsecureAzureConfig warns for http AZURE_DEVOPS_URL, not https', () => {
|
||||
const prev = process.env.AZURE_DEVOPS_URL;
|
||||
warnSpy.mockClear();
|
||||
try {
|
||||
process.env.AZURE_DEVOPS_URL = 'http://tfs.corp.example';
|
||||
warnIfInsecureAzureConfig();
|
||||
expect(warnSpy).toHaveBeenCalledTimes(1);
|
||||
warnSpy.mockClear();
|
||||
process.env.AZURE_DEVOPS_URL = 'https://tfs.corp.example';
|
||||
warnIfInsecureAzureConfig();
|
||||
expect(warnSpy).not.toHaveBeenCalled();
|
||||
} finally {
|
||||
if (prev === undefined) delete process.env.AZURE_DEVOPS_URL;
|
||||
else process.env.AZURE_DEVOPS_URL = prev;
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('extractRepoName — Azure DevOps URLs', () => {
|
||||
it('extracts name from self-hosted Azure DevOps URL', () => {
|
||||
expect(
|
||||
extractRepoName('http://azuredevops.example.com/DefaultCollection/MyProject/_git/MyRepo'),
|
||||
).toBe('MyRepo');
|
||||
});
|
||||
|
||||
it('extracts name from dev.azure.com URL', () => {
|
||||
expect(extractRepoName('https://dev.azure.com/org/project/_git/myrepo')).toBe('myrepo');
|
||||
});
|
||||
|
||||
it('extracts name from visualstudio.com URL', () => {
|
||||
expect(extractRepoName('https://myorg.visualstudio.com/project/_git/myrepo')).toBe('myrepo');
|
||||
});
|
||||
});
|
||||
|
||||
describe('validateGitUrl — Azure DevOps URLs', () => {
|
||||
it('allows self-hosted Azure DevOps Server URLs', () => {
|
||||
expect(() =>
|
||||
validateGitUrl('http://azuredevops.example.com/DefaultCollection/Project/_git/Repo'),
|
||||
).not.toThrow();
|
||||
});
|
||||
|
||||
it('allows dev.azure.com URLs', () => {
|
||||
expect(() => validateGitUrl('https://dev.azure.com/org/project/_git/repo')).not.toThrow();
|
||||
});
|
||||
});
|
||||
|
||||
describe('normalizeGitUrlForCompare', () => {
|
||||
it('strips trailing .git', () => {
|
||||
expect(normalizeGitUrlForCompare('https://github.com/owner/repo.git')).toBe(
|
||||
|
|
|
|||
|
|
@ -5,7 +5,12 @@
|
|||
* 8.3 short-name form before passing them to KuzuDB's native layer.
|
||||
*/
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import { toNativeSafePath, cleanupNativePathJunctions } from '../../src/core/lbug/lbug-config.js';
|
||||
import path from 'path';
|
||||
import {
|
||||
toNativeSafePath,
|
||||
cleanupNativePathJunctions,
|
||||
resolveNativeSafeStorageDir,
|
||||
} from '../../src/core/lbug/lbug-config.js';
|
||||
|
||||
describe('toNativeSafePath', () => {
|
||||
it('returns ASCII paths unchanged on any platform', () => {
|
||||
|
|
@ -61,3 +66,52 @@ describe('toNativeSafePath', () => {
|
|||
});
|
||||
}
|
||||
});
|
||||
|
||||
describe('resolveNativeSafeStorageDir (#2202)', () => {
|
||||
it('returns <storage>/<subdir> for an ASCII storage path on any platform', () => {
|
||||
const storage = path.join('repo', '.gitnexus');
|
||||
expect(resolveNativeSafeStorageDir(storage, 'csv')).toBe(path.join(storage, 'csv'));
|
||||
expect(resolveNativeSafeStorageDir(storage, 'pdg-csv')).toBe(path.join(storage, 'pdg-csv'));
|
||||
});
|
||||
|
||||
it('keeps csv and pdg-csv distinct (no collision between structural and streamed dirs)', () => {
|
||||
const storage = path.join('repo', '.gitnexus');
|
||||
expect(resolveNativeSafeStorageDir(storage, 'csv')).not.toBe(
|
||||
resolveNativeSafeStorageDir(storage, 'pdg-csv'),
|
||||
);
|
||||
});
|
||||
|
||||
if (process.platform !== 'win32') {
|
||||
it('does NOT relocate a non-ASCII storage path off Windows (platform gate)', () => {
|
||||
const storage = path.join('repo', '用户', '.gitnexus');
|
||||
// Non-win32: the relocation never fires regardless of non-ASCII chars.
|
||||
expect(resolveNativeSafeStorageDir(storage, 'pdg-csv')).toBe(path.join(storage, 'pdg-csv'));
|
||||
});
|
||||
}
|
||||
|
||||
if (process.platform === 'win32') {
|
||||
it('relocates a non-ASCII storage path to a unique mkdtemp os.tmpdir() dir per subdir', () => {
|
||||
const fs = require('fs');
|
||||
const storage = 'C:\\Project\\中文\\.gitnexus';
|
||||
// mkdtemp creates the dirs — track + clean them up.
|
||||
const csv = resolveNativeSafeStorageDir(storage, 'csv');
|
||||
const pdg = resolveNativeSafeStorageDir(storage, 'pdg-csv');
|
||||
try {
|
||||
// Both relocated under os.tmpdir(), ASCII-prefixed, and distinct (each
|
||||
// mkdtemp call returns a fresh random suffix — never a predictable name).
|
||||
expect(pdg.includes('gitnexus-pdg-csv-')).toBe(true);
|
||||
expect(csv.includes('gitnexus-csv-')).toBe(true);
|
||||
expect(csv).not.toBe(pdg);
|
||||
// Two calls for the same (storage, subdir) yield DIFFERENT dirs (random).
|
||||
const csv2 = resolveNativeSafeStorageDir(storage, 'csv');
|
||||
expect(csv2).not.toBe(csv);
|
||||
// The relocated paths are not under the original non-ASCII storage path.
|
||||
expect(pdg.includes('中文')).toBe(false);
|
||||
fs.rmSync(csv2, { recursive: true, force: true });
|
||||
} finally {
|
||||
fs.rmSync(csv, { recursive: true, force: true });
|
||||
fs.rmSync(pdg, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
|
|
|
|||
331
gitnexus/test/unit/lbug/pdg-emit-sink.test.ts
Normal file
331
gitnexus/test/unit/lbug/pdg-emit-sink.test.ts
Normal file
|
|
@ -0,0 +1,331 @@
|
|||
/**
|
||||
* PdgEmitSink unit tests (issue #2202 U2).
|
||||
*
|
||||
* Verifies the streaming PDG emit sink:
|
||||
* - routes BasicBlock nodes + PDG edges to bounded CSV-on-disk;
|
||||
* - delegates structural nodes/edges + the whole-program TAINT_PATH edge to
|
||||
* the real graph (never streamed);
|
||||
* - is byte-identical (set-wise) to the whole-graph `streamAllCSVsToDisk`
|
||||
* emit for the same node/edge set (the issue's byte-identity acceptance);
|
||||
* - never accumulates the PDG layer in the in-memory graph (the RSS bound).
|
||||
*/
|
||||
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
|
||||
import fs from 'node:fs';
|
||||
import fsp from 'node:fs/promises';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
|
||||
import { createKnowledgeGraph } from '../../../src/core/graph/graph.js';
|
||||
import { streamAllCSVsToDisk, buildBasicBlockRow } from '../../../src/core/lbug/csv-generator.js';
|
||||
import { PdgEmitSink } from '../../../src/core/lbug/pdg-emit-sink.js';
|
||||
import type { GraphNode, GraphRelationship } from 'gitnexus-shared';
|
||||
|
||||
const bbNode = (fp: string, idx: number, line: number): GraphNode => ({
|
||||
id: `BasicBlock:${fp}:1:0:${idx}`,
|
||||
label: 'BasicBlock',
|
||||
properties: { name: '', filePath: fp, startLine: line, endLine: line + 1, text: `blk ${idx}` },
|
||||
});
|
||||
|
||||
const pdgEdge = (
|
||||
fp: string,
|
||||
from: number,
|
||||
to: number,
|
||||
type: GraphRelationship['type'],
|
||||
reason: string,
|
||||
): GraphRelationship => ({
|
||||
id: `${type}:${fp}:${from}->${to}`,
|
||||
sourceId: `BasicBlock:${fp}:1:0:${from}`,
|
||||
targetId: `BasicBlock:${fp}:1:0:${to}`,
|
||||
type,
|
||||
confidence: 1,
|
||||
reason,
|
||||
});
|
||||
|
||||
/** Sorted non-empty lines of a CSV file (order-independent comparison). */
|
||||
const sortedLines = async (csvPath: string): Promise<string[]> => {
|
||||
const text = await fsp.readFile(csvPath, 'utf8');
|
||||
return text
|
||||
.split('\n')
|
||||
.filter((l) => l.length > 0)
|
||||
.sort();
|
||||
};
|
||||
|
||||
let tmpRoot: string;
|
||||
|
||||
beforeEach(() => {
|
||||
tmpRoot = fs.mkdtempSync(path.join(os.tmpdir(), 'pdg-sink-'));
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
fs.rmSync(tmpRoot, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
describe('PdgEmitSink — routing', () => {
|
||||
it('routes BasicBlock nodes and PDG edges to CSV, never to the real graph', () => {
|
||||
const real = createKnowledgeGraph();
|
||||
const sink = new PdgEmitSink(real, path.join(tmpRoot, 'pdg-csv'));
|
||||
|
||||
sink.addNode(bbNode('a.ts', 0, 1));
|
||||
sink.addNode(bbNode('a.ts', 1, 5));
|
||||
sink.addRelationship(pdgEdge('a.ts', 0, 1, 'CFG', 'seq'));
|
||||
sink.addRelationship(pdgEdge('a.ts', 0, 1, 'REACHING_DEF', 'x:1:0'));
|
||||
|
||||
// PDG layer must not land in the in-memory graph (the RSS bound).
|
||||
expect(real.nodeCount).toBe(0);
|
||||
expect(real.relationshipCount).toBe(0);
|
||||
expect(sink.nodeCount).toBe(0);
|
||||
|
||||
sink.finalize();
|
||||
});
|
||||
|
||||
it('delegates structural nodes, CALLS, and the whole-program TAINT_PATH edge to the real graph', () => {
|
||||
const real = createKnowledgeGraph();
|
||||
const sink = new PdgEmitSink(real, path.join(tmpRoot, 'pdg-csv'));
|
||||
|
||||
sink.addNode({
|
||||
id: 'Function:a.ts:fn:1',
|
||||
label: 'Function',
|
||||
properties: { name: 'fn', filePath: 'a.ts', startLine: 1, endLine: 9 },
|
||||
});
|
||||
sink.addRelationship({
|
||||
id: 'CALLS:1',
|
||||
sourceId: 'Function:a.ts:fn:1',
|
||||
targetId: 'Function:a.ts:fn2:9',
|
||||
type: 'CALLS',
|
||||
confidence: 1,
|
||||
reason: '',
|
||||
});
|
||||
// TAINT_PATH is a whole-program (Function→Function) edge — NOT streamed.
|
||||
sink.addRelationship({
|
||||
id: 'TAINT_PATH:1',
|
||||
sourceId: 'Function:a.ts:fn:1',
|
||||
targetId: 'Function:a.ts:fn2:9',
|
||||
type: 'TAINT_PATH',
|
||||
confidence: 0.9,
|
||||
reason: 'src->sink',
|
||||
});
|
||||
|
||||
expect(real.nodeCount).toBe(1);
|
||||
expect(real.relationshipCount).toBe(2);
|
||||
expect(real.getNode('Function:a.ts:fn:1')).toBeDefined();
|
||||
|
||||
const manifest = sink.finalize();
|
||||
// No BasicBlock node CSV was created (no BasicBlock nodes were routed).
|
||||
expect(manifest.nodeFiles.size).toBe(0);
|
||||
expect(manifest.relsByPair.size).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('PdgEmitSink — byte-identity vs whole-graph emit', () => {
|
||||
it('streamed CSV line set equals streamAllCSVsToDisk for the same nodes/edges', async () => {
|
||||
const fp = 'a.ts';
|
||||
const nodes = [bbNode(fp, 0, 1), bbNode(fp, 1, 5), bbNode(fp, 2, 9)];
|
||||
const edges: GraphRelationship[] = [
|
||||
pdgEdge(fp, 0, 1, 'CFG', 'seq'),
|
||||
pdgEdge(fp, 1, 2, 'CFG', 'cond-true'),
|
||||
pdgEdge(fp, 0, 2, 'REACHING_DEF', 'x:1:0'),
|
||||
pdgEdge(fp, 1, 2, 'CDG', 'T'),
|
||||
pdgEdge(fp, 0, 1, 'POST_DOMINATE', ''),
|
||||
pdgEdge(fp, 0, 2, 'TAINTED', 'taint'),
|
||||
pdgEdge(fp, 1, 2, 'SANITIZES', 'clean'),
|
||||
];
|
||||
|
||||
// Whole-graph path: add to a plain graph, run streamAllCSVsToDisk.
|
||||
const wholeGraph = createKnowledgeGraph();
|
||||
for (const n of nodes) wholeGraph.addNode(n);
|
||||
for (const e of edges) wholeGraph.addRelationship(e);
|
||||
const wholeDir = path.join(tmpRoot, 'csv');
|
||||
await streamAllCSVsToDisk(wholeGraph, path.join(tmpRoot, 'no-such-repo'), wholeDir);
|
||||
|
||||
// Streamed path: route the same set through the sink.
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir);
|
||||
for (const n of nodes) sink.addNode(n);
|
||||
for (const e of edges) sink.addRelationship(e);
|
||||
const manifest = sink.finalize();
|
||||
|
||||
// BasicBlock node CSV: identical line set.
|
||||
expect(await sortedLines(path.join(pdgDir, 'basicblock.csv'))).toEqual(
|
||||
await sortedLines(path.join(wholeDir, 'basicblock.csv')),
|
||||
);
|
||||
|
||||
// PDG edges all route to the BasicBlock|BasicBlock pair file: identical set.
|
||||
expect(await sortedLines(path.join(pdgDir, 'rel_BasicBlock_BasicBlock.csv'))).toEqual(
|
||||
await sortedLines(path.join(wholeDir, 'rel_BasicBlock_BasicBlock.csv')),
|
||||
);
|
||||
|
||||
// Manifest reports the streamed files + row counts.
|
||||
expect(manifest.nodeFiles.get('BasicBlock')?.rows).toBe(nodes.length);
|
||||
expect(manifest.relsByPair.get('BasicBlock|BasicBlock')?.rows).toBe(edges.length);
|
||||
});
|
||||
|
||||
it('emits rows via the shared builder (buildBasicBlockRow)', async () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir);
|
||||
const n = bbNode('a.ts', 0, 3);
|
||||
sink.addNode(n);
|
||||
sink.finalize();
|
||||
const lines = await sortedLines(path.join(pdgDir, 'basicblock.csv'));
|
||||
// header + one data row; the data row is exactly buildBasicBlockRow(n).
|
||||
expect(lines).toContain(buildBasicBlockRow(n));
|
||||
});
|
||||
});
|
||||
|
||||
describe('PdgEmitSink — bounded retention', () => {
|
||||
it('flushes incrementally so the graph never holds the PDG layer', async () => {
|
||||
const real = createKnowledgeGraph();
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const CHUNK = 2;
|
||||
const sink = new PdgEmitSink(real, pdgDir, CHUNK); // tiny chunk to force flushes
|
||||
|
||||
// TOTAL is intentionally NOT a multiple of CHUNK so the final partial chunk
|
||||
// is genuinely still buffered (unflushed) at the mid-stream read. With a
|
||||
// multiple (e.g. 50 % 2 === 0) the last addRow's flush would have written
|
||||
// every row and the "mid-stream" assertion would prove nothing (#2202
|
||||
// review #7).
|
||||
const TOTAL = 51;
|
||||
const REMAINDER = TOTAL % CHUNK; // 1 — must be non-zero
|
||||
expect(REMAINDER).toBeGreaterThan(0);
|
||||
for (let i = 0; i < TOTAL; i++) sink.addNode(bbNode('a.ts', i, i));
|
||||
|
||||
// Mid-stream (before finalize): exactly the whole flushed chunks are on
|
||||
// disk; the partial last chunk (REMAINDER rows) is still buffered in memory,
|
||||
// proving the writer streams to the OS and never buffers the whole layer.
|
||||
const midText = fs.readFileSync(path.join(pdgDir, 'basicblock.csv'), 'utf8');
|
||||
const midDataRows = midText.split('\n').filter((l) => l.length > 0).length - 1; // minus header
|
||||
expect(midDataRows).toBe(TOTAL - REMAINDER); // 50 flushed, 1 still buffered
|
||||
expect(TOTAL - midDataRows).toBe(REMAINDER); // exactly the unflushed remainder
|
||||
expect(TOTAL - midDataRows).toBeLessThanOrEqual(CHUNK); // unflushed is bounded by one chunk
|
||||
|
||||
// The in-memory graph never received a single BasicBlock.
|
||||
expect(real.nodeCount).toBe(0);
|
||||
|
||||
const manifest = sink.finalize();
|
||||
expect(manifest.nodeFiles.get('BasicBlock')?.rows).toBe(TOTAL);
|
||||
const finalRows = (await sortedLines(path.join(pdgDir, 'basicblock.csv'))).length - 1;
|
||||
expect(finalRows).toBe(TOTAL); // finalize flushed the buffered remainder
|
||||
});
|
||||
|
||||
it('finalize twice throws', () => {
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), path.join(tmpRoot, 'pdg-csv'));
|
||||
sink.finalize();
|
||||
expect(() => sink.finalize()).toThrow(/twice/);
|
||||
});
|
||||
});
|
||||
|
||||
describe('PdgEmitSink — pass-through contract (dedup is the caller’s)', () => {
|
||||
// The sink does NOT dedup by id (that would retain every id → O(total ids)
|
||||
// memory, undermining the O(chunk) bound). Cross-pass dedup is done upstream,
|
||||
// per file, in run.ts (a file imported by two language passes is emitted
|
||||
// once). The sink is a faithful pass-through: it writes every id it is given
|
||||
// and must not be fed duplicates. See #2202 finding #1 + the run-loop / Vue+TS
|
||||
// integration coverage for the cross-pass dedup itself.
|
||||
it('writes every BasicBlock it is given (no id dedup in the sink)', async () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir);
|
||||
const n = bbNode('a.ts', 0, 1);
|
||||
sink.addNode(n);
|
||||
sink.addNode(n); // sink does not dedup — both rows are written
|
||||
const manifest = sink.finalize();
|
||||
expect(manifest.nodeFiles.get('BasicBlock')?.rows).toBe(2);
|
||||
expect((await sortedLines(path.join(pdgDir, 'basicblock.csv'))).length - 1).toBe(2);
|
||||
});
|
||||
|
||||
it('writes every PDG edge it is given (no id dedup in the sink)', () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir);
|
||||
const e = pdgEdge('a.ts', 0, 1, 'CFG', 'seq');
|
||||
sink.addRelationship(e);
|
||||
sink.addRelationship(e);
|
||||
const manifest = sink.finalize();
|
||||
expect(manifest.relsByPair.get('BasicBlock|BasicBlock')?.rows).toBe(2);
|
||||
});
|
||||
|
||||
it('skips a PDG edge whose endpoint label is not a node table', () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir);
|
||||
// sourceId prefix "Bogus" is not in NODE_TABLES → skipped (mirrors RelPairRouter).
|
||||
sink.addRelationship({
|
||||
id: 'CFG:bogus',
|
||||
sourceId: 'Bogus:a.ts:0',
|
||||
targetId: 'BasicBlock:a.ts:1:0:1',
|
||||
type: 'CFG',
|
||||
confidence: 1,
|
||||
reason: 'seq',
|
||||
});
|
||||
const manifest = sink.finalize();
|
||||
expect(manifest.relsByPair.size).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('PdgEmitSink — IO failure poisoning (#2202 review #4/#6)', () => {
|
||||
// A streamed-write failure is an IO fault, not the CFG-logic error the emit
|
||||
// loop's per-file try/catch is built to swallow. The sink poisons the failing
|
||||
// writer (or records an open failure) so finalize fails loudly instead of
|
||||
// returning a truncated manifest that the bulk COPY would silently load.
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
it('poisons the writer on a mid-stream write failure (disk-full) → finalize throws', () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir, 1); // flush every row
|
||||
|
||||
// The writer opens fine (real openSync); the flush's writeSync fails.
|
||||
const spy = vi.spyOn(fs, 'writeSync').mockImplementation(() => {
|
||||
throw new Error('ENOSPC: no space left on device');
|
||||
});
|
||||
// The throw propagates to the immediate caller (the emit loop, which would
|
||||
// swallow it as a per-file CFG error) — that is exactly why finalize must
|
||||
// re-check poison below.
|
||||
expect(() => sink.addNode(bbNode('a.ts', 0, 1))).toThrow(/ENOSPC/);
|
||||
spy.mockRestore();
|
||||
|
||||
expect(() => sink.finalize()).toThrow(/IO error|ENOSPC/);
|
||||
});
|
||||
|
||||
it('records an openSync failure (EMFILE) → finalize throws even if the caller swallowed it', () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv'); // ctor mkdir/rm run before the spy
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir);
|
||||
|
||||
const spy = vi.spyOn(fs, 'openSync').mockImplementation(() => {
|
||||
throw new Error('EMFILE: too many open files');
|
||||
});
|
||||
expect(() => sink.addNode(bbNode('a.ts', 0, 1))).toThrow(/EMFILE/);
|
||||
spy.mockRestore();
|
||||
|
||||
expect(() => sink.finalize()).toThrow(/IO error|EMFILE/);
|
||||
});
|
||||
|
||||
it('surfaces a final-flush IO failure from finalize (rows buffered, never mid-flushed)', () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir, 1000); // big chunk → no mid flush
|
||||
sink.addNode(bbNode('a.ts', 0, 1)); // buffered only (1 < 1000)
|
||||
|
||||
// The only writeSync happens during the final flush inside finalize → close.
|
||||
const spy = vi.spyOn(fs, 'writeSync').mockImplementation(() => {
|
||||
throw new Error('ENOSPC: disk full at close');
|
||||
});
|
||||
// close() never throws (it records poison); finalize reports it.
|
||||
expect(() => sink.finalize()).toThrow(/IO error|ENOSPC/);
|
||||
spy.mockRestore();
|
||||
});
|
||||
|
||||
it('a poisoned writer stops accepting rows (no unbounded buffering on a dead fd)', () => {
|
||||
const pdgDir = path.join(tmpRoot, 'pdg-csv');
|
||||
const sink = new PdgEmitSink(createKnowledgeGraph(), pdgDir, 1);
|
||||
|
||||
const spy = vi.spyOn(fs, 'writeSync').mockImplementation(() => {
|
||||
throw new Error('ENOSPC');
|
||||
});
|
||||
expect(() => sink.addNode(bbNode('a.ts', 0, 1))).toThrow(/ENOSPC/); // poisons the writer
|
||||
// Subsequent rows are dropped silently at the writer (it is dead) — they do
|
||||
// not re-throw and do not accumulate; the run still fails at finalize.
|
||||
expect(() => sink.addNode(bbNode('a.ts', 1, 2))).not.toThrow();
|
||||
expect(() => sink.addNode(bbNode('a.ts', 2, 3))).not.toThrow();
|
||||
spy.mockRestore();
|
||||
|
||||
expect(() => sink.finalize()).toThrow(/IO error|ENOSPC/);
|
||||
});
|
||||
});
|
||||
176
gitnexus/test/unit/rel-pair-routing.test.ts
Normal file
176
gitnexus/test/unit/rel-pair-routing.test.ts
Normal file
|
|
@ -0,0 +1,176 @@
|
|||
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
||||
import { EventEmitter } from 'events';
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
import { RelPairRouter, getNodeLabel } from '../../src/core/lbug/rel-pair-routing.js';
|
||||
|
||||
/**
|
||||
* Unit tests for RelPairRouter (#2203 U2) — the production per-pair emit path.
|
||||
*
|
||||
* Mirrors test/unit/rel-csv-split.test.ts: drives the router with an injected
|
||||
* mock WriteStream factory so the error, backpressure, and teardown paths are
|
||||
* exercised without LadybugDB or real disk streams. These paths are otherwise
|
||||
* unreachable in the integration suite (which only hits the no-backpressure
|
||||
* happy path), so this is the coverage for the router's failure modes.
|
||||
*/
|
||||
|
||||
// Controllable backpressure + error injection (same shape as the split oracle's mock).
|
||||
class MockWriteStream extends EventEmitter {
|
||||
public chunks: string[] = [];
|
||||
public destroyed = false;
|
||||
public ended = false;
|
||||
public blocked = false;
|
||||
public maxDrainListenersSeen = 0;
|
||||
// State flags + events so `stream/promises.finished(ws)` (used by the
|
||||
// router's close()) resolves against this mock instead of hanging.
|
||||
public writable = true;
|
||||
public writableEnded = false;
|
||||
public writableFinished = false;
|
||||
|
||||
write(chunk: string): boolean {
|
||||
this.chunks.push(chunk);
|
||||
const count = this.listenerCount('drain');
|
||||
if (count > this.maxDrainListenersSeen) this.maxDrainListenersSeen = count;
|
||||
return !this.blocked;
|
||||
}
|
||||
|
||||
end(cb?: (err?: Error) => void): this {
|
||||
this.ended = true;
|
||||
this.writableEnded = true;
|
||||
this.writableFinished = true;
|
||||
this.writable = false;
|
||||
if (cb) cb();
|
||||
queueMicrotask(() => {
|
||||
this.emit('finish');
|
||||
this.emit('close');
|
||||
});
|
||||
return this;
|
||||
}
|
||||
|
||||
destroy(): this {
|
||||
this.destroyed = true;
|
||||
return this;
|
||||
}
|
||||
|
||||
unblock(): void {
|
||||
this.blocked = false;
|
||||
this.emit('drain');
|
||||
}
|
||||
|
||||
triggerError(err: Error): void {
|
||||
this.emit('error', err);
|
||||
}
|
||||
}
|
||||
|
||||
const HEADER = '"from","to","type","confidence","reason","step"';
|
||||
const VALID = new Set<string>(['File', 'Function', 'Community', 'Process']);
|
||||
|
||||
const row = (from: string, to: string, type = 'CALLS'): string =>
|
||||
`"${from}","${to}","${type}",1.0,"auto",0`;
|
||||
|
||||
function mockFactory(streams: MockWriteStream[], opts?: { blocked?: boolean }) {
|
||||
return (() => {
|
||||
const ws = new MockWriteStream();
|
||||
if (opts?.blocked) ws.blocked = true;
|
||||
streams.push(ws);
|
||||
return ws;
|
||||
}) as unknown as (filePath: string) => import('fs').WriteStream;
|
||||
}
|
||||
|
||||
let tmpDir: string;
|
||||
|
||||
beforeEach(() => {
|
||||
tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'rel-pair-routing-test-'));
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
fs.rmSync(tmpDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 50 });
|
||||
});
|
||||
|
||||
describe('getNodeLabel', () => {
|
||||
it('maps comm_/proc_ prefixes and otherwise splits on the first colon', () => {
|
||||
expect(getNodeLabel('comm_42')).toBe('Community');
|
||||
expect(getNodeLabel('proc_7')).toBe('Process');
|
||||
expect(getNodeLabel('Function:src/a.ts:f:1')).toBe('Function');
|
||||
expect(getNodeLabel('File:src/a.ts')).toBe('File');
|
||||
});
|
||||
});
|
||||
|
||||
describe('RelPairRouter', () => {
|
||||
it('routes valid edges to per-pair files (header first) and skips invalid-label edges', async () => {
|
||||
const streams: MockWriteStream[] = [];
|
||||
const router = new RelPairRouter(tmpDir, HEADER, VALID, mockFactory(streams));
|
||||
|
||||
const route = async (from: string, to: string) => {
|
||||
const p = router.route(from, to, row(from, to));
|
||||
if (p) await p;
|
||||
};
|
||||
await route('File:a', 'Function:a:f:1');
|
||||
await route('File:a', 'Function:a:g:2'); // same pair
|
||||
await route('Function:a:f:1', 'Function:a:g:2'); // different pair
|
||||
await route('Bogus:x', 'File:a'); // invalid FROM label → skipped
|
||||
await route('File:a', 'Bogus:y'); // invalid TO label → skipped (other branch)
|
||||
await router.close();
|
||||
|
||||
expect(router.skipped).toBe(2);
|
||||
expect(router.total).toBe(3);
|
||||
expect([...router.byPair.keys()].sort()).toEqual(['File|Function', 'Function|Function']);
|
||||
expect(router.byPair.get('File|Function')!.rows).toBe(2);
|
||||
// Header is the first chunk written to each pair stream.
|
||||
expect(streams[0].chunks[0]).toBe(HEADER + '\n');
|
||||
expect(streams.every((s) => s.ended)).toBe(true);
|
||||
});
|
||||
|
||||
it('returns a drain promise under backpressure and completes once unblocked', async () => {
|
||||
const streams: MockWriteStream[] = [];
|
||||
const router = new RelPairRouter(
|
||||
tmpDir,
|
||||
HEADER,
|
||||
VALID,
|
||||
mockFactory(streams, { blocked: true }),
|
||||
);
|
||||
|
||||
const pending = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1'));
|
||||
expect(pending).toBeInstanceOf(Promise); // header write hit backpressure
|
||||
streams[0].unblock();
|
||||
await pending;
|
||||
|
||||
expect(streams[0].maxDrainListenersSeen).toBeLessThanOrEqual(1);
|
||||
expect(streams[0].chunks[0]).toBe(HEADER + '\n');
|
||||
expect(router.total).toBe(1);
|
||||
});
|
||||
|
||||
it('on a stream error: route() throws the real error, lastError exposes it, close() rejects + destroys', async () => {
|
||||
const streams: MockWriteStream[] = [];
|
||||
const router = new RelPairRouter(tmpDir, HEADER, VALID, mockFactory(streams));
|
||||
|
||||
const first = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1'));
|
||||
if (first) await first;
|
||||
|
||||
const err = new Error('EMFILE: too many open files');
|
||||
streams[0].triggerError(err);
|
||||
|
||||
// The next route surfaces the REAL error, not a generic AbortError.
|
||||
expect(() => router.route('File:a', 'Function:a:g:2', row('File:a', 'Function:a:g:2'))).toThrow(
|
||||
'EMFILE',
|
||||
);
|
||||
expect(router.lastError).toBe(err);
|
||||
await expect(router.close()).rejects.toThrow('EMFILE');
|
||||
expect(streams[0].destroyed).toBe(true);
|
||||
});
|
||||
|
||||
it('destroy() tears down every open pair stream', async () => {
|
||||
const streams: MockWriteStream[] = [];
|
||||
const router = new RelPairRouter(tmpDir, HEADER, VALID, mockFactory(streams));
|
||||
|
||||
const a = router.route('File:a', 'Function:a:f:1', row('File:a', 'Function:a:f:1'));
|
||||
if (a) await a;
|
||||
const b = router.route('Community:1', 'Community:2', row('Community:1', 'Community:2'));
|
||||
if (b) await b;
|
||||
|
||||
router.destroy();
|
||||
expect(streams.length).toBe(2);
|
||||
expect(streams.every((s) => s.destroyed)).toBe(true);
|
||||
});
|
||||
});
|
||||
99
gitnexus/test/unit/stream-pdg-emit-config.test.ts
Normal file
99
gitnexus/test/unit/stream-pdg-emit-config.test.ts
Normal file
|
|
@ -0,0 +1,99 @@
|
|||
/**
|
||||
* Streaming PDG-emit config gating (issue #2202 U3).
|
||||
*
|
||||
* Verifies:
|
||||
* - `resolveStreamPdgEmit` engages only when pdg + full-rebuild (force) + an
|
||||
* enable signal (explicit option OR GITNEXUS_STREAM_PDG_EMIT) all hold;
|
||||
* - `resolvePdgEmitChunkSize` prefers the explicit option, falls back to
|
||||
* GITNEXUS_PDG_EMIT_CHUNK_SIZE, else undefined;
|
||||
* - the memory-only streaming knobs are NOT stamped into RepoMeta.pdg, so
|
||||
* changing them never trips `pdgModeMismatch` (would force needless full
|
||||
* writebacks otherwise).
|
||||
*/
|
||||
import { describe, it, expect, afterEach, vi } from 'vitest';
|
||||
import {
|
||||
resolveStreamPdgEmit,
|
||||
resolvePdgEmitChunkSize,
|
||||
pdgModeMismatch,
|
||||
resolvePdgConfig,
|
||||
} from '../../src/core/run-analyze.js';
|
||||
|
||||
afterEach(() => {
|
||||
vi.unstubAllEnvs();
|
||||
});
|
||||
|
||||
describe('resolveStreamPdgEmit — gating', () => {
|
||||
it('engages only with pdg + force + an enable signal', () => {
|
||||
// explicit option
|
||||
expect(resolveStreamPdgEmit({ pdg: true, force: true, streamPdgEmit: true })).toBe(true);
|
||||
// env toggle
|
||||
vi.stubEnv('GITNEXUS_STREAM_PDG_EMIT', '1');
|
||||
expect(resolveStreamPdgEmit({ pdg: true, force: true })).toBe(true);
|
||||
});
|
||||
|
||||
it('does NOT engage without --force (incremental writeback reads BasicBlocks back)', () => {
|
||||
expect(resolveStreamPdgEmit({ pdg: true, force: false, streamPdgEmit: true })).toBe(false);
|
||||
expect(resolveStreamPdgEmit({ pdg: true, streamPdgEmit: true })).toBe(false);
|
||||
});
|
||||
|
||||
it('does NOT engage without --pdg (nothing to stream)', () => {
|
||||
expect(resolveStreamPdgEmit({ pdg: false, force: true, streamPdgEmit: true })).toBe(false);
|
||||
});
|
||||
|
||||
it('does NOT engage with no enable signal', () => {
|
||||
expect(resolveStreamPdgEmit({ pdg: true, force: true })).toBe(false);
|
||||
vi.stubEnv('GITNEXUS_STREAM_PDG_EMIT', '0');
|
||||
expect(resolveStreamPdgEmit({ pdg: true, force: true })).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('resolvePdgEmitChunkSize', () => {
|
||||
it('prefers the explicit option over the env var', () => {
|
||||
vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', '128');
|
||||
expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: 999 })).toBe(999);
|
||||
});
|
||||
|
||||
it('falls back to the env var, then to undefined', () => {
|
||||
vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', '128');
|
||||
expect(resolvePdgEmitChunkSize({})).toBe(128);
|
||||
vi.unstubAllEnvs();
|
||||
expect(resolvePdgEmitChunkSize({})).toBeUndefined();
|
||||
});
|
||||
|
||||
it('rejects non-positive / non-integer env values (falls back to undefined)', () => {
|
||||
for (const bad of ['0', '-5', 'abc', '1.5', '']) {
|
||||
vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', bad);
|
||||
expect(resolvePdgEmitChunkSize({})).toBeUndefined();
|
||||
}
|
||||
});
|
||||
|
||||
it('rejects an explicit non-positive option (0/negative is not nullish — would defeat buffering)', () => {
|
||||
expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: 0 })).toBeUndefined();
|
||||
expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: -10 })).toBeUndefined();
|
||||
// ...but an explicit 0 still falls back to a valid env value when present.
|
||||
vi.stubEnv('GITNEXUS_PDG_EMIT_CHUNK_SIZE', '256');
|
||||
expect(resolvePdgEmitChunkSize({ pdgEmitChunkSize: 0 })).toBe(256);
|
||||
});
|
||||
});
|
||||
|
||||
describe('streaming knobs are NOT emit-affecting (no pdgModeMismatch)', () => {
|
||||
it('changing streamPdgEmit / pdgEmitChunkSize does not trip pdgModeMismatch', () => {
|
||||
const base = { pdg: true as const };
|
||||
const recorded = resolvePdgConfig(base);
|
||||
// Same pdg config, but with the streaming knobs flipped — must NOT mismatch.
|
||||
expect(
|
||||
pdgModeMismatch(recorded, {
|
||||
...base,
|
||||
streamPdgEmit: true,
|
||||
pdgEmitChunkSize: 64,
|
||||
} as Parameters<typeof pdgModeMismatch>[1]),
|
||||
).toBe(false);
|
||||
});
|
||||
|
||||
it('streaming knobs are absent from the resolved RepoMeta.pdg stamp', () => {
|
||||
const stamp = resolvePdgConfig({ pdg: true });
|
||||
expect(stamp).toBeDefined();
|
||||
expect(stamp).not.toHaveProperty('streamPdgEmit');
|
||||
expect(stamp).not.toHaveProperty('pdgEmitChunkSize');
|
||||
});
|
||||
});
|
||||
16
package-lock.json
generated
16
package-lock.json
generated
|
|
@ -1868,10 +1868,20 @@
|
|||
"license": "MIT"
|
||||
},
|
||||
"node_modules/js-yaml": {
|
||||
"version": "4.1.1",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.1.1.tgz",
|
||||
"integrity": "sha512-qQKT4zQxXl8lLwBtHMWwaTcGfFOZviOJet3Oy/xmGk2gZH677CJM9EvtfdSkgWcATZhj/55JZ0rmy3myCT5lsA==",
|
||||
"version": "4.2.0",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.2.0.tgz",
|
||||
"integrity": "sha512-ePWsvanv0DWuDRsW8dnt+R4jQ31SCRCQ7hhNcPXZPsoBZiemuZNYGf7adZdqX2D86j6rvKp3RpCxVTSb8WQlOw==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/puzrin"
|
||||
},
|
||||
{
|
||||
"type": "github",
|
||||
"url": "https://github.com/sponsors/nodeca"
|
||||
}
|
||||
],
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"argparse": "^2.0.1"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue