diff --git a/gitnexus-shared/src/language-detection.ts b/gitnexus-shared/src/language-detection.ts index b17c1d629..b3f34a168 100644 --- a/gitnexus-shared/src/language-detection.ts +++ b/gitnexus-shared/src/language-detection.ts @@ -163,6 +163,8 @@ const AUXILIARY_BASENAME_MAP: Record = { */ export const getSyntaxLanguageFromFilename = (filePath: string): string => { if (isBladeTemplateFilename(filePath)) return 'markup'; + // Notebooks are ingested as Python; the on-disk bytes are JSON. + if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) return 'json'; const lang = getLanguageFromFilename(filePath); if (lang) return SYNTAX_MAP[lang]; diff --git a/gitnexus/src/core/embeddings/ast-utils.ts b/gitnexus/src/core/embeddings/ast-utils.ts index 3e393f2c1..8a2448ea0 100644 --- a/gitnexus/src/core/embeddings/ast-utils.ts +++ b/gitnexus/src/core/embeddings/ast-utils.ts @@ -39,13 +39,16 @@ export const ensureAndParse = async (content: string, filePath: string): Promise // C++ UE macros, Dart extension types) would leave embeddings looking at an // error-recovered tree. Resolved from `language` so the transform and the // parser always come from the same provider. Length-preserving, so node - // offsets still index `content`. + // offsets still index `content` except for `.ipynb`, which is replaced by + // concatenated code-cell Python (same as the parse worker). let parseContent = getProvider(language).preprocessSource?.(content, filePath) ?? content; if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) { const extracted = extractNotebookPython(content); - if (!extracted) return null; - parseContent = getProvider(language).preprocessSource?.(extracted.pythonSource, filePath) ?? - extracted.pythonSource; + if (extracted) { + parseContent = + getProvider(language).preprocessSource?.(extracted.pythonSource, filePath) ?? + extracted.pythonSource; + } } return parseSourceSafe(parserInstance, parseContent); diff --git a/gitnexus/src/core/ingestion/cfg/collect.ts b/gitnexus/src/core/ingestion/cfg/collect.ts index 62d20ecb6..4e43fbeea 100644 --- a/gitnexus/src/core/ingestion/cfg/collect.ts +++ b/gitnexus/src/core/ingestion/cfg/collect.ts @@ -15,7 +15,7 @@ */ import type { SyntaxNode } from '../utils/ast-helpers.js'; import { CfgNestingDepthError } from './cfg-builder.js'; -import type { CfgVisitor, FunctionCfg } from './types.js'; +import type { CfgVisitor, FunctionCfg, SiteRecord } from './types.js'; /** * Default per-function source-line cap used by the worker when the `--pdg` run @@ -84,18 +84,26 @@ function shiftCfgLines(cfg: FunctionCfg, offset: number): FunctionCfg { } function remapCfgLines(cfg: FunctionCfg, mapLine: (row: number) => number): FunctionCfg { + // CFG visitors store 1-based source lines; `mapLine` maps 0-based tree-sitter rows. + const map1 = (line: number): number => mapLine(line - 1) + 1; + const mapSite = (site: SiteRecord): SiteRecord => + site.at !== undefined ? { ...site, at: [map1(site.at[0]), site.at[1]] } : site; return { ...cfg, - functionStartLine: mapLine(cfg.functionStartLine), - functionEndLine: mapLine(cfg.functionEndLine), + functionStartLine: map1(cfg.functionStartLine), + functionEndLine: map1(cfg.functionEndLine), blocks: cfg.blocks.map((b) => ({ ...b, - startLine: mapLine(b.startLine), - endLine: mapLine(b.endLine), - statements: b.statements?.map((s) => ({ ...s, line: mapLine(s.line) })), + startLine: map1(b.startLine), + endLine: map1(b.endLine), + statements: b.statements?.map((s) => ({ + ...s, + line: map1(s.line), + sites: s.sites?.map(mapSite), + })), })), bindings: cfg.bindings?.map((bd) => - bd.declLine > 0 ? { ...bd, declLine: mapLine(bd.declLine) } : bd, + bd.declLine > 0 ? { ...bd, declLine: map1(bd.declLine) } : bd, ), }; } diff --git a/gitnexus/src/core/ingestion/ipynb-extractor.ts b/gitnexus/src/core/ingestion/ipynb-extractor.ts index 21e1d7db1..0423b0aef 100644 --- a/gitnexus/src/core/ingestion/ipynb-extractor.ts +++ b/gitnexus/src/core/ingestion/ipynb-extractor.ts @@ -198,7 +198,6 @@ export function extractNotebookPython(content: string): NotebookPythonExtraction const chunks: string[] = []; const segments: NotebookLineSegment[] = []; - let extractLine = 0; let searchFrom = 0; for (const rawCell of nb.cells) { @@ -223,31 +222,22 @@ export function extractNotebookPython(content: string): NotebookPythonExtraction continue; } - const text = lines.join('\n'); + let text = lines.join('\n'); + if (text.endsWith('\n')) text = text.slice(0, -1); if (text.trim().length === 0) { - chunks.push(''); - const start = extractLine; - extractLine += 1; - segments.push({ - extractStartLine: start, - extractEndLine: start, - jsonStartLine, - jsonEndLine, - }); continue; } if (chunks.length > 0) { - chunks.push('\n'); - extractLine += 1; + chunks.push('\n\n'); } - const extractStartLine = extractLine; chunks.push(text); - const addedLines = text.split('\n').length; - extractLine += addedLines; + const assembled = chunks.join(''); + const extractStartLine = indexToLine(assembled, assembled.length - text.length); + const extractEndLine = indexToLine(assembled, assembled.length - 1); segments.push({ extractStartLine, - extractEndLine: extractLine - 1, + extractEndLine, jsonStartLine, jsonEndLine, }); @@ -272,14 +262,29 @@ export function mapExtractLine(row: number, segments: readonly NotebookLineSegme return last.jsonEndLine; } +const extractCache = new Map< + string, + { content: string; result: NotebookPythonExtraction | null } +>(); + +/** Memoize extraction for FTS/CSV (same file, many symbols). */ +export function extractNotebookPythonCached( + filePath: string, + content: string, +): NotebookPythonExtraction | null { + const hit = extractCache.get(filePath); + if (hit && hit.content === content) return hit.result; + const result = extractNotebookPython(content); + extractCache.set(filePath, { content, result }); + return result; +} + /** Python snippet for a graph span stored in JSON file coordinates. */ -export function notebookPythonSnippet( - fileContent: string, +export function notebookPythonSnippetFromExtract( + extracted: NotebookPythonExtraction, startLine: number, endLine: number, ): string | null { - const extracted = extractNotebookPython(fileContent); - if (!extracted) return null; const pyLines = extracted.pythonSource.split('\n'); const out: string[] = []; for (const seg of extracted.segments) { @@ -293,3 +298,16 @@ export function notebookPythonSnippet( if (out.length === 0) return null; return out.join('\n'); } + +export function notebookPythonSnippet( + fileContent: string, + startLine: number, + endLine: number, + filePath?: string, +): string | null { + const extracted = filePath + ? extractNotebookPythonCached(filePath, fileContent) + : extractNotebookPython(fileContent); + if (!extracted) return null; + return notebookPythonSnippetFromExtract(extracted, startLine, endLine); +} diff --git a/gitnexus/src/core/ingestion/languages/python/captures.ts b/gitnexus/src/core/ingestion/languages/python/captures.ts index 1ae0cfdb2..e1395df11 100644 --- a/gitnexus/src/core/ingestion/languages/python/captures.ts +++ b/gitnexus/src/core/ingestion/languages/python/captures.ts @@ -15,7 +15,7 @@ * Pure given the input source text. No I/O, no globals consulted. */ -import type { Capture, CaptureMatch } from 'gitnexus-shared'; +import type { Capture, CaptureMatch, Range } from 'gitnexus-shared'; import { nodeToCapture, syntheticCapture, @@ -24,7 +24,11 @@ import { } from '../../utils/ast-helpers.js'; import { splitImportStatement } from './import-decomposer.js'; import { getPythonParser, getPythonScopeQuery } from './query.js'; -import { extractNotebookPython } from '../../ipynb-extractor.js'; +import { + extractNotebookPython, + mapExtractLine, + type NotebookLineSegment, +} from '../../ipynb-extractor.js'; import { synthesizeConstructorFieldTypeBindings, synthesizeReceiverTypeBinding, @@ -64,14 +68,18 @@ export function emitPythonScopeCaptures( ): readonly CaptureMatch[] { let parseText = sourceText; let tree = cachedTree as ReturnType['parse']> | undefined; - if ( - filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb') && - sourceMeta?.sourceKind !== 'pre-extracted-script' - ) { + let notebookSegments: readonly NotebookLineSegment[] | undefined; + if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) { const extracted = extractNotebookPython(sourceText); - if (extracted === null) return []; - parseText = extracted.pythonSource; - tree = undefined; + if (extracted === null) { + if (sourceMeta?.sourceKind !== 'pre-extracted-script') return []; + } else { + parseText = extracted.pythonSource; + notebookSegments = extracted.segments; + if (sourceMeta?.sourceKind !== 'pre-extracted-script') { + tree = undefined; + } + } } // Skip the parse when the caller (the scope-resolution orchestrator's // `treeCache`) already produced a Tree for this source — empty under @@ -238,9 +246,31 @@ export function emitPythonScopeCaptures( out.push(...synthesizePythonInheritanceReferences(tree.rootNode)); out.push(...synthesizeCallableFlowCaptures(tree.rootNode, PYTHON_CALLABLE_CAPTURE_OPTIONS)); + if (notebookSegments !== undefined) { + return out.map((match) => remapCaptureMatch(match, notebookSegments)); + } return out; } +function remapRange(range: Range, segments: readonly NotebookLineSegment[]): Range { + return { + ...range, + startLine: mapExtractLine(range.startLine - 1, segments) + 1, + endLine: mapExtractLine(range.endLine - 1, segments) + 1, + }; +} + +function remapCaptureMatch( + match: CaptureMatch, + segments: readonly NotebookLineSegment[], +): CaptureMatch { + const next: Record = {}; + for (const [key, cap] of Object.entries(match)) { + next[key] = { ...cap, range: remapRange(cap.range, segments) }; + } + return next; +} + /** * Synthesize `@reference.inherits` captures from Python class superclass * lists so the registry-primary scope-resolution path emits EXTENDS edges diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index 9a2c6d1e6..1d80e4322 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -1685,14 +1685,14 @@ const processFileGroup = ( let scopeExtractionFailed = false; const parsedFile = extractParsedFile( provider, - parseContent, + notebookSegments ? file.content : parseContent, file.path, (message) => { scopeExtractionFailed = true; reportWarning(message); }, tree, - scopeSourceKind, + notebookSegments ? 'full-file' : scopeSourceKind, ); if (scopeExtractionFailed) (result.scopeExtractionFailures ??= []).push(file.path); if (parsedFile !== undefined) { diff --git a/gitnexus/src/core/lbug/csv-generator.ts b/gitnexus/src/core/lbug/csv-generator.ts index fdfd6ecc8..99d9bb483 100644 --- a/gitnexus/src/core/lbug/csv-generator.ts +++ b/gitnexus/src/core/lbug/csv-generator.ts @@ -23,7 +23,10 @@ import { VALID_NODE_TABLES, parseRelationSchemaPairs, RelPairRouter } from './re import { parseTruthyEnv } from '../ingestion/utils/env.js'; import { SYMBOL_NODE_LABELS } from '../ingestion/utils/symbol-labels.js'; import { applyCjkSegmentationIfEnabled } from '../search/cjk-segmentation.js'; -import { notebookPythonSnippet } from '../ingestion/ipynb-extractor.js'; +import { + notebookPythonSnippetFromExtract, + extractNotebookPythonCached, +} from '../ingestion/ipynb-extractor.js'; /** Computed once — `RELATION_SCHEMA` is a static template literal. Exported so * the streamed sinks (`GraphEmitSink`, `PdgEmitSink`) share this parse @@ -330,7 +333,10 @@ const extractContent = async ( .replace(/\\/g, '/') .toLowerCase(); if (notebookPath.endsWith('.ipynb')) { - const reconstructed = notebookPythonSnippet(content, startLine, endLine); + const extracted = extractNotebookPythonCached(notebookPath, content); + const reconstructed = extracted + ? notebookPythonSnippetFromExtract(extracted, startLine, endLine) + : null; if (reconstructed) { const MAX_SNIPPET = 5000; const capped = diff --git a/gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts b/gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts index ffb2070d4..c0b0c12ce 100644 --- a/gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts +++ b/gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts @@ -77,6 +77,16 @@ describe('Jupyter notebook Python pipeline', () => { n.properties.filePath.replace(/\\/g, '/').endsWith('julia.ipynb') && n.name === 'train', ), ).toHaveLength(0); + expect( + getNodesByLabelFull(result, 'File').filter((n) => + n.properties.filePath.replace(/\\/g, '/').endsWith('julia.ipynb'), + ), + ).toHaveLength(0); + expect( + getNodesByLabelFull(result, 'File').filter((n) => + n.properties.filePath.replace(/\\/g, '/').endsWith('broken.ipynb'), + ), + ).toHaveLength(0); } finally { fs.rmSync(root, { recursive: true, force: true }); } diff --git a/gitnexus/test/unit/ingestion-utils.test.ts b/gitnexus/test/unit/ingestion-utils.test.ts index 79605d024..344bbe311 100644 --- a/gitnexus/test/unit/ingestion-utils.test.ts +++ b/gitnexus/test/unit/ingestion-utils.test.ts @@ -54,6 +54,7 @@ describe('getLanguageFromFilename', () => { it('detects .ipynb files as Python', () => { expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python); expect(getProviderForFile('notebooks/analysis.ipynb')?.id).toBe(SupportedLanguages.Python); + expect(getSyntaxLanguageFromFilename('analysis.ipynb')).toBe('json'); }); }); diff --git a/gitnexus/test/unit/ipynb-extractor.test.ts b/gitnexus/test/unit/ipynb-extractor.test.ts index 23c48a94f..35bb6d98b 100644 --- a/gitnexus/test/unit/ipynb-extractor.test.ts +++ b/gitnexus/test/unit/ipynb-extractor.test.ts @@ -15,8 +15,7 @@ function notebook(opts: { opts.language === undefined ? undefined : { display_name: 'Python', language: opts.language, name: 'python' }; - const language_info = - opts.languageInfo === undefined ? undefined : { name: opts.languageInfo }; + const language_info = opts.languageInfo === undefined ? undefined : { name: opts.languageInfo }; return JSON.stringify( { nbformat: 4, @@ -64,15 +63,21 @@ describe('extractNotebookPython', () => { language: 'python', cells: [ { cell_type: 'code', metadata: {}, source: ['x = 1\n'], outputs: [] }, - { cell_type: 'code', metadata: {}, source: ['def train():\n', ' return x\n'], outputs: [] }, + { + cell_type: 'code', + metadata: {}, + source: ['def train():\n', ' return x\n'], + outputs: [], + }, ], }); const result = extractNotebookPython(content); expect(result).not.toBeNull(); expect(result!.pythonSource).toMatch(/x = 1\n+def train/); expect(result!.segments).toHaveLength(2); - expect(result!.segments[1].extractStartLine).toBeGreaterThan(result!.segments[0].extractStartLine); - expect(result!.segments[1].jsonStartLine).toBeGreaterThan(result!.segments[0].jsonStartLine); + const defRow = result!.pythonSource.split('\n').findIndex((l) => l.startsWith('def train')); + expect(defRow).toBe(result!.segments[1].extractStartLine); + expect(mapExtractLine(defRow, result!.segments)).toBe(result!.segments[1].jsonStartLine); }); it('accepts source as a single string', () => { @@ -199,7 +204,9 @@ describe('notebookPythonSnippet', () => { it('returns Python def train not JSON cell_type', () => { const content = notebook({ language: 'python', - cells: [{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }], + cells: [ + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], }); const extracted = extractNotebookPython(content)!; const snippet = notebookPythonSnippet( diff --git a/gitnexus/test/unit/ipynb-python-captures.test.ts b/gitnexus/test/unit/ipynb-python-captures.test.ts index 505674de7..f03a759ce 100644 --- a/gitnexus/test/unit/ipynb-python-captures.test.ts +++ b/gitnexus/test/unit/ipynb-python-captures.test.ts @@ -37,6 +37,9 @@ describe('Python notebook scope captures', () => { }); it('extractParsedFile yields a Function for train', () => { + const captured = emitPythonScopeCaptures(notebook, 'analysis.ipynb'); + const fnCapture = captured.find((m) => m['@scope.function'] !== undefined); + expect(fnCapture?.['@scope.function']?.range.startLine).toBeGreaterThan(1); const parsed = extractParsedFile(pythonProvider, notebook, 'analysis.ipynb'); expect(parsed).toBeDefined(); expect(parsed!.localDefs.some((d) => d.qualifiedName === 'train' || d.name === 'train')).toBe(