diff --git a/docs/languages/jupyter-notebook.md b/docs/languages/jupyter-notebook.md new file mode 100644 index 000000000..c6b79090b --- /dev/null +++ b/docs/languages/jupyter-notebook.md @@ -0,0 +1,38 @@ +# Jupyter notebook (.ipynb) indexing + +Status: implemented (Python code cells) + +GitNexus indexes Jupyter notebooks by extracting Python code cells and parsing them with the existing Python language provider. Notebooks are not executed. + +## Goal + +After `analyze`, functions, classes, and imports defined in Python code cells are queryable like ordinary `.py` files. + +## Compatibility + +- `.ipynb` is detected as Python (`gitnexus-shared` `EXTENSION_MAP` and `pythonProvider.extensions`). +- Extraction lives in `gitnexus/src/core/ingestion/ipynb-extractor.ts`. Shared ingestion modules do not name nbformat AST types. +- Notebooks are **not** Python import targets. A notebook may import `.py` modules; `import some_notebook` does not resolve to an `.ipynb`. +- Files over the walker size cap (default 512KB, `GITNEXUS_MAX_FILE_SIZE`) are skipped like any other oversized file. Output-heavy notebooks may need a higher cap. Outputs are not stripped in this slice. +- Group-layer FastAPI/Flask/Django scanners that require a `.py` suffix still ignore notebooks. + +## Kernel and magics + +- Skip the file when `kernelspec.language` or `language_info.name` is present and is not a Python-family name (`python`, `python2`, `python3`, `ipython`). Disagreeing fields skip the file. +- If those fields are absent, code cells are treated as Python unless a cell's own language metadata says otherwise. +- A cell whose first non-empty line is a cell magic (`%%`) is skipped. +- Line magics (`%`) and shell (`!`) lines are commented in place so JSON line mapping stays affine. + +## Line numbers + +Graph `startLine` / `endLine` are 0-based coordinates in the on-disk `.ipynb` JSON. FTS/MCP symbol snippets reconstruct cell Python; they do not slice raw JSON. + +Concatenating cells in document order is notebook semantics. An earlier cell with a syntax error may cause later definitions to be missed; that is a documented limitation, not a per-cell fallback parse. + +## Tests + +- `gitnexus/test/unit/ipynb-extractor.test.ts` +- `gitnexus/test/unit/ingestion-utils.test.ts` (`.ipynb` detection) +- `gitnexus/test/integration/ipynb-python-pipeline.test.ts` + +No Jupyter, nbconvert, or nbformat runtime dependency. diff --git a/gitnexus-shared/src/language-detection.ts b/gitnexus-shared/src/language-detection.ts index 3fe27f7b7..b17c1d629 100644 --- a/gitnexus-shared/src/language-detection.ts +++ b/gitnexus-shared/src/language-detection.ts @@ -29,7 +29,7 @@ const RUBY_EXTENSIONLESS_FILES = new Set([ const EXTENSION_MAP: Record = { [SupportedLanguages.JavaScript]: ['.js', '.jsx', '.mjs', '.cjs'], [SupportedLanguages.TypeScript]: ['.ts', '.tsx', '.mts', '.cts'], - [SupportedLanguages.Python]: ['.py'], + [SupportedLanguages.Python]: ['.py', '.ipynb'], [SupportedLanguages.Java]: ['.java'], [SupportedLanguages.C]: ['.c'], [SupportedLanguages.ObjectiveC]: ['.m', '.mm'], diff --git a/gitnexus/src/core/embeddings/ast-utils.ts b/gitnexus/src/core/embeddings/ast-utils.ts index b29ca80fb..3e393f2c1 100644 --- a/gitnexus/src/core/embeddings/ast-utils.ts +++ b/gitnexus/src/core/embeddings/ast-utils.ts @@ -11,6 +11,7 @@ import { } from '../tree-sitter/parser-loader.js'; import { parseSourceSafe } from '../tree-sitter/safe-parse.js'; import { getLanguageForFileContent, getProvider } from '../ingestion/languages/index.js'; +import { extractNotebookPython } from '../ingestion/ipynb-extractor.js'; const parserCache = new Map(); @@ -39,7 +40,13 @@ export const ensureAndParse = async (content: string, filePath: string): Promise // error-recovered tree. Resolved from `language` so the transform and the // parser always come from the same provider. Length-preserving, so node // offsets still index `content`. - const parseContent = getProvider(language).preprocessSource?.(content, filePath) ?? content; + let parseContent = getProvider(language).preprocessSource?.(content, filePath) ?? content; + if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) { + const extracted = extractNotebookPython(content); + if (!extracted) return null; + parseContent = getProvider(language).preprocessSource?.(extracted.pythonSource, filePath) ?? + extracted.pythonSource; + } return parseSourceSafe(parserInstance, parseContent); }; diff --git a/gitnexus/src/core/ingestion/cfg/collect.ts b/gitnexus/src/core/ingestion/cfg/collect.ts index bdf1b095d..62d20ecb6 100644 --- a/gitnexus/src/core/ingestion/cfg/collect.ts +++ b/gitnexus/src/core/ingestion/cfg/collect.ts @@ -83,12 +83,30 @@ function shiftCfgLines(cfg: FunctionCfg, offset: number): FunctionCfg { }; } +function remapCfgLines(cfg: FunctionCfg, mapLine: (row: number) => number): FunctionCfg { + return { + ...cfg, + functionStartLine: mapLine(cfg.functionStartLine), + functionEndLine: mapLine(cfg.functionEndLine), + blocks: cfg.blocks.map((b) => ({ + ...b, + startLine: mapLine(b.startLine), + endLine: mapLine(b.endLine), + statements: b.statements?.map((s) => ({ ...s, line: mapLine(s.line) })), + })), + bindings: cfg.bindings?.map((bd) => + bd.declLine > 0 ? { ...bd, declLine: mapLine(bd.declLine) } : bd, + ), + }; +} + export function collectFunctionCfgs( root: SyntaxNode, visitor: CfgVisitor, filePath: string, maxFunctionLines = 0, lineOffset = 0, + mapLine?: (row: number) => number, ): CollectedCfgs { const cfgs: FunctionCfg[] = []; let tooManyLines = 0; @@ -109,7 +127,9 @@ export function collectFunctionCfgs( // and silently drop every remaining function's CFG (#2195). try { const cfg = visitor.buildFunctionCfg(node, filePath); - if (cfg) cfgs.push(shiftCfgLines(cfg, lineOffset)); + if (cfg) { + cfgs.push(mapLine ? remapCfgLines(cfg, mapLine) : shiftCfgLines(cfg, lineOffset)); + } } catch (err) { if (err instanceof CfgNestingDepthError) tooDeeplyNested++; else buildError++; diff --git a/gitnexus/src/core/ingestion/ipynb-extractor.ts b/gitnexus/src/core/ingestion/ipynb-extractor.ts new file mode 100644 index 000000000..65486a4de --- /dev/null +++ b/gitnexus/src/core/ingestion/ipynb-extractor.ts @@ -0,0 +1,298 @@ +/** + * Jupyter notebook (.ipynb) Python extractor. + * + * Pulls code-cell source from nbformat JSON so the Python tree-sitter + * grammar can parse it. Pure — no I/O, no tree-sitter, worker-safe. + * + * Graph coordinates stay 0-based lines in the on-disk JSON file. Extract + * buffer lines map through {@link mapExtractLine}. + */ + +export interface NotebookLineSegment { + readonly extractStartLine: number; + readonly extractEndLine: number; + readonly jsonStartLine: number; + readonly jsonEndLine: number; +} + +export interface NotebookPythonExtraction { + readonly pythonSource: string; + readonly segments: readonly NotebookLineSegment[]; +} + +const PYTHON_FAMILY = new Set(['python', 'python2', 'python3', 'ipython']); + +export function isPythonFamilyLanguage(name: string | undefined | null): boolean { + if (name === undefined || name === null) return false; + const n = name.trim().toLowerCase(); + if (PYTHON_FAMILY.has(n)) return true; + return /^python\d/.test(n); +} + +function indexToLine(content: string, index: number): number { + let line = 0; + const end = Math.max(0, Math.min(index, content.length)); + for (let i = 0; i < end; i++) { + if (content.charCodeAt(i) === 10) line++; + } + return line; +} + +function skipWs(content: string, i: number): number { + while (i < content.length) { + const c = content.charCodeAt(i); + if (c === 32 || c === 9 || c === 10 || c === 13) i++; + else break; + } + return i; +} + +/** Span of a JSON string or array value starting at the first `"` or `[`. */ +function jsonValueSpan(content: string, start: number): { start: number; end: number } | null { + const i = skipWs(content, start); + if (i >= content.length) return null; + if (content[i] === '"') { + let j = i + 1; + while (j < content.length) { + if (content[j] === '\\') { + j += 2; + continue; + } + if (content[j] === '"') return { start: i, end: j + 1 }; + j++; + } + return null; + } + if (content[i] === '[') { + let depth = 1; + let j = i + 1; + let inStr = false; + while (j < content.length && depth > 0) { + const ch = content[j]; + if (inStr) { + if (ch === '\\') j += 2; + else { + if (ch === '"') inStr = false; + j++; + } + continue; + } + if (ch === '"') inStr = true; + else if (ch === '[') depth++; + else if (ch === ']') depth--; + j++; + } + return { start: i, end: j }; + } + return null; +} + +function findNextCodeCellSourceSpan( + content: string, + from: number, +): { span: { start: number; end: number }; nextFrom: number } | null { + let search = from; + while (search < content.length) { + const typeKey = content.indexOf('"cell_type"', search); + if (typeKey < 0) return null; + const colon = content.indexOf(':', typeKey + 11); + if (colon < 0) return null; + const valueStart = skipWs(content, colon + 1); + if (content.slice(valueStart, valueStart + 6) !== '"code"') { + search = typeKey + 11; + continue; + } + const sourceKey = content.indexOf('"source"', valueStart); + if (sourceKey < 0) return null; + const srcColon = content.indexOf(':', sourceKey + 8); + if (srcColon < 0) return null; + const span = jsonValueSpan(content, srcColon + 1); + if (!span) return null; + return { span, nextFrom: span.end }; + } + return null; +} + +function flattenSource(source: unknown): string { + if (typeof source === 'string') return source; + if (Array.isArray(source)) { + return source.map((part) => (typeof part === 'string' ? part : '')).join(''); + } + return ''; +} + +function cellLanguage(cell: Record): string | undefined { + const meta = cell.metadata; + if (!meta || typeof meta !== 'object') return undefined; + const m = meta as Record; + if (typeof m.language === 'string') return m.language; + const vscode = m.vscode; + if (vscode && typeof vscode === 'object') { + const langId = (vscode as Record).languageId; + if (typeof langId === 'string') return langId; + } + return undefined; +} + +function notebookLanguageFields(nb: Record): { + kernelspec?: string; + languageInfo?: string; +} { + const metadata = nb.metadata; + if (!metadata || typeof metadata !== 'object') return {}; + const md = metadata as Record; + const ks = md.kernelspec; + const li = md.language_info; + return { + kernelspec: + ks && typeof ks === 'object' && typeof (ks as Record).language === 'string' + ? String((ks as Record).language) + : undefined, + languageInfo: + li && typeof li === 'object' && typeof (li as Record).name === 'string' + ? String((li as Record).name) + : undefined, + }; +} + +function kernelShouldSkip(nb: Record): boolean { + const { kernelspec, languageInfo } = notebookLanguageFields(nb); + const fields = [kernelspec, languageInfo].filter((x): x is string => x !== undefined); + if (fields.length === 0) return false; + if (fields.some((f) => !isPythonFamilyLanguage(f))) return true; + if (fields.length === 2 && fields[0].toLowerCase() !== fields[1].toLowerCase()) { + const a = isPythonFamilyLanguage(fields[0]); + const b = isPythonFamilyLanguage(fields[1]); + if (a !== b) return true; + } + return false; +} + +function processCellLines(raw: string): { skipCell: boolean; lines: string[] } { + const lines = raw.split('\n'); + const firstNonEmpty = lines.find((l) => l.trim().length > 0); + if (firstNonEmpty !== undefined && firstNonEmpty.trimStart().startsWith('%%')) { + return { skipCell: true, lines: [''] }; + } + const out = lines.map((line) => { + const t = line.trimStart(); + if (t.startsWith('%') || t.startsWith('!')) { + return `# ${line}`; + } + return line; + }); + return { skipCell: false, lines: out }; +} + +export function extractNotebookPython(content: string): NotebookPythonExtraction | null { + let parsed: unknown; + try { + parsed = JSON.parse(content); + } catch { + return null; + } + if (!parsed || typeof parsed !== 'object') return null; + const nb = parsed as Record; + if (!Array.isArray(nb.cells)) return null; + if (kernelShouldSkip(nb)) return null; + + const chunks: string[] = []; + const segments: NotebookLineSegment[] = []; + let extractLine = 0; + let searchFrom = 0; + let codeCellIndex = 0; + + for (const rawCell of nb.cells) { + if (!rawCell || typeof rawCell !== 'object') continue; + const cell = rawCell as Record; + if (cell.cell_type !== 'code') continue; + + const located = findNextCodeCellSourceSpan(content, searchFrom); + if (!located) return null; + searchFrom = located.nextFrom; + codeCellIndex++; + + const lang = cellLanguage(cell); + if (lang !== undefined && !isPythonFamilyLanguage(lang)) { + continue; + } + + const { skipCell, lines } = processCellLines(flattenSource(cell.source)); + const jsonStartLine = indexToLine(content, located.span.start); + const jsonEndLine = Math.max(jsonStartLine, indexToLine(content, located.span.end - 1)); + + if (skipCell) { + continue; + } + + const text = lines.join('\n'); + if (text.trim().length === 0) { + chunks.push(''); + const start = extractLine; + extractLine += 1; + segments.push({ + extractStartLine: start, + extractEndLine: start, + jsonStartLine, + jsonEndLine, + }); + continue; + } + + if (chunks.length > 0) { + chunks.push('\n'); + extractLine += 1; + } + const extractStartLine = extractLine; + chunks.push(text); + const addedLines = text.split('\n').length; + extractLine += addedLines; + segments.push({ + extractStartLine, + extractEndLine: extractLine - 1, + jsonStartLine, + jsonEndLine, + }); + } + + void codeCellIndex; + const pythonSource = chunks.join(''); + if (pythonSource.trim().length === 0) return null; + return { pythonSource, segments }; +} + +export function mapExtractLine(row: number, segments: readonly NotebookLineSegment[]): number { + if (segments.length === 0) return row; + for (const seg of segments) { + if (row >= seg.extractStartLine && row <= seg.extractEndLine) { + const delta = row - seg.extractStartLine; + const jsonSpan = seg.jsonEndLine - seg.jsonStartLine; + return seg.jsonStartLine + Math.min(delta, jsonSpan); + } + } + if (row < segments[0].extractStartLine) return segments[0].jsonStartLine; + const last = segments[segments.length - 1]; + return last.jsonEndLine; +} + +/** Python snippet for a graph span stored in JSON file coordinates. */ +export function notebookPythonSnippet( + fileContent: string, + startLine: number, + endLine: number, +): string | null { + const extracted = extractNotebookPython(fileContent); + if (!extracted) return null; + const pyLines = extracted.pythonSource.split('\n'); + const out: string[] = []; + for (const seg of extracted.segments) { + for (let extract = seg.extractStartLine; extract <= seg.extractEndLine; extract++) { + const jsonLine = mapExtractLine(extract, extracted.segments); + if (jsonLine >= startLine && jsonLine <= endLine) { + out.push(pyLines[extract] ?? ''); + } + } + } + if (out.length === 0) return null; + return out.join('\n'); +} diff --git a/gitnexus/src/core/ingestion/languages/python.ts b/gitnexus/src/core/ingestion/languages/python.ts index ce8d0fa90..d07902f66 100644 --- a/gitnexus/src/core/ingestion/languages/python.ts +++ b/gitnexus/src/core/ingestion/languages/python.ts @@ -107,7 +107,7 @@ function normalizePythonStringLiteral(text: string): string | undefined { export const pythonProvider = defineLanguage({ id: SupportedLanguages.Python, - extensions: ['.py'], + extensions: ['.py', '.ipynb'], entryPointPatterns: [/^app$/, /^(get|post|put|delete|patch)_/i, /^api_/, /^view_/], astFrameworkPatterns: [ { diff --git a/gitnexus/src/core/ingestion/languages/python/captures.ts b/gitnexus/src/core/ingestion/languages/python/captures.ts index a6d19fdc4..1ae0cfdb2 100644 --- a/gitnexus/src/core/ingestion/languages/python/captures.ts +++ b/gitnexus/src/core/ingestion/languages/python/captures.ts @@ -24,6 +24,7 @@ import { } from '../../utils/ast-helpers.js'; import { splitImportStatement } from './import-decomposer.js'; import { getPythonParser, getPythonScopeQuery } from './query.js'; +import { extractNotebookPython } from '../../ipynb-extractor.js'; import { synthesizeConstructorFieldTypeBindings, synthesizeReceiverTypeBinding, @@ -57,23 +58,34 @@ const PYTHON_CALLABLE_CAPTURE_OPTIONS = { export function emitPythonScopeCaptures( sourceText: string, - _filePath: string, + filePath: string, cachedTree?: unknown, + sourceMeta?: { sourceKind?: 'full-file' | 'pre-extracted-script' }, ): readonly CaptureMatch[] { + let parseText = sourceText; + let tree = cachedTree as ReturnType['parse']> | undefined; + if ( + filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb') && + sourceMeta?.sourceKind !== 'pre-extracted-script' + ) { + const extracted = extractNotebookPython(sourceText); + if (extracted === null) return []; + parseText = extracted.pythonSource; + tree = undefined; + } // Skip the parse when the caller (the scope-resolution orchestrator's // `treeCache`) already produced a Tree for this source — empty under // worker-pool runs, so cache miss = re-parse. The cachedTree parameter // is typed as `unknown` at the // contract layer (see `LanguageProvider.emitScopeCaptures`); cast // here at the use site. - let tree = cachedTree as ReturnType['parse']> | undefined; if (tree === undefined) { try { - tree = parseSourceSafe(getPythonParser(), sourceText, undefined, { - bufferSize: getTreeSitterBufferSize(sourceText), + tree = parseSourceSafe(getPythonParser(), parseText, undefined, { + bufferSize: getTreeSitterBufferSize(parseText), }); } catch (err) { - throw scopeExtractionError('parse', _filePath, err); + throw scopeExtractionError('parse', filePath, err); } recordCacheMiss(); } else { @@ -84,7 +96,7 @@ export function emitPythonScopeCaptures( try { rawMatches = getPythonScopeQuery().matches(tree.rootNode); } catch (err) { - throw scopeExtractionError('scope query', _filePath, err); + throw scopeExtractionError('scope query', filePath, err); } const out: CaptureMatch[] = []; diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index 1145b38e7..9a2c6d1e6 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -135,6 +135,11 @@ import { extractTemplateComponents, isVueSetupTopLevel, } from '../vue-sfc-extractor.js'; +import { + extractNotebookPython, + mapExtractLine, + type NotebookLineSegment, +} from '../ipynb-extractor.js'; import type { NodeLabel, ParameterTypeClass } from 'gitnexus-shared'; import type { FieldInfo, FieldExtractorContext } from '../field-types.js'; import type { MethodInfo, MethodExtractorContext } from '../method-types.js'; @@ -1597,6 +1602,9 @@ const processFileGroup = ( let scopeSourceKind: ScopeCaptureSourceKind = 'full-file'; let lineOffset = 0; let isVueSetup = false; + let notebookSegments: readonly NotebookLineSegment[] | undefined; + const mapRow = (row: number): number => + notebookSegments ? mapExtractLine(row, notebookSegments) : row + lineOffset; if (language === SupportedLanguages.Vue) { const extracted = extractVueScript(file.content); if (!extracted) continue; // skip .vue files with no script block @@ -1604,6 +1612,15 @@ const processFileGroup = ( scopeSourceKind = 'pre-extracted-script'; lineOffset = extracted.lineOffset; isVueSetup = extracted.isSetup; + } else if ( + language === SupportedLanguages.Python && + file.path.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb') + ) { + const extracted = extractNotebookPython(file.content); + if (!extracted) continue; + parseContent = extracted.pythonSource; + scopeSourceKind = 'pre-extracted-script'; + notebookSegments = extracted.segments; } // Per-language source-text transform (e.g., UE macro stripping for C++). @@ -1722,6 +1739,7 @@ const processFileGroup = ( // `lineOffset` in the file — shift the CFG into file coordinates so // it joins its graph node and BasicBlock lines map to source. lineOffset, + notebookSegments ? mapRow : undefined, ); if (cfgs.length) withChannels = { ...withChannels, cfgSideChannel: cfgs }; // Surface per-function CFG skips per-language (#2195): merged + logged @@ -2483,10 +2501,10 @@ const processFileGroup = ( : definitionNode?.startPosition; const startLine = startPosition !== undefined - ? startPosition.row + lineOffset + ? mapRow(startPosition.row) : nameNode - ? nameNode.startPosition.row + lineOffset - : lineOffset; + ? mapRow(nameNode.startPosition.row) + : mapRow(0); const startColumn = startPosition?.column ?? nameNode?.startPosition.column ?? 0; // Compute enclosing class BEFORE node ID — needed to qualify method IDs @@ -2958,7 +2976,7 @@ const processFileGroup = ( filePath: file.path, toolName: nodeName, description: (dec.arg || description || '').slice(0, 200), - lineNumber: definitionNode.startPosition.row + lineOffset, + lineNumber: mapRow(definitionNode.startPosition.row), handlerNodeId: nodeId, }); } @@ -3063,7 +3081,7 @@ const processFileGroup = ( (objectLiteralBindingInfo?.ownerName || isArrayContainedObjectCallable) ? { startColumn } : {}), - endLine: definitionNode ? definitionNode.endPosition.row + lineOffset : startLine, + endLine: definitionNode ? mapRow(definitionNode.endPosition.row) : startLine, language: language, isExported, ...(qualifiedTypeName !== undefined ? { qualifiedName: qualifiedTypeName } : {}), diff --git a/gitnexus/src/core/lbug/csv-generator.ts b/gitnexus/src/core/lbug/csv-generator.ts index 5e652319c..fdfd6ecc8 100644 --- a/gitnexus/src/core/lbug/csv-generator.ts +++ b/gitnexus/src/core/lbug/csv-generator.ts @@ -23,6 +23,7 @@ import { VALID_NODE_TABLES, parseRelationSchemaPairs, RelPairRouter } from './re import { parseTruthyEnv } from '../ingestion/utils/env.js'; import { SYMBOL_NODE_LABELS } from '../ingestion/utils/symbol-labels.js'; import { applyCjkSegmentationIfEnabled } from '../search/cjk-segmentation.js'; +import { notebookPythonSnippet } from '../ingestion/ipynb-extractor.js'; /** Computed once — `RELATION_SCHEMA` is a static template literal. Exported so * the streamed sinks (`GraphEmitSink`, `PdgEmitSink`) share this parse @@ -325,6 +326,21 @@ const extractContent = async ( const endLine = node.properties.endLine; if (startLine === undefined || endLine === undefined) return ''; + const notebookPath = String(filePath ?? '') + .replace(/\\/g, '/') + .toLowerCase(); + if (notebookPath.endsWith('.ipynb')) { + const reconstructed = notebookPythonSnippet(content, startLine, endLine); + if (reconstructed) { + const MAX_SNIPPET = 5000; + const capped = + reconstructed.length > MAX_SNIPPET + ? reconstructed.slice(0, MAX_SNIPPET) + '\n... [truncated]' + : reconstructed; + return normalizeFtsText(applyCjkSegmentationIfEnabled(capped)); + } + } + const lines = prepared.lines; const exactSymbolContent = EXACT_SYMBOL_CONTENT_LABELS.has(node.label); const start = Math.max(0, exactSymbolContent ? startLine : startLine - 2); diff --git a/gitnexus/src/storage/parse-cache.ts b/gitnexus/src/storage/parse-cache.ts index 7e91cee4c..2c95bfb69 100644 --- a/gitnexus/src/storage/parse-cache.ts +++ b/gitnexus/src/storage/parse-cache.ts @@ -778,7 +778,12 @@ import { copyV8CacheIfPresent, tryLoadV8Cache, writeV8CacheFile } from './v8-sid // v104 (#3339 review): TS/JS pair-HOC queries now name object-pair // `mutation(withAuth(arrow))` handlers. Warm caches replay the pre-fix // capture set (anonymous arrows, no Function name), so both stores re-extract. -const SCHEMA_BUMP = 104; +// v104 (#3339 review): TS/JS pair-HOC queries now name object-pair +// `mutation(withAuth(arrow))` handlers. Warm caches replay the pre-fix +// capture set (anonymous arrows, no Function name), so both stores re-extract. +// v105 (#3371): `.ipynb` code cells are extracted to Python before parse. +// Warm caches keyed on raw JSON would replay empty/failed Python parses. +const SCHEMA_BUMP = 105; const GITNEXUS_PKG_VERSION = (() => { try { // package.json sits at gitnexus/package.json — two levels up from diff --git a/gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts b/gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts new file mode 100644 index 000000000..ffb2070d4 --- /dev/null +++ b/gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts @@ -0,0 +1,84 @@ +/** + * Jupyter notebooks are indexed as Python by extracting code cells. + */ +import { describe, it, expect } from 'vitest'; +import path from 'path'; +import fs from 'node:fs'; +import os from 'node:os'; +import { + getNodesByLabel, + getNodesByLabelFull, + getRelationships, + runPipelineFromRepo, + writeFixtureRepo, +} from './helpers.js'; + +function pythonNotebook(cells: Array<{ source: string | string[]; language?: string }>): string { + return JSON.stringify( + { + nbformat: 4, + nbformat_minor: 5, + metadata: { + kernelspec: { display_name: 'Python 3', language: 'python', name: 'python3' }, + language_info: { name: 'python' }, + }, + cells: cells.map((c) => ({ + cell_type: 'code', + metadata: c.language ? { language: c.language } : {}, + source: c.source, + outputs: [], + })), + }, + null, + 2, + ); +} + +describe('Jupyter notebook Python pipeline', () => { + it('indexes notebook functions, imports sibling py, and skip-fails closed', async () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-ipynb-')); + try { + writeFixtureRepo(root, { + 'lib.py': 'def helper():\n return 1\n', + 'analysis.ipynb': pythonNotebook([ + { source: ['from lib import helper\n'] }, + { source: ['def train():\n', ' return helper()\n'] }, + ]), + 'broken.ipynb': '{not json', + 'julia.ipynb': JSON.stringify({ + nbformat: 4, + nbformat_minor: 5, + metadata: { + kernelspec: { language: 'julia', name: 'julia', display_name: 'Julia' }, + }, + cells: [ + { cell_type: 'code', metadata: {}, source: ['function train() end\n'], outputs: [] }, + ], + }), + }); + const result = await runPipelineFromRepo(root, () => {}, { skipGraphPhases: true }); + const functions = getNodesByLabel(result, 'Function'); + expect(functions).toContain('train'); + expect(functions).toContain('helper'); + const train = getNodesByLabelFull(result, 'Function').find((n) => n.name === 'train'); + expect(train?.properties.filePath.replace(/\\/g, '/')).toMatch(/analysis\.ipynb$/); + expect(train?.properties.startLine).toBeGreaterThan(0); + const imports = getRelationships(result, 'IMPORTS'); + expect( + imports.some( + (e) => + e.sourceFilePath.replace(/\\/g, '/').endsWith('analysis.ipynb') && + (e.target === 'helper' || e.targetFilePath.replace(/\\/g, '/').endsWith('lib.py')), + ), + ).toBe(true); + expect( + getNodesByLabelFull(result, 'Function').filter( + (n) => + n.properties.filePath.replace(/\\/g, '/').endsWith('julia.ipynb') && n.name === 'train', + ), + ).toHaveLength(0); + } finally { + fs.rmSync(root, { recursive: true, force: true }); + } + }, 60000); +}); diff --git a/gitnexus/test/unit/incremental-parse-cache.test.ts b/gitnexus/test/unit/incremental-parse-cache.test.ts index 879af7a38..cbf77be34 100644 --- a/gitnexus/test/unit/incremental-parse-cache.test.ts +++ b/gitnexus/test/unit/incremental-parse-cache.test.ts @@ -288,8 +288,12 @@ describe('PARSE_CACHE_VERSION', () => { // Moved 103 -> 104 for #3339 review: pair-HOC queries name // `mutation(withAuth(arrow))` object-pair handlers. Warm caches replay // anonymous arrows, so both stores re-extract. - it('pins SCHEMA_BUMP to 104 so concurrent bumps cannot silently collide (#2766, #3015, #3088, #2885, #3128, #2865, #3130, #1432, #3161, #3179, #3219, #3190, #3253, #3273, #3339)', () => { - expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(104); + // Moved 103 -> 104 for #3339 review: pair-HOC queries name + // `mutation(withAuth(arrow))` object-pair handlers. Warm caches replay + // anonymous arrows, so both stores re-extract. + // Moved 104 -> 105 for #3371: notebook code-cell extraction before Python parse. + it('pins SCHEMA_BUMP to 105 so concurrent bumps cannot silently collide (#2766, #3015, #3088, #2885, #3128, #2865, #3130, #1432, #3161, #3179, #3219, #3190, #3253, #3273, #3339, #3371)', () => { + expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(105); expect(PARSE_CACHE_BUCKET_COUNT).toBe(128); // The PREVIOUS version must fail the reuse gate, not merely differ from the // current one — a hardcoded number outside the conflict hunk rebases cleanly @@ -298,6 +302,7 @@ describe('PARSE_CACHE_VERSION', () => { for (const taken of [ 59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103, + 104, ]) { expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).not.toBe(taken); } diff --git a/gitnexus/test/unit/ingestion-utils.test.ts b/gitnexus/test/unit/ingestion-utils.test.ts index db16bfd34..79605d024 100644 --- a/gitnexus/test/unit/ingestion-utils.test.ts +++ b/gitnexus/test/unit/ingestion-utils.test.ts @@ -50,6 +50,11 @@ describe('getLanguageFromFilename', () => { it('detects .py files', () => { expect(getLanguageFromFilename('main.py')).toBe(SupportedLanguages.Python); }); + + it('detects .ipynb files as Python', () => { + expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python); + expect(getProviderForFile('notebooks/analysis.ipynb')?.id).toBe(SupportedLanguages.Python); + }); }); describe('Java', () => { diff --git a/gitnexus/test/unit/ipynb-extractor.test.ts b/gitnexus/test/unit/ipynb-extractor.test.ts new file mode 100644 index 000000000..23c48a94f --- /dev/null +++ b/gitnexus/test/unit/ipynb-extractor.test.ts @@ -0,0 +1,213 @@ +import { describe, it, expect } from 'vitest'; +import { + extractNotebookPython, + mapExtractLine, + notebookPythonSnippet, + isPythonFamilyLanguage, +} from '../../src/core/ingestion/ipynb-extractor.js'; + +function notebook(opts: { + language?: string; + languageInfo?: string; + cells: Array>; +}): string { + const kernelspec = + opts.language === undefined + ? undefined + : { display_name: 'Python', language: opts.language, name: 'python' }; + const language_info = + opts.languageInfo === undefined ? undefined : { name: opts.languageInfo }; + return JSON.stringify( + { + nbformat: 4, + nbformat_minor: 5, + metadata: { + ...(kernelspec ? { kernelspec } : {}), + ...(language_info ? { language_info } : {}), + }, + cells: opts.cells, + }, + null, + 2, + ); +} + +describe('isPythonFamilyLanguage', () => { + it('accepts python3 and ipython', () => { + expect(isPythonFamilyLanguage('python3')).toBe(true); + expect(isPythonFamilyLanguage('IPython')).toBe(true); + expect(isPythonFamilyLanguage('julia')).toBe(false); + }); +}); + +describe('extractNotebookPython', () => { + it('extracts def train from a Python v4 notebook', () => { + const content = notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: {}, + source: ['def train():\n', ' pass\n'], + outputs: [], + }, + ], + }); + const result = extractNotebookPython(content); + expect(result).not.toBeNull(); + expect(result!.pythonSource).toContain('def train():'); + expect(result!.segments).toHaveLength(1); + }); + + it('keeps two code cells in order with two segments', () => { + const content = notebook({ + language: 'python', + cells: [ + { cell_type: 'code', metadata: {}, source: ['x = 1\n'], outputs: [] }, + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' return x\n'], outputs: [] }, + ], + }); + const result = extractNotebookPython(content); + expect(result).not.toBeNull(); + expect(result!.pythonSource).toMatch(/x = 1\n+def train/); + expect(result!.segments).toHaveLength(2); + expect(result!.segments[1].extractStartLine).toBeGreaterThan(result!.segments[0].extractStartLine); + expect(result!.segments[1].jsonStartLine).toBeGreaterThan(result!.segments[0].jsonStartLine); + }); + + it('accepts source as a single string', () => { + const content = notebook({ + language: 'python', + cells: [{ cell_type: 'code', metadata: {}, source: 'y = 2\n', outputs: [] }], + }); + expect(extractNotebookPython(content)?.pythonSource).toContain('y = 2'); + }); + + it('returns null for markdown-only notebooks', () => { + const content = notebook({ + language: 'python', + cells: [{ cell_type: 'markdown', metadata: {}, source: ['# hi\n'] }], + }); + expect(extractNotebookPython(content)).toBeNull(); + }); + + it('returns null for invalid JSON', () => { + expect(extractNotebookPython('{not json')).toBeNull(); + }); + + it('returns null for a Julia kernelspec', () => { + const content = notebook({ + language: 'julia', + cells: [{ cell_type: 'code', metadata: {}, source: ['1 + 1\n'], outputs: [] }], + }); + expect(extractNotebookPython(content)).toBeNull(); + }); + + it('extracts python3 kernelspec', () => { + const content = notebook({ + language: 'python3', + cells: [{ cell_type: 'code', metadata: {}, source: ['a = 1\n'], outputs: [] }], + }); + expect(extractNotebookPython(content)?.pythonSource).toContain('a = 1'); + }); + + it('comments line magics in place', () => { + const content = notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: {}, + source: ['%time\n', 'x = 1\n', '!ls\n'], + outputs: [], + }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource.split('\n').filter((l) => l.length > 0)).toEqual([ + '# %time', + 'x = 1', + '# !ls', + ]); + }); + + it('skips a %%bash cell and still extracts later train', () => { + const content = notebook({ + language: 'python', + cells: [ + { cell_type: 'code', metadata: {}, source: ['%%bash\n', 'echo hi\n'], outputs: [] }, + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('def train'); + expect(result!.pythonSource).not.toContain('echo hi'); + }); + + it('maps identical duplicate cells to later JSON lines', () => { + const src = ['print(1)\n']; + const content = notebook({ + language: 'python', + cells: [ + { cell_type: 'code', metadata: {}, source: src, outputs: [] }, + { cell_type: 'code', metadata: {}, source: src, outputs: [] }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.segments).toHaveLength(2); + expect(result!.segments[1].jsonStartLine).toBeGreaterThan(result!.segments[0].jsonStartLine); + }); + + it('skips an R-language code cell and keeps Python cells', () => { + const content = notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: { language: 'R' }, + source: ['x <- 1\n'], + outputs: [], + }, + { cell_type: 'code', metadata: {}, source: ['z = 3\n'], outputs: [] }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('z = 3'); + expect(result!.pythonSource).not.toContain('x <- 1'); + }); +}); + +describe('mapExtractLine', () => { + it('maps a second-cell extract row onto that cell JSON line', () => { + const content = notebook({ + language: 'python', + cells: [ + { cell_type: 'markdown', metadata: {}, source: ['# intro\n'] }, + { cell_type: 'code', metadata: {}, source: ['a = 1\n'], outputs: [] }, + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }); + const extracted = extractNotebookPython(content)!; + const trainSeg = extracted.segments[1]; + const mapped = mapExtractLine(trainSeg.extractStartLine, extracted.segments); + expect(mapped).toBe(trainSeg.jsonStartLine); + expect(mapped).toBeGreaterThan(extracted.segments[0].jsonStartLine); + }); +}); + +describe('notebookPythonSnippet', () => { + it('returns Python def train not JSON cell_type', () => { + const content = notebook({ + language: 'python', + cells: [{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }], + }); + const extracted = extractNotebookPython(content)!; + const snippet = notebookPythonSnippet( + content, + extracted.segments[0].jsonStartLine, + extracted.segments[0].jsonEndLine, + ); + expect(snippet).toContain('def train'); + expect(snippet).not.toContain('cell_type'); + }); +}); diff --git a/gitnexus/test/unit/ipynb-python-captures.test.ts b/gitnexus/test/unit/ipynb-python-captures.test.ts new file mode 100644 index 000000000..505674de7 --- /dev/null +++ b/gitnexus/test/unit/ipynb-python-captures.test.ts @@ -0,0 +1,46 @@ +import { describe, it, expect } from 'vitest'; +import { emitPythonScopeCaptures } from '../../src/core/ingestion/languages/python/captures.js'; +import { pythonProvider } from '../../src/core/ingestion/languages/python.js'; +import { extractParsedFile } from '../../src/core/ingestion/scope-extractor-bridge.js'; +import { getLanguageFromFilename, SupportedLanguages } from 'gitnexus-shared'; +import { getProviderForFile } from '../../src/core/ingestion/languages/index.js'; + +const notebook = JSON.stringify( + { + nbformat: 4, + nbformat_minor: 5, + metadata: { + kernelspec: { language: 'python', name: 'python3', display_name: 'Python' }, + }, + cells: [ + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }, + null, + 2, +); + +describe('Python notebook scope captures', () => { + it('detects .ipynb as Python', () => { + expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python); + expect(getProviderForFile('n.ipynb')?.id).toBe(SupportedLanguages.Python); + }); + + it('does not parse raw JSON as Python on the full-file path', () => { + const matches = emitPythonScopeCaptures(notebook, 'analysis.ipynb'); + const names = matches.flatMap((m) => Object.keys(m)); + expect(names.some((k) => k.includes('function'))).toBe(true); + }); + + it('returns no captures for invalid notebook JSON', () => { + expect(emitPythonScopeCaptures('{not json', 'broken.ipynb')).toEqual([]); + }); + + it('extractParsedFile yields a Function for train', () => { + const parsed = extractParsedFile(pythonProvider, notebook, 'analysis.ipynb'); + expect(parsed).toBeDefined(); + expect(parsed!.localDefs.some((d) => d.qualifiedName === 'train' || d.name === 'train')).toBe( + true, + ); + }); +});