From 274ec3df6ea2165ff0d6021523eadefa0eba3c67 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Gerg=C5=91=20Magyar?= Date: Sun, 27 Sep 2026 06:54:37 +0100 Subject: [PATCH] feat: index Jupyter notebooks as Python (#3381) --- docs/languages/jupyter-notebook.md | 44 ++ gitnexus-shared/src/index.ts | 1 + gitnexus-shared/src/language-detection.ts | 7 +- gitnexus/src/core/embeddings/ast-utils.ts | 17 +- gitnexus/src/core/ingestion/cfg/collect.ts | 32 +- .../src/core/ingestion/ipynb-extractor.ts | 697 ++++++++++++++++++ .../src/core/ingestion/language-provider.ts | 3 + .../src/core/ingestion/languages/python.ts | 2 +- .../ingestion/languages/python/captures.ts | 90 ++- .../core/ingestion/scope-extractor-bridge.ts | 7 +- .../core/ingestion/workers/parse-worker.ts | 31 +- gitnexus/src/core/lbug/csv-generator.ts | 18 +- gitnexus/src/storage/parse-cache.ts | 4 +- .../resolvers/ipynb-python-pipeline.test.ts | 93 +++ gitnexus/test/unit/ast-utils.test.ts | 29 + .../test/unit/incremental-parse-cache.test.ts | 7 +- gitnexus/test/unit/ingestion-utils.test.ts | 8 + gitnexus/test/unit/ipynb-extractor.test.ts | 428 +++++++++++ .../test/unit/ipynb-python-captures.test.ts | 52 ++ 19 files changed, 1540 insertions(+), 30 deletions(-) create mode 100644 docs/languages/jupyter-notebook.md create mode 100644 gitnexus/src/core/ingestion/ipynb-extractor.ts create mode 100644 gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts create mode 100644 gitnexus/test/unit/ipynb-extractor.test.ts create mode 100644 gitnexus/test/unit/ipynb-python-captures.test.ts diff --git a/docs/languages/jupyter-notebook.md b/docs/languages/jupyter-notebook.md new file mode 100644 index 000000000..f69c77aaf --- /dev/null +++ b/docs/languages/jupyter-notebook.md @@ -0,0 +1,44 @@ +# Jupyter notebook (.ipynb) indexing + +Status: implemented (Python code cells) + +GitNexus indexes Jupyter notebooks by extracting Python code cells and parsing them with the existing Python language provider. Notebooks are not executed. + +## Goal + +After `analyze`, functions, classes, and imports defined in Python code cells are queryable like ordinary `.py` files. + +## Compatibility + +- `.ipynb` is detected as Python (`gitnexus-shared` `EXTENSION_MAP` and `pythonProvider.extensions`). +- Extraction lives in `gitnexus/src/core/ingestion/ipynb-extractor.ts`. Shared ingestion modules do not name nbformat AST types. +- Notebooks are **not** Python import targets. A notebook may import `.py` modules; `import some_notebook` does not resolve to an `.ipynb`. +- Files over the walker size cap (default 512KB, `GITNEXUS_MAX_FILE_SIZE`) are skipped like any other oversized file. Output-heavy notebooks may need a higher cap. Outputs are not stripped in this slice. +- Group-layer FastAPI/Flask/Django scanners that require a `.py` suffix still ignore notebooks. + +## Kernel and magics + +- Skip the file when `kernelspec.language` or `language_info.name` is present and is not a Python-family name (`python`, `python2`, `python3`, `ipython`, `python 3`, `ipython3`). `python` and `python3` together still index. +- If `kernelspec.language` is missing, `kernelspec.name` is used only when it is an obvious language id (`python3`, `ir`, `julia-1.8`). Conda env names are ignored. +- If `language_info.name` is missing, `language_info.file_extension` (`.py` vs `.r` / `.jl`) is used the same way. +- If those fields are absent, code cells are treated as Python unless a cell's own `language`, `metadata.language`, or `metadata.vscode.languageId` says otherwise. +- A cell whose first non-empty line is a foreign cell magic (`%%bash`, `%%html`, `%%sql`, `%%R`) is skipped. Python-body cell magics (`%%time`, `%%timeit`, `%%capture`, `%%prun`, `%%debug`, `%%px`, `%%python`) stay, with the magic line commented. +- `%run`, `%load`, and `%loadpy` of a local `.py` path become `import module` on that same line so the notebook links to the file. URLs, `..` paths, and `.ipynb` targets stay comments. The notebook is not executed. +- Sage, SageMath, MicroPython, Pyodide, PyPy, and PySpark kernels are indexed as Python. SQL, R, and Julia cells are not. +- A cell with an unclosed string or bracket is commented out so it cannot hide later cells. Those cells are concatenated in order; one syntax error no longer drops the rest of the notebook. +- Line magics (`%`), shell (`!`), and IPython help (`train?`, `?train`) are commented in place so JSON line mapping stays affine. +- nbformat v4 `cells` / `source` is the normal path. nbformat v3 `worksheets[].cells` and code-cell `input` are accepted. A leading UTF-8 BOM is accepted. Markdown and raw cells are not code. + +## Line numbers + +Graph `startLine` / `endLine` are 0-based coordinates in the on-disk `.ipynb` JSON. FTS/MCP symbol snippets reconstruct cell Python; they do not slice raw JSON. + +Concatenating cells in document order is notebook semantics. An earlier cell with a syntax error may cause later definitions to be missed; that is a documented limitation, not a per-cell fallback parse. + +## Tests + +- `gitnexus/test/unit/ipynb-extractor.test.ts` +- `gitnexus/test/unit/ingestion-utils.test.ts` (`.ipynb` detection) +- `gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts` + +No Jupyter, nbconvert, or nbformat runtime dependency. diff --git a/gitnexus-shared/src/index.ts b/gitnexus-shared/src/index.ts index de49223bf..b5fbe37a5 100644 --- a/gitnexus-shared/src/index.ts +++ b/gitnexus-shared/src/index.ts @@ -22,6 +22,7 @@ export { getLanguageFromFilename, getSyntaxLanguageFromFilename, isBladeTemplateFilename, + isNotebookFilename, } from './language-detection.js'; export type { MroStrategy } from './mro-strategy.js'; diff --git a/gitnexus-shared/src/language-detection.ts b/gitnexus-shared/src/language-detection.ts index 3fe27f7b7..583a0fd5f 100644 --- a/gitnexus-shared/src/language-detection.ts +++ b/gitnexus-shared/src/language-detection.ts @@ -29,7 +29,7 @@ const RUBY_EXTENSIONLESS_FILES = new Set([ const EXTENSION_MAP: Record = { [SupportedLanguages.JavaScript]: ['.js', '.jsx', '.mjs', '.cjs'], [SupportedLanguages.TypeScript]: ['.ts', '.tsx', '.mts', '.cts'], - [SupportedLanguages.Python]: ['.py'], + [SupportedLanguages.Python]: ['.py', '.ipynb'], [SupportedLanguages.Java]: ['.java'], [SupportedLanguages.C]: ['.c'], [SupportedLanguages.ObjectiveC]: ['.m', '.mm'], @@ -76,6 +76,10 @@ for (const [lang, exts] of Object.entries(EXTENSION_MAP) as [ export const isBladeTemplateFilename = (filePath: string): boolean => filePath.replace(/\\/g, '/').toLowerCase().endsWith('.blade.php'); +/** Jupyter notebooks: ingested as Python; on-disk bytes are JSON. */ +export const isNotebookFilename = (filePath: string): boolean => + filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb'); + /** * Map file extension to SupportedLanguage enum. * Returns null if the file extension is not recognized. @@ -163,6 +167,7 @@ const AUXILIARY_BASENAME_MAP: Record = { */ export const getSyntaxLanguageFromFilename = (filePath: string): string => { if (isBladeTemplateFilename(filePath)) return 'markup'; + if (isNotebookFilename(filePath)) return 'json'; const lang = getLanguageFromFilename(filePath); if (lang) return SYNTAX_MAP[lang]; diff --git a/gitnexus/src/core/embeddings/ast-utils.ts b/gitnexus/src/core/embeddings/ast-utils.ts index b29ca80fb..f0478b452 100644 --- a/gitnexus/src/core/embeddings/ast-utils.ts +++ b/gitnexus/src/core/embeddings/ast-utils.ts @@ -11,6 +11,7 @@ import { } from '../tree-sitter/parser-loader.js'; import { parseSourceSafe } from '../tree-sitter/safe-parse.js'; import { getLanguageForFileContent, getProvider } from '../ingestion/languages/index.js'; +import { extractNotebookPython, isNotebookPath } from '../ingestion/ipynb-extractor.js'; const parserCache = new Map(); @@ -38,9 +39,21 @@ export const ensureAndParse = async (content: string, filePath: string): Promise // C++ UE macros, Dart extension types) would leave embeddings looking at an // error-recovered tree. Resolved from `language` so the transform and the // parser always come from the same provider. Length-preserving, so node - // offsets still index `content`. - const parseContent = getProvider(language).preprocessSource?.(content, filePath) ?? content; + // offsets still index `content` except for `.ipynb`, which is replaced by + // concatenated code-cell Python (same as the parse worker). + const provider = getProvider(language); + if (isNotebookPath(filePath)) { + const body = content.charCodeAt(0) === 0xfeff ? content.slice(1) : content; + if (body.trimStart().startsWith('{')) { + const extracted = extractNotebookPython(content); + if (!extracted) return null; + const parseContent = + provider.preprocessSource?.(extracted.pythonSource, filePath) ?? extracted.pythonSource; + return parseSourceSafe(parserInstance, parseContent); + } + } + const parseContent = provider.preprocessSource?.(content, filePath) ?? content; return parseSourceSafe(parserInstance, parseContent); }; diff --git a/gitnexus/src/core/ingestion/cfg/collect.ts b/gitnexus/src/core/ingestion/cfg/collect.ts index bdf1b095d..4e43fbeea 100644 --- a/gitnexus/src/core/ingestion/cfg/collect.ts +++ b/gitnexus/src/core/ingestion/cfg/collect.ts @@ -15,7 +15,7 @@ */ import type { SyntaxNode } from '../utils/ast-helpers.js'; import { CfgNestingDepthError } from './cfg-builder.js'; -import type { CfgVisitor, FunctionCfg } from './types.js'; +import type { CfgVisitor, FunctionCfg, SiteRecord } from './types.js'; /** * Default per-function source-line cap used by the worker when the `--pdg` run @@ -83,12 +83,38 @@ function shiftCfgLines(cfg: FunctionCfg, offset: number): FunctionCfg { }; } +function remapCfgLines(cfg: FunctionCfg, mapLine: (row: number) => number): FunctionCfg { + // CFG visitors store 1-based source lines; `mapLine` maps 0-based tree-sitter rows. + const map1 = (line: number): number => mapLine(line - 1) + 1; + const mapSite = (site: SiteRecord): SiteRecord => + site.at !== undefined ? { ...site, at: [map1(site.at[0]), site.at[1]] } : site; + return { + ...cfg, + functionStartLine: map1(cfg.functionStartLine), + functionEndLine: map1(cfg.functionEndLine), + blocks: cfg.blocks.map((b) => ({ + ...b, + startLine: map1(b.startLine), + endLine: map1(b.endLine), + statements: b.statements?.map((s) => ({ + ...s, + line: map1(s.line), + sites: s.sites?.map(mapSite), + })), + })), + bindings: cfg.bindings?.map((bd) => + bd.declLine > 0 ? { ...bd, declLine: map1(bd.declLine) } : bd, + ), + }; +} + export function collectFunctionCfgs( root: SyntaxNode, visitor: CfgVisitor, filePath: string, maxFunctionLines = 0, lineOffset = 0, + mapLine?: (row: number) => number, ): CollectedCfgs { const cfgs: FunctionCfg[] = []; let tooManyLines = 0; @@ -109,7 +135,9 @@ export function collectFunctionCfgs( // and silently drop every remaining function's CFG (#2195). try { const cfg = visitor.buildFunctionCfg(node, filePath); - if (cfg) cfgs.push(shiftCfgLines(cfg, lineOffset)); + if (cfg) { + cfgs.push(mapLine ? remapCfgLines(cfg, mapLine) : shiftCfgLines(cfg, lineOffset)); + } } catch (err) { if (err instanceof CfgNestingDepthError) tooDeeplyNested++; else buildError++; diff --git a/gitnexus/src/core/ingestion/ipynb-extractor.ts b/gitnexus/src/core/ingestion/ipynb-extractor.ts new file mode 100644 index 000000000..b4e6dadda --- /dev/null +++ b/gitnexus/src/core/ingestion/ipynb-extractor.ts @@ -0,0 +1,697 @@ +/** + * Jupyter notebook (.ipynb) Python extractor. + * + * Pulls code-cell source from nbformat JSON so the Python tree-sitter + * grammar can parse it. Extraction helpers are I/O-free and worker-safe. + * `extractNotebookPythonCached` is an optional process-local LRU for FTS. + * + * Graph coordinates stay 0-based lines in the on-disk JSON file. Extract + * buffer lines map through {@link mapExtractLine}. + */ + +import { isNotebookFilename } from 'gitnexus-shared'; +import { buildLineIndex, lineFromOffset } from '../embeddings/line-index.js'; + +export interface NotebookLineSegment { + readonly extractStartLine: number; + readonly extractEndLine: number; + readonly jsonStartLine: number; + readonly jsonEndLine: number; +} + +export interface NotebookPythonExtraction { + readonly pythonSource: string; + readonly segments: readonly NotebookLineSegment[]; +} + +const PYTHON_FAMILY = new Set([ + 'python', + 'python2', + 'python3', + 'ipython', + 'sage', + 'sagemath', + 'micropython', + 'pyodide', + 'pypy', + 'pyspark', +]); + +export function isNotebookPath(filePath: string): boolean { + return isNotebookFilename(filePath); +} + +export function isPythonFamilyLanguage(name: string | undefined | null): boolean { + if (name === undefined || name === null) return false; + const n = name.trim().toLowerCase(); + if (PYTHON_FAMILY.has(n)) return true; + if (n.startsWith('ipython')) return true; + return /^python(?:\d|\b)/.test(n); +} + +/** Kernel spec names that are a language id, not a conda env label. */ +const NON_PYTHON_KERNEL = + /^(?:julia|ir|r|scala|rust|ruby|javascript|nodejs|node|bash|sh|sql|csharp|fsharp|go|java|kotlin|swift|php|perl|lua|octave|matlab|sas|stata|haskell|clojure|elixir|groovy|powershell|sos|cpp|cxx|dart|typescript|wolfram|scheme|racket|ocaml|fortran|gnuplot|sqlite|mysql|postgresql|tsql)(?:[-_][\w.-]+|\d[\w.-]*)?$/i; + +function kernelNameLanguage(name: string): string | undefined { + if (isPythonFamilyLanguage(name) || NON_PYTHON_KERNEL.test(name.trim())) return name; + return undefined; +} + +function extensionLanguage(ext: string): string | undefined { + const e = ext.trim().toLowerCase(); + if (e === '.py' || e === '.pyi' || e === '.ipy') return 'python'; + if ( + e === '.r' || + e === '.jl' || + e === '.scala' || + e === '.js' || + e === '.java' || + e === '.go' || + e === '.rs' || + e === '.rb' || + e === '.php' || + e === '.kt' || + e === '.swift' || + e === '.cs' + ) { + return ext; + } + return undefined; +} + +function skipWs(content: string, i: number): number { + while (i < content.length) { + const c = content.charCodeAt(i); + if (c === 32 || c === 9 || c === 10 || c === 13) i++; + else break; + } + return i; +} + +/** Span of a JSON string or array value starting at the first `"` or `[`. */ +function jsonValueSpan(content: string, start: number): { start: number; end: number } | null { + const i = skipWs(content, start); + if (i >= content.length) return null; + if (content[i] === '"') { + let j = i + 1; + while (j < content.length) { + if (content[j] === '\\') { + j += 2; + continue; + } + if (content[j] === '"') return { start: i, end: j + 1 }; + j++; + } + return null; + } + if (content[i] === '[') { + let depth = 1; + let j = i + 1; + let inStr = false; + while (j < content.length && depth > 0) { + const ch = content[j]; + if (inStr) { + if (ch === '\\') j += 2; + else { + if (ch === '"') inStr = false; + j++; + } + continue; + } + if (ch === '"') inStr = true; + else if (ch === '[') depth++; + else if (ch === ']') depth--; + j++; + } + return { start: i, end: j }; + } + return null; +} + +function jsonBraceSpan( + content: string, + start: number, + open: '{' | '[', + close: '}' | ']', +): { start: number; end: number } | null { + const i = skipWs(content, start); + if (i >= content.length || content[i] !== open) return null; + let depth = 1; + let j = i + 1; + let inStr = false; + while (j < content.length && depth > 0) { + const ch = content[j]; + if (inStr) { + if (ch === '\\') j += 2; + else { + if (ch === '"') inStr = false; + j++; + } + continue; + } + if (ch === '"') inStr = true; + else if (ch === open) depth++; + else if (ch === close) depth--; + j++; + } + if (depth !== 0) return null; + return { start: i, end: j }; +} + +function findDepth1Key( + content: string, + objStart: number, + objEnd: number, + key: string, + match: 'first' | 'last' = 'first', +): number { + let depth = 0; + let inStr = false; + let j = objStart; + let found = -1; + while (j < objEnd) { + const ch = content[j]; + if (inStr) { + if (ch === '\\') j += 2; + else { + if (ch === '"') inStr = false; + j++; + } + continue; + } + if (ch === '"') { + if (depth === 1 && content.startsWith(key, j)) { + if (match === 'first') return j; + found = j; + } + inStr = true; + j++; + continue; + } + if (ch === '{' || ch === '[') depth++; + else if (ch === '}' || ch === ']') depth--; + j++; + } + return found; +} + +function arraySpanAfterKey( + content: string, + objStart: number, + objEnd: number, + key: '"cells"' | '"worksheets"', + match: 'first' | 'last', +): { start: number; end: number } | null { + const keyAt = findDepth1Key(content, objStart, objEnd, key, match); + if (keyAt < 0) return null; + const colon = content.indexOf(':', keyAt + key.length); + if (colon < 0 || colon >= objEnd) return null; + return jsonBraceSpan(content, colon + 1, '[', ']'); +} + +function codeCellArraySpans(content: string, origin: number): { start: number; end: number }[] { + const root = jsonBraceSpan(content, origin, '{', '}'); + if (!root) return []; + const cells = arraySpanAfterKey(content, root.start, root.end, '"cells"', 'last'); + if (cells) return [cells]; + const worksheets = arraySpanAfterKey(content, root.start, root.end, '"worksheets"', 'last'); + if (!worksheets) return []; + const spans: { start: number; end: number }[] = []; + let search = worksheets.start + 1; + while (search < worksheets.end) { + const brace = nextUnquotedChar(content, search, worksheets.end, '{'); + if (brace < 0) break; + const obj = jsonBraceSpan(content, brace, '{', '}'); + if (!obj || obj.end > worksheets.end) { + search = brace + 1; + continue; + } + const nested = arraySpanAfterKey(content, obj.start, obj.end, '"cells"', 'first'); + if (nested) spans.push(nested); + search = obj.end; + } + return spans; +} + +function readJsonString(content: string, start: number): string | null { + const i = skipWs(content, start); + if (content[i] !== '"') return null; + let j = i + 1; + let out = ''; + while (j < content.length) { + const ch = content[j]; + if (ch === '\\') { + out += content[j + 1] ?? ''; + j += 2; + continue; + } + if (ch === '"') return out; + out += ch; + j++; + } + return null; +} + +function nextUnquotedChar(content: string, from: number, until: number, needle: '{' | '"'): number { + let inStr = false; + let j = from; + while (j < until) { + const ch = content[j]; + if (inStr) { + if (ch === '\\') j += 2; + else { + if (ch === '"') inStr = false; + j++; + } + continue; + } + if (ch === '"') { + if (needle === '"') return j; + inStr = true; + j++; + continue; + } + if (ch === needle) return j; + j++; + } + return -1; +} + +function firstQuoteInValue(content: string, span: { start: number; end: number }): number { + const i = skipWs(content, span.start); + if (i < span.end && content[i] === '"') return i; + if (i < span.end && content[i] === '[') { + const q = nextUnquotedChar(content, i + 1, span.end, '"'); + if (q >= 0) return q; + } + return span.start; +} + +function findNextCodeCellSourceSpan( + content: string, + from: number, + cells: { start: number; end: number }, +): { span: { start: number; end: number } | null; nextFrom: number } | null { + let search = Math.max(from, cells.start + 1); + while (search < cells.end) { + const brace = nextUnquotedChar(content, search, cells.end, '{'); + if (brace < 0) return null; + const obj = jsonBraceSpan(content, brace, '{', '}'); + if (!obj || obj.end > cells.end) { + search = brace + 1; + continue; + } + const typeKey = findDepth1Key(content, obj.start, obj.end, '"cell_type"'); + if (typeKey < 0) { + search = obj.end; + continue; + } + const colon = content.indexOf(':', typeKey + 11); + if (colon < 0 || colon >= obj.end) { + search = obj.end; + continue; + } + const valueStart = skipWs(content, colon + 1); + const cellType = readJsonString(content, valueStart); + if (cellType === null || cellType.toLowerCase() !== 'code') { + search = obj.end; + continue; + } + let sourceKey = findDepth1Key(content, obj.start, obj.end, '"source"'); + let keyLen = 8; + if (sourceKey < 0) { + sourceKey = findDepth1Key(content, obj.start, obj.end, '"input"'); + keyLen = 7; + } + if (sourceKey < 0) { + return { span: null, nextFrom: obj.end }; + } + const srcColon = content.indexOf(':', sourceKey + keyLen); + if (srcColon < 0 || srcColon >= obj.end) { + return { span: null, nextFrom: obj.end }; + } + const span = jsonValueSpan(content, srcColon + 1); + if (!span) { + return { span: null, nextFrom: obj.end }; + } + return { span, nextFrom: obj.end }; + } + return null; +} + +function flattenSource(source: unknown): string { + if (typeof source === 'string') return source; + if (Array.isArray(source)) { + return source.map((part) => (typeof part === 'string' ? part : '')).join(''); + } + return ''; +} + +function cellLanguage(cell: Record): string | undefined { + if (typeof cell.language === 'string') return cell.language; + const meta = cell.metadata; + if (!meta || typeof meta !== 'object') return undefined; + const m = meta as Record; + if (typeof m.language === 'string') return m.language; + const vscode = m.vscode; + if (vscode && typeof vscode === 'object') { + const langId = (vscode as Record).languageId; + if (typeof langId === 'string') return langId; + } + return undefined; +} + +function notebookLanguageFields(nb: Record): { + kernelspec?: string; + languageInfo?: string; +} { + const metadata = nb.metadata; + if (!metadata || typeof metadata !== 'object') return {}; + const md = metadata as Record; + const ks = md.kernelspec; + const li = md.language_info; + return { + kernelspec: + ks && typeof ks === 'object' + ? typeof (ks as Record).language === 'string' + ? String((ks as Record).language) + : kernelNameLanguage(String((ks as Record).name ?? '')) + : undefined, + languageInfo: + li && typeof li === 'object' + ? typeof (li as Record).name === 'string' + ? String((li as Record).name) + : extensionLanguage(String((li as Record).file_extension ?? '')) + : undefined, + }; +} + +function kernelShouldSkip(nb: Record): boolean { + const { kernelspec, languageInfo } = notebookLanguageFields(nb); + const fields = [kernelspec, languageInfo].filter((x): x is string => x !== undefined); + if (fields.length === 0) return false; + return fields.some((f) => !isPythonFamilyLanguage(f)); +} + +/** Cell magics whose body is still Python. Foreign %% cells are skipped. */ +const PYTHON_CELL_MAGICS = new Set([ + 'time', + 'timeit', + 'capture', + 'prun', + 'lprun', + 'mprun', + 'debug', + 'px', + 'python', + 'python2', + 'python3', + 'ipython', + 'pypy', +]); + +function cellMagicName(line: string): string { + const token = line.trimStart().slice(2).trim().split(/\s+/)[0] ?? ''; + return token.toLowerCase(); +} + +function isIpythonHelpLine(line: string): boolean { + const t = line.trim(); + if (t.length === 0 || t.startsWith('#')) return false; + return /^\?\??\S/.test(t) || /^[A-Za-z_][\w.]*(?:\?\?|\?)\s*$/.test(t); +} + +function moduleSpecFromRunPath(raw: string): string | null { + let path = raw + .trim() + .replace(/^['"]|['"]$/g, '') + .replace(/\\/g, '/'); + if (path.startsWith('http://') || path.startsWith('https://') || path.endsWith('.ipynb')) { + return null; + } + path = path.replace(/^\.\//, ''); + if (path.endsWith('.py')) path = path.slice(0, -3); + const parts = path.split('/').filter((part) => part.length > 0); + if (parts.length === 0 || parts.some((part) => part === '..' || !/^[A-Za-z_]\w*$/.test(part))) { + return null; + } + return parts.join('.'); +} + +function rewriteRunOrLoad(line: string): string | null { + const trimmed = line.trimStart(); + const indent = line.slice(0, line.length - trimmed.length); + const match = trimmed.match(/^%(?:run|load|loadpy)\s+(\S+)/); + if (!match) return null; + const spec = moduleSpecFromRunPath(match[1]); + if (!spec) return null; + return `${indent}import ${spec} # ${trimmed}`; +} + +function isBalancedPython(source: string): boolean { + let quote: "'" | '"' | null = null; + let triple = false; + let paren = 0; + let bracket = 0; + let brace = 0; + for (let i = 0; i < source.length; i++) { + const ch = source[i]; + if (quote) { + if (ch === '\\' && !triple) { + i++; + continue; + } + if (triple && source.startsWith(quote.repeat(3), i)) { + quote = null; + triple = false; + i += 2; + continue; + } + if (!triple && ch === quote) quote = null; + continue; + } + if (ch === '#') { + const nl = source.indexOf('\n', i); + if (nl < 0) break; + i = nl; + continue; + } + if ((ch === '"' || ch === "'") && source.startsWith(ch.repeat(3), i)) { + quote = ch; + triple = true; + i += 2; + continue; + } + if (ch === '"' || ch === "'") { + quote = ch; + continue; + } + if (ch === '(') paren++; + else if (ch === ')') paren--; + else if (ch === '[') bracket++; + else if (ch === ']') bracket--; + else if (ch === '{') brace++; + else if (ch === '}') brace--; + if (paren < 0 || bracket < 0 || brace < 0) return false; + } + return quote === null && paren === 0 && bracket === 0 && brace === 0; +} + +function processCellLines(raw: string): { skipCell: boolean; lines: string[] } { + const lines = raw.replace(/\r\n/g, '\n').replace(/\r/g, '\n').split('\n'); + const firstNonEmpty = lines.find((l) => l.trim().length > 0); + const magic = + firstNonEmpty !== undefined && firstNonEmpty.trimStart().startsWith('%%') + ? cellMagicName(firstNonEmpty) + : undefined; + if (magic !== undefined && !PYTHON_CELL_MAGICS.has(magic) && !isPythonFamilyLanguage(magic)) { + return { skipCell: true, lines: [''] }; + } + const rewritten = lines.map((line) => { + const t = line.trimStart(); + if (isIpythonHelpLine(line) || t.startsWith('!')) return `# ${line}`; + const runImport = rewriteRunOrLoad(line); + if (runImport) return runImport; + if (t.startsWith('%')) return `# ${line}`; + return line; + }); + if (isBalancedPython(rewritten.join('\n'))) { + return { skipCell: false, lines: rewritten }; + } + return { + skipCell: false, + lines: rewritten.map((line) => (line.trim().length === 0 ? line : `# ${line}`)), + }; +} + +function notebookCodeCells(nb: Record): unknown[] | null { + if (Array.isArray(nb.cells)) return nb.cells; + if (!Array.isArray(nb.worksheets)) return null; + const cells: unknown[] = []; + for (const worksheet of nb.worksheets) { + if (!worksheet || typeof worksheet !== 'object') continue; + const nested = (worksheet as Record).cells; + if (Array.isArray(nested)) cells.push(...nested); + } + return cells.length > 0 ? cells : null; +} + +function cellSource(cell: Record): unknown { + return cell.source !== undefined ? cell.source : cell.input; +} + +export function extractNotebookPython(content: string): NotebookPythonExtraction | null { + const origin = content.charCodeAt(0) === 0xfeff ? 1 : 0; + let parsed: unknown; + try { + parsed = JSON.parse(origin === 0 ? content : content.slice(origin)); + } catch { + return null; + } + if (!parsed || typeof parsed !== 'object') return null; + const nb = parsed as Record; + const codeCells = notebookCodeCells(nb); + if (!codeCells) return null; + if (kernelShouldSkip(nb)) return null; + + const spans = codeCellArraySpans(content, origin); + if (spans.length === 0) return null; + + const lineStarts = buildLineIndex(content); + const chunks: string[] = []; + const segments: NotebookLineSegment[] = []; + let spanIndex = 0; + let searchFrom = 0; + + for (const rawCell of codeCells) { + if (!rawCell || typeof rawCell !== 'object') continue; + const cell = rawCell as Record; + if (String(cell.cell_type).toLowerCase() !== 'code') continue; + + let located: { span: { start: number; end: number } | null; nextFrom: number } | null = null; + while (spanIndex < spans.length) { + located = findNextCodeCellSourceSpan(content, searchFrom, spans[spanIndex]); + if (located) break; + spanIndex++; + searchFrom = 0; + } + if (!located) return null; + searchFrom = located.nextFrom; + if (!located.span) continue; + + const lang = cellLanguage(cell); + if (lang !== undefined && !isPythonFamilyLanguage(lang)) { + continue; + } + + const { skipCell, lines } = processCellLines(flattenSource(cellSource(cell))); + const jsonStartLine = lineFromOffset(lineStarts, firstQuoteInValue(content, located.span)); + const jsonEndLine = Math.max(jsonStartLine, lineFromOffset(lineStarts, located.span.end - 1)); + + if (skipCell) { + continue; + } + + let text = lines.join('\n'); + const hadTrailingNewline = text.endsWith('\n'); + if (hadTrailingNewline) text = text.slice(0, -1); + if (text.trim().length === 0) { + continue; + } + + if (chunks.length > 0) { + chunks.push('\n\n'); + } + const lineCount = hadTrailingNewline ? Math.max(1, lines.length - 1) : lines.length; + const extractStartLine = + segments.length === 0 ? 0 : segments[segments.length - 1].extractEndLine + 2; + chunks.push(text); + segments.push({ + extractStartLine, + extractEndLine: extractStartLine + lineCount - 1, + jsonStartLine, + jsonEndLine, + }); + } + + const pythonSource = chunks.join(''); + if (pythonSource.trim().length === 0) return null; + return { pythonSource, segments }; +} + +export function mapExtractLine(row: number, segments: readonly NotebookLineSegment[]): number { + if (segments.length === 0) return row; + for (const seg of segments) { + if (row >= seg.extractStartLine && row <= seg.extractEndLine) { + const delta = row - seg.extractStartLine; + const jsonSpan = seg.jsonEndLine - seg.jsonStartLine; + return seg.jsonStartLine + Math.min(delta, jsonSpan); + } + } + if (row < segments[0].extractStartLine) return segments[0].jsonStartLine; + const last = segments[segments.length - 1]; + return last.jsonEndLine; +} + +const extractCache = new Map< + string, + { content: string; result: NotebookPythonExtraction | null } +>(); + +/** Memoize extraction for FTS/CSV (same file, many symbols). */ +const EXTRACT_CACHE_LIMIT = 32; + +export function extractNotebookPythonCached( + filePath: string, + content: string, +): NotebookPythonExtraction | null { + const hit = extractCache.get(filePath); + if (hit && hit.content === content) { + extractCache.delete(filePath); + extractCache.set(filePath, hit); + return hit.result; + } + const result = extractNotebookPython(content); + if (extractCache.size >= EXTRACT_CACHE_LIMIT && !extractCache.has(filePath)) { + const oldest = extractCache.keys().next().value; + if (oldest !== undefined) extractCache.delete(oldest); + } + extractCache.set(filePath, { content, result }); + return result; +} + +/** Python snippet for a graph span stored in JSON file coordinates. */ +export function notebookPythonSnippetFromExtract( + extracted: NotebookPythonExtraction, + startLine: number, + endLine: number, +): string | null { + const pyLines = extracted.pythonSource.split('\n'); + const out: string[] = []; + for (const seg of extracted.segments) { + if (seg.jsonEndLine < startLine || seg.jsonStartLine > endLine) continue; + for (let extract = seg.extractStartLine; extract <= seg.extractEndLine; extract++) { + const jsonSpan = seg.jsonEndLine - seg.jsonStartLine; + const jsonLine = seg.jsonStartLine + Math.min(extract - seg.extractStartLine, jsonSpan); + if (jsonLine >= startLine && jsonLine <= endLine) { + out.push(pyLines[extract] ?? ''); + } + } + } + if (out.length === 0) return null; + return out.join('\n'); +} + +export function notebookPythonSnippet( + fileContent: string, + startLine: number, + endLine: number, + filePath?: string, +): string | null { + const extracted = filePath + ? extractNotebookPythonCached(filePath, fileContent) + : extractNotebookPython(fileContent); + if (!extracted) return null; + return notebookPythonSnippetFromExtract(extracted, startLine, endLine); +} diff --git a/gitnexus/src/core/ingestion/language-provider.ts b/gitnexus/src/core/ingestion/language-provider.ts index 6b5fb2d76..64c4736e7 100644 --- a/gitnexus/src/core/ingestion/language-provider.ts +++ b/gitnexus/src/core/ingestion/language-provider.ts @@ -26,6 +26,7 @@ import type { WorkspaceIndex, } from 'gitnexus-shared'; import type { LanguageTypeConfig } from './type-extractors/types.js'; +import type { NotebookLineSegment } from './ipynb-extractor.js'; import type { CallRouter } from './call-routing.js'; import type { CallExtractor } from './call-types.js'; import type { ClassExtractor } from './class-types.js'; @@ -828,6 +829,8 @@ interface LanguageProviderConfig { */ sourceMeta?: { readonly sourceKind?: 'full-file' | 'pre-extracted-script'; + /** Python `.ipynb` only: JSON line segments for the pre-extracted buffer. */ + readonly notebookSegments?: readonly NotebookLineSegment[]; }, ) => readonly CaptureMatch[]; diff --git a/gitnexus/src/core/ingestion/languages/python.ts b/gitnexus/src/core/ingestion/languages/python.ts index ce8d0fa90..d07902f66 100644 --- a/gitnexus/src/core/ingestion/languages/python.ts +++ b/gitnexus/src/core/ingestion/languages/python.ts @@ -107,7 +107,7 @@ function normalizePythonStringLiteral(text: string): string | undefined { export const pythonProvider = defineLanguage({ id: SupportedLanguages.Python, - extensions: ['.py'], + extensions: ['.py', '.ipynb'], entryPointPatterns: [/^app$/, /^(get|post|put|delete|patch)_/i, /^api_/, /^view_/], astFrameworkPatterns: [ { diff --git a/gitnexus/src/core/ingestion/languages/python/captures.ts b/gitnexus/src/core/ingestion/languages/python/captures.ts index e1aefd57b..5b212e525 100644 --- a/gitnexus/src/core/ingestion/languages/python/captures.ts +++ b/gitnexus/src/core/ingestion/languages/python/captures.ts @@ -12,10 +12,10 @@ * binding, and `__init__` assignments from annotated parameters emit * class-scoped instance-field bindings (see `receiver-binding.ts`). * - * Pure given the input source text. No I/O, no globals consulted. + * No I/O. A `.ipynb` path also depends on `filePath` and `sourceMeta`, not only the source text. */ -import type { Capture, CaptureMatch } from 'gitnexus-shared'; +import type { Capture, CaptureMatch, Range } from 'gitnexus-shared'; import { nodeToCapture, syntheticCapture, @@ -24,6 +24,12 @@ import { } from '../../utils/ast-helpers.js'; import { splitImportStatement } from './import-decomposer.js'; import { getPythonParser, getPythonScopeQuery } from './query.js'; +import { + extractNotebookPython, + isNotebookPath, + mapExtractLine, + type NotebookLineSegment, +} from '../../ipynb-extractor.js'; import { synthesizeConstructorFieldTypeBindings, synthesizeReceiverTypeBinding, @@ -71,23 +77,36 @@ const PYTHON_CALLABLE_CAPTURE_OPTIONS = { export function emitPythonScopeCaptures( sourceText: string, - _filePath: string, + filePath: string, cachedTree?: unknown, + sourceMeta?: { + sourceKind?: 'full-file' | 'pre-extracted-script'; + notebookSegments?: readonly NotebookLineSegment[]; + }, ): readonly CaptureMatch[] { + let parseText = sourceText; + let tree = cachedTree as ReturnType['parse']> | undefined; + let notebookSegments: readonly NotebookLineSegment[] | undefined; + if (isNotebookPath(filePath)) { + const resolved = resolveNotebookCaptureSource(sourceText, tree, sourceMeta); + if (resolved === null) return []; + parseText = resolved.parseText; + tree = resolved.tree; + notebookSegments = resolved.notebookSegments; + } // Skip the parse when the caller (the scope-resolution orchestrator's // `treeCache`) already produced a Tree for this source — empty under // worker-pool runs, so cache miss = re-parse. The cachedTree parameter // is typed as `unknown` at the // contract layer (see `LanguageProvider.emitScopeCaptures`); cast // here at the use site. - let tree = cachedTree as ReturnType['parse']> | undefined; if (tree === undefined) { try { - tree = parseSourceSafe(getPythonParser(), sourceText, undefined, { - bufferSize: getTreeSitterBufferSize(sourceText), + tree = parseSourceSafe(getPythonParser(), parseText, undefined, { + bufferSize: getTreeSitterBufferSize(parseText), }); } catch (err) { - throw scopeExtractionError('parse', _filePath, err); + throw scopeExtractionError('parse', filePath, err); } recordCacheMiss(); } else { @@ -98,7 +117,7 @@ export function emitPythonScopeCaptures( try { rawMatches = getPythonScopeQuery().matches(tree.rootNode); } catch (err) { - throw scopeExtractionError('scope query', _filePath, err); + throw scopeExtractionError('scope query', filePath, err); } const out: CaptureMatch[] = []; @@ -240,9 +259,64 @@ export function emitPythonScopeCaptures( out.push(...synthesizePythonInheritanceReferences(tree.rootNode)); out.push(...synthesizeCallableFlowCaptures(tree.rootNode, PYTHON_CALLABLE_CAPTURE_OPTIONS)); + if (notebookSegments !== undefined) { + return out.map((match) => remapCaptureMatch(match, notebookSegments)); + } return out; } +function resolveNotebookCaptureSource( + sourceText: string, + cachedTree: ReturnType['parse']> | undefined, + sourceMeta?: { + sourceKind?: 'full-file' | 'pre-extracted-script'; + notebookSegments?: readonly NotebookLineSegment[]; + }, +): { + parseText: string; + tree: ReturnType['parse']> | undefined; + notebookSegments?: readonly NotebookLineSegment[]; +} | null { + if (sourceMeta?.notebookSegments) { + return { + parseText: sourceText, + tree: cachedTree, + notebookSegments: sourceMeta.notebookSegments, + }; + } + const extracted = extractNotebookPython(sourceText); + if (extracted === null) { + if (sourceMeta?.sourceKind === 'pre-extracted-script') { + return { parseText: sourceText, tree: cachedTree }; + } + return null; + } + return { + parseText: extracted.pythonSource, + tree: sourceMeta?.sourceKind === 'pre-extracted-script' ? cachedTree : undefined, + notebookSegments: extracted.segments, + }; +} + +function remapRange(range: Range, segments: readonly NotebookLineSegment[]): Range { + return { + ...range, + startLine: mapExtractLine(range.startLine - 1, segments) + 1, + endLine: mapExtractLine(range.endLine - 1, segments) + 1, + }; +} + +function remapCaptureMatch( + match: CaptureMatch, + segments: readonly NotebookLineSegment[], +): CaptureMatch { + const next: Record = {}; + for (const [key, cap] of Object.entries(match)) { + next[key] = { ...cap, range: remapRange(cap.range, segments) }; + } + return next; +} + /** * Synthesize `@reference.inherits` captures from Python class superclass * lists so the registry-primary scope-resolution path emits EXTENDS edges diff --git a/gitnexus/src/core/ingestion/scope-extractor-bridge.ts b/gitnexus/src/core/ingestion/scope-extractor-bridge.ts index 17f639941..c95009a74 100644 --- a/gitnexus/src/core/ingestion/scope-extractor-bridge.ts +++ b/gitnexus/src/core/ingestion/scope-extractor-bridge.ts @@ -27,6 +27,7 @@ import type { ParsedFile } from 'gitnexus-shared'; import { extract as extractScope } from './scope-extractor.js'; import type { LanguageProvider } from './language-provider.js'; +import type { NotebookLineSegment } from './ipynb-extractor.js'; import { logger } from '../logger.js'; /** Callback used to report scope-extraction warnings to the host (worker or direct). */ @@ -45,6 +46,7 @@ export function extractParsedFile( onWarn?: ScopeBridgeWarn, cachedTree?: unknown, sourceKind: ScopeCaptureSourceKind = 'full-file', + notebookSegments?: readonly NotebookLineSegment[], ): ParsedFile | undefined { if (provider.emitScopeCaptures === undefined) return undefined; if (sourceText.trim().length === 0) return undefined; @@ -58,7 +60,10 @@ export function extractParsedFile( cachedTree === undefined ? (provider.preprocessSource?.(sourceText, filePath) ?? sourceText) : sourceText; - const captures = provider.emitScopeCaptures(parseText, filePath, cachedTree, { sourceKind }); + const captures = provider.emitScopeCaptures(parseText, filePath, cachedTree, { + sourceKind, + ...(notebookSegments ? { notebookSegments } : {}), + }); return extractScope(captures, filePath, provider); } catch (err) { const message = `scope extraction failed for ${filePath}: ${ diff --git a/gitnexus/src/core/ingestion/workers/parse-worker.ts b/gitnexus/src/core/ingestion/workers/parse-worker.ts index 1145b38e7..2c397b5d8 100644 --- a/gitnexus/src/core/ingestion/workers/parse-worker.ts +++ b/gitnexus/src/core/ingestion/workers/parse-worker.ts @@ -135,6 +135,12 @@ import { extractTemplateComponents, isVueSetupTopLevel, } from '../vue-sfc-extractor.js'; +import { + extractNotebookPython, + isNotebookPath, + mapExtractLine, + type NotebookLineSegment, +} from '../ipynb-extractor.js'; import type { NodeLabel, ParameterTypeClass } from 'gitnexus-shared'; import type { FieldInfo, FieldExtractorContext } from '../field-types.js'; import type { MethodInfo, MethodExtractorContext } from '../method-types.js'; @@ -1597,6 +1603,9 @@ const processFileGroup = ( let scopeSourceKind: ScopeCaptureSourceKind = 'full-file'; let lineOffset = 0; let isVueSetup = false; + let notebookSegments: readonly NotebookLineSegment[] | undefined; + const mapRow = (row: number): number => + notebookSegments ? mapExtractLine(row, notebookSegments) : row + lineOffset; if (language === SupportedLanguages.Vue) { const extracted = extractVueScript(file.content); if (!extracted) continue; // skip .vue files with no script block @@ -1604,6 +1613,12 @@ const processFileGroup = ( scopeSourceKind = 'pre-extracted-script'; lineOffset = extracted.lineOffset; isVueSetup = extracted.isSetup; + } else if (language === SupportedLanguages.Python && isNotebookPath(file.path)) { + const extracted = extractNotebookPython(file.content); + if (!extracted) continue; + parseContent = extracted.pythonSource; + scopeSourceKind = 'pre-extracted-script'; + notebookSegments = extracted.segments; } // Per-language source-text transform (e.g., UE macro stripping for C++). @@ -1676,6 +1691,7 @@ const processFileGroup = ( }, tree, scopeSourceKind, + notebookSegments, ); if (scopeExtractionFailed) (result.scopeExtractionFailures ??= []).push(file.path); if (parsedFile !== undefined) { @@ -1722,6 +1738,7 @@ const processFileGroup = ( // `lineOffset` in the file — shift the CFG into file coordinates so // it joins its graph node and BasicBlock lines map to source. lineOffset, + notebookSegments ? mapRow : undefined, ); if (cfgs.length) withChannels = { ...withChannels, cfgSideChannel: cfgs }; // Surface per-function CFG skips per-language (#2195): merged + logged @@ -1878,7 +1895,7 @@ const processFileGroup = ( sourceId: srcId, receiverText, propertyName, - line: captureMap['assignment'].startPosition.row + 1, + line: mapRow(captureMap['assignment'].startPosition.row) + 1, ...(receiverTypeName ? { receiverTypeName } : {}), }); } @@ -1913,7 +1930,7 @@ const processFileGroup = ( filePath: file.path, httpMethod, decoratorName, - lineNumber: decoratorNode.startPosition.row + lineOffset, + lineNumber: mapRow(decoratorNode.startPosition.row), ...(decoratorReceiver ? { decoratorReceiver } : {}), ...(handlerName ? { handlerName } : {}), }; @@ -2483,10 +2500,10 @@ const processFileGroup = ( : definitionNode?.startPosition; const startLine = startPosition !== undefined - ? startPosition.row + lineOffset + ? mapRow(startPosition.row) : nameNode - ? nameNode.startPosition.row + lineOffset - : lineOffset; + ? mapRow(nameNode.startPosition.row) + : mapRow(0); const startColumn = startPosition?.column ?? nameNode?.startPosition.column ?? 0; // Compute enclosing class BEFORE node ID — needed to qualify method IDs @@ -2958,7 +2975,7 @@ const processFileGroup = ( filePath: file.path, toolName: nodeName, description: (dec.arg || description || '').slice(0, 200), - lineNumber: definitionNode.startPosition.row + lineOffset, + lineNumber: mapRow(definitionNode.startPosition.row), handlerNodeId: nodeId, }); } @@ -3063,7 +3080,7 @@ const processFileGroup = ( (objectLiteralBindingInfo?.ownerName || isArrayContainedObjectCallable) ? { startColumn } : {}), - endLine: definitionNode ? definitionNode.endPosition.row + lineOffset : startLine, + endLine: definitionNode ? mapRow(definitionNode.endPosition.row) : startLine, language: language, isExported, ...(qualifiedTypeName !== undefined ? { qualifiedName: qualifiedTypeName } : {}), diff --git a/gitnexus/src/core/lbug/csv-generator.ts b/gitnexus/src/core/lbug/csv-generator.ts index 5e652319c..aafb556f6 100644 --- a/gitnexus/src/core/lbug/csv-generator.ts +++ b/gitnexus/src/core/lbug/csv-generator.ts @@ -23,6 +23,7 @@ import { VALID_NODE_TABLES, parseRelationSchemaPairs, RelPairRouter } from './re import { parseTruthyEnv } from '../ingestion/utils/env.js'; import { SYMBOL_NODE_LABELS } from '../ingestion/utils/symbol-labels.js'; import { applyCjkSegmentationIfEnabled } from '../search/cjk-segmentation.js'; +import { isNotebookPath, notebookPythonSnippet } from '../ingestion/ipynb-extractor.js'; /** Computed once — `RELATION_SCHEMA` is a static template literal. Exported so * the streamed sinks (`GraphEmitSink`, `PdgEmitSink`) share this parse @@ -325,15 +326,24 @@ const extractContent = async ( const endLine = node.properties.endLine; if (startLine === undefined || endLine === undefined) return ''; + const MAX_SNIPPET = 5000; + const capSnippet = (text: string): string => + text.length > MAX_SNIPPET ? text.slice(0, MAX_SNIPPET) + '\n... [truncated]' : text; + + const notebookPath = String(filePath ?? ''); + if (isNotebookPath(notebookPath)) { + const reconstructed = notebookPythonSnippet(content, startLine, endLine, notebookPath); + if (reconstructed) { + return normalizeFtsText(applyCjkSegmentationIfEnabled(capSnippet(reconstructed))); + } + } + const lines = prepared.lines; const exactSymbolContent = EXACT_SYMBOL_CONTENT_LABELS.has(node.label); const start = Math.max(0, exactSymbolContent ? startLine : startLine - 2); const end = Math.min(lines.length - 1, exactSymbolContent ? endLine : endLine + 2); const snippet = lines.slice(start, end + 1).join('\n'); - const MAX_SNIPPET = 5000; - const capped = - snippet.length > MAX_SNIPPET ? snippet.slice(0, MAX_SNIPPET) + '\n... [truncated]' : snippet; - return normalizeFtsText(applyCjkSegmentationIfEnabled(capped)); + return normalizeFtsText(applyCjkSegmentationIfEnabled(capSnippet(snippet))); }; // ============================================================================ diff --git a/gitnexus/src/storage/parse-cache.ts b/gitnexus/src/storage/parse-cache.ts index 0f7ce849c..6c75bf39d 100644 --- a/gitnexus/src/storage/parse-cache.ts +++ b/gitnexus/src/storage/parse-cache.ts @@ -785,7 +785,9 @@ import { copyV8CacheIfPresent, tryLoadV8Cache, writeV8CacheFile } from './v8-sid // Same v112: the `valueAlternatives` provider hook extends it to Kotlin // `?:`/`if`, Swift/Dart `??`/`?:`, and Python `x if c else y`, and keeps a // Ruby multi-statement `if` one opaque source. -const SCHEMA_BUMP = 112; +// v113 (#3371): `.ipynb` code cells are extracted to Python before parse. +// Warm caches keyed on raw JSON would replay empty/failed Python parses. +const SCHEMA_BUMP = 113; const GITNEXUS_PKG_VERSION = (() => { try { // package.json sits at gitnexus/package.json — two levels up from diff --git a/gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts b/gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts new file mode 100644 index 000000000..b03845dac --- /dev/null +++ b/gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts @@ -0,0 +1,93 @@ +/** + * Jupyter notebooks are indexed as Python by extracting code cells. + */ +import { describe, it, expect } from 'vitest'; +import path from 'path'; +import fs from 'node:fs'; +import os from 'node:os'; +import { + getNodesByLabel, + getNodesByLabelFull, + getRelationships, + runPipelineFromRepo, + writeFixtureRepo, +} from './helpers.js'; +import { extractNotebookPython } from '../../../src/core/ingestion/ipynb-extractor.js'; + +function pythonNotebook(cells: Array<{ source: string | string[]; language?: string }>): string { + return JSON.stringify( + { + nbformat: 4, + nbformat_minor: 5, + metadata: { + kernelspec: { display_name: 'Python 3', language: 'python', name: 'python3' }, + language_info: { name: 'python' }, + }, + cells: cells.map((c) => ({ + cell_type: 'code', + metadata: c.language ? { language: c.language } : {}, + source: c.source, + outputs: [], + })), + }, + null, + 2, + ); +} + +describe('Jupyter notebook Python pipeline', () => { + it('indexes notebook functions, imports sibling py, and skip-fails closed', async () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-ipynb-')); + try { + writeFixtureRepo(root, { + 'lib.py': 'def helper():\n return 1\n', + 'analysis.ipynb': pythonNotebook([ + { source: ['from lib import helper\n'] }, + { source: ['def train():\n', ' return helper()\n'] }, + ]), + 'broken.ipynb': '{not json', + 'julia.ipynb': JSON.stringify({ + nbformat: 4, + nbformat_minor: 5, + metadata: { + kernelspec: { language: 'julia', name: 'julia', display_name: 'Julia' }, + }, + cells: [ + { cell_type: 'code', metadata: {}, source: ['function train() end\n'], outputs: [] }, + ], + }), + }); + const result = await runPipelineFromRepo(root, () => {}, { skipGraphPhases: true }); + const functions = getNodesByLabel(result, 'Function'); + expect(functions).toContain('train'); + expect(functions).toContain('helper'); + const train = getNodesByLabelFull(result, 'Function').find((n) => n.name === 'train'); + expect(train?.properties.filePath.replace(/\\/g, '/')).toMatch(/analysis\.ipynb$/); + const extracted = extractNotebookPython( + fs.readFileSync(path.join(root, 'analysis.ipynb'), 'utf8'), + )!; + expect(train?.properties.startLine).toBe(extracted.segments[1].jsonStartLine); + const imports = getRelationships(result, 'IMPORTS'); + expect( + imports.some( + (e) => + e.sourceFilePath.replace(/\\/g, '/').endsWith('analysis.ipynb') && + (e.target === 'helper' || e.targetFilePath.replace(/\\/g, '/').endsWith('lib.py')), + ), + ).toBe(true); + expect( + getNodesByLabelFull(result, 'Function').filter( + (n) => + n.properties.filePath.replace(/\\/g, '/').endsWith('julia.ipynb') && n.name === 'train', + ), + ).toHaveLength(0); + expect( + getNodesByLabelFull(result, 'Function').filter((n) => + n.properties.filePath.replace(/\\/g, '/').endsWith('broken.ipynb'), + ), + ).toHaveLength(0); + } finally { + fs.rmSync(root, { recursive: true, force: true }); + } + }, 60000); +}); diff --git a/gitnexus/test/unit/ast-utils.test.ts b/gitnexus/test/unit/ast-utils.test.ts index e2895592c..5611325a1 100644 --- a/gitnexus/test/unit/ast-utils.test.ts +++ b/gitnexus/test/unit/ast-utils.test.ts @@ -123,4 +123,33 @@ describe('ensureAndParse', () => { expect(objcParse).toHaveBeenCalledTimes(2); expect(cppParse).toHaveBeenCalledTimes(1); }); + + it('parses extracted notebook Python and returns null when extraction fails', async () => { + parseSourceSafeSpy.mockClear(); + const pyParse = vi.fn().mockReturnValue({ lang: 'py', rootNode: { type: 'module' } }); + createParserForLanguage.mockImplementation(async (language: string) => { + if (language === 'python') return { parse: pyParse }; + throw new Error(`unexpected language ${language}`); + }); + + const { ensureAndParse } = await import('../../src/core/embeddings/ast-utils.js'); + const nb = JSON.stringify({ + nbformat: 4, + nbformat_minor: 5, + metadata: { kernelspec: { language: 'python', name: 'python3', display_name: 'Python' } }, + cells: [ + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }); + + await ensureAndParse(nb, 'analysis.ipynb'); + expect(parseSourceSafeSpy).toHaveBeenCalled(); + const parsedText = parseSourceSafeSpy.mock.calls.at(-1)?.[1] as string; + expect(parsedText).toContain('def train'); + expect(parsedText).not.toContain('cell_type'); + + parseSourceSafeSpy.mockClear(); + expect(await ensureAndParse('{not json', 'broken.ipynb')).toBeNull(); + expect(parseSourceSafeSpy).not.toHaveBeenCalled(); + }); }); diff --git a/gitnexus/test/unit/incremental-parse-cache.test.ts b/gitnexus/test/unit/incremental-parse-cache.test.ts index 55549332e..c08498613 100644 --- a/gitnexus/test/unit/incremental-parse-cache.test.ts +++ b/gitnexus/test/unit/incremental-parse-cache.test.ts @@ -290,8 +290,9 @@ describe('PARSE_CACHE_VERSION', () => { // anonymous arrows, so both stores re-extract. // Moved 104 -> 112 for #3354: callable-value flow follows `??`/`||`/`?:` // branches. 105-111 are claimed by open PR #3326. - it('pins SCHEMA_BUMP to 112 so concurrent bumps cannot silently collide (#2766, #3015, #3088, #2885, #3128, #2865, #3130, #1432, #3161, #3179, #3219, #3190, #3253, #3273, #3339, #3354)', () => { - expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(112); + // Moved 112 -> 113 for #3371: notebook code-cell extraction before Python parse. + it('pins SCHEMA_BUMP to 113 so concurrent bumps cannot silently collide (#2766, #3015, #3088, #2885, #3128, #2865, #3130, #1432, #3161, #3179, #3219, #3190, #3253, #3273, #3339, #3354, #3371)', () => { + expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(113); expect(PARSE_CACHE_BUCKET_COUNT).toBe(128); // The PREVIOUS version must fail the reuse gate, not merely differ from the // current one — a hardcoded number outside the conflict hunk rebases cleanly @@ -300,7 +301,7 @@ describe('PARSE_CACHE_VERSION', () => { for (const taken of [ 59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, 81, 82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103, - 104, 105, 106, 107, 108, 109, 110, 111, + 104, 105, 106, 107, 108, 109, 110, 111, 112, ]) { expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).not.toBe(taken); } diff --git a/gitnexus/test/unit/ingestion-utils.test.ts b/gitnexus/test/unit/ingestion-utils.test.ts index db16bfd34..2a3582773 100644 --- a/gitnexus/test/unit/ingestion-utils.test.ts +++ b/gitnexus/test/unit/ingestion-utils.test.ts @@ -3,6 +3,7 @@ import { getLanguageFromFilename, getSyntaxLanguageFromFilename, isBladeTemplateFilename, + isNotebookFilename, SupportedLanguages, } from 'gitnexus-shared'; import { getProvider, getProviderForFile } from '../../src/core/ingestion/languages/index.js'; @@ -50,6 +51,13 @@ describe('getLanguageFromFilename', () => { it('detects .py files', () => { expect(getLanguageFromFilename('main.py')).toBe(SupportedLanguages.Python); }); + + it('detects .ipynb files as Python', () => { + expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python); + expect(getProviderForFile('notebooks/analysis.ipynb')?.id).toBe(SupportedLanguages.Python); + expect(isNotebookFilename('notebooks/analysis.ipynb')).toBe(true); + expect(getSyntaxLanguageFromFilename('analysis.ipynb')).toBe('json'); + }); }); describe('Java', () => { diff --git a/gitnexus/test/unit/ipynb-extractor.test.ts b/gitnexus/test/unit/ipynb-extractor.test.ts new file mode 100644 index 000000000..9335ce0cd --- /dev/null +++ b/gitnexus/test/unit/ipynb-extractor.test.ts @@ -0,0 +1,428 @@ +import { describe, it, expect } from 'vitest'; +import { + extractNotebookPython, + mapExtractLine, + notebookPythonSnippet, + isPythonFamilyLanguage, +} from '../../src/core/ingestion/ipynb-extractor.js'; + +function notebook(opts: { + language?: string; + languageInfo?: string; + cells: Array>; +}): string { + const kernelspec = + opts.language === undefined + ? undefined + : { display_name: 'Python', language: opts.language, name: 'python' }; + const language_info = opts.languageInfo === undefined ? undefined : { name: opts.languageInfo }; + return JSON.stringify( + { + nbformat: 4, + nbformat_minor: 5, + metadata: { + ...(kernelspec ? { kernelspec } : {}), + ...(language_info ? { language_info } : {}), + }, + cells: opts.cells, + }, + null, + 2, + ); +} + +describe('isPythonFamilyLanguage', () => { + it('accepts python3 and ipython', () => { + expect(isPythonFamilyLanguage('python3')).toBe(true); + expect(isPythonFamilyLanguage('IPython')).toBe(true); + expect(isPythonFamilyLanguage('julia')).toBe(false); + }); +}); + +describe('extractNotebookPython', () => { + it('maps source that is serialized before cell_type', () => { + const content = JSON.stringify({ + nbformat: 4, + nbformat_minor: 5, + metadata: { kernelspec: { language: 'python', name: 'python3', display_name: 'Python' } }, + cells: [ + { + source: ['def train():\n', ' pass\n'], + cell_type: 'code', + metadata: {}, + outputs: [], + }, + ], + }); + const result = extractNotebookPython(content); + expect(result?.pythonSource).toContain('def train'); + expect(result?.segments).toHaveLength(1); + }); + + it('extracts def train from a Python v4 notebook', () => { + const content = notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: {}, + source: ['def train():\n', ' pass\n'], + outputs: [], + }, + ], + }); + const result = extractNotebookPython(content); + expect(result).not.toBeNull(); + expect(result!.pythonSource).toContain('def train():'); + expect(result!.segments).toHaveLength(1); + const defJsonLine = content.split('\n').findIndex((l) => l.includes('def train')); + expect(result!.segments[0].jsonStartLine).toBe(defJsonLine); + }); + + it('keeps two code cells in order with two segments', () => { + const content = notebook({ + language: 'python', + cells: [ + { cell_type: 'code', metadata: {}, source: ['x = 1\n'], outputs: [] }, + { + cell_type: 'code', + metadata: {}, + source: ['def train():\n', ' return x\n'], + outputs: [], + }, + ], + }); + const result = extractNotebookPython(content); + expect(result).not.toBeNull(); + expect(result!.pythonSource).toMatch(/x = 1\n+def train/); + expect(result!.segments).toHaveLength(2); + const defRow = result!.pythonSource.split('\n').findIndex((l) => l.startsWith('def train')); + expect(defRow).toBe(result!.segments[1].extractStartLine); + expect(mapExtractLine(defRow, result!.segments)).toBe(result!.segments[1].jsonStartLine); + }); + + it('accepts source as a single string', () => { + const content = notebook({ + language: 'python', + cells: [{ cell_type: 'code', metadata: {}, source: 'y = 2\n', outputs: [] }], + }); + expect(extractNotebookPython(content)?.pythonSource).toContain('y = 2'); + }); + + it('returns null for markdown-only notebooks', () => { + const content = notebook({ + language: 'python', + cells: [{ cell_type: 'markdown', metadata: {}, source: ['# hi\n'] }], + }); + expect(extractNotebookPython(content)).toBeNull(); + }); + + it('returns null for invalid JSON', () => { + expect(extractNotebookPython('{not json')).toBeNull(); + }); + + it('returns null for a Julia kernelspec', () => { + const content = notebook({ + language: 'julia', + cells: [{ cell_type: 'code', metadata: {}, source: ['1 + 1\n'], outputs: [] }], + }); + expect(extractNotebookPython(content)).toBeNull(); + }); + + it('extracts python3 kernelspec', () => { + const content = notebook({ + language: 'python3', + cells: [{ cell_type: 'code', metadata: {}, source: ['a = 1\n'], outputs: [] }], + }); + expect(extractNotebookPython(content)?.pythonSource).toContain('a = 1'); + }); + + it('comments line magics in place', () => { + const content = notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: {}, + source: ['%time\n', 'x = 1\n', '!ls\n'], + outputs: [], + }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource.split('\n').filter((l) => l.length > 0)).toEqual([ + '# %time', + 'x = 1', + '# !ls', + ]); + }); + + it('skips a %%bash cell and still extracts later train', () => { + const content = notebook({ + language: 'python', + cells: [ + { cell_type: 'code', metadata: {}, source: ['%%bash\n', 'echo hi\n'], outputs: [] }, + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('def train'); + expect(result!.pythonSource).not.toContain('echo hi'); + }); + + it('maps identical duplicate cells to later JSON lines', () => { + const src = ['print(1)\n']; + const content = notebook({ + language: 'python', + cells: [ + { cell_type: 'code', metadata: {}, source: src, outputs: [] }, + { cell_type: 'code', metadata: {}, source: src, outputs: [] }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.segments).toHaveLength(2); + expect(result!.segments[1].jsonStartLine).toBeGreaterThan(result!.segments[0].jsonStartLine); + }); + + it('skips an R-language code cell and keeps Python cells', () => { + const content = notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: { language: 'R' }, + source: ['x <- 1\n'], + outputs: [], + }, + { cell_type: 'code', metadata: {}, source: ['z = 3\n'], outputs: [] }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('z = 3'); + expect(result!.pythonSource).not.toContain('x <- 1'); + }); +}); + +describe('mapExtractLine', () => { + it('maps a second-cell extract row onto that cell JSON line', () => { + const content = notebook({ + language: 'python', + cells: [ + { cell_type: 'markdown', metadata: {}, source: ['# intro\n'] }, + { cell_type: 'code', metadata: {}, source: ['a = 1\n'], outputs: [] }, + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }); + const extracted = extractNotebookPython(content)!; + const trainSeg = extracted.segments[1]; + const mapped = mapExtractLine(trainSeg.extractStartLine, extracted.segments); + expect(mapped).toBe(trainSeg.jsonStartLine); + expect(mapped).toBeGreaterThan(extracted.segments[0].jsonStartLine); + }); +}); + +describe('notebookPythonSnippet', () => { + it('returns Python def train not JSON cell_type', () => { + const content = notebook({ + language: 'python', + cells: [ + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }); + const extracted = extractNotebookPython(content)!; + const snippet = notebookPythonSnippet( + content, + extracted.segments[0].jsonStartLine, + extracted.segments[0].jsonEndLine, + ); + expect(snippet).toContain('def train'); + expect(snippet).not.toContain('cell_type'); + }); +}); + +describe('extractNotebookPython edge cases', () => { + it('skips a code cell without source and keeps later Python', () => { + const content = notebook({ + language: 'python', + cells: [ + { cell_type: 'code', metadata: {}, source: ['def ok():\n', ' pass\n'], outputs: [] }, + { cell_type: 'code', metadata: {}, outputs: [] }, + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('def ok'); + expect(result!.pythonSource).toContain('def train'); + }); + + it('returns null when language_info is julia without kernelspec', () => { + const content = notebook({ + languageInfo: 'julia', + cells: [{ cell_type: 'code', metadata: {}, source: ['1 + 1\n'], outputs: [] }], + }); + expect(extractNotebookPython(content)).toBeNull(); + }); + + it('maps coordinates to the last cells array when the key is duplicated', () => { + const decoy = JSON.stringify( + [{ cell_type: 'code', metadata: {}, source: ['def decoy():\n', ' pass\n'], outputs: [] }], + null, + 2, + ); + const real = JSON.stringify( + [{ cell_type: 'code', metadata: {}, source: ['def real():\n', ' pass\n'], outputs: [] }], + null, + 2, + ); + const content = `{ + "nbformat": 4, + "nbformat_minor": 5, + "metadata": { "kernelspec": { "language": "python", "name": "python3", "display_name": "Python" } }, + "cells": ${decoy}, + "cells": ${real} +}`; + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('def real'); + expect(result!.pythonSource).not.toContain('decoy'); + const realLine = content.split('\n').findIndex((l) => l.includes('def real')); + expect(result!.segments[0].jsonStartLine).toBe(realLine); + }); + + it('keeps Python under %%time and skips %%bash', () => { + const content = notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: {}, + source: ['%%time\n', 'def train():\n', ' pass\n'], + outputs: [], + }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('def train'); + expect(result!.pythonSource).toContain('# %%time'); + }); + + it('comments IPython help lines', () => { + const content = notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: {}, + source: ['train?\n', 'def train():\n', ' pass\n'], + outputs: [], + }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource.split('\n').filter((l) => l.length > 0)[0]).toBe('# train?'); + expect(result!.pythonSource).toContain('def train'); + }); + + it('parses a UTF-8 BOM notebook', () => { + const content = + '\uFEFF' + + notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: {}, + source: ['def train():\n', ' pass\n'], + outputs: [], + }, + ], + }); + expect(extractNotebookPython(content)?.pythonSource).toContain('def train'); + }); + + it('treats python and python3 metadata as the same family', () => { + const content = notebook({ + language: 'python', + languageInfo: 'python3', + cells: [{ cell_type: 'code', metadata: {}, source: ['a = 1\n'], outputs: [] }], + }); + expect(extractNotebookPython(content)?.pythonSource).toContain('a = 1'); + }); + + it('skips a notebook whose kernelspec name is R and has no language field', () => { + const content = JSON.stringify({ + nbformat: 4, + nbformat_minor: 5, + metadata: { kernelspec: { name: 'ir', display_name: 'R' } }, + cells: [{ cell_type: 'code', metadata: {}, source: ['x <- 1\n'], outputs: [] }], + }); + expect(extractNotebookPython(content)).toBeNull(); + }); + + it('extracts nbformat v3 worksheets via input', () => { + const content = JSON.stringify( + { + nbformat: 3, + nbformat_minor: 0, + metadata: { name: 'legacy' }, + worksheets: [ + { + cells: [ + { + cell_type: 'code', + language: 'python', + input: ['def train():\n', ' pass\n'], + outputs: [], + }, + ], + }, + ], + }, + null, + 2, + ); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('def train'); + const line = content.split('\n').findIndex((l) => l.includes('def train')); + expect(result!.segments[0].jsonStartLine).toBe(line); + }); + + it('keeps later cells when an earlier cell has an unclosed string', () => { + const content = notebook({ + language: 'python', + cells: [ + { cell_type: 'code', metadata: {}, source: ['text = """unterminated\n'], outputs: [] }, + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('def train'); + expect(result!.pythonSource).not.toMatch(/^text = """/m); + }); + + it('turns %run of a local module into an import', () => { + const content = notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: {}, + source: ['%run ./lib.py\n', 'def train():\n', ' pass\n'], + outputs: [], + }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('import lib # %run ./lib.py'); + expect(result!.pythonSource).toContain('def train'); + }); + + it('indexes a Sage kernel as Python', () => { + const content = notebook({ + language: 'sage', + cells: [ + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }); + expect(extractNotebookPython(content)?.pythonSource).toContain('def train'); + }); +}); diff --git a/gitnexus/test/unit/ipynb-python-captures.test.ts b/gitnexus/test/unit/ipynb-python-captures.test.ts new file mode 100644 index 000000000..e6074a73c --- /dev/null +++ b/gitnexus/test/unit/ipynb-python-captures.test.ts @@ -0,0 +1,52 @@ +import { describe, it, expect } from 'vitest'; +import { emitPythonScopeCaptures } from '../../src/core/ingestion/languages/python/captures.js'; +import { pythonProvider } from '../../src/core/ingestion/languages/python.js'; +import { extractParsedFile } from '../../src/core/ingestion/scope-extractor-bridge.js'; +import { extractNotebookPython, mapExtractLine } from '../../src/core/ingestion/ipynb-extractor.js'; +import { getLanguageFromFilename, SupportedLanguages } from 'gitnexus-shared'; +import { getProviderForFile } from '../../src/core/ingestion/languages/index.js'; + +const notebook = JSON.stringify( + { + nbformat: 4, + nbformat_minor: 5, + metadata: { + kernelspec: { language: 'python', name: 'python3', display_name: 'Python' }, + }, + cells: [ + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }, + null, + 2, +); + +describe('Python notebook scope captures', () => { + it('detects .ipynb as Python', () => { + expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python); + expect(getProviderForFile('n.ipynb')?.id).toBe(SupportedLanguages.Python); + }); + + it('does not parse raw JSON as Python on the full-file path', () => { + const matches = emitPythonScopeCaptures(notebook, 'analysis.ipynb'); + const names = matches.flatMap((m) => Object.keys(m)); + expect(names.some((k) => k.includes('function'))).toBe(true); + }); + + it('returns no captures for invalid notebook JSON', () => { + expect(emitPythonScopeCaptures('{not json', 'broken.ipynb')).toEqual([]); + }); + + it('extractParsedFile yields a Function for train', () => { + const captured = emitPythonScopeCaptures(notebook, 'analysis.ipynb'); + const fnCapture = captured.find((m) => m['@scope.function'] !== undefined); + const extracted = extractNotebookPython(notebook)!; + const expectedJson = mapExtractLine(extracted.segments[0].extractStartLine, extracted.segments); + expect(fnCapture?.['@scope.function']?.range.startLine).toBe(expectedJson + 1); + const parsed = extractParsedFile(pythonProvider, notebook, 'analysis.ipynb'); + expect(parsed).toBeDefined(); + expect(parsed!.localDefs.some((d) => d.qualifiedName === 'train' || d.name === 'train')).toBe( + true, + ); + }); +});