From 5bf6dd8d9967657ade30fd2cdf667cc0d935c53b Mon Sep 17 00:00:00 2001 From: Gergo Magyar Date: Fri, 25 Sep 2026 18:20:12 +0000 Subject: [PATCH] feat: cover real Jupyter notebook shapes in Python extraction Index Sage and Pyodide kernels, keep later cells when one cell is broken, and link %run of a local module without executing a kernel. --- docs/languages/jupyter-notebook.md | 16 +- gitnexus-shared/src/index.ts | 1 + gitnexus-shared/src/language-detection.ts | 7 +- gitnexus/src/core/embeddings/ast-utils.ts | 9 +- .../src/core/ingestion/ipynb-extractor.ts | 383 +++++++++++++----- .../src/core/ingestion/language-provider.ts | 8 +- .../ingestion/languages/python/captures.ts | 52 ++- .../core/ingestion/scope-extractor-bridge.ts | 8 +- gitnexus/src/core/lbug/csv-generator.ts | 27 +- .../test/unit/incremental-parse-cache.test.ts | 3 - gitnexus/test/unit/ingestion-utils.test.ts | 2 + gitnexus/test/unit/ipynb-extractor.test.ts | 138 +++++++ 12 files changed, 502 insertions(+), 152 deletions(-) diff --git a/docs/languages/jupyter-notebook.md b/docs/languages/jupyter-notebook.md index c6b79090b..f69c77aaf 100644 --- a/docs/languages/jupyter-notebook.md +++ b/docs/languages/jupyter-notebook.md @@ -18,10 +18,16 @@ After `analyze`, functions, classes, and imports defined in Python code cells ar ## Kernel and magics -- Skip the file when `kernelspec.language` or `language_info.name` is present and is not a Python-family name (`python`, `python2`, `python3`, `ipython`). Disagreeing fields skip the file. -- If those fields are absent, code cells are treated as Python unless a cell's own language metadata says otherwise. -- A cell whose first non-empty line is a cell magic (`%%`) is skipped. -- Line magics (`%`) and shell (`!`) lines are commented in place so JSON line mapping stays affine. +- Skip the file when `kernelspec.language` or `language_info.name` is present and is not a Python-family name (`python`, `python2`, `python3`, `ipython`, `python 3`, `ipython3`). `python` and `python3` together still index. +- If `kernelspec.language` is missing, `kernelspec.name` is used only when it is an obvious language id (`python3`, `ir`, `julia-1.8`). Conda env names are ignored. +- If `language_info.name` is missing, `language_info.file_extension` (`.py` vs `.r` / `.jl`) is used the same way. +- If those fields are absent, code cells are treated as Python unless a cell's own `language`, `metadata.language`, or `metadata.vscode.languageId` says otherwise. +- A cell whose first non-empty line is a foreign cell magic (`%%bash`, `%%html`, `%%sql`, `%%R`) is skipped. Python-body cell magics (`%%time`, `%%timeit`, `%%capture`, `%%prun`, `%%debug`, `%%px`, `%%python`) stay, with the magic line commented. +- `%run`, `%load`, and `%loadpy` of a local `.py` path become `import module` on that same line so the notebook links to the file. URLs, `..` paths, and `.ipynb` targets stay comments. The notebook is not executed. +- Sage, SageMath, MicroPython, Pyodide, PyPy, and PySpark kernels are indexed as Python. SQL, R, and Julia cells are not. +- A cell with an unclosed string or bracket is commented out so it cannot hide later cells. Those cells are concatenated in order; one syntax error no longer drops the rest of the notebook. +- Line magics (`%`), shell (`!`), and IPython help (`train?`, `?train`) are commented in place so JSON line mapping stays affine. +- nbformat v4 `cells` / `source` is the normal path. nbformat v3 `worksheets[].cells` and code-cell `input` are accepted. A leading UTF-8 BOM is accepted. Markdown and raw cells are not code. ## Line numbers @@ -33,6 +39,6 @@ Concatenating cells in document order is notebook semantics. An earlier cell wit - `gitnexus/test/unit/ipynb-extractor.test.ts` - `gitnexus/test/unit/ingestion-utils.test.ts` (`.ipynb` detection) -- `gitnexus/test/integration/ipynb-python-pipeline.test.ts` +- `gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts` No Jupyter, nbconvert, or nbformat runtime dependency. diff --git a/gitnexus-shared/src/index.ts b/gitnexus-shared/src/index.ts index de49223bf..b5fbe37a5 100644 --- a/gitnexus-shared/src/index.ts +++ b/gitnexus-shared/src/index.ts @@ -22,6 +22,7 @@ export { getLanguageFromFilename, getSyntaxLanguageFromFilename, isBladeTemplateFilename, + isNotebookFilename, } from './language-detection.js'; export type { MroStrategy } from './mro-strategy.js'; diff --git a/gitnexus-shared/src/language-detection.ts b/gitnexus-shared/src/language-detection.ts index b3f34a168..583a0fd5f 100644 --- a/gitnexus-shared/src/language-detection.ts +++ b/gitnexus-shared/src/language-detection.ts @@ -76,6 +76,10 @@ for (const [lang, exts] of Object.entries(EXTENSION_MAP) as [ export const isBladeTemplateFilename = (filePath: string): boolean => filePath.replace(/\\/g, '/').toLowerCase().endsWith('.blade.php'); +/** Jupyter notebooks: ingested as Python; on-disk bytes are JSON. */ +export const isNotebookFilename = (filePath: string): boolean => + filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb'); + /** * Map file extension to SupportedLanguage enum. * Returns null if the file extension is not recognized. @@ -163,8 +167,7 @@ const AUXILIARY_BASENAME_MAP: Record = { */ export const getSyntaxLanguageFromFilename = (filePath: string): string => { if (isBladeTemplateFilename(filePath)) return 'markup'; - // Notebooks are ingested as Python; the on-disk bytes are JSON. - if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) return 'json'; + if (isNotebookFilename(filePath)) return 'json'; const lang = getLanguageFromFilename(filePath); if (lang) return SYNTAX_MAP[lang]; diff --git a/gitnexus/src/core/embeddings/ast-utils.ts b/gitnexus/src/core/embeddings/ast-utils.ts index a0a15d7ae..0fa85d18f 100644 --- a/gitnexus/src/core/embeddings/ast-utils.ts +++ b/gitnexus/src/core/embeddings/ast-utils.ts @@ -41,15 +41,16 @@ export const ensureAndParse = async (content: string, filePath: string): Promise // parser always come from the same provider. Length-preserving, so node // offsets still index `content` except for `.ipynb`, which is replaced by // concatenated code-cell Python (same as the parse worker). - let parseContent = getProvider(language).preprocessSource?.(content, filePath) ?? content; + const provider = getProvider(language); if (isNotebookPath(filePath)) { const extracted = extractNotebookPython(content); if (!extracted) return null; - parseContent = - getProvider(language).preprocessSource?.(extracted.pythonSource, filePath) ?? - extracted.pythonSource; + const parseContent = + provider.preprocessSource?.(extracted.pythonSource, filePath) ?? extracted.pythonSource; + return parseSourceSafe(parserInstance, parseContent); } + const parseContent = provider.preprocessSource?.(content, filePath) ?? content; return parseSourceSafe(parserInstance, parseContent); }; diff --git a/gitnexus/src/core/ingestion/ipynb-extractor.ts b/gitnexus/src/core/ingestion/ipynb-extractor.ts index 075b664e6..d6a8f9478 100644 --- a/gitnexus/src/core/ingestion/ipynb-extractor.ts +++ b/gitnexus/src/core/ingestion/ipynb-extractor.ts @@ -9,6 +9,9 @@ * buffer lines map through {@link mapExtractLine}. */ +import { isNotebookFilename } from 'gitnexus-shared'; +import { buildLineIndex, lineFromOffset } from '../embeddings/line-index.js'; + export interface NotebookLineSegment { readonly extractStartLine: number; readonly extractEndLine: number; @@ -21,42 +24,60 @@ export interface NotebookPythonExtraction { readonly segments: readonly NotebookLineSegment[]; } -const PYTHON_FAMILY = new Set(['python', 'python2', 'python3', 'ipython']); +const PYTHON_FAMILY = new Set([ + 'python', + 'python2', + 'python3', + 'ipython', + 'sage', + 'sagemath', + 'micropython', + 'pyodide', + 'pypy', + 'pyspark', +]); export function isNotebookPath(filePath: string): boolean { - return filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb'); + return isNotebookFilename(filePath); } export function isPythonFamilyLanguage(name: string | undefined | null): boolean { if (name === undefined || name === null) return false; const n = name.trim().toLowerCase(); if (PYTHON_FAMILY.has(n)) return true; - return /^python\d/.test(n); + if (n.startsWith('ipython')) return true; + return /^python(?:\d|\b)/.test(n); } -function buildLineStarts(content: string): number[] { - const starts = [0]; - for (let i = 0; i < content.length; i++) { - if (content.charCodeAt(i) === 10) starts.push(i + 1); - } - return starts; +/** Kernel spec names that are a language id, not a conda env label. */ +const NON_PYTHON_KERNEL = + /^(?:julia|ir|r|scala|rust|ruby|javascript|nodejs|node|bash|sh|sql|csharp|fsharp|go|java|kotlin|swift|php|perl|lua|octave|matlab|sas|stata|haskell|clojure|elixir|groovy|powershell|sos|cpp|cxx|dart|typescript|wolfram|scheme|racket|ocaml|fortran|gnuplot|sqlite|mysql|postgresql|tsql)(?:[-_][\w.-]+|\d[\w.-]*)?$/i; + +function kernelNameLanguage(name: string): string | undefined { + if (isPythonFamilyLanguage(name) || NON_PYTHON_KERNEL.test(name.trim())) return name; + return undefined; } -function indexToLine(lineStarts: readonly number[], index: number): number { - if (index <= 0) return 0; - let lo = 0; - let hi = lineStarts.length - 1; - let ans = 0; - while (lo <= hi) { - const mid = (lo + hi) >> 1; - if (lineStarts[mid] <= index) { - ans = mid; - lo = mid + 1; - } else { - hi = mid - 1; - } +function extensionLanguage(ext: string): string | undefined { + const e = ext.trim().toLowerCase(); + if (e === '.py' || e === '.pyi' || e === '.ipy') return 'python'; + if ( + e === '.r' || + e === '.jl' || + e === '.scala' || + e === '.js' || + e === '.java' || + e === '.go' || + e === '.rs' || + e === '.rb' || + e === '.php' || + e === '.kt' || + e === '.swift' || + e === '.cs' + ) { + return ext; } - return ans; + return undefined; } function skipWs(content: string, i: number): number { @@ -138,34 +159,13 @@ function jsonBraceSpan( return { start: i, end: j }; } -function findDepth1Key(content: string, objStart: number, objEnd: number, key: string): number { - let depth = 0; - let inStr = false; - let j = objStart; - while (j < objEnd) { - const ch = content[j]; - if (inStr) { - if (ch === '\\') j += 2; - else { - if (ch === '"') inStr = false; - j++; - } - continue; - } - if (ch === '"') { - if (depth === 1 && content.startsWith(key, j)) return j; - inStr = true; - j++; - continue; - } - if (ch === '{' || ch === '[') depth++; - else if (ch === '}' || ch === ']') depth--; - j++; - } - return -1; -} - -function findLastDepth1Key(content: string, objStart: number, objEnd: number, key: string): number { +function findDepth1Key( + content: string, + objStart: number, + objEnd: number, + key: string, + match: 'first' | 'last' = 'first', +): number { let depth = 0; let inStr = false; let j = objStart; @@ -181,7 +181,10 @@ function findLastDepth1Key(content: string, objStart: number, objEnd: number, ke continue; } if (ch === '"') { - if (depth === 1 && content.startsWith(key, j)) found = j; + if (depth === 1 && content.startsWith(key, j)) { + if (match === 'first') return j; + found = j; + } inStr = true; j++; continue; @@ -193,16 +196,63 @@ function findLastDepth1Key(content: string, objStart: number, objEnd: number, ke return found; } -function findCellsArraySpan(content: string): { start: number; end: number } | null { - const root = jsonBraceSpan(content, 0, '{', '}'); - if (!root) return null; - const cellsKey = findLastDepth1Key(content, root.start, root.end, '"cells"'); - if (cellsKey < 0) return null; - const colon = content.indexOf(':', cellsKey + 7); - if (colon < 0 || colon >= root.end) return null; +function arraySpanAfterKey( + content: string, + objStart: number, + objEnd: number, + key: '"cells"' | '"worksheets"', + match: 'first' | 'last', +): { start: number; end: number } | null { + const keyAt = findDepth1Key(content, objStart, objEnd, key, match); + if (keyAt < 0) return null; + const colon = content.indexOf(':', keyAt + key.length); + if (colon < 0 || colon >= objEnd) return null; return jsonBraceSpan(content, colon + 1, '[', ']'); } +function codeCellArraySpans(content: string, origin: number): { start: number; end: number }[] { + const root = jsonBraceSpan(content, origin, '{', '}'); + if (!root) return []; + const cells = arraySpanAfterKey(content, root.start, root.end, '"cells"', 'last'); + if (cells) return [cells]; + const worksheets = arraySpanAfterKey(content, root.start, root.end, '"worksheets"', 'last'); + if (!worksheets) return []; + const spans: { start: number; end: number }[] = []; + let search = worksheets.start + 1; + while (search < worksheets.end) { + const brace = nextUnquotedChar(content, search, worksheets.end, '{'); + if (brace < 0) break; + const obj = jsonBraceSpan(content, brace, '{', '}'); + if (!obj || obj.end > worksheets.end) { + search = brace + 1; + continue; + } + const nested = arraySpanAfterKey(content, obj.start, obj.end, '"cells"', 'first'); + if (nested) spans.push(nested); + search = obj.end; + } + return spans; +} + +function readJsonString(content: string, start: number): string | null { + const i = skipWs(content, start); + if (content[i] !== '"') return null; + let j = i + 1; + let out = ''; + while (j < content.length) { + const ch = content[j]; + if (ch === '\\') { + out += content[j + 1] ?? ''; + j += 2; + continue; + } + if (ch === '"') return out; + out += ch; + j++; + } + return null; +} + function nextUnquotedChar(content: string, from: number, until: number, needle: '{' | '"'): number { let inStr = false; let j = from; @@ -263,15 +313,21 @@ function findNextCodeCellSourceSpan( continue; } const valueStart = skipWs(content, colon + 1); - if (content.slice(valueStart, valueStart + 6) !== '"code"') { + const cellType = readJsonString(content, valueStart); + if (cellType === null || cellType.toLowerCase() !== 'code') { search = obj.end; continue; } - const sourceKey = findDepth1Key(content, obj.start, obj.end, '"source"'); + let sourceKey = findDepth1Key(content, obj.start, obj.end, '"source"'); + let keyLen = 8; + if (sourceKey < 0) { + sourceKey = findDepth1Key(content, obj.start, obj.end, '"input"'); + keyLen = 7; + } if (sourceKey < 0) { return { span: null, nextFrom: obj.end }; } - const srcColon = content.indexOf(':', sourceKey + 8); + const srcColon = content.indexOf(':', sourceKey + keyLen); if (srcColon < 0 || srcColon >= obj.end) { return { span: null, nextFrom: obj.end }; } @@ -293,6 +349,7 @@ function flattenSource(source: unknown): string { } function cellLanguage(cell: Record): string | undefined { + if (typeof cell.language === 'string') return cell.language; const meta = cell.metadata; if (!meta || typeof meta !== 'object') return undefined; const m = meta as Record; @@ -316,12 +373,16 @@ function notebookLanguageFields(nb: Record): { const li = md.language_info; return { kernelspec: - ks && typeof ks === 'object' && typeof (ks as Record).language === 'string' - ? String((ks as Record).language) + ks && typeof ks === 'object' + ? typeof (ks as Record).language === 'string' + ? String((ks as Record).language) + : kernelNameLanguage(String((ks as Record).name ?? '')) : undefined, languageInfo: - li && typeof li === 'object' && typeof (li as Record).name === 'string' - ? String((li as Record).name) + li && typeof li === 'object' + ? typeof (li as Record).name === 'string' + ? String((li as Record).name) + : extensionLanguage(String((li as Record).file_extension ?? '')) : undefined, }; } @@ -330,57 +391,191 @@ function kernelShouldSkip(nb: Record): boolean { const { kernelspec, languageInfo } = notebookLanguageFields(nb); const fields = [kernelspec, languageInfo].filter((x): x is string => x !== undefined); if (fields.length === 0) return false; - if (fields.some((f) => !isPythonFamilyLanguage(f))) return true; - if (fields.length === 2 && fields[0].toLowerCase() !== fields[1].toLowerCase()) { - const a = isPythonFamilyLanguage(fields[0]); - const b = isPythonFamilyLanguage(fields[1]); - if (a !== b) return true; + return fields.some((f) => !isPythonFamilyLanguage(f)); +} + +/** Cell magics whose body is still Python. Foreign %% cells are skipped. */ +const PYTHON_CELL_MAGICS = new Set([ + 'time', + 'timeit', + 'capture', + 'prun', + 'lprun', + 'mprun', + 'debug', + 'px', + 'python', + 'python2', + 'python3', + 'ipython', + 'pypy', +]); + +function cellMagicName(line: string): string { + const token = line.trimStart().slice(2).trim().split(/\s+/)[0] ?? ''; + return token.toLowerCase(); +} + +function isIpythonHelpLine(line: string): boolean { + const t = line.trim(); + if (t.length === 0 || t.startsWith('#')) return false; + return /^\?\??\S/.test(t) || /^[A-Za-z_][\w.]*(?:\?\?|\?)\s*$/.test(t); +} + +function moduleSpecFromRunPath(raw: string): string | null { + let path = raw + .trim() + .replace(/^['"]|['"]$/g, '') + .replace(/\\/g, '/'); + if (path.startsWith('http://') || path.startsWith('https://') || path.endsWith('.ipynb')) { + return null; } - return false; + path = path.replace(/^\.\//, ''); + if (path.endsWith('.py')) path = path.slice(0, -3); + const parts = path.split('/').filter((part) => part.length > 0); + if (parts.length === 0 || parts.some((part) => part === '..' || !/^[A-Za-z_]\w*$/.test(part))) { + return null; + } + return parts.join('.'); +} + +function rewriteRunOrLoad(line: string): string | null { + const trimmed = line.trimStart(); + const indent = line.slice(0, line.length - trimmed.length); + const match = trimmed.match(/^%(?:run|load|loadpy)\s+(\S+)/); + if (!match) return null; + const spec = moduleSpecFromRunPath(match[1]); + if (!spec) return null; + return `${indent}import ${spec} # ${trimmed}`; +} + +function isBalancedPython(source: string): boolean { + let quote: "'" | '"' | null = null; + let triple = false; + let paren = 0; + let bracket = 0; + let brace = 0; + for (let i = 0; i < source.length; i++) { + const ch = source[i]; + if (quote) { + if (ch === '\\' && !triple) { + i++; + continue; + } + if (triple && source.startsWith(quote.repeat(3), i)) { + quote = null; + triple = false; + i += 2; + continue; + } + if (!triple && ch === quote) quote = null; + continue; + } + if (ch === '#') { + const nl = source.indexOf('\n', i); + if (nl < 0) break; + i = nl; + continue; + } + if ((ch === '"' || ch === "'") && source.startsWith(ch.repeat(3), i)) { + quote = ch; + triple = true; + i += 2; + continue; + } + if (ch === '"' || ch === "'") { + quote = ch; + continue; + } + if (ch === '(') paren++; + else if (ch === ')') paren--; + else if (ch === '[') bracket++; + else if (ch === ']') bracket--; + else if (ch === '{') brace++; + else if (ch === '}') brace--; + if (paren < 0 || bracket < 0 || brace < 0) return false; + } + return quote === null && paren === 0 && bracket === 0 && brace === 0; } function processCellLines(raw: string): { skipCell: boolean; lines: string[] } { - const lines = raw.split('\n'); + const lines = raw.replace(/\r\n/g, '\n').replace(/\r/g, '\n').split('\n'); const firstNonEmpty = lines.find((l) => l.trim().length > 0); - if (firstNonEmpty !== undefined && firstNonEmpty.trimStart().startsWith('%%')) { + const magic = + firstNonEmpty !== undefined && firstNonEmpty.trimStart().startsWith('%%') + ? cellMagicName(firstNonEmpty) + : undefined; + if (magic !== undefined && !PYTHON_CELL_MAGICS.has(magic) && !isPythonFamilyLanguage(magic)) { return { skipCell: true, lines: [''] }; } - const out = lines.map((line) => { + const rewritten = lines.map((line) => { const t = line.trimStart(); - if (t.startsWith('%') || t.startsWith('!')) { - return `# ${line}`; - } + if (isIpythonHelpLine(line) || t.startsWith('!')) return `# ${line}`; + const runImport = rewriteRunOrLoad(line); + if (runImport) return runImport; + if (t.startsWith('%')) return `# ${line}`; return line; }); - return { skipCell: false, lines: out }; + if (isBalancedPython(rewritten.join('\n'))) { + return { skipCell: false, lines: rewritten }; + } + return { + skipCell: false, + lines: rewritten.map((line) => (line.trim().length === 0 ? line : `# ${line}`)), + }; +} + +function notebookCodeCells(nb: Record): unknown[] | null { + if (Array.isArray(nb.cells)) return nb.cells; + if (!Array.isArray(nb.worksheets)) return null; + const cells: unknown[] = []; + for (const worksheet of nb.worksheets) { + if (!worksheet || typeof worksheet !== 'object') continue; + const nested = (worksheet as Record).cells; + if (Array.isArray(nested)) cells.push(...nested); + } + return cells.length > 0 ? cells : null; +} + +function cellSource(cell: Record): unknown { + return cell.source !== undefined ? cell.source : cell.input; } export function extractNotebookPython(content: string): NotebookPythonExtraction | null { + const origin = content.charCodeAt(0) === 0xfeff ? 1 : 0; let parsed: unknown; try { - parsed = JSON.parse(content); + parsed = JSON.parse(origin === 0 ? content : content.slice(origin)); } catch { return null; } if (!parsed || typeof parsed !== 'object') return null; const nb = parsed as Record; - if (!Array.isArray(nb.cells)) return null; + const codeCells = notebookCodeCells(nb); + if (!codeCells) return null; if (kernelShouldSkip(nb)) return null; - const cells = findCellsArraySpan(content); - if (!cells) return null; + const spans = codeCellArraySpans(content, origin); + if (spans.length === 0) return null; - const lineStarts = buildLineStarts(content); + const lineStarts = buildLineIndex(content); const chunks: string[] = []; const segments: NotebookLineSegment[] = []; + let spanIndex = 0; let searchFrom = 0; - for (const rawCell of nb.cells) { + for (const rawCell of codeCells) { if (!rawCell || typeof rawCell !== 'object') continue; const cell = rawCell as Record; - if (cell.cell_type !== 'code') continue; + if (String(cell.cell_type).toLowerCase() !== 'code') continue; - const located = findNextCodeCellSourceSpan(content, searchFrom, cells); + let located: { span: { start: number; end: number } | null; nextFrom: number } | null = null; + while (spanIndex < spans.length) { + located = findNextCodeCellSourceSpan(content, searchFrom, spans[spanIndex]); + if (located) break; + spanIndex++; + searchFrom = 0; + } if (!located) return null; searchFrom = located.nextFrom; if (!located.span) continue; @@ -390,16 +585,17 @@ export function extractNotebookPython(content: string): NotebookPythonExtraction continue; } - const { skipCell, lines } = processCellLines(flattenSource(cell.source)); - const jsonStartLine = indexToLine(lineStarts, firstQuoteInValue(content, located.span)); - const jsonEndLine = Math.max(jsonStartLine, indexToLine(lineStarts, located.span.end - 1)); + const { skipCell, lines } = processCellLines(flattenSource(cellSource(cell))); + const jsonStartLine = lineFromOffset(lineStarts, firstQuoteInValue(content, located.span)); + const jsonEndLine = Math.max(jsonStartLine, lineFromOffset(lineStarts, located.span.end - 1)); if (skipCell) { continue; } let text = lines.join('\n'); - if (text.endsWith('\n')) text = text.slice(0, -1); + const hadTrailingNewline = text.endsWith('\n'); + if (hadTrailingNewline) text = text.slice(0, -1); if (text.trim().length === 0) { continue; } @@ -407,7 +603,7 @@ export function extractNotebookPython(content: string): NotebookPythonExtraction if (chunks.length > 0) { chunks.push('\n\n'); } - const lineCount = text.split('\n').length; + const lineCount = hadTrailingNewline ? Math.max(1, lines.length - 1) : lines.length; const extractStartLine = segments.length === 0 ? 0 : segments[segments.length - 1].extractEndLine + 2; chunks.push(text); @@ -472,7 +668,8 @@ export function notebookPythonSnippetFromExtract( for (const seg of extracted.segments) { if (seg.jsonEndLine < startLine || seg.jsonStartLine > endLine) continue; for (let extract = seg.extractStartLine; extract <= seg.extractEndLine; extract++) { - const jsonLine = mapExtractLine(extract, extracted.segments); + const jsonSpan = seg.jsonEndLine - seg.jsonStartLine; + const jsonLine = seg.jsonStartLine + Math.min(extract - seg.extractStartLine, jsonSpan); if (jsonLine >= startLine && jsonLine <= endLine) { out.push(pyLines[extract] ?? ''); } diff --git a/gitnexus/src/core/ingestion/language-provider.ts b/gitnexus/src/core/ingestion/language-provider.ts index 3ed0a1a09..64c4736e7 100644 --- a/gitnexus/src/core/ingestion/language-provider.ts +++ b/gitnexus/src/core/ingestion/language-provider.ts @@ -26,6 +26,7 @@ import type { WorkspaceIndex, } from 'gitnexus-shared'; import type { LanguageTypeConfig } from './type-extractors/types.js'; +import type { NotebookLineSegment } from './ipynb-extractor.js'; import type { CallRouter } from './call-routing.js'; import type { CallExtractor } from './call-types.js'; import type { ClassExtractor } from './class-types.js'; @@ -829,12 +830,7 @@ interface LanguageProviderConfig { sourceMeta?: { readonly sourceKind?: 'full-file' | 'pre-extracted-script'; /** Python `.ipynb` only: JSON line segments for the pre-extracted buffer. */ - readonly notebookSegments?: readonly { - readonly extractStartLine: number; - readonly extractEndLine: number; - readonly jsonStartLine: number; - readonly jsonEndLine: number; - }[]; + readonly notebookSegments?: readonly NotebookLineSegment[]; }, ) => readonly CaptureMatch[]; diff --git a/gitnexus/src/core/ingestion/languages/python/captures.ts b/gitnexus/src/core/ingestion/languages/python/captures.ts index 131ae7591..717140305 100644 --- a/gitnexus/src/core/ingestion/languages/python/captures.ts +++ b/gitnexus/src/core/ingestion/languages/python/captures.ts @@ -74,20 +74,11 @@ export function emitPythonScopeCaptures( let tree = cachedTree as ReturnType['parse']> | undefined; let notebookSegments: readonly NotebookLineSegment[] | undefined; if (isNotebookPath(filePath)) { - if (sourceMeta?.notebookSegments) { - notebookSegments = sourceMeta.notebookSegments; - } else { - const extracted = extractNotebookPython(sourceText); - if (extracted === null) { - if (sourceMeta?.sourceKind !== 'pre-extracted-script') return []; - } else { - parseText = extracted.pythonSource; - notebookSegments = extracted.segments; - if (sourceMeta?.sourceKind !== 'pre-extracted-script') { - tree = undefined; - } - } - } + const resolved = resolveNotebookCaptureSource(sourceText, tree, sourceMeta); + if (resolved === null) return []; + parseText = resolved.parseText; + tree = resolved.tree; + notebookSegments = resolved.notebookSegments; } // Skip the parse when the caller (the scope-resolution orchestrator's // `treeCache`) already produced a Tree for this source — empty under @@ -260,6 +251,39 @@ export function emitPythonScopeCaptures( return out; } +function resolveNotebookCaptureSource( + sourceText: string, + cachedTree: ReturnType['parse']> | undefined, + sourceMeta?: { + sourceKind?: 'full-file' | 'pre-extracted-script'; + notebookSegments?: readonly NotebookLineSegment[]; + }, +): { + parseText: string; + tree: ReturnType['parse']> | undefined; + notebookSegments?: readonly NotebookLineSegment[]; +} | null { + if (sourceMeta?.notebookSegments) { + return { + parseText: sourceText, + tree: cachedTree, + notebookSegments: sourceMeta.notebookSegments, + }; + } + const extracted = extractNotebookPython(sourceText); + if (extracted === null) { + if (sourceMeta?.sourceKind === 'pre-extracted-script') { + return { parseText: sourceText, tree: cachedTree }; + } + return null; + } + return { + parseText: extracted.pythonSource, + tree: sourceMeta?.sourceKind === 'pre-extracted-script' ? cachedTree : undefined, + notebookSegments: extracted.segments, + }; +} + function remapRange(range: Range, segments: readonly NotebookLineSegment[]): Range { return { ...range, diff --git a/gitnexus/src/core/ingestion/scope-extractor-bridge.ts b/gitnexus/src/core/ingestion/scope-extractor-bridge.ts index 9999cec03..c95009a74 100644 --- a/gitnexus/src/core/ingestion/scope-extractor-bridge.ts +++ b/gitnexus/src/core/ingestion/scope-extractor-bridge.ts @@ -27,6 +27,7 @@ import type { ParsedFile } from 'gitnexus-shared'; import { extract as extractScope } from './scope-extractor.js'; import type { LanguageProvider } from './language-provider.js'; +import type { NotebookLineSegment } from './ipynb-extractor.js'; import { logger } from '../logger.js'; /** Callback used to report scope-extraction warnings to the host (worker or direct). */ @@ -45,12 +46,7 @@ export function extractParsedFile( onWarn?: ScopeBridgeWarn, cachedTree?: unknown, sourceKind: ScopeCaptureSourceKind = 'full-file', - notebookSegments?: readonly { - readonly extractStartLine: number; - readonly extractEndLine: number; - readonly jsonStartLine: number; - readonly jsonEndLine: number; - }[], + notebookSegments?: readonly NotebookLineSegment[], ): ParsedFile | undefined { if (provider.emitScopeCaptures === undefined) return undefined; if (sourceText.trim().length === 0) return undefined; diff --git a/gitnexus/src/core/lbug/csv-generator.ts b/gitnexus/src/core/lbug/csv-generator.ts index 1a1ab4513..aafb556f6 100644 --- a/gitnexus/src/core/lbug/csv-generator.ts +++ b/gitnexus/src/core/lbug/csv-generator.ts @@ -23,11 +23,7 @@ import { VALID_NODE_TABLES, parseRelationSchemaPairs, RelPairRouter } from './re import { parseTruthyEnv } from '../ingestion/utils/env.js'; import { SYMBOL_NODE_LABELS } from '../ingestion/utils/symbol-labels.js'; import { applyCjkSegmentationIfEnabled } from '../search/cjk-segmentation.js'; -import { - extractNotebookPythonCached, - isNotebookPath, - notebookPythonSnippetFromExtract, -} from '../ingestion/ipynb-extractor.js'; +import { isNotebookPath, notebookPythonSnippet } from '../ingestion/ipynb-extractor.js'; /** Computed once — `RELATION_SCHEMA` is a static template literal. Exported so * the streamed sinks (`GraphEmitSink`, `PdgEmitSink`) share this parse @@ -330,19 +326,15 @@ const extractContent = async ( const endLine = node.properties.endLine; if (startLine === undefined || endLine === undefined) return ''; + const MAX_SNIPPET = 5000; + const capSnippet = (text: string): string => + text.length > MAX_SNIPPET ? text.slice(0, MAX_SNIPPET) + '\n... [truncated]' : text; + const notebookPath = String(filePath ?? ''); if (isNotebookPath(notebookPath)) { - const extracted = extractNotebookPythonCached(notebookPath, content); - const reconstructed = extracted - ? notebookPythonSnippetFromExtract(extracted, startLine, endLine) - : null; + const reconstructed = notebookPythonSnippet(content, startLine, endLine, notebookPath); if (reconstructed) { - const MAX_SNIPPET = 5000; - const capped = - reconstructed.length > MAX_SNIPPET - ? reconstructed.slice(0, MAX_SNIPPET) + '\n... [truncated]' - : reconstructed; - return normalizeFtsText(applyCjkSegmentationIfEnabled(capped)); + return normalizeFtsText(applyCjkSegmentationIfEnabled(capSnippet(reconstructed))); } } @@ -351,10 +343,7 @@ const extractContent = async ( const start = Math.max(0, exactSymbolContent ? startLine : startLine - 2); const end = Math.min(lines.length - 1, exactSymbolContent ? endLine : endLine + 2); const snippet = lines.slice(start, end + 1).join('\n'); - const MAX_SNIPPET = 5000; - const capped = - snippet.length > MAX_SNIPPET ? snippet.slice(0, MAX_SNIPPET) + '\n... [truncated]' : snippet; - return normalizeFtsText(applyCjkSegmentationIfEnabled(capped)); + return normalizeFtsText(applyCjkSegmentationIfEnabled(capSnippet(snippet))); }; // ============================================================================ diff --git a/gitnexus/test/unit/incremental-parse-cache.test.ts b/gitnexus/test/unit/incremental-parse-cache.test.ts index cbf77be34..114b6fa18 100644 --- a/gitnexus/test/unit/incremental-parse-cache.test.ts +++ b/gitnexus/test/unit/incremental-parse-cache.test.ts @@ -288,9 +288,6 @@ describe('PARSE_CACHE_VERSION', () => { // Moved 103 -> 104 for #3339 review: pair-HOC queries name // `mutation(withAuth(arrow))` object-pair handlers. Warm caches replay // anonymous arrows, so both stores re-extract. - // Moved 103 -> 104 for #3339 review: pair-HOC queries name - // `mutation(withAuth(arrow))` object-pair handlers. Warm caches replay - // anonymous arrows, so both stores re-extract. // Moved 104 -> 105 for #3371: notebook code-cell extraction before Python parse. it('pins SCHEMA_BUMP to 105 so concurrent bumps cannot silently collide (#2766, #3015, #3088, #2885, #3128, #2865, #3130, #1432, #3161, #3179, #3219, #3190, #3253, #3273, #3339, #3371)', () => { expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(105); diff --git a/gitnexus/test/unit/ingestion-utils.test.ts b/gitnexus/test/unit/ingestion-utils.test.ts index 344bbe311..2a3582773 100644 --- a/gitnexus/test/unit/ingestion-utils.test.ts +++ b/gitnexus/test/unit/ingestion-utils.test.ts @@ -3,6 +3,7 @@ import { getLanguageFromFilename, getSyntaxLanguageFromFilename, isBladeTemplateFilename, + isNotebookFilename, SupportedLanguages, } from 'gitnexus-shared'; import { getProvider, getProviderForFile } from '../../src/core/ingestion/languages/index.js'; @@ -54,6 +55,7 @@ describe('getLanguageFromFilename', () => { it('detects .ipynb files as Python', () => { expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python); expect(getProviderForFile('notebooks/analysis.ipynb')?.id).toBe(SupportedLanguages.Python); + expect(isNotebookFilename('notebooks/analysis.ipynb')).toBe(true); expect(getSyntaxLanguageFromFilename('analysis.ipynb')).toBe('json'); }); }); diff --git a/gitnexus/test/unit/ipynb-extractor.test.ts b/gitnexus/test/unit/ipynb-extractor.test.ts index 5befb06cf..9335ce0cd 100644 --- a/gitnexus/test/unit/ipynb-extractor.test.ts +++ b/gitnexus/test/unit/ipynb-extractor.test.ts @@ -287,4 +287,142 @@ describe('extractNotebookPython edge cases', () => { const realLine = content.split('\n').findIndex((l) => l.includes('def real')); expect(result!.segments[0].jsonStartLine).toBe(realLine); }); + + it('keeps Python under %%time and skips %%bash', () => { + const content = notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: {}, + source: ['%%time\n', 'def train():\n', ' pass\n'], + outputs: [], + }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('def train'); + expect(result!.pythonSource).toContain('# %%time'); + }); + + it('comments IPython help lines', () => { + const content = notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: {}, + source: ['train?\n', 'def train():\n', ' pass\n'], + outputs: [], + }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource.split('\n').filter((l) => l.length > 0)[0]).toBe('# train?'); + expect(result!.pythonSource).toContain('def train'); + }); + + it('parses a UTF-8 BOM notebook', () => { + const content = + '\uFEFF' + + notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: {}, + source: ['def train():\n', ' pass\n'], + outputs: [], + }, + ], + }); + expect(extractNotebookPython(content)?.pythonSource).toContain('def train'); + }); + + it('treats python and python3 metadata as the same family', () => { + const content = notebook({ + language: 'python', + languageInfo: 'python3', + cells: [{ cell_type: 'code', metadata: {}, source: ['a = 1\n'], outputs: [] }], + }); + expect(extractNotebookPython(content)?.pythonSource).toContain('a = 1'); + }); + + it('skips a notebook whose kernelspec name is R and has no language field', () => { + const content = JSON.stringify({ + nbformat: 4, + nbformat_minor: 5, + metadata: { kernelspec: { name: 'ir', display_name: 'R' } }, + cells: [{ cell_type: 'code', metadata: {}, source: ['x <- 1\n'], outputs: [] }], + }); + expect(extractNotebookPython(content)).toBeNull(); + }); + + it('extracts nbformat v3 worksheets via input', () => { + const content = JSON.stringify( + { + nbformat: 3, + nbformat_minor: 0, + metadata: { name: 'legacy' }, + worksheets: [ + { + cells: [ + { + cell_type: 'code', + language: 'python', + input: ['def train():\n', ' pass\n'], + outputs: [], + }, + ], + }, + ], + }, + null, + 2, + ); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('def train'); + const line = content.split('\n').findIndex((l) => l.includes('def train')); + expect(result!.segments[0].jsonStartLine).toBe(line); + }); + + it('keeps later cells when an earlier cell has an unclosed string', () => { + const content = notebook({ + language: 'python', + cells: [ + { cell_type: 'code', metadata: {}, source: ['text = """unterminated\n'], outputs: [] }, + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('def train'); + expect(result!.pythonSource).not.toMatch(/^text = """/m); + }); + + it('turns %run of a local module into an import', () => { + const content = notebook({ + language: 'python', + cells: [ + { + cell_type: 'code', + metadata: {}, + source: ['%run ./lib.py\n', 'def train():\n', ' pass\n'], + outputs: [], + }, + ], + }); + const result = extractNotebookPython(content); + expect(result!.pythonSource).toContain('import lib # %run ./lib.py'); + expect(result!.pythonSource).toContain('def train'); + }); + + it('indexes a Sage kernel as Python', () => { + const content = notebook({ + language: 'sage', + cells: [ + { cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }, + ], + }); + expect(extractNotebookPython(content)?.pythonSource).toContain('def train'); + }); });