feat: cover real Jupyter notebook shapes in Python extraction

Index Sage and Pyodide kernels, keep later cells when one cell is broken, and link %run of a local module without executing a kernel.
This commit is contained in:
Gergo Magyar 2026-09-25 18:20:12 +00:00
parent 5535df854d
commit 5bf6dd8d99
12 changed files with 502 additions and 152 deletions

View file

@ -18,10 +18,16 @@ After `analyze`, functions, classes, and imports defined in Python code cells ar
## Kernel and magics
- Skip the file when `kernelspec.language` or `language_info.name` is present and is not a Python-family name (`python`, `python2`, `python3`, `ipython`). Disagreeing fields skip the file.
- If those fields are absent, code cells are treated as Python unless a cell's own language metadata says otherwise.
- A cell whose first non-empty line is a cell magic (`%%`) is skipped.
- Line magics (`%`) and shell (`!`) lines are commented in place so JSON line mapping stays affine.
- Skip the file when `kernelspec.language` or `language_info.name` is present and is not a Python-family name (`python`, `python2`, `python3`, `ipython`, `python 3`, `ipython3`). `python` and `python3` together still index.
- If `kernelspec.language` is missing, `kernelspec.name` is used only when it is an obvious language id (`python3`, `ir`, `julia-1.8`). Conda env names are ignored.
- If `language_info.name` is missing, `language_info.file_extension` (`.py` vs `.r` / `.jl`) is used the same way.
- If those fields are absent, code cells are treated as Python unless a cell's own `language`, `metadata.language`, or `metadata.vscode.languageId` says otherwise.
- A cell whose first non-empty line is a foreign cell magic (`%%bash`, `%%html`, `%%sql`, `%%R`) is skipped. Python-body cell magics (`%%time`, `%%timeit`, `%%capture`, `%%prun`, `%%debug`, `%%px`, `%%python`) stay, with the magic line commented.
- `%run`, `%load`, and `%loadpy` of a local `.py` path become `import module` on that same line so the notebook links to the file. URLs, `..` paths, and `.ipynb` targets stay comments. The notebook is not executed.
- Sage, SageMath, MicroPython, Pyodide, PyPy, and PySpark kernels are indexed as Python. SQL, R, and Julia cells are not.
- A cell with an unclosed string or bracket is commented out so it cannot hide later cells. Those cells are concatenated in order; one syntax error no longer drops the rest of the notebook.
- Line magics (`%`), shell (`!`), and IPython help (`train?`, `?train`) are commented in place so JSON line mapping stays affine.
- nbformat v4 `cells` / `source` is the normal path. nbformat v3 `worksheets[].cells` and code-cell `input` are accepted. A leading UTF-8 BOM is accepted. Markdown and raw cells are not code.
## Line numbers
@ -33,6 +39,6 @@ Concatenating cells in document order is notebook semantics. An earlier cell wit
- `gitnexus/test/unit/ipynb-extractor.test.ts`
- `gitnexus/test/unit/ingestion-utils.test.ts` (`.ipynb` detection)
- `gitnexus/test/integration/ipynb-python-pipeline.test.ts`
- `gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts`
No Jupyter, nbconvert, or nbformat runtime dependency.

View file

@ -22,6 +22,7 @@ export {
getLanguageFromFilename,
getSyntaxLanguageFromFilename,
isBladeTemplateFilename,
isNotebookFilename,
} from './language-detection.js';
export type { MroStrategy } from './mro-strategy.js';

View file

@ -76,6 +76,10 @@ for (const [lang, exts] of Object.entries(EXTENSION_MAP) as [
export const isBladeTemplateFilename = (filePath: string): boolean =>
filePath.replace(/\\/g, '/').toLowerCase().endsWith('.blade.php');
/** Jupyter notebooks: ingested as Python; on-disk bytes are JSON. */
export const isNotebookFilename = (filePath: string): boolean =>
filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb');
/**
* Map file extension to SupportedLanguage enum.
* Returns null if the file extension is not recognized.
@ -163,8 +167,7 @@ const AUXILIARY_BASENAME_MAP: Record<string, string> = {
*/
export const getSyntaxLanguageFromFilename = (filePath: string): string => {
if (isBladeTemplateFilename(filePath)) return 'markup';
// Notebooks are ingested as Python; the on-disk bytes are JSON.
if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) return 'json';
if (isNotebookFilename(filePath)) return 'json';
const lang = getLanguageFromFilename(filePath);
if (lang) return SYNTAX_MAP[lang];

View file

@ -41,15 +41,16 @@ export const ensureAndParse = async (content: string, filePath: string): Promise
// parser always come from the same provider. Length-preserving, so node
// offsets still index `content` except for `.ipynb`, which is replaced by
// concatenated code-cell Python (same as the parse worker).
let parseContent = getProvider(language).preprocessSource?.(content, filePath) ?? content;
const provider = getProvider(language);
if (isNotebookPath(filePath)) {
const extracted = extractNotebookPython(content);
if (!extracted) return null;
parseContent =
getProvider(language).preprocessSource?.(extracted.pythonSource, filePath) ??
extracted.pythonSource;
const parseContent =
provider.preprocessSource?.(extracted.pythonSource, filePath) ?? extracted.pythonSource;
return parseSourceSafe(parserInstance, parseContent);
}
const parseContent = provider.preprocessSource?.(content, filePath) ?? content;
return parseSourceSafe(parserInstance, parseContent);
};

View file

@ -9,6 +9,9 @@
* buffer lines map through {@link mapExtractLine}.
*/
import { isNotebookFilename } from 'gitnexus-shared';
import { buildLineIndex, lineFromOffset } from '../embeddings/line-index.js';
export interface NotebookLineSegment {
readonly extractStartLine: number;
readonly extractEndLine: number;
@ -21,42 +24,60 @@ export interface NotebookPythonExtraction {
readonly segments: readonly NotebookLineSegment[];
}
const PYTHON_FAMILY = new Set(['python', 'python2', 'python3', 'ipython']);
const PYTHON_FAMILY = new Set([
'python',
'python2',
'python3',
'ipython',
'sage',
'sagemath',
'micropython',
'pyodide',
'pypy',
'pyspark',
]);
export function isNotebookPath(filePath: string): boolean {
return filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb');
return isNotebookFilename(filePath);
}
export function isPythonFamilyLanguage(name: string | undefined | null): boolean {
if (name === undefined || name === null) return false;
const n = name.trim().toLowerCase();
if (PYTHON_FAMILY.has(n)) return true;
return /^python\d/.test(n);
if (n.startsWith('ipython')) return true;
return /^python(?:\d|\b)/.test(n);
}
function buildLineStarts(content: string): number[] {
const starts = [0];
for (let i = 0; i < content.length; i++) {
if (content.charCodeAt(i) === 10) starts.push(i + 1);
}
return starts;
/** Kernel spec names that are a language id, not a conda env label. */
const NON_PYTHON_KERNEL =
/^(?:julia|ir|r|scala|rust|ruby|javascript|nodejs|node|bash|sh|sql|csharp|fsharp|go|java|kotlin|swift|php|perl|lua|octave|matlab|sas|stata|haskell|clojure|elixir|groovy|powershell|sos|cpp|cxx|dart|typescript|wolfram|scheme|racket|ocaml|fortran|gnuplot|sqlite|mysql|postgresql|tsql)(?:[-_][\w.-]+|\d[\w.-]*)?$/i;
function kernelNameLanguage(name: string): string | undefined {
if (isPythonFamilyLanguage(name) || NON_PYTHON_KERNEL.test(name.trim())) return name;
return undefined;
}
function indexToLine(lineStarts: readonly number[], index: number): number {
if (index <= 0) return 0;
let lo = 0;
let hi = lineStarts.length - 1;
let ans = 0;
while (lo <= hi) {
const mid = (lo + hi) >> 1;
if (lineStarts[mid] <= index) {
ans = mid;
lo = mid + 1;
} else {
hi = mid - 1;
}
function extensionLanguage(ext: string): string | undefined {
const e = ext.trim().toLowerCase();
if (e === '.py' || e === '.pyi' || e === '.ipy') return 'python';
if (
e === '.r' ||
e === '.jl' ||
e === '.scala' ||
e === '.js' ||
e === '.java' ||
e === '.go' ||
e === '.rs' ||
e === '.rb' ||
e === '.php' ||
e === '.kt' ||
e === '.swift' ||
e === '.cs'
) {
return ext;
}
return ans;
return undefined;
}
function skipWs(content: string, i: number): number {
@ -138,34 +159,13 @@ function jsonBraceSpan(
return { start: i, end: j };
}
function findDepth1Key(content: string, objStart: number, objEnd: number, key: string): number {
let depth = 0;
let inStr = false;
let j = objStart;
while (j < objEnd) {
const ch = content[j];
if (inStr) {
if (ch === '\\') j += 2;
else {
if (ch === '"') inStr = false;
j++;
}
continue;
}
if (ch === '"') {
if (depth === 1 && content.startsWith(key, j)) return j;
inStr = true;
j++;
continue;
}
if (ch === '{' || ch === '[') depth++;
else if (ch === '}' || ch === ']') depth--;
j++;
}
return -1;
}
function findLastDepth1Key(content: string, objStart: number, objEnd: number, key: string): number {
function findDepth1Key(
content: string,
objStart: number,
objEnd: number,
key: string,
match: 'first' | 'last' = 'first',
): number {
let depth = 0;
let inStr = false;
let j = objStart;
@ -181,7 +181,10 @@ function findLastDepth1Key(content: string, objStart: number, objEnd: number, ke
continue;
}
if (ch === '"') {
if (depth === 1 && content.startsWith(key, j)) found = j;
if (depth === 1 && content.startsWith(key, j)) {
if (match === 'first') return j;
found = j;
}
inStr = true;
j++;
continue;
@ -193,16 +196,63 @@ function findLastDepth1Key(content: string, objStart: number, objEnd: number, ke
return found;
}
function findCellsArraySpan(content: string): { start: number; end: number } | null {
const root = jsonBraceSpan(content, 0, '{', '}');
if (!root) return null;
const cellsKey = findLastDepth1Key(content, root.start, root.end, '"cells"');
if (cellsKey < 0) return null;
const colon = content.indexOf(':', cellsKey + 7);
if (colon < 0 || colon >= root.end) return null;
function arraySpanAfterKey(
content: string,
objStart: number,
objEnd: number,
key: '"cells"' | '"worksheets"',
match: 'first' | 'last',
): { start: number; end: number } | null {
const keyAt = findDepth1Key(content, objStart, objEnd, key, match);
if (keyAt < 0) return null;
const colon = content.indexOf(':', keyAt + key.length);
if (colon < 0 || colon >= objEnd) return null;
return jsonBraceSpan(content, colon + 1, '[', ']');
}
function codeCellArraySpans(content: string, origin: number): { start: number; end: number }[] {
const root = jsonBraceSpan(content, origin, '{', '}');
if (!root) return [];
const cells = arraySpanAfterKey(content, root.start, root.end, '"cells"', 'last');
if (cells) return [cells];
const worksheets = arraySpanAfterKey(content, root.start, root.end, '"worksheets"', 'last');
if (!worksheets) return [];
const spans: { start: number; end: number }[] = [];
let search = worksheets.start + 1;
while (search < worksheets.end) {
const brace = nextUnquotedChar(content, search, worksheets.end, '{');
if (brace < 0) break;
const obj = jsonBraceSpan(content, brace, '{', '}');
if (!obj || obj.end > worksheets.end) {
search = brace + 1;
continue;
}
const nested = arraySpanAfterKey(content, obj.start, obj.end, '"cells"', 'first');
if (nested) spans.push(nested);
search = obj.end;
}
return spans;
}
function readJsonString(content: string, start: number): string | null {
const i = skipWs(content, start);
if (content[i] !== '"') return null;
let j = i + 1;
let out = '';
while (j < content.length) {
const ch = content[j];
if (ch === '\\') {
out += content[j + 1] ?? '';
j += 2;
continue;
}
if (ch === '"') return out;
out += ch;
j++;
}
return null;
}
function nextUnquotedChar(content: string, from: number, until: number, needle: '{' | '"'): number {
let inStr = false;
let j = from;
@ -263,15 +313,21 @@ function findNextCodeCellSourceSpan(
continue;
}
const valueStart = skipWs(content, colon + 1);
if (content.slice(valueStart, valueStart + 6) !== '"code"') {
const cellType = readJsonString(content, valueStart);
if (cellType === null || cellType.toLowerCase() !== 'code') {
search = obj.end;
continue;
}
const sourceKey = findDepth1Key(content, obj.start, obj.end, '"source"');
let sourceKey = findDepth1Key(content, obj.start, obj.end, '"source"');
let keyLen = 8;
if (sourceKey < 0) {
sourceKey = findDepth1Key(content, obj.start, obj.end, '"input"');
keyLen = 7;
}
if (sourceKey < 0) {
return { span: null, nextFrom: obj.end };
}
const srcColon = content.indexOf(':', sourceKey + 8);
const srcColon = content.indexOf(':', sourceKey + keyLen);
if (srcColon < 0 || srcColon >= obj.end) {
return { span: null, nextFrom: obj.end };
}
@ -293,6 +349,7 @@ function flattenSource(source: unknown): string {
}
function cellLanguage(cell: Record<string, unknown>): string | undefined {
if (typeof cell.language === 'string') return cell.language;
const meta = cell.metadata;
if (!meta || typeof meta !== 'object') return undefined;
const m = meta as Record<string, unknown>;
@ -316,12 +373,16 @@ function notebookLanguageFields(nb: Record<string, unknown>): {
const li = md.language_info;
return {
kernelspec:
ks && typeof ks === 'object' && typeof (ks as Record<string, unknown>).language === 'string'
? String((ks as Record<string, unknown>).language)
ks && typeof ks === 'object'
? typeof (ks as Record<string, unknown>).language === 'string'
? String((ks as Record<string, unknown>).language)
: kernelNameLanguage(String((ks as Record<string, unknown>).name ?? ''))
: undefined,
languageInfo:
li && typeof li === 'object' && typeof (li as Record<string, unknown>).name === 'string'
? String((li as Record<string, unknown>).name)
li && typeof li === 'object'
? typeof (li as Record<string, unknown>).name === 'string'
? String((li as Record<string, unknown>).name)
: extensionLanguage(String((li as Record<string, unknown>).file_extension ?? ''))
: undefined,
};
}
@ -330,57 +391,191 @@ function kernelShouldSkip(nb: Record<string, unknown>): boolean {
const { kernelspec, languageInfo } = notebookLanguageFields(nb);
const fields = [kernelspec, languageInfo].filter((x): x is string => x !== undefined);
if (fields.length === 0) return false;
if (fields.some((f) => !isPythonFamilyLanguage(f))) return true;
if (fields.length === 2 && fields[0].toLowerCase() !== fields[1].toLowerCase()) {
const a = isPythonFamilyLanguage(fields[0]);
const b = isPythonFamilyLanguage(fields[1]);
if (a !== b) return true;
return fields.some((f) => !isPythonFamilyLanguage(f));
}
/** Cell magics whose body is still Python. Foreign %% cells are skipped. */
const PYTHON_CELL_MAGICS = new Set([
'time',
'timeit',
'capture',
'prun',
'lprun',
'mprun',
'debug',
'px',
'python',
'python2',
'python3',
'ipython',
'pypy',
]);
function cellMagicName(line: string): string {
const token = line.trimStart().slice(2).trim().split(/\s+/)[0] ?? '';
return token.toLowerCase();
}
function isIpythonHelpLine(line: string): boolean {
const t = line.trim();
if (t.length === 0 || t.startsWith('#')) return false;
return /^\?\??\S/.test(t) || /^[A-Za-z_][\w.]*(?:\?\?|\?)\s*$/.test(t);
}
function moduleSpecFromRunPath(raw: string): string | null {
let path = raw
.trim()
.replace(/^['"]|['"]$/g, '')
.replace(/\\/g, '/');
if (path.startsWith('http://') || path.startsWith('https://') || path.endsWith('.ipynb')) {
return null;
}
return false;
path = path.replace(/^\.\//, '');
if (path.endsWith('.py')) path = path.slice(0, -3);
const parts = path.split('/').filter((part) => part.length > 0);
if (parts.length === 0 || parts.some((part) => part === '..' || !/^[A-Za-z_]\w*$/.test(part))) {
return null;
}
return parts.join('.');
}
function rewriteRunOrLoad(line: string): string | null {
const trimmed = line.trimStart();
const indent = line.slice(0, line.length - trimmed.length);
const match = trimmed.match(/^%(?:run|load|loadpy)\s+(\S+)/);
if (!match) return null;
const spec = moduleSpecFromRunPath(match[1]);
if (!spec) return null;
return `${indent}import ${spec} # ${trimmed}`;
}
function isBalancedPython(source: string): boolean {
let quote: "'" | '"' | null = null;
let triple = false;
let paren = 0;
let bracket = 0;
let brace = 0;
for (let i = 0; i < source.length; i++) {
const ch = source[i];
if (quote) {
if (ch === '\\' && !triple) {
i++;
continue;
}
if (triple && source.startsWith(quote.repeat(3), i)) {
quote = null;
triple = false;
i += 2;
continue;
}
if (!triple && ch === quote) quote = null;
continue;
}
if (ch === '#') {
const nl = source.indexOf('\n', i);
if (nl < 0) break;
i = nl;
continue;
}
if ((ch === '"' || ch === "'") && source.startsWith(ch.repeat(3), i)) {
quote = ch;
triple = true;
i += 2;
continue;
}
if (ch === '"' || ch === "'") {
quote = ch;
continue;
}
if (ch === '(') paren++;
else if (ch === ')') paren--;
else if (ch === '[') bracket++;
else if (ch === ']') bracket--;
else if (ch === '{') brace++;
else if (ch === '}') brace--;
if (paren < 0 || bracket < 0 || brace < 0) return false;
}
return quote === null && paren === 0 && bracket === 0 && brace === 0;
}
function processCellLines(raw: string): { skipCell: boolean; lines: string[] } {
const lines = raw.split('\n');
const lines = raw.replace(/\r\n/g, '\n').replace(/\r/g, '\n').split('\n');
const firstNonEmpty = lines.find((l) => l.trim().length > 0);
if (firstNonEmpty !== undefined && firstNonEmpty.trimStart().startsWith('%%')) {
const magic =
firstNonEmpty !== undefined && firstNonEmpty.trimStart().startsWith('%%')
? cellMagicName(firstNonEmpty)
: undefined;
if (magic !== undefined && !PYTHON_CELL_MAGICS.has(magic) && !isPythonFamilyLanguage(magic)) {
return { skipCell: true, lines: [''] };
}
const out = lines.map((line) => {
const rewritten = lines.map((line) => {
const t = line.trimStart();
if (t.startsWith('%') || t.startsWith('!')) {
return `# ${line}`;
}
if (isIpythonHelpLine(line) || t.startsWith('!')) return `# ${line}`;
const runImport = rewriteRunOrLoad(line);
if (runImport) return runImport;
if (t.startsWith('%')) return `# ${line}`;
return line;
});
return { skipCell: false, lines: out };
if (isBalancedPython(rewritten.join('\n'))) {
return { skipCell: false, lines: rewritten };
}
return {
skipCell: false,
lines: rewritten.map((line) => (line.trim().length === 0 ? line : `# ${line}`)),
};
}
function notebookCodeCells(nb: Record<string, unknown>): unknown[] | null {
if (Array.isArray(nb.cells)) return nb.cells;
if (!Array.isArray(nb.worksheets)) return null;
const cells: unknown[] = [];
for (const worksheet of nb.worksheets) {
if (!worksheet || typeof worksheet !== 'object') continue;
const nested = (worksheet as Record<string, unknown>).cells;
if (Array.isArray(nested)) cells.push(...nested);
}
return cells.length > 0 ? cells : null;
}
function cellSource(cell: Record<string, unknown>): unknown {
return cell.source !== undefined ? cell.source : cell.input;
}
export function extractNotebookPython(content: string): NotebookPythonExtraction | null {
const origin = content.charCodeAt(0) === 0xfeff ? 1 : 0;
let parsed: unknown;
try {
parsed = JSON.parse(content);
parsed = JSON.parse(origin === 0 ? content : content.slice(origin));
} catch {
return null;
}
if (!parsed || typeof parsed !== 'object') return null;
const nb = parsed as Record<string, unknown>;
if (!Array.isArray(nb.cells)) return null;
const codeCells = notebookCodeCells(nb);
if (!codeCells) return null;
if (kernelShouldSkip(nb)) return null;
const cells = findCellsArraySpan(content);
if (!cells) return null;
const spans = codeCellArraySpans(content, origin);
if (spans.length === 0) return null;
const lineStarts = buildLineStarts(content);
const lineStarts = buildLineIndex(content);
const chunks: string[] = [];
const segments: NotebookLineSegment[] = [];
let spanIndex = 0;
let searchFrom = 0;
for (const rawCell of nb.cells) {
for (const rawCell of codeCells) {
if (!rawCell || typeof rawCell !== 'object') continue;
const cell = rawCell as Record<string, unknown>;
if (cell.cell_type !== 'code') continue;
if (String(cell.cell_type).toLowerCase() !== 'code') continue;
const located = findNextCodeCellSourceSpan(content, searchFrom, cells);
let located: { span: { start: number; end: number } | null; nextFrom: number } | null = null;
while (spanIndex < spans.length) {
located = findNextCodeCellSourceSpan(content, searchFrom, spans[spanIndex]);
if (located) break;
spanIndex++;
searchFrom = 0;
}
if (!located) return null;
searchFrom = located.nextFrom;
if (!located.span) continue;
@ -390,16 +585,17 @@ export function extractNotebookPython(content: string): NotebookPythonExtraction
continue;
}
const { skipCell, lines } = processCellLines(flattenSource(cell.source));
const jsonStartLine = indexToLine(lineStarts, firstQuoteInValue(content, located.span));
const jsonEndLine = Math.max(jsonStartLine, indexToLine(lineStarts, located.span.end - 1));
const { skipCell, lines } = processCellLines(flattenSource(cellSource(cell)));
const jsonStartLine = lineFromOffset(lineStarts, firstQuoteInValue(content, located.span));
const jsonEndLine = Math.max(jsonStartLine, lineFromOffset(lineStarts, located.span.end - 1));
if (skipCell) {
continue;
}
let text = lines.join('\n');
if (text.endsWith('\n')) text = text.slice(0, -1);
const hadTrailingNewline = text.endsWith('\n');
if (hadTrailingNewline) text = text.slice(0, -1);
if (text.trim().length === 0) {
continue;
}
@ -407,7 +603,7 @@ export function extractNotebookPython(content: string): NotebookPythonExtraction
if (chunks.length > 0) {
chunks.push('\n\n');
}
const lineCount = text.split('\n').length;
const lineCount = hadTrailingNewline ? Math.max(1, lines.length - 1) : lines.length;
const extractStartLine =
segments.length === 0 ? 0 : segments[segments.length - 1].extractEndLine + 2;
chunks.push(text);
@ -472,7 +668,8 @@ export function notebookPythonSnippetFromExtract(
for (const seg of extracted.segments) {
if (seg.jsonEndLine < startLine || seg.jsonStartLine > endLine) continue;
for (let extract = seg.extractStartLine; extract <= seg.extractEndLine; extract++) {
const jsonLine = mapExtractLine(extract, extracted.segments);
const jsonSpan = seg.jsonEndLine - seg.jsonStartLine;
const jsonLine = seg.jsonStartLine + Math.min(extract - seg.extractStartLine, jsonSpan);
if (jsonLine >= startLine && jsonLine <= endLine) {
out.push(pyLines[extract] ?? '');
}

View file

@ -26,6 +26,7 @@ import type {
WorkspaceIndex,
} from 'gitnexus-shared';
import type { LanguageTypeConfig } from './type-extractors/types.js';
import type { NotebookLineSegment } from './ipynb-extractor.js';
import type { CallRouter } from './call-routing.js';
import type { CallExtractor } from './call-types.js';
import type { ClassExtractor } from './class-types.js';
@ -829,12 +830,7 @@ interface LanguageProviderConfig {
sourceMeta?: {
readonly sourceKind?: 'full-file' | 'pre-extracted-script';
/** Python `.ipynb` only: JSON line segments for the pre-extracted buffer. */
readonly notebookSegments?: readonly {
readonly extractStartLine: number;
readonly extractEndLine: number;
readonly jsonStartLine: number;
readonly jsonEndLine: number;
}[];
readonly notebookSegments?: readonly NotebookLineSegment[];
},
) => readonly CaptureMatch[];

View file

@ -74,20 +74,11 @@ export function emitPythonScopeCaptures(
let tree = cachedTree as ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined;
let notebookSegments: readonly NotebookLineSegment[] | undefined;
if (isNotebookPath(filePath)) {
if (sourceMeta?.notebookSegments) {
notebookSegments = sourceMeta.notebookSegments;
} else {
const extracted = extractNotebookPython(sourceText);
if (extracted === null) {
if (sourceMeta?.sourceKind !== 'pre-extracted-script') return [];
} else {
parseText = extracted.pythonSource;
notebookSegments = extracted.segments;
if (sourceMeta?.sourceKind !== 'pre-extracted-script') {
tree = undefined;
}
}
}
const resolved = resolveNotebookCaptureSource(sourceText, tree, sourceMeta);
if (resolved === null) return [];
parseText = resolved.parseText;
tree = resolved.tree;
notebookSegments = resolved.notebookSegments;
}
// Skip the parse when the caller (the scope-resolution orchestrator's
// `treeCache`) already produced a Tree for this source — empty under
@ -260,6 +251,39 @@ export function emitPythonScopeCaptures(
return out;
}
function resolveNotebookCaptureSource(
sourceText: string,
cachedTree: ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined,
sourceMeta?: {
sourceKind?: 'full-file' | 'pre-extracted-script';
notebookSegments?: readonly NotebookLineSegment[];
},
): {
parseText: string;
tree: ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined;
notebookSegments?: readonly NotebookLineSegment[];
} | null {
if (sourceMeta?.notebookSegments) {
return {
parseText: sourceText,
tree: cachedTree,
notebookSegments: sourceMeta.notebookSegments,
};
}
const extracted = extractNotebookPython(sourceText);
if (extracted === null) {
if (sourceMeta?.sourceKind === 'pre-extracted-script') {
return { parseText: sourceText, tree: cachedTree };
}
return null;
}
return {
parseText: extracted.pythonSource,
tree: sourceMeta?.sourceKind === 'pre-extracted-script' ? cachedTree : undefined,
notebookSegments: extracted.segments,
};
}
function remapRange(range: Range, segments: readonly NotebookLineSegment[]): Range {
return {
...range,

View file

@ -27,6 +27,7 @@
import type { ParsedFile } from 'gitnexus-shared';
import { extract as extractScope } from './scope-extractor.js';
import type { LanguageProvider } from './language-provider.js';
import type { NotebookLineSegment } from './ipynb-extractor.js';
import { logger } from '../logger.js';
/** Callback used to report scope-extraction warnings to the host (worker or direct). */
@ -45,12 +46,7 @@ export function extractParsedFile(
onWarn?: ScopeBridgeWarn,
cachedTree?: unknown,
sourceKind: ScopeCaptureSourceKind = 'full-file',
notebookSegments?: readonly {
readonly extractStartLine: number;
readonly extractEndLine: number;
readonly jsonStartLine: number;
readonly jsonEndLine: number;
}[],
notebookSegments?: readonly NotebookLineSegment[],
): ParsedFile | undefined {
if (provider.emitScopeCaptures === undefined) return undefined;
if (sourceText.trim().length === 0) return undefined;

View file

@ -23,11 +23,7 @@ import { VALID_NODE_TABLES, parseRelationSchemaPairs, RelPairRouter } from './re
import { parseTruthyEnv } from '../ingestion/utils/env.js';
import { SYMBOL_NODE_LABELS } from '../ingestion/utils/symbol-labels.js';
import { applyCjkSegmentationIfEnabled } from '../search/cjk-segmentation.js';
import {
extractNotebookPythonCached,
isNotebookPath,
notebookPythonSnippetFromExtract,
} from '../ingestion/ipynb-extractor.js';
import { isNotebookPath, notebookPythonSnippet } from '../ingestion/ipynb-extractor.js';
/** Computed once — `RELATION_SCHEMA` is a static template literal. Exported so
* the streamed sinks (`GraphEmitSink`, `PdgEmitSink`) share this parse
@ -330,19 +326,15 @@ const extractContent = async (
const endLine = node.properties.endLine;
if (startLine === undefined || endLine === undefined) return '';
const MAX_SNIPPET = 5000;
const capSnippet = (text: string): string =>
text.length > MAX_SNIPPET ? text.slice(0, MAX_SNIPPET) + '\n... [truncated]' : text;
const notebookPath = String(filePath ?? '');
if (isNotebookPath(notebookPath)) {
const extracted = extractNotebookPythonCached(notebookPath, content);
const reconstructed = extracted
? notebookPythonSnippetFromExtract(extracted, startLine, endLine)
: null;
const reconstructed = notebookPythonSnippet(content, startLine, endLine, notebookPath);
if (reconstructed) {
const MAX_SNIPPET = 5000;
const capped =
reconstructed.length > MAX_SNIPPET
? reconstructed.slice(0, MAX_SNIPPET) + '\n... [truncated]'
: reconstructed;
return normalizeFtsText(applyCjkSegmentationIfEnabled(capped));
return normalizeFtsText(applyCjkSegmentationIfEnabled(capSnippet(reconstructed)));
}
}
@ -351,10 +343,7 @@ const extractContent = async (
const start = Math.max(0, exactSymbolContent ? startLine : startLine - 2);
const end = Math.min(lines.length - 1, exactSymbolContent ? endLine : endLine + 2);
const snippet = lines.slice(start, end + 1).join('\n');
const MAX_SNIPPET = 5000;
const capped =
snippet.length > MAX_SNIPPET ? snippet.slice(0, MAX_SNIPPET) + '\n... [truncated]' : snippet;
return normalizeFtsText(applyCjkSegmentationIfEnabled(capped));
return normalizeFtsText(applyCjkSegmentationIfEnabled(capSnippet(snippet)));
};
// ============================================================================

View file

@ -288,9 +288,6 @@ describe('PARSE_CACHE_VERSION', () => {
// Moved 103 -> 104 for #3339 review: pair-HOC queries name
// `mutation(withAuth(arrow))` object-pair handlers. Warm caches replay
// anonymous arrows, so both stores re-extract.
// Moved 103 -> 104 for #3339 review: pair-HOC queries name
// `mutation(withAuth(arrow))` object-pair handlers. Warm caches replay
// anonymous arrows, so both stores re-extract.
// Moved 104 -> 105 for #3371: notebook code-cell extraction before Python parse.
it('pins SCHEMA_BUMP to 105 so concurrent bumps cannot silently collide (#2766, #3015, #3088, #2885, #3128, #2865, #3130, #1432, #3161, #3179, #3219, #3190, #3253, #3273, #3339, #3371)', () => {
expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(105);

View file

@ -3,6 +3,7 @@ import {
getLanguageFromFilename,
getSyntaxLanguageFromFilename,
isBladeTemplateFilename,
isNotebookFilename,
SupportedLanguages,
} from 'gitnexus-shared';
import { getProvider, getProviderForFile } from '../../src/core/ingestion/languages/index.js';
@ -54,6 +55,7 @@ describe('getLanguageFromFilename', () => {
it('detects .ipynb files as Python', () => {
expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python);
expect(getProviderForFile('notebooks/analysis.ipynb')?.id).toBe(SupportedLanguages.Python);
expect(isNotebookFilename('notebooks/analysis.ipynb')).toBe(true);
expect(getSyntaxLanguageFromFilename('analysis.ipynb')).toBe('json');
});
});

View file

@ -287,4 +287,142 @@ describe('extractNotebookPython edge cases', () => {
const realLine = content.split('\n').findIndex((l) => l.includes('def real'));
expect(result!.segments[0].jsonStartLine).toBe(realLine);
});
it('keeps Python under %%time and skips %%bash', () => {
const content = notebook({
language: 'python',
cells: [
{
cell_type: 'code',
metadata: {},
source: ['%%time\n', 'def train():\n', ' pass\n'],
outputs: [],
},
],
});
const result = extractNotebookPython(content);
expect(result!.pythonSource).toContain('def train');
expect(result!.pythonSource).toContain('# %%time');
});
it('comments IPython help lines', () => {
const content = notebook({
language: 'python',
cells: [
{
cell_type: 'code',
metadata: {},
source: ['train?\n', 'def train():\n', ' pass\n'],
outputs: [],
},
],
});
const result = extractNotebookPython(content);
expect(result!.pythonSource.split('\n').filter((l) => l.length > 0)[0]).toBe('# train?');
expect(result!.pythonSource).toContain('def train');
});
it('parses a UTF-8 BOM notebook', () => {
const content =
'\uFEFF' +
notebook({
language: 'python',
cells: [
{
cell_type: 'code',
metadata: {},
source: ['def train():\n', ' pass\n'],
outputs: [],
},
],
});
expect(extractNotebookPython(content)?.pythonSource).toContain('def train');
});
it('treats python and python3 metadata as the same family', () => {
const content = notebook({
language: 'python',
languageInfo: 'python3',
cells: [{ cell_type: 'code', metadata: {}, source: ['a = 1\n'], outputs: [] }],
});
expect(extractNotebookPython(content)?.pythonSource).toContain('a = 1');
});
it('skips a notebook whose kernelspec name is R and has no language field', () => {
const content = JSON.stringify({
nbformat: 4,
nbformat_minor: 5,
metadata: { kernelspec: { name: 'ir', display_name: 'R' } },
cells: [{ cell_type: 'code', metadata: {}, source: ['x <- 1\n'], outputs: [] }],
});
expect(extractNotebookPython(content)).toBeNull();
});
it('extracts nbformat v3 worksheets via input', () => {
const content = JSON.stringify(
{
nbformat: 3,
nbformat_minor: 0,
metadata: { name: 'legacy' },
worksheets: [
{
cells: [
{
cell_type: 'code',
language: 'python',
input: ['def train():\n', ' pass\n'],
outputs: [],
},
],
},
],
},
null,
2,
);
const result = extractNotebookPython(content);
expect(result!.pythonSource).toContain('def train');
const line = content.split('\n').findIndex((l) => l.includes('def train'));
expect(result!.segments[0].jsonStartLine).toBe(line);
});
it('keeps later cells when an earlier cell has an unclosed string', () => {
const content = notebook({
language: 'python',
cells: [
{ cell_type: 'code', metadata: {}, source: ['text = """unterminated\n'], outputs: [] },
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
],
});
const result = extractNotebookPython(content);
expect(result!.pythonSource).toContain('def train');
expect(result!.pythonSource).not.toMatch(/^text = """/m);
});
it('turns %run of a local module into an import', () => {
const content = notebook({
language: 'python',
cells: [
{
cell_type: 'code',
metadata: {},
source: ['%run ./lib.py\n', 'def train():\n', ' pass\n'],
outputs: [],
},
],
});
const result = extractNotebookPython(content);
expect(result!.pythonSource).toContain('import lib # %run ./lib.py');
expect(result!.pythonSource).toContain('def train');
});
it('indexes a Sage kernel as Python', () => {
const content = notebook({
language: 'sage',
cells: [
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
],
});
expect(extractNotebookPython(content)?.pythonSource).toContain('def train');
});
});