feat: index Jupyter notebooks as Python via code-cell extraction

Extract .ipynb code cells in-process (Vue-style) and parse them with the
existing Python provider so notebook functions and imports join the graph
without executing kernels. Fixes #3371.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Gergo Magyar 2026-09-25 14:49:09 +00:00
parent 233ca28492
commit d2acefe016
15 changed files with 785 additions and 18 deletions

View file

@ -0,0 +1,38 @@
# Jupyter notebook (.ipynb) indexing
Status: implemented (Python code cells)
GitNexus indexes Jupyter notebooks by extracting Python code cells and parsing them with the existing Python language provider. Notebooks are not executed.
## Goal
After `analyze`, functions, classes, and imports defined in Python code cells are queryable like ordinary `.py` files.
## Compatibility
- `.ipynb` is detected as Python (`gitnexus-shared` `EXTENSION_MAP` and `pythonProvider.extensions`).
- Extraction lives in `gitnexus/src/core/ingestion/ipynb-extractor.ts`. Shared ingestion modules do not name nbformat AST types.
- Notebooks are **not** Python import targets. A notebook may import `.py` modules; `import some_notebook` does not resolve to an `.ipynb`.
- Files over the walker size cap (default 512KB, `GITNEXUS_MAX_FILE_SIZE`) are skipped like any other oversized file. Output-heavy notebooks may need a higher cap. Outputs are not stripped in this slice.
- Group-layer FastAPI/Flask/Django scanners that require a `.py` suffix still ignore notebooks.
## Kernel and magics
- Skip the file when `kernelspec.language` or `language_info.name` is present and is not a Python-family name (`python`, `python2`, `python3`, `ipython`). Disagreeing fields skip the file.
- If those fields are absent, code cells are treated as Python unless a cell's own language metadata says otherwise.
- A cell whose first non-empty line is a cell magic (`%%`) is skipped.
- Line magics (`%`) and shell (`!`) lines are commented in place so JSON line mapping stays affine.
## Line numbers
Graph `startLine` / `endLine` are 0-based coordinates in the on-disk `.ipynb` JSON. FTS/MCP symbol snippets reconstruct cell Python; they do not slice raw JSON.
Concatenating cells in document order is notebook semantics. An earlier cell with a syntax error may cause later definitions to be missed; that is a documented limitation, not a per-cell fallback parse.
## Tests
- `gitnexus/test/unit/ipynb-extractor.test.ts`
- `gitnexus/test/unit/ingestion-utils.test.ts` (`.ipynb` detection)
- `gitnexus/test/integration/ipynb-python-pipeline.test.ts`
No Jupyter, nbconvert, or nbformat runtime dependency.

View file

@ -29,7 +29,7 @@ const RUBY_EXTENSIONLESS_FILES = new Set([
const EXTENSION_MAP: Record<SupportedLanguages, readonly string[]> = {
[SupportedLanguages.JavaScript]: ['.js', '.jsx', '.mjs', '.cjs'],
[SupportedLanguages.TypeScript]: ['.ts', '.tsx', '.mts', '.cts'],
[SupportedLanguages.Python]: ['.py'],
[SupportedLanguages.Python]: ['.py', '.ipynb'],
[SupportedLanguages.Java]: ['.java'],
[SupportedLanguages.C]: ['.c'],
[SupportedLanguages.ObjectiveC]: ['.m', '.mm'],

View file

@ -11,6 +11,7 @@ import {
} from '../tree-sitter/parser-loader.js';
import { parseSourceSafe } from '../tree-sitter/safe-parse.js';
import { getLanguageForFileContent, getProvider } from '../ingestion/languages/index.js';
import { extractNotebookPython } from '../ingestion/ipynb-extractor.js';
const parserCache = new Map<string, any>();
@ -39,7 +40,13 @@ export const ensureAndParse = async (content: string, filePath: string): Promise
// error-recovered tree. Resolved from `language` so the transform and the
// parser always come from the same provider. Length-preserving, so node
// offsets still index `content`.
const parseContent = getProvider(language).preprocessSource?.(content, filePath) ?? content;
let parseContent = getProvider(language).preprocessSource?.(content, filePath) ?? content;
if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) {
const extracted = extractNotebookPython(content);
if (!extracted) return null;
parseContent = getProvider(language).preprocessSource?.(extracted.pythonSource, filePath) ??
extracted.pythonSource;
}
return parseSourceSafe(parserInstance, parseContent);
};

View file

@ -83,12 +83,30 @@ function shiftCfgLines(cfg: FunctionCfg, offset: number): FunctionCfg {
};
}
function remapCfgLines(cfg: FunctionCfg, mapLine: (row: number) => number): FunctionCfg {
return {
...cfg,
functionStartLine: mapLine(cfg.functionStartLine),
functionEndLine: mapLine(cfg.functionEndLine),
blocks: cfg.blocks.map((b) => ({
...b,
startLine: mapLine(b.startLine),
endLine: mapLine(b.endLine),
statements: b.statements?.map((s) => ({ ...s, line: mapLine(s.line) })),
})),
bindings: cfg.bindings?.map((bd) =>
bd.declLine > 0 ? { ...bd, declLine: mapLine(bd.declLine) } : bd,
),
};
}
export function collectFunctionCfgs(
root: SyntaxNode,
visitor: CfgVisitor<SyntaxNode>,
filePath: string,
maxFunctionLines = 0,
lineOffset = 0,
mapLine?: (row: number) => number,
): CollectedCfgs {
const cfgs: FunctionCfg[] = [];
let tooManyLines = 0;
@ -109,7 +127,9 @@ export function collectFunctionCfgs(
// and silently drop every remaining function's CFG (#2195).
try {
const cfg = visitor.buildFunctionCfg(node, filePath);
if (cfg) cfgs.push(shiftCfgLines(cfg, lineOffset));
if (cfg) {
cfgs.push(mapLine ? remapCfgLines(cfg, mapLine) : shiftCfgLines(cfg, lineOffset));
}
} catch (err) {
if (err instanceof CfgNestingDepthError) tooDeeplyNested++;
else buildError++;

View file

@ -0,0 +1,298 @@
/**
* Jupyter notebook (.ipynb) Python extractor.
*
* Pulls code-cell source from nbformat JSON so the Python tree-sitter
* grammar can parse it. Pure — no I/O, no tree-sitter, worker-safe.
*
* Graph coordinates stay 0-based lines in the on-disk JSON file. Extract
* buffer lines map through {@link mapExtractLine}.
*/
export interface NotebookLineSegment {
readonly extractStartLine: number;
readonly extractEndLine: number;
readonly jsonStartLine: number;
readonly jsonEndLine: number;
}
export interface NotebookPythonExtraction {
readonly pythonSource: string;
readonly segments: readonly NotebookLineSegment[];
}
const PYTHON_FAMILY = new Set(['python', 'python2', 'python3', 'ipython']);
export function isPythonFamilyLanguage(name: string | undefined | null): boolean {
if (name === undefined || name === null) return false;
const n = name.trim().toLowerCase();
if (PYTHON_FAMILY.has(n)) return true;
return /^python\d/.test(n);
}
function indexToLine(content: string, index: number): number {
let line = 0;
const end = Math.max(0, Math.min(index, content.length));
for (let i = 0; i < end; i++) {
if (content.charCodeAt(i) === 10) line++;
}
return line;
}
function skipWs(content: string, i: number): number {
while (i < content.length) {
const c = content.charCodeAt(i);
if (c === 32 || c === 9 || c === 10 || c === 13) i++;
else break;
}
return i;
}
/** Span of a JSON string or array value starting at the first `"` or `[`. */
function jsonValueSpan(content: string, start: number): { start: number; end: number } | null {
const i = skipWs(content, start);
if (i >= content.length) return null;
if (content[i] === '"') {
let j = i + 1;
while (j < content.length) {
if (content[j] === '\\') {
j += 2;
continue;
}
if (content[j] === '"') return { start: i, end: j + 1 };
j++;
}
return null;
}
if (content[i] === '[') {
let depth = 1;
let j = i + 1;
let inStr = false;
while (j < content.length && depth > 0) {
const ch = content[j];
if (inStr) {
if (ch === '\\') j += 2;
else {
if (ch === '"') inStr = false;
j++;
}
continue;
}
if (ch === '"') inStr = true;
else if (ch === '[') depth++;
else if (ch === ']') depth--;
j++;
}
return { start: i, end: j };
}
return null;
}
function findNextCodeCellSourceSpan(
content: string,
from: number,
): { span: { start: number; end: number }; nextFrom: number } | null {
let search = from;
while (search < content.length) {
const typeKey = content.indexOf('"cell_type"', search);
if (typeKey < 0) return null;
const colon = content.indexOf(':', typeKey + 11);
if (colon < 0) return null;
const valueStart = skipWs(content, colon + 1);
if (content.slice(valueStart, valueStart + 6) !== '"code"') {
search = typeKey + 11;
continue;
}
const sourceKey = content.indexOf('"source"', valueStart);
if (sourceKey < 0) return null;
const srcColon = content.indexOf(':', sourceKey + 8);
if (srcColon < 0) return null;
const span = jsonValueSpan(content, srcColon + 1);
if (!span) return null;
return { span, nextFrom: span.end };
}
return null;
}
function flattenSource(source: unknown): string {
if (typeof source === 'string') return source;
if (Array.isArray(source)) {
return source.map((part) => (typeof part === 'string' ? part : '')).join('');
}
return '';
}
function cellLanguage(cell: Record<string, unknown>): string | undefined {
const meta = cell.metadata;
if (!meta || typeof meta !== 'object') return undefined;
const m = meta as Record<string, unknown>;
if (typeof m.language === 'string') return m.language;
const vscode = m.vscode;
if (vscode && typeof vscode === 'object') {
const langId = (vscode as Record<string, unknown>).languageId;
if (typeof langId === 'string') return langId;
}
return undefined;
}
function notebookLanguageFields(nb: Record<string, unknown>): {
kernelspec?: string;
languageInfo?: string;
} {
const metadata = nb.metadata;
if (!metadata || typeof metadata !== 'object') return {};
const md = metadata as Record<string, unknown>;
const ks = md.kernelspec;
const li = md.language_info;
return {
kernelspec:
ks && typeof ks === 'object' && typeof (ks as Record<string, unknown>).language === 'string'
? String((ks as Record<string, unknown>).language)
: undefined,
languageInfo:
li && typeof li === 'object' && typeof (li as Record<string, unknown>).name === 'string'
? String((li as Record<string, unknown>).name)
: undefined,
};
}
function kernelShouldSkip(nb: Record<string, unknown>): boolean {
const { kernelspec, languageInfo } = notebookLanguageFields(nb);
const fields = [kernelspec, languageInfo].filter((x): x is string => x !== undefined);
if (fields.length === 0) return false;
if (fields.some((f) => !isPythonFamilyLanguage(f))) return true;
if (fields.length === 2 && fields[0].toLowerCase() !== fields[1].toLowerCase()) {
const a = isPythonFamilyLanguage(fields[0]);
const b = isPythonFamilyLanguage(fields[1]);
if (a !== b) return true;
}
return false;
}
function processCellLines(raw: string): { skipCell: boolean; lines: string[] } {
const lines = raw.split('\n');
const firstNonEmpty = lines.find((l) => l.trim().length > 0);
if (firstNonEmpty !== undefined && firstNonEmpty.trimStart().startsWith('%%')) {
return { skipCell: true, lines: [''] };
}
const out = lines.map((line) => {
const t = line.trimStart();
if (t.startsWith('%') || t.startsWith('!')) {
return `# ${line}`;
}
return line;
});
return { skipCell: false, lines: out };
}
export function extractNotebookPython(content: string): NotebookPythonExtraction | null {
let parsed: unknown;
try {
parsed = JSON.parse(content);
} catch {
return null;
}
if (!parsed || typeof parsed !== 'object') return null;
const nb = parsed as Record<string, unknown>;
if (!Array.isArray(nb.cells)) return null;
if (kernelShouldSkip(nb)) return null;
const chunks: string[] = [];
const segments: NotebookLineSegment[] = [];
let extractLine = 0;
let searchFrom = 0;
let codeCellIndex = 0;
for (const rawCell of nb.cells) {
if (!rawCell || typeof rawCell !== 'object') continue;
const cell = rawCell as Record<string, unknown>;
if (cell.cell_type !== 'code') continue;
const located = findNextCodeCellSourceSpan(content, searchFrom);
if (!located) return null;
searchFrom = located.nextFrom;
codeCellIndex++;
const lang = cellLanguage(cell);
if (lang !== undefined && !isPythonFamilyLanguage(lang)) {
continue;
}
const { skipCell, lines } = processCellLines(flattenSource(cell.source));
const jsonStartLine = indexToLine(content, located.span.start);
const jsonEndLine = Math.max(jsonStartLine, indexToLine(content, located.span.end - 1));
if (skipCell) {
continue;
}
const text = lines.join('\n');
if (text.trim().length === 0) {
chunks.push('');
const start = extractLine;
extractLine += 1;
segments.push({
extractStartLine: start,
extractEndLine: start,
jsonStartLine,
jsonEndLine,
});
continue;
}
if (chunks.length > 0) {
chunks.push('\n');
extractLine += 1;
}
const extractStartLine = extractLine;
chunks.push(text);
const addedLines = text.split('\n').length;
extractLine += addedLines;
segments.push({
extractStartLine,
extractEndLine: extractLine - 1,
jsonStartLine,
jsonEndLine,
});
}
void codeCellIndex;
const pythonSource = chunks.join('');
if (pythonSource.trim().length === 0) return null;
return { pythonSource, segments };
}
export function mapExtractLine(row: number, segments: readonly NotebookLineSegment[]): number {
if (segments.length === 0) return row;
for (const seg of segments) {
if (row >= seg.extractStartLine && row <= seg.extractEndLine) {
const delta = row - seg.extractStartLine;
const jsonSpan = seg.jsonEndLine - seg.jsonStartLine;
return seg.jsonStartLine + Math.min(delta, jsonSpan);
}
}
if (row < segments[0].extractStartLine) return segments[0].jsonStartLine;
const last = segments[segments.length - 1];
return last.jsonEndLine;
}
/** Python snippet for a graph span stored in JSON file coordinates. */
export function notebookPythonSnippet(
fileContent: string,
startLine: number,
endLine: number,
): string | null {
const extracted = extractNotebookPython(fileContent);
if (!extracted) return null;
const pyLines = extracted.pythonSource.split('\n');
const out: string[] = [];
for (const seg of extracted.segments) {
for (let extract = seg.extractStartLine; extract <= seg.extractEndLine; extract++) {
const jsonLine = mapExtractLine(extract, extracted.segments);
if (jsonLine >= startLine && jsonLine <= endLine) {
out.push(pyLines[extract] ?? '');
}
}
}
if (out.length === 0) return null;
return out.join('\n');
}

View file

@ -107,7 +107,7 @@ function normalizePythonStringLiteral(text: string): string | undefined {
export const pythonProvider = defineLanguage({
id: SupportedLanguages.Python,
extensions: ['.py'],
extensions: ['.py', '.ipynb'],
entryPointPatterns: [/^app$/, /^(get|post|put|delete|patch)_/i, /^api_/, /^view_/],
astFrameworkPatterns: [
{

View file

@ -24,6 +24,7 @@ import {
} from '../../utils/ast-helpers.js';
import { splitImportStatement } from './import-decomposer.js';
import { getPythonParser, getPythonScopeQuery } from './query.js';
import { extractNotebookPython } from '../../ipynb-extractor.js';
import {
synthesizeConstructorFieldTypeBindings,
synthesizeReceiverTypeBinding,
@ -57,23 +58,34 @@ const PYTHON_CALLABLE_CAPTURE_OPTIONS = {
export function emitPythonScopeCaptures(
sourceText: string,
_filePath: string,
filePath: string,
cachedTree?: unknown,
sourceMeta?: { sourceKind?: 'full-file' | 'pre-extracted-script' },
): readonly CaptureMatch[] {
let parseText = sourceText;
let tree = cachedTree as ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined;
if (
filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb') &&
sourceMeta?.sourceKind !== 'pre-extracted-script'
) {
const extracted = extractNotebookPython(sourceText);
if (extracted === null) return [];
parseText = extracted.pythonSource;
tree = undefined;
}
// Skip the parse when the caller (the scope-resolution orchestrator's
// `treeCache`) already produced a Tree for this source — empty under
// worker-pool runs, so cache miss = re-parse. The cachedTree parameter
// is typed as `unknown` at the
// contract layer (see `LanguageProvider.emitScopeCaptures`); cast
// here at the use site.
let tree = cachedTree as ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined;
if (tree === undefined) {
try {
tree = parseSourceSafe(getPythonParser(), sourceText, undefined, {
bufferSize: getTreeSitterBufferSize(sourceText),
tree = parseSourceSafe(getPythonParser(), parseText, undefined, {
bufferSize: getTreeSitterBufferSize(parseText),
});
} catch (err) {
throw scopeExtractionError('parse', _filePath, err);
throw scopeExtractionError('parse', filePath, err);
}
recordCacheMiss();
} else {
@ -84,7 +96,7 @@ export function emitPythonScopeCaptures(
try {
rawMatches = getPythonScopeQuery().matches(tree.rootNode);
} catch (err) {
throw scopeExtractionError('scope query', _filePath, err);
throw scopeExtractionError('scope query', filePath, err);
}
const out: CaptureMatch[] = [];

View file

@ -135,6 +135,11 @@ import {
extractTemplateComponents,
isVueSetupTopLevel,
} from '../vue-sfc-extractor.js';
import {
extractNotebookPython,
mapExtractLine,
type NotebookLineSegment,
} from '../ipynb-extractor.js';
import type { NodeLabel, ParameterTypeClass } from 'gitnexus-shared';
import type { FieldInfo, FieldExtractorContext } from '../field-types.js';
import type { MethodInfo, MethodExtractorContext } from '../method-types.js';
@ -1597,6 +1602,9 @@ const processFileGroup = (
let scopeSourceKind: ScopeCaptureSourceKind = 'full-file';
let lineOffset = 0;
let isVueSetup = false;
let notebookSegments: readonly NotebookLineSegment[] | undefined;
const mapRow = (row: number): number =>
notebookSegments ? mapExtractLine(row, notebookSegments) : row + lineOffset;
if (language === SupportedLanguages.Vue) {
const extracted = extractVueScript(file.content);
if (!extracted) continue; // skip .vue files with no script block
@ -1604,6 +1612,15 @@ const processFileGroup = (
scopeSourceKind = 'pre-extracted-script';
lineOffset = extracted.lineOffset;
isVueSetup = extracted.isSetup;
} else if (
language === SupportedLanguages.Python &&
file.path.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')
) {
const extracted = extractNotebookPython(file.content);
if (!extracted) continue;
parseContent = extracted.pythonSource;
scopeSourceKind = 'pre-extracted-script';
notebookSegments = extracted.segments;
}
// Per-language source-text transform (e.g., UE macro stripping for C++).
@ -1722,6 +1739,7 @@ const processFileGroup = (
// `lineOffset` in the file — shift the CFG into file coordinates so
// it joins its graph node and BasicBlock lines map to source.
lineOffset,
notebookSegments ? mapRow : undefined,
);
if (cfgs.length) withChannels = { ...withChannels, cfgSideChannel: cfgs };
// Surface per-function CFG skips per-language (#2195): merged + logged
@ -2483,10 +2501,10 @@ const processFileGroup = (
: definitionNode?.startPosition;
const startLine =
startPosition !== undefined
? startPosition.row + lineOffset
? mapRow(startPosition.row)
: nameNode
? nameNode.startPosition.row + lineOffset
: lineOffset;
? mapRow(nameNode.startPosition.row)
: mapRow(0);
const startColumn = startPosition?.column ?? nameNode?.startPosition.column ?? 0;
// Compute enclosing class BEFORE node ID — needed to qualify method IDs
@ -2958,7 +2976,7 @@ const processFileGroup = (
filePath: file.path,
toolName: nodeName,
description: (dec.arg || description || '').slice(0, 200),
lineNumber: definitionNode.startPosition.row + lineOffset,
lineNumber: mapRow(definitionNode.startPosition.row),
handlerNodeId: nodeId,
});
}
@ -3063,7 +3081,7 @@ const processFileGroup = (
(objectLiteralBindingInfo?.ownerName || isArrayContainedObjectCallable)
? { startColumn }
: {}),
endLine: definitionNode ? definitionNode.endPosition.row + lineOffset : startLine,
endLine: definitionNode ? mapRow(definitionNode.endPosition.row) : startLine,
language: language,
isExported,
...(qualifiedTypeName !== undefined ? { qualifiedName: qualifiedTypeName } : {}),

View file

@ -23,6 +23,7 @@ import { VALID_NODE_TABLES, parseRelationSchemaPairs, RelPairRouter } from './re
import { parseTruthyEnv } from '../ingestion/utils/env.js';
import { SYMBOL_NODE_LABELS } from '../ingestion/utils/symbol-labels.js';
import { applyCjkSegmentationIfEnabled } from '../search/cjk-segmentation.js';
import { notebookPythonSnippet } from '../ingestion/ipynb-extractor.js';
/** Computed once — `RELATION_SCHEMA` is a static template literal. Exported so
* the streamed sinks (`GraphEmitSink`, `PdgEmitSink`) share this parse
@ -325,6 +326,21 @@ const extractContent = async (
const endLine = node.properties.endLine;
if (startLine === undefined || endLine === undefined) return '';
const notebookPath = String(filePath ?? '')
.replace(/\\/g, '/')
.toLowerCase();
if (notebookPath.endsWith('.ipynb')) {
const reconstructed = notebookPythonSnippet(content, startLine, endLine);
if (reconstructed) {
const MAX_SNIPPET = 5000;
const capped =
reconstructed.length > MAX_SNIPPET
? reconstructed.slice(0, MAX_SNIPPET) + '\n... [truncated]'
: reconstructed;
return normalizeFtsText(applyCjkSegmentationIfEnabled(capped));
}
}
const lines = prepared.lines;
const exactSymbolContent = EXACT_SYMBOL_CONTENT_LABELS.has(node.label);
const start = Math.max(0, exactSymbolContent ? startLine : startLine - 2);

View file

@ -778,7 +778,12 @@ import { copyV8CacheIfPresent, tryLoadV8Cache, writeV8CacheFile } from './v8-sid
// v104 (#3339 review): TS/JS pair-HOC queries now name object-pair
// `mutation(withAuth(arrow))` handlers. Warm caches replay the pre-fix
// capture set (anonymous arrows, no Function name), so both stores re-extract.
const SCHEMA_BUMP = 104;
// v104 (#3339 review): TS/JS pair-HOC queries now name object-pair
// `mutation(withAuth(arrow))` handlers. Warm caches replay the pre-fix
// capture set (anonymous arrows, no Function name), so both stores re-extract.
// v105 (#3371): `.ipynb` code cells are extracted to Python before parse.
// Warm caches keyed on raw JSON would replay empty/failed Python parses.
const SCHEMA_BUMP = 105;
const GITNEXUS_PKG_VERSION = (() => {
try {
// package.json sits at gitnexus/package.json — two levels up from

View file

@ -0,0 +1,84 @@
/**
* Jupyter notebooks are indexed as Python by extracting code cells.
*/
import { describe, it, expect } from 'vitest';
import path from 'path';
import fs from 'node:fs';
import os from 'node:os';
import {
getNodesByLabel,
getNodesByLabelFull,
getRelationships,
runPipelineFromRepo,
writeFixtureRepo,
} from './helpers.js';
function pythonNotebook(cells: Array<{ source: string | string[]; language?: string }>): string {
return JSON.stringify(
{
nbformat: 4,
nbformat_minor: 5,
metadata: {
kernelspec: { display_name: 'Python 3', language: 'python', name: 'python3' },
language_info: { name: 'python' },
},
cells: cells.map((c) => ({
cell_type: 'code',
metadata: c.language ? { language: c.language } : {},
source: c.source,
outputs: [],
})),
},
null,
2,
);
}
describe('Jupyter notebook Python pipeline', () => {
it('indexes notebook functions, imports sibling py, and skip-fails closed', async () => {
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-ipynb-'));
try {
writeFixtureRepo(root, {
'lib.py': 'def helper():\n return 1\n',
'analysis.ipynb': pythonNotebook([
{ source: ['from lib import helper\n'] },
{ source: ['def train():\n', ' return helper()\n'] },
]),
'broken.ipynb': '{not json',
'julia.ipynb': JSON.stringify({
nbformat: 4,
nbformat_minor: 5,
metadata: {
kernelspec: { language: 'julia', name: 'julia', display_name: 'Julia' },
},
cells: [
{ cell_type: 'code', metadata: {}, source: ['function train() end\n'], outputs: [] },
],
}),
});
const result = await runPipelineFromRepo(root, () => {}, { skipGraphPhases: true });
const functions = getNodesByLabel(result, 'Function');
expect(functions).toContain('train');
expect(functions).toContain('helper');
const train = getNodesByLabelFull(result, 'Function').find((n) => n.name === 'train');
expect(train?.properties.filePath.replace(/\\/g, '/')).toMatch(/analysis\.ipynb$/);
expect(train?.properties.startLine).toBeGreaterThan(0);
const imports = getRelationships(result, 'IMPORTS');
expect(
imports.some(
(e) =>
e.sourceFilePath.replace(/\\/g, '/').endsWith('analysis.ipynb') &&
(e.target === 'helper' || e.targetFilePath.replace(/\\/g, '/').endsWith('lib.py')),
),
).toBe(true);
expect(
getNodesByLabelFull(result, 'Function').filter(
(n) =>
n.properties.filePath.replace(/\\/g, '/').endsWith('julia.ipynb') && n.name === 'train',
),
).toHaveLength(0);
} finally {
fs.rmSync(root, { recursive: true, force: true });
}
}, 60000);
});

View file

@ -288,8 +288,12 @@ describe('PARSE_CACHE_VERSION', () => {
// Moved 103 -> 104 for #3339 review: pair-HOC queries name
// `mutation(withAuth(arrow))` object-pair handlers. Warm caches replay
// anonymous arrows, so both stores re-extract.
it('pins SCHEMA_BUMP to 104 so concurrent bumps cannot silently collide (#2766, #3015, #3088, #2885, #3128, #2865, #3130, #1432, #3161, #3179, #3219, #3190, #3253, #3273, #3339)', () => {
expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(104);
// Moved 103 -> 104 for #3339 review: pair-HOC queries name
// `mutation(withAuth(arrow))` object-pair handlers. Warm caches replay
// anonymous arrows, so both stores re-extract.
// Moved 104 -> 105 for #3371: notebook code-cell extraction before Python parse.
it('pins SCHEMA_BUMP to 105 so concurrent bumps cannot silently collide (#2766, #3015, #3088, #2885, #3128, #2865, #3130, #1432, #3161, #3179, #3219, #3190, #3253, #3273, #3339, #3371)', () => {
expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(105);
expect(PARSE_CACHE_BUCKET_COUNT).toBe(128);
// The PREVIOUS version must fail the reuse gate, not merely differ from the
// current one — a hardcoded number outside the conflict hunk rebases cleanly
@ -298,6 +302,7 @@ describe('PARSE_CACHE_VERSION', () => {
for (const taken of [
59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, 81,
82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103,
104,
]) {
expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).not.toBe(taken);
}

View file

@ -50,6 +50,11 @@ describe('getLanguageFromFilename', () => {
it('detects .py files', () => {
expect(getLanguageFromFilename('main.py')).toBe(SupportedLanguages.Python);
});
it('detects .ipynb files as Python', () => {
expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python);
expect(getProviderForFile('notebooks/analysis.ipynb')?.id).toBe(SupportedLanguages.Python);
});
});
describe('Java', () => {

View file

@ -0,0 +1,213 @@
import { describe, it, expect } from 'vitest';
import {
extractNotebookPython,
mapExtractLine,
notebookPythonSnippet,
isPythonFamilyLanguage,
} from '../../src/core/ingestion/ipynb-extractor.js';
function notebook(opts: {
language?: string;
languageInfo?: string;
cells: Array<Record<string, unknown>>;
}): string {
const kernelspec =
opts.language === undefined
? undefined
: { display_name: 'Python', language: opts.language, name: 'python' };
const language_info =
opts.languageInfo === undefined ? undefined : { name: opts.languageInfo };
return JSON.stringify(
{
nbformat: 4,
nbformat_minor: 5,
metadata: {
...(kernelspec ? { kernelspec } : {}),
...(language_info ? { language_info } : {}),
},
cells: opts.cells,
},
null,
2,
);
}
describe('isPythonFamilyLanguage', () => {
it('accepts python3 and ipython', () => {
expect(isPythonFamilyLanguage('python3')).toBe(true);
expect(isPythonFamilyLanguage('IPython')).toBe(true);
expect(isPythonFamilyLanguage('julia')).toBe(false);
});
});
describe('extractNotebookPython', () => {
it('extracts def train from a Python v4 notebook', () => {
const content = notebook({
language: 'python',
cells: [
{
cell_type: 'code',
metadata: {},
source: ['def train():\n', ' pass\n'],
outputs: [],
},
],
});
const result = extractNotebookPython(content);
expect(result).not.toBeNull();
expect(result!.pythonSource).toContain('def train():');
expect(result!.segments).toHaveLength(1);
});
it('keeps two code cells in order with two segments', () => {
const content = notebook({
language: 'python',
cells: [
{ cell_type: 'code', metadata: {}, source: ['x = 1\n'], outputs: [] },
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' return x\n'], outputs: [] },
],
});
const result = extractNotebookPython(content);
expect(result).not.toBeNull();
expect(result!.pythonSource).toMatch(/x = 1\n+def train/);
expect(result!.segments).toHaveLength(2);
expect(result!.segments[1].extractStartLine).toBeGreaterThan(result!.segments[0].extractStartLine);
expect(result!.segments[1].jsonStartLine).toBeGreaterThan(result!.segments[0].jsonStartLine);
});
it('accepts source as a single string', () => {
const content = notebook({
language: 'python',
cells: [{ cell_type: 'code', metadata: {}, source: 'y = 2\n', outputs: [] }],
});
expect(extractNotebookPython(content)?.pythonSource).toContain('y = 2');
});
it('returns null for markdown-only notebooks', () => {
const content = notebook({
language: 'python',
cells: [{ cell_type: 'markdown', metadata: {}, source: ['# hi\n'] }],
});
expect(extractNotebookPython(content)).toBeNull();
});
it('returns null for invalid JSON', () => {
expect(extractNotebookPython('{not json')).toBeNull();
});
it('returns null for a Julia kernelspec', () => {
const content = notebook({
language: 'julia',
cells: [{ cell_type: 'code', metadata: {}, source: ['1 + 1\n'], outputs: [] }],
});
expect(extractNotebookPython(content)).toBeNull();
});
it('extracts python3 kernelspec', () => {
const content = notebook({
language: 'python3',
cells: [{ cell_type: 'code', metadata: {}, source: ['a = 1\n'], outputs: [] }],
});
expect(extractNotebookPython(content)?.pythonSource).toContain('a = 1');
});
it('comments line magics in place', () => {
const content = notebook({
language: 'python',
cells: [
{
cell_type: 'code',
metadata: {},
source: ['%time\n', 'x = 1\n', '!ls\n'],
outputs: [],
},
],
});
const result = extractNotebookPython(content);
expect(result!.pythonSource.split('\n').filter((l) => l.length > 0)).toEqual([
'# %time',
'x = 1',
'# !ls',
]);
});
it('skips a %%bash cell and still extracts later train', () => {
const content = notebook({
language: 'python',
cells: [
{ cell_type: 'code', metadata: {}, source: ['%%bash\n', 'echo hi\n'], outputs: [] },
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
],
});
const result = extractNotebookPython(content);
expect(result!.pythonSource).toContain('def train');
expect(result!.pythonSource).not.toContain('echo hi');
});
it('maps identical duplicate cells to later JSON lines', () => {
const src = ['print(1)\n'];
const content = notebook({
language: 'python',
cells: [
{ cell_type: 'code', metadata: {}, source: src, outputs: [] },
{ cell_type: 'code', metadata: {}, source: src, outputs: [] },
],
});
const result = extractNotebookPython(content);
expect(result!.segments).toHaveLength(2);
expect(result!.segments[1].jsonStartLine).toBeGreaterThan(result!.segments[0].jsonStartLine);
});
it('skips an R-language code cell and keeps Python cells', () => {
const content = notebook({
language: 'python',
cells: [
{
cell_type: 'code',
metadata: { language: 'R' },
source: ['x <- 1\n'],
outputs: [],
},
{ cell_type: 'code', metadata: {}, source: ['z = 3\n'], outputs: [] },
],
});
const result = extractNotebookPython(content);
expect(result!.pythonSource).toContain('z = 3');
expect(result!.pythonSource).not.toContain('x <- 1');
});
});
describe('mapExtractLine', () => {
it('maps a second-cell extract row onto that cell JSON line', () => {
const content = notebook({
language: 'python',
cells: [
{ cell_type: 'markdown', metadata: {}, source: ['# intro\n'] },
{ cell_type: 'code', metadata: {}, source: ['a = 1\n'], outputs: [] },
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
],
});
const extracted = extractNotebookPython(content)!;
const trainSeg = extracted.segments[1];
const mapped = mapExtractLine(trainSeg.extractStartLine, extracted.segments);
expect(mapped).toBe(trainSeg.jsonStartLine);
expect(mapped).toBeGreaterThan(extracted.segments[0].jsonStartLine);
});
});
describe('notebookPythonSnippet', () => {
it('returns Python def train not JSON cell_type', () => {
const content = notebook({
language: 'python',
cells: [{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }],
});
const extracted = extractNotebookPython(content)!;
const snippet = notebookPythonSnippet(
content,
extracted.segments[0].jsonStartLine,
extracted.segments[0].jsonEndLine,
);
expect(snippet).toContain('def train');
expect(snippet).not.toContain('cell_type');
});
});

View file

@ -0,0 +1,46 @@
import { describe, it, expect } from 'vitest';
import { emitPythonScopeCaptures } from '../../src/core/ingestion/languages/python/captures.js';
import { pythonProvider } from '../../src/core/ingestion/languages/python.js';
import { extractParsedFile } from '../../src/core/ingestion/scope-extractor-bridge.js';
import { getLanguageFromFilename, SupportedLanguages } from 'gitnexus-shared';
import { getProviderForFile } from '../../src/core/ingestion/languages/index.js';
const notebook = JSON.stringify(
{
nbformat: 4,
nbformat_minor: 5,
metadata: {
kernelspec: { language: 'python', name: 'python3', display_name: 'Python' },
},
cells: [
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
],
},
null,
2,
);
describe('Python notebook scope captures', () => {
it('detects .ipynb as Python', () => {
expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python);
expect(getProviderForFile('n.ipynb')?.id).toBe(SupportedLanguages.Python);
});
it('does not parse raw JSON as Python on the full-file path', () => {
const matches = emitPythonScopeCaptures(notebook, 'analysis.ipynb');
const names = matches.flatMap((m) => Object.keys(m));
expect(names.some((k) => k.includes('function'))).toBe(true);
});
it('returns no captures for invalid notebook JSON', () => {
expect(emitPythonScopeCaptures('{not json', 'broken.ipynb')).toEqual([]);
});
it('extractParsedFile yields a Function for train', () => {
const parsed = extractParsedFile(pythonProvider, notebook, 'analysis.ipynb');
expect(parsed).toBeDefined();
expect(parsed!.localDefs.some((d) => d.qualifiedName === 'train' || d.name === 'train')).toBe(
true,
);
});
});