fix(review): keep notebook graph lines aligned with JSON

Join cells so tree-sitter rows match segment maps, remap CFG and
scope captures from 1-based extract lines, and highlight .ipynb as JSON.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Gergo Magyar 2026-09-25 15:27:50 +00:00
parent 939e0e4639
commit 020bf6e4dc
11 changed files with 139 additions and 51 deletions

View file

@ -163,6 +163,8 @@ const AUXILIARY_BASENAME_MAP: Record<string, string> = {
*/
export const getSyntaxLanguageFromFilename = (filePath: string): string => {
if (isBladeTemplateFilename(filePath)) return 'markup';
// Notebooks are ingested as Python; the on-disk bytes are JSON.
if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) return 'json';
const lang = getLanguageFromFilename(filePath);
if (lang) return SYNTAX_MAP[lang];

View file

@ -39,13 +39,16 @@ export const ensureAndParse = async (content: string, filePath: string): Promise
// C++ UE macros, Dart extension types) would leave embeddings looking at an
// error-recovered tree. Resolved from `language` so the transform and the
// parser always come from the same provider. Length-preserving, so node
// offsets still index `content`.
// offsets still index `content` except for `.ipynb`, which is replaced by
// concatenated code-cell Python (same as the parse worker).
let parseContent = getProvider(language).preprocessSource?.(content, filePath) ?? content;
if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) {
const extracted = extractNotebookPython(content);
if (!extracted) return null;
parseContent = getProvider(language).preprocessSource?.(extracted.pythonSource, filePath) ??
extracted.pythonSource;
if (extracted) {
parseContent =
getProvider(language).preprocessSource?.(extracted.pythonSource, filePath) ??
extracted.pythonSource;
}
}
return parseSourceSafe(parserInstance, parseContent);

View file

@ -15,7 +15,7 @@
*/
import type { SyntaxNode } from '../utils/ast-helpers.js';
import { CfgNestingDepthError } from './cfg-builder.js';
import type { CfgVisitor, FunctionCfg } from './types.js';
import type { CfgVisitor, FunctionCfg, SiteRecord } from './types.js';
/**
* Default per-function source-line cap used by the worker when the `--pdg` run
@ -84,18 +84,26 @@ function shiftCfgLines(cfg: FunctionCfg, offset: number): FunctionCfg {
}
function remapCfgLines(cfg: FunctionCfg, mapLine: (row: number) => number): FunctionCfg {
// CFG visitors store 1-based source lines; `mapLine` maps 0-based tree-sitter rows.
const map1 = (line: number): number => mapLine(line - 1) + 1;
const mapSite = (site: SiteRecord): SiteRecord =>
site.at !== undefined ? { ...site, at: [map1(site.at[0]), site.at[1]] } : site;
return {
...cfg,
functionStartLine: mapLine(cfg.functionStartLine),
functionEndLine: mapLine(cfg.functionEndLine),
functionStartLine: map1(cfg.functionStartLine),
functionEndLine: map1(cfg.functionEndLine),
blocks: cfg.blocks.map((b) => ({
...b,
startLine: mapLine(b.startLine),
endLine: mapLine(b.endLine),
statements: b.statements?.map((s) => ({ ...s, line: mapLine(s.line) })),
startLine: map1(b.startLine),
endLine: map1(b.endLine),
statements: b.statements?.map((s) => ({
...s,
line: map1(s.line),
sites: s.sites?.map(mapSite),
})),
})),
bindings: cfg.bindings?.map((bd) =>
bd.declLine > 0 ? { ...bd, declLine: mapLine(bd.declLine) } : bd,
bd.declLine > 0 ? { ...bd, declLine: map1(bd.declLine) } : bd,
),
};
}

View file

@ -198,7 +198,6 @@ export function extractNotebookPython(content: string): NotebookPythonExtraction
const chunks: string[] = [];
const segments: NotebookLineSegment[] = [];
let extractLine = 0;
let searchFrom = 0;
for (const rawCell of nb.cells) {
@ -223,31 +222,22 @@ export function extractNotebookPython(content: string): NotebookPythonExtraction
continue;
}
const text = lines.join('\n');
let text = lines.join('\n');
if (text.endsWith('\n')) text = text.slice(0, -1);
if (text.trim().length === 0) {
chunks.push('');
const start = extractLine;
extractLine += 1;
segments.push({
extractStartLine: start,
extractEndLine: start,
jsonStartLine,
jsonEndLine,
});
continue;
}
if (chunks.length > 0) {
chunks.push('\n');
extractLine += 1;
chunks.push('\n\n');
}
const extractStartLine = extractLine;
chunks.push(text);
const addedLines = text.split('\n').length;
extractLine += addedLines;
const assembled = chunks.join('');
const extractStartLine = indexToLine(assembled, assembled.length - text.length);
const extractEndLine = indexToLine(assembled, assembled.length - 1);
segments.push({
extractStartLine,
extractEndLine: extractLine - 1,
extractEndLine,
jsonStartLine,
jsonEndLine,
});
@ -272,14 +262,29 @@ export function mapExtractLine(row: number, segments: readonly NotebookLineSegme
return last.jsonEndLine;
}
const extractCache = new Map<
string,
{ content: string; result: NotebookPythonExtraction | null }
>();
/** Memoize extraction for FTS/CSV (same file, many symbols). */
export function extractNotebookPythonCached(
filePath: string,
content: string,
): NotebookPythonExtraction | null {
const hit = extractCache.get(filePath);
if (hit && hit.content === content) return hit.result;
const result = extractNotebookPython(content);
extractCache.set(filePath, { content, result });
return result;
}
/** Python snippet for a graph span stored in JSON file coordinates. */
export function notebookPythonSnippet(
fileContent: string,
export function notebookPythonSnippetFromExtract(
extracted: NotebookPythonExtraction,
startLine: number,
endLine: number,
): string | null {
const extracted = extractNotebookPython(fileContent);
if (!extracted) return null;
const pyLines = extracted.pythonSource.split('\n');
const out: string[] = [];
for (const seg of extracted.segments) {
@ -293,3 +298,16 @@ export function notebookPythonSnippet(
if (out.length === 0) return null;
return out.join('\n');
}
export function notebookPythonSnippet(
fileContent: string,
startLine: number,
endLine: number,
filePath?: string,
): string | null {
const extracted = filePath
? extractNotebookPythonCached(filePath, fileContent)
: extractNotebookPython(fileContent);
if (!extracted) return null;
return notebookPythonSnippetFromExtract(extracted, startLine, endLine);
}

View file

@ -15,7 +15,7 @@
* Pure given the input source text. No I/O, no globals consulted.
*/
import type { Capture, CaptureMatch } from 'gitnexus-shared';
import type { Capture, CaptureMatch, Range } from 'gitnexus-shared';
import {
nodeToCapture,
syntheticCapture,
@ -24,7 +24,11 @@ import {
} from '../../utils/ast-helpers.js';
import { splitImportStatement } from './import-decomposer.js';
import { getPythonParser, getPythonScopeQuery } from './query.js';
import { extractNotebookPython } from '../../ipynb-extractor.js';
import {
extractNotebookPython,
mapExtractLine,
type NotebookLineSegment,
} from '../../ipynb-extractor.js';
import {
synthesizeConstructorFieldTypeBindings,
synthesizeReceiverTypeBinding,
@ -64,14 +68,18 @@ export function emitPythonScopeCaptures(
): readonly CaptureMatch[] {
let parseText = sourceText;
let tree = cachedTree as ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined;
if (
filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb') &&
sourceMeta?.sourceKind !== 'pre-extracted-script'
) {
let notebookSegments: readonly NotebookLineSegment[] | undefined;
if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) {
const extracted = extractNotebookPython(sourceText);
if (extracted === null) return [];
parseText = extracted.pythonSource;
tree = undefined;
if (extracted === null) {
if (sourceMeta?.sourceKind !== 'pre-extracted-script') return [];
} else {
parseText = extracted.pythonSource;
notebookSegments = extracted.segments;
if (sourceMeta?.sourceKind !== 'pre-extracted-script') {
tree = undefined;
}
}
}
// Skip the parse when the caller (the scope-resolution orchestrator's
// `treeCache`) already produced a Tree for this source — empty under
@ -238,9 +246,31 @@ export function emitPythonScopeCaptures(
out.push(...synthesizePythonInheritanceReferences(tree.rootNode));
out.push(...synthesizeCallableFlowCaptures(tree.rootNode, PYTHON_CALLABLE_CAPTURE_OPTIONS));
if (notebookSegments !== undefined) {
return out.map((match) => remapCaptureMatch(match, notebookSegments));
}
return out;
}
function remapRange(range: Range, segments: readonly NotebookLineSegment[]): Range {
return {
...range,
startLine: mapExtractLine(range.startLine - 1, segments) + 1,
endLine: mapExtractLine(range.endLine - 1, segments) + 1,
};
}
function remapCaptureMatch(
match: CaptureMatch,
segments: readonly NotebookLineSegment[],
): CaptureMatch {
const next: Record<string, Capture> = {};
for (const [key, cap] of Object.entries(match)) {
next[key] = { ...cap, range: remapRange(cap.range, segments) };
}
return next;
}
/**
* Synthesize `@reference.inherits` captures from Python class superclass
* lists so the registry-primary scope-resolution path emits EXTENDS edges

View file

@ -1685,14 +1685,14 @@ const processFileGroup = (
let scopeExtractionFailed = false;
const parsedFile = extractParsedFile(
provider,
parseContent,
notebookSegments ? file.content : parseContent,
file.path,
(message) => {
scopeExtractionFailed = true;
reportWarning(message);
},
tree,
scopeSourceKind,
notebookSegments ? 'full-file' : scopeSourceKind,
);
if (scopeExtractionFailed) (result.scopeExtractionFailures ??= []).push(file.path);
if (parsedFile !== undefined) {

View file

@ -23,7 +23,10 @@ import { VALID_NODE_TABLES, parseRelationSchemaPairs, RelPairRouter } from './re
import { parseTruthyEnv } from '../ingestion/utils/env.js';
import { SYMBOL_NODE_LABELS } from '../ingestion/utils/symbol-labels.js';
import { applyCjkSegmentationIfEnabled } from '../search/cjk-segmentation.js';
import { notebookPythonSnippet } from '../ingestion/ipynb-extractor.js';
import {
notebookPythonSnippetFromExtract,
extractNotebookPythonCached,
} from '../ingestion/ipynb-extractor.js';
/** Computed once — `RELATION_SCHEMA` is a static template literal. Exported so
* the streamed sinks (`GraphEmitSink`, `PdgEmitSink`) share this parse
@ -330,7 +333,10 @@ const extractContent = async (
.replace(/\\/g, '/')
.toLowerCase();
if (notebookPath.endsWith('.ipynb')) {
const reconstructed = notebookPythonSnippet(content, startLine, endLine);
const extracted = extractNotebookPythonCached(notebookPath, content);
const reconstructed = extracted
? notebookPythonSnippetFromExtract(extracted, startLine, endLine)
: null;
if (reconstructed) {
const MAX_SNIPPET = 5000;
const capped =

View file

@ -77,6 +77,16 @@ describe('Jupyter notebook Python pipeline', () => {
n.properties.filePath.replace(/\\/g, '/').endsWith('julia.ipynb') && n.name === 'train',
),
).toHaveLength(0);
expect(
getNodesByLabelFull(result, 'File').filter((n) =>
n.properties.filePath.replace(/\\/g, '/').endsWith('julia.ipynb'),
),
).toHaveLength(0);
expect(
getNodesByLabelFull(result, 'File').filter((n) =>
n.properties.filePath.replace(/\\/g, '/').endsWith('broken.ipynb'),
),
).toHaveLength(0);
} finally {
fs.rmSync(root, { recursive: true, force: true });
}

View file

@ -54,6 +54,7 @@ describe('getLanguageFromFilename', () => {
it('detects .ipynb files as Python', () => {
expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python);
expect(getProviderForFile('notebooks/analysis.ipynb')?.id).toBe(SupportedLanguages.Python);
expect(getSyntaxLanguageFromFilename('analysis.ipynb')).toBe('json');
});
});

View file

@ -15,8 +15,7 @@ function notebook(opts: {
opts.language === undefined
? undefined
: { display_name: 'Python', language: opts.language, name: 'python' };
const language_info =
opts.languageInfo === undefined ? undefined : { name: opts.languageInfo };
const language_info = opts.languageInfo === undefined ? undefined : { name: opts.languageInfo };
return JSON.stringify(
{
nbformat: 4,
@ -64,15 +63,21 @@ describe('extractNotebookPython', () => {
language: 'python',
cells: [
{ cell_type: 'code', metadata: {}, source: ['x = 1\n'], outputs: [] },
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' return x\n'], outputs: [] },
{
cell_type: 'code',
metadata: {},
source: ['def train():\n', ' return x\n'],
outputs: [],
},
],
});
const result = extractNotebookPython(content);
expect(result).not.toBeNull();
expect(result!.pythonSource).toMatch(/x = 1\n+def train/);
expect(result!.segments).toHaveLength(2);
expect(result!.segments[1].extractStartLine).toBeGreaterThan(result!.segments[0].extractStartLine);
expect(result!.segments[1].jsonStartLine).toBeGreaterThan(result!.segments[0].jsonStartLine);
const defRow = result!.pythonSource.split('\n').findIndex((l) => l.startsWith('def train'));
expect(defRow).toBe(result!.segments[1].extractStartLine);
expect(mapExtractLine(defRow, result!.segments)).toBe(result!.segments[1].jsonStartLine);
});
it('accepts source as a single string', () => {
@ -199,7 +204,9 @@ describe('notebookPythonSnippet', () => {
it('returns Python def train not JSON cell_type', () => {
const content = notebook({
language: 'python',
cells: [{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }],
cells: [
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
],
});
const extracted = extractNotebookPython(content)!;
const snippet = notebookPythonSnippet(

View file

@ -37,6 +37,9 @@ describe('Python notebook scope captures', () => {
});
it('extractParsedFile yields a Function for train', () => {
const captured = emitPythonScopeCaptures(notebook, 'analysis.ipynb');
const fnCapture = captured.find((m) => m['@scope.function'] !== undefined);
expect(fnCapture?.['@scope.function']?.range.startLine).toBeGreaterThan(1);
const parsed = extractParsedFile(pythonProvider, notebook, 'analysis.ipynb');
expect(parsed).toBeDefined();
expect(parsed!.localDefs.some((d) => d.qualifiedName === 'train' || d.name === 'train')).toBe(