mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-02 02:11:29 +00:00
fix(review): keep notebook graph lines aligned with JSON
Join cells so tree-sitter rows match segment maps, remap CFG and scope captures from 1-based extract lines, and highlight .ipynb as JSON. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
939e0e4639
commit
020bf6e4dc
11 changed files with 139 additions and 51 deletions
|
|
@ -163,6 +163,8 @@ const AUXILIARY_BASENAME_MAP: Record<string, string> = {
|
|||
*/
|
||||
export const getSyntaxLanguageFromFilename = (filePath: string): string => {
|
||||
if (isBladeTemplateFilename(filePath)) return 'markup';
|
||||
// Notebooks are ingested as Python; the on-disk bytes are JSON.
|
||||
if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) return 'json';
|
||||
|
||||
const lang = getLanguageFromFilename(filePath);
|
||||
if (lang) return SYNTAX_MAP[lang];
|
||||
|
|
|
|||
|
|
@ -39,13 +39,16 @@ export const ensureAndParse = async (content: string, filePath: string): Promise
|
|||
// C++ UE macros, Dart extension types) would leave embeddings looking at an
|
||||
// error-recovered tree. Resolved from `language` so the transform and the
|
||||
// parser always come from the same provider. Length-preserving, so node
|
||||
// offsets still index `content`.
|
||||
// offsets still index `content` except for `.ipynb`, which is replaced by
|
||||
// concatenated code-cell Python (same as the parse worker).
|
||||
let parseContent = getProvider(language).preprocessSource?.(content, filePath) ?? content;
|
||||
if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) {
|
||||
const extracted = extractNotebookPython(content);
|
||||
if (!extracted) return null;
|
||||
parseContent = getProvider(language).preprocessSource?.(extracted.pythonSource, filePath) ??
|
||||
extracted.pythonSource;
|
||||
if (extracted) {
|
||||
parseContent =
|
||||
getProvider(language).preprocessSource?.(extracted.pythonSource, filePath) ??
|
||||
extracted.pythonSource;
|
||||
}
|
||||
}
|
||||
|
||||
return parseSourceSafe(parserInstance, parseContent);
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@
|
|||
*/
|
||||
import type { SyntaxNode } from '../utils/ast-helpers.js';
|
||||
import { CfgNestingDepthError } from './cfg-builder.js';
|
||||
import type { CfgVisitor, FunctionCfg } from './types.js';
|
||||
import type { CfgVisitor, FunctionCfg, SiteRecord } from './types.js';
|
||||
|
||||
/**
|
||||
* Default per-function source-line cap used by the worker when the `--pdg` run
|
||||
|
|
@ -84,18 +84,26 @@ function shiftCfgLines(cfg: FunctionCfg, offset: number): FunctionCfg {
|
|||
}
|
||||
|
||||
function remapCfgLines(cfg: FunctionCfg, mapLine: (row: number) => number): FunctionCfg {
|
||||
// CFG visitors store 1-based source lines; `mapLine` maps 0-based tree-sitter rows.
|
||||
const map1 = (line: number): number => mapLine(line - 1) + 1;
|
||||
const mapSite = (site: SiteRecord): SiteRecord =>
|
||||
site.at !== undefined ? { ...site, at: [map1(site.at[0]), site.at[1]] } : site;
|
||||
return {
|
||||
...cfg,
|
||||
functionStartLine: mapLine(cfg.functionStartLine),
|
||||
functionEndLine: mapLine(cfg.functionEndLine),
|
||||
functionStartLine: map1(cfg.functionStartLine),
|
||||
functionEndLine: map1(cfg.functionEndLine),
|
||||
blocks: cfg.blocks.map((b) => ({
|
||||
...b,
|
||||
startLine: mapLine(b.startLine),
|
||||
endLine: mapLine(b.endLine),
|
||||
statements: b.statements?.map((s) => ({ ...s, line: mapLine(s.line) })),
|
||||
startLine: map1(b.startLine),
|
||||
endLine: map1(b.endLine),
|
||||
statements: b.statements?.map((s) => ({
|
||||
...s,
|
||||
line: map1(s.line),
|
||||
sites: s.sites?.map(mapSite),
|
||||
})),
|
||||
})),
|
||||
bindings: cfg.bindings?.map((bd) =>
|
||||
bd.declLine > 0 ? { ...bd, declLine: mapLine(bd.declLine) } : bd,
|
||||
bd.declLine > 0 ? { ...bd, declLine: map1(bd.declLine) } : bd,
|
||||
),
|
||||
};
|
||||
}
|
||||
|
|
|
|||
|
|
@ -198,7 +198,6 @@ export function extractNotebookPython(content: string): NotebookPythonExtraction
|
|||
|
||||
const chunks: string[] = [];
|
||||
const segments: NotebookLineSegment[] = [];
|
||||
let extractLine = 0;
|
||||
let searchFrom = 0;
|
||||
|
||||
for (const rawCell of nb.cells) {
|
||||
|
|
@ -223,31 +222,22 @@ export function extractNotebookPython(content: string): NotebookPythonExtraction
|
|||
continue;
|
||||
}
|
||||
|
||||
const text = lines.join('\n');
|
||||
let text = lines.join('\n');
|
||||
if (text.endsWith('\n')) text = text.slice(0, -1);
|
||||
if (text.trim().length === 0) {
|
||||
chunks.push('');
|
||||
const start = extractLine;
|
||||
extractLine += 1;
|
||||
segments.push({
|
||||
extractStartLine: start,
|
||||
extractEndLine: start,
|
||||
jsonStartLine,
|
||||
jsonEndLine,
|
||||
});
|
||||
continue;
|
||||
}
|
||||
|
||||
if (chunks.length > 0) {
|
||||
chunks.push('\n');
|
||||
extractLine += 1;
|
||||
chunks.push('\n\n');
|
||||
}
|
||||
const extractStartLine = extractLine;
|
||||
chunks.push(text);
|
||||
const addedLines = text.split('\n').length;
|
||||
extractLine += addedLines;
|
||||
const assembled = chunks.join('');
|
||||
const extractStartLine = indexToLine(assembled, assembled.length - text.length);
|
||||
const extractEndLine = indexToLine(assembled, assembled.length - 1);
|
||||
segments.push({
|
||||
extractStartLine,
|
||||
extractEndLine: extractLine - 1,
|
||||
extractEndLine,
|
||||
jsonStartLine,
|
||||
jsonEndLine,
|
||||
});
|
||||
|
|
@ -272,14 +262,29 @@ export function mapExtractLine(row: number, segments: readonly NotebookLineSegme
|
|||
return last.jsonEndLine;
|
||||
}
|
||||
|
||||
const extractCache = new Map<
|
||||
string,
|
||||
{ content: string; result: NotebookPythonExtraction | null }
|
||||
>();
|
||||
|
||||
/** Memoize extraction for FTS/CSV (same file, many symbols). */
|
||||
export function extractNotebookPythonCached(
|
||||
filePath: string,
|
||||
content: string,
|
||||
): NotebookPythonExtraction | null {
|
||||
const hit = extractCache.get(filePath);
|
||||
if (hit && hit.content === content) return hit.result;
|
||||
const result = extractNotebookPython(content);
|
||||
extractCache.set(filePath, { content, result });
|
||||
return result;
|
||||
}
|
||||
|
||||
/** Python snippet for a graph span stored in JSON file coordinates. */
|
||||
export function notebookPythonSnippet(
|
||||
fileContent: string,
|
||||
export function notebookPythonSnippetFromExtract(
|
||||
extracted: NotebookPythonExtraction,
|
||||
startLine: number,
|
||||
endLine: number,
|
||||
): string | null {
|
||||
const extracted = extractNotebookPython(fileContent);
|
||||
if (!extracted) return null;
|
||||
const pyLines = extracted.pythonSource.split('\n');
|
||||
const out: string[] = [];
|
||||
for (const seg of extracted.segments) {
|
||||
|
|
@ -293,3 +298,16 @@ export function notebookPythonSnippet(
|
|||
if (out.length === 0) return null;
|
||||
return out.join('\n');
|
||||
}
|
||||
|
||||
export function notebookPythonSnippet(
|
||||
fileContent: string,
|
||||
startLine: number,
|
||||
endLine: number,
|
||||
filePath?: string,
|
||||
): string | null {
|
||||
const extracted = filePath
|
||||
? extractNotebookPythonCached(filePath, fileContent)
|
||||
: extractNotebookPython(fileContent);
|
||||
if (!extracted) return null;
|
||||
return notebookPythonSnippetFromExtract(extracted, startLine, endLine);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@
|
|||
* Pure given the input source text. No I/O, no globals consulted.
|
||||
*/
|
||||
|
||||
import type { Capture, CaptureMatch } from 'gitnexus-shared';
|
||||
import type { Capture, CaptureMatch, Range } from 'gitnexus-shared';
|
||||
import {
|
||||
nodeToCapture,
|
||||
syntheticCapture,
|
||||
|
|
@ -24,7 +24,11 @@ import {
|
|||
} from '../../utils/ast-helpers.js';
|
||||
import { splitImportStatement } from './import-decomposer.js';
|
||||
import { getPythonParser, getPythonScopeQuery } from './query.js';
|
||||
import { extractNotebookPython } from '../../ipynb-extractor.js';
|
||||
import {
|
||||
extractNotebookPython,
|
||||
mapExtractLine,
|
||||
type NotebookLineSegment,
|
||||
} from '../../ipynb-extractor.js';
|
||||
import {
|
||||
synthesizeConstructorFieldTypeBindings,
|
||||
synthesizeReceiverTypeBinding,
|
||||
|
|
@ -64,14 +68,18 @@ export function emitPythonScopeCaptures(
|
|||
): readonly CaptureMatch[] {
|
||||
let parseText = sourceText;
|
||||
let tree = cachedTree as ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined;
|
||||
if (
|
||||
filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb') &&
|
||||
sourceMeta?.sourceKind !== 'pre-extracted-script'
|
||||
) {
|
||||
let notebookSegments: readonly NotebookLineSegment[] | undefined;
|
||||
if (filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb')) {
|
||||
const extracted = extractNotebookPython(sourceText);
|
||||
if (extracted === null) return [];
|
||||
parseText = extracted.pythonSource;
|
||||
tree = undefined;
|
||||
if (extracted === null) {
|
||||
if (sourceMeta?.sourceKind !== 'pre-extracted-script') return [];
|
||||
} else {
|
||||
parseText = extracted.pythonSource;
|
||||
notebookSegments = extracted.segments;
|
||||
if (sourceMeta?.sourceKind !== 'pre-extracted-script') {
|
||||
tree = undefined;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Skip the parse when the caller (the scope-resolution orchestrator's
|
||||
// `treeCache`) already produced a Tree for this source — empty under
|
||||
|
|
@ -238,9 +246,31 @@ export function emitPythonScopeCaptures(
|
|||
out.push(...synthesizePythonInheritanceReferences(tree.rootNode));
|
||||
out.push(...synthesizeCallableFlowCaptures(tree.rootNode, PYTHON_CALLABLE_CAPTURE_OPTIONS));
|
||||
|
||||
if (notebookSegments !== undefined) {
|
||||
return out.map((match) => remapCaptureMatch(match, notebookSegments));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function remapRange(range: Range, segments: readonly NotebookLineSegment[]): Range {
|
||||
return {
|
||||
...range,
|
||||
startLine: mapExtractLine(range.startLine - 1, segments) + 1,
|
||||
endLine: mapExtractLine(range.endLine - 1, segments) + 1,
|
||||
};
|
||||
}
|
||||
|
||||
function remapCaptureMatch(
|
||||
match: CaptureMatch,
|
||||
segments: readonly NotebookLineSegment[],
|
||||
): CaptureMatch {
|
||||
const next: Record<string, Capture> = {};
|
||||
for (const [key, cap] of Object.entries(match)) {
|
||||
next[key] = { ...cap, range: remapRange(cap.range, segments) };
|
||||
}
|
||||
return next;
|
||||
}
|
||||
|
||||
/**
|
||||
* Synthesize `@reference.inherits` captures from Python class superclass
|
||||
* lists so the registry-primary scope-resolution path emits EXTENDS edges
|
||||
|
|
|
|||
|
|
@ -1685,14 +1685,14 @@ const processFileGroup = (
|
|||
let scopeExtractionFailed = false;
|
||||
const parsedFile = extractParsedFile(
|
||||
provider,
|
||||
parseContent,
|
||||
notebookSegments ? file.content : parseContent,
|
||||
file.path,
|
||||
(message) => {
|
||||
scopeExtractionFailed = true;
|
||||
reportWarning(message);
|
||||
},
|
||||
tree,
|
||||
scopeSourceKind,
|
||||
notebookSegments ? 'full-file' : scopeSourceKind,
|
||||
);
|
||||
if (scopeExtractionFailed) (result.scopeExtractionFailures ??= []).push(file.path);
|
||||
if (parsedFile !== undefined) {
|
||||
|
|
|
|||
|
|
@ -23,7 +23,10 @@ import { VALID_NODE_TABLES, parseRelationSchemaPairs, RelPairRouter } from './re
|
|||
import { parseTruthyEnv } from '../ingestion/utils/env.js';
|
||||
import { SYMBOL_NODE_LABELS } from '../ingestion/utils/symbol-labels.js';
|
||||
import { applyCjkSegmentationIfEnabled } from '../search/cjk-segmentation.js';
|
||||
import { notebookPythonSnippet } from '../ingestion/ipynb-extractor.js';
|
||||
import {
|
||||
notebookPythonSnippetFromExtract,
|
||||
extractNotebookPythonCached,
|
||||
} from '../ingestion/ipynb-extractor.js';
|
||||
|
||||
/** Computed once — `RELATION_SCHEMA` is a static template literal. Exported so
|
||||
* the streamed sinks (`GraphEmitSink`, `PdgEmitSink`) share this parse
|
||||
|
|
@ -330,7 +333,10 @@ const extractContent = async (
|
|||
.replace(/\\/g, '/')
|
||||
.toLowerCase();
|
||||
if (notebookPath.endsWith('.ipynb')) {
|
||||
const reconstructed = notebookPythonSnippet(content, startLine, endLine);
|
||||
const extracted = extractNotebookPythonCached(notebookPath, content);
|
||||
const reconstructed = extracted
|
||||
? notebookPythonSnippetFromExtract(extracted, startLine, endLine)
|
||||
: null;
|
||||
if (reconstructed) {
|
||||
const MAX_SNIPPET = 5000;
|
||||
const capped =
|
||||
|
|
|
|||
|
|
@ -77,6 +77,16 @@ describe('Jupyter notebook Python pipeline', () => {
|
|||
n.properties.filePath.replace(/\\/g, '/').endsWith('julia.ipynb') && n.name === 'train',
|
||||
),
|
||||
).toHaveLength(0);
|
||||
expect(
|
||||
getNodesByLabelFull(result, 'File').filter((n) =>
|
||||
n.properties.filePath.replace(/\\/g, '/').endsWith('julia.ipynb'),
|
||||
),
|
||||
).toHaveLength(0);
|
||||
expect(
|
||||
getNodesByLabelFull(result, 'File').filter((n) =>
|
||||
n.properties.filePath.replace(/\\/g, '/').endsWith('broken.ipynb'),
|
||||
),
|
||||
).toHaveLength(0);
|
||||
} finally {
|
||||
fs.rmSync(root, { recursive: true, force: true });
|
||||
}
|
||||
|
|
|
|||
|
|
@ -54,6 +54,7 @@ describe('getLanguageFromFilename', () => {
|
|||
it('detects .ipynb files as Python', () => {
|
||||
expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python);
|
||||
expect(getProviderForFile('notebooks/analysis.ipynb')?.id).toBe(SupportedLanguages.Python);
|
||||
expect(getSyntaxLanguageFromFilename('analysis.ipynb')).toBe('json');
|
||||
});
|
||||
});
|
||||
|
||||
|
|
|
|||
|
|
@ -15,8 +15,7 @@ function notebook(opts: {
|
|||
opts.language === undefined
|
||||
? undefined
|
||||
: { display_name: 'Python', language: opts.language, name: 'python' };
|
||||
const language_info =
|
||||
opts.languageInfo === undefined ? undefined : { name: opts.languageInfo };
|
||||
const language_info = opts.languageInfo === undefined ? undefined : { name: opts.languageInfo };
|
||||
return JSON.stringify(
|
||||
{
|
||||
nbformat: 4,
|
||||
|
|
@ -64,15 +63,21 @@ describe('extractNotebookPython', () => {
|
|||
language: 'python',
|
||||
cells: [
|
||||
{ cell_type: 'code', metadata: {}, source: ['x = 1\n'], outputs: [] },
|
||||
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' return x\n'], outputs: [] },
|
||||
{
|
||||
cell_type: 'code',
|
||||
metadata: {},
|
||||
source: ['def train():\n', ' return x\n'],
|
||||
outputs: [],
|
||||
},
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result).not.toBeNull();
|
||||
expect(result!.pythonSource).toMatch(/x = 1\n+def train/);
|
||||
expect(result!.segments).toHaveLength(2);
|
||||
expect(result!.segments[1].extractStartLine).toBeGreaterThan(result!.segments[0].extractStartLine);
|
||||
expect(result!.segments[1].jsonStartLine).toBeGreaterThan(result!.segments[0].jsonStartLine);
|
||||
const defRow = result!.pythonSource.split('\n').findIndex((l) => l.startsWith('def train'));
|
||||
expect(defRow).toBe(result!.segments[1].extractStartLine);
|
||||
expect(mapExtractLine(defRow, result!.segments)).toBe(result!.segments[1].jsonStartLine);
|
||||
});
|
||||
|
||||
it('accepts source as a single string', () => {
|
||||
|
|
@ -199,7 +204,9 @@ describe('notebookPythonSnippet', () => {
|
|||
it('returns Python def train not JSON cell_type', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] }],
|
||||
cells: [
|
||||
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
|
||||
],
|
||||
});
|
||||
const extracted = extractNotebookPython(content)!;
|
||||
const snippet = notebookPythonSnippet(
|
||||
|
|
|
|||
|
|
@ -37,6 +37,9 @@ describe('Python notebook scope captures', () => {
|
|||
});
|
||||
|
||||
it('extractParsedFile yields a Function for train', () => {
|
||||
const captured = emitPythonScopeCaptures(notebook, 'analysis.ipynb');
|
||||
const fnCapture = captured.find((m) => m['@scope.function'] !== undefined);
|
||||
expect(fnCapture?.['@scope.function']?.range.startLine).toBeGreaterThan(1);
|
||||
const parsed = extractParsedFile(pythonProvider, notebook, 'analysis.ipynb');
|
||||
expect(parsed).toBeDefined();
|
||||
expect(parsed!.localDefs.some((d) => d.qualifiedName === 'train' || d.name === 'train')).toBe(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue