mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-07 02:58:02 +00:00
feat: index Jupyter notebooks as Python (#3381)
This commit is contained in:
parent
51fb64c976
commit
274ec3df6e
19 changed files with 1540 additions and 30 deletions
44
docs/languages/jupyter-notebook.md
Normal file
44
docs/languages/jupyter-notebook.md
Normal file
|
|
@ -0,0 +1,44 @@
|
|||
# Jupyter notebook (.ipynb) indexing
|
||||
|
||||
Status: implemented (Python code cells)
|
||||
|
||||
GitNexus indexes Jupyter notebooks by extracting Python code cells and parsing them with the existing Python language provider. Notebooks are not executed.
|
||||
|
||||
## Goal
|
||||
|
||||
After `analyze`, functions, classes, and imports defined in Python code cells are queryable like ordinary `.py` files.
|
||||
|
||||
## Compatibility
|
||||
|
||||
- `.ipynb` is detected as Python (`gitnexus-shared` `EXTENSION_MAP` and `pythonProvider.extensions`).
|
||||
- Extraction lives in `gitnexus/src/core/ingestion/ipynb-extractor.ts`. Shared ingestion modules do not name nbformat AST types.
|
||||
- Notebooks are **not** Python import targets. A notebook may import `.py` modules; `import some_notebook` does not resolve to an `.ipynb`.
|
||||
- Files over the walker size cap (default 512KB, `GITNEXUS_MAX_FILE_SIZE`) are skipped like any other oversized file. Output-heavy notebooks may need a higher cap. Outputs are not stripped in this slice.
|
||||
- Group-layer FastAPI/Flask/Django scanners that require a `.py` suffix still ignore notebooks.
|
||||
|
||||
## Kernel and magics
|
||||
|
||||
- Skip the file when `kernelspec.language` or `language_info.name` is present and is not a Python-family name (`python`, `python2`, `python3`, `ipython`, `python 3`, `ipython3`). `python` and `python3` together still index.
|
||||
- If `kernelspec.language` is missing, `kernelspec.name` is used only when it is an obvious language id (`python3`, `ir`, `julia-1.8`). Conda env names are ignored.
|
||||
- If `language_info.name` is missing, `language_info.file_extension` (`.py` vs `.r` / `.jl`) is used the same way.
|
||||
- If those fields are absent, code cells are treated as Python unless a cell's own `language`, `metadata.language`, or `metadata.vscode.languageId` says otherwise.
|
||||
- A cell whose first non-empty line is a foreign cell magic (`%%bash`, `%%html`, `%%sql`, `%%R`) is skipped. Python-body cell magics (`%%time`, `%%timeit`, `%%capture`, `%%prun`, `%%debug`, `%%px`, `%%python`) stay, with the magic line commented.
|
||||
- `%run`, `%load`, and `%loadpy` of a local `.py` path become `import module` on that same line so the notebook links to the file. URLs, `..` paths, and `.ipynb` targets stay comments. The notebook is not executed.
|
||||
- Sage, SageMath, MicroPython, Pyodide, PyPy, and PySpark kernels are indexed as Python. SQL, R, and Julia cells are not.
|
||||
- A cell with an unclosed string or bracket is commented out so it cannot hide later cells. Those cells are concatenated in order; one syntax error no longer drops the rest of the notebook.
|
||||
- Line magics (`%`), shell (`!`), and IPython help (`train?`, `?train`) are commented in place so JSON line mapping stays affine.
|
||||
- nbformat v4 `cells` / `source` is the normal path. nbformat v3 `worksheets[].cells` and code-cell `input` are accepted. A leading UTF-8 BOM is accepted. Markdown and raw cells are not code.
|
||||
|
||||
## Line numbers
|
||||
|
||||
Graph `startLine` / `endLine` are 0-based coordinates in the on-disk `.ipynb` JSON. FTS/MCP symbol snippets reconstruct cell Python; they do not slice raw JSON.
|
||||
|
||||
Concatenating cells in document order is notebook semantics. An earlier cell with a syntax error may cause later definitions to be missed; that is a documented limitation, not a per-cell fallback parse.
|
||||
|
||||
## Tests
|
||||
|
||||
- `gitnexus/test/unit/ipynb-extractor.test.ts`
|
||||
- `gitnexus/test/unit/ingestion-utils.test.ts` (`.ipynb` detection)
|
||||
- `gitnexus/test/integration/resolvers/ipynb-python-pipeline.test.ts`
|
||||
|
||||
No Jupyter, nbconvert, or nbformat runtime dependency.
|
||||
|
|
@ -22,6 +22,7 @@ export {
|
|||
getLanguageFromFilename,
|
||||
getSyntaxLanguageFromFilename,
|
||||
isBladeTemplateFilename,
|
||||
isNotebookFilename,
|
||||
} from './language-detection.js';
|
||||
export type { MroStrategy } from './mro-strategy.js';
|
||||
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ const RUBY_EXTENSIONLESS_FILES = new Set([
|
|||
const EXTENSION_MAP: Record<SupportedLanguages, readonly string[]> = {
|
||||
[SupportedLanguages.JavaScript]: ['.js', '.jsx', '.mjs', '.cjs'],
|
||||
[SupportedLanguages.TypeScript]: ['.ts', '.tsx', '.mts', '.cts'],
|
||||
[SupportedLanguages.Python]: ['.py'],
|
||||
[SupportedLanguages.Python]: ['.py', '.ipynb'],
|
||||
[SupportedLanguages.Java]: ['.java'],
|
||||
[SupportedLanguages.C]: ['.c'],
|
||||
[SupportedLanguages.ObjectiveC]: ['.m', '.mm'],
|
||||
|
|
@ -76,6 +76,10 @@ for (const [lang, exts] of Object.entries(EXTENSION_MAP) as [
|
|||
export const isBladeTemplateFilename = (filePath: string): boolean =>
|
||||
filePath.replace(/\\/g, '/').toLowerCase().endsWith('.blade.php');
|
||||
|
||||
/** Jupyter notebooks: ingested as Python; on-disk bytes are JSON. */
|
||||
export const isNotebookFilename = (filePath: string): boolean =>
|
||||
filePath.replace(/\\/g, '/').toLowerCase().endsWith('.ipynb');
|
||||
|
||||
/**
|
||||
* Map file extension to SupportedLanguage enum.
|
||||
* Returns null if the file extension is not recognized.
|
||||
|
|
@ -163,6 +167,7 @@ const AUXILIARY_BASENAME_MAP: Record<string, string> = {
|
|||
*/
|
||||
export const getSyntaxLanguageFromFilename = (filePath: string): string => {
|
||||
if (isBladeTemplateFilename(filePath)) return 'markup';
|
||||
if (isNotebookFilename(filePath)) return 'json';
|
||||
|
||||
const lang = getLanguageFromFilename(filePath);
|
||||
if (lang) return SYNTAX_MAP[lang];
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ import {
|
|||
} from '../tree-sitter/parser-loader.js';
|
||||
import { parseSourceSafe } from '../tree-sitter/safe-parse.js';
|
||||
import { getLanguageForFileContent, getProvider } from '../ingestion/languages/index.js';
|
||||
import { extractNotebookPython, isNotebookPath } from '../ingestion/ipynb-extractor.js';
|
||||
|
||||
const parserCache = new Map<string, any>();
|
||||
|
||||
|
|
@ -38,9 +39,21 @@ export const ensureAndParse = async (content: string, filePath: string): Promise
|
|||
// C++ UE macros, Dart extension types) would leave embeddings looking at an
|
||||
// error-recovered tree. Resolved from `language` so the transform and the
|
||||
// parser always come from the same provider. Length-preserving, so node
|
||||
// offsets still index `content`.
|
||||
const parseContent = getProvider(language).preprocessSource?.(content, filePath) ?? content;
|
||||
// offsets still index `content` except for `.ipynb`, which is replaced by
|
||||
// concatenated code-cell Python (same as the parse worker).
|
||||
const provider = getProvider(language);
|
||||
if (isNotebookPath(filePath)) {
|
||||
const body = content.charCodeAt(0) === 0xfeff ? content.slice(1) : content;
|
||||
if (body.trimStart().startsWith('{')) {
|
||||
const extracted = extractNotebookPython(content);
|
||||
if (!extracted) return null;
|
||||
const parseContent =
|
||||
provider.preprocessSource?.(extracted.pythonSource, filePath) ?? extracted.pythonSource;
|
||||
return parseSourceSafe(parserInstance, parseContent);
|
||||
}
|
||||
}
|
||||
|
||||
const parseContent = provider.preprocessSource?.(content, filePath) ?? content;
|
||||
return parseSourceSafe(parserInstance, parseContent);
|
||||
};
|
||||
|
||||
|
|
|
|||
|
|
@ -15,7 +15,7 @@
|
|||
*/
|
||||
import type { SyntaxNode } from '../utils/ast-helpers.js';
|
||||
import { CfgNestingDepthError } from './cfg-builder.js';
|
||||
import type { CfgVisitor, FunctionCfg } from './types.js';
|
||||
import type { CfgVisitor, FunctionCfg, SiteRecord } from './types.js';
|
||||
|
||||
/**
|
||||
* Default per-function source-line cap used by the worker when the `--pdg` run
|
||||
|
|
@ -83,12 +83,38 @@ function shiftCfgLines(cfg: FunctionCfg, offset: number): FunctionCfg {
|
|||
};
|
||||
}
|
||||
|
||||
function remapCfgLines(cfg: FunctionCfg, mapLine: (row: number) => number): FunctionCfg {
|
||||
// CFG visitors store 1-based source lines; `mapLine` maps 0-based tree-sitter rows.
|
||||
const map1 = (line: number): number => mapLine(line - 1) + 1;
|
||||
const mapSite = (site: SiteRecord): SiteRecord =>
|
||||
site.at !== undefined ? { ...site, at: [map1(site.at[0]), site.at[1]] } : site;
|
||||
return {
|
||||
...cfg,
|
||||
functionStartLine: map1(cfg.functionStartLine),
|
||||
functionEndLine: map1(cfg.functionEndLine),
|
||||
blocks: cfg.blocks.map((b) => ({
|
||||
...b,
|
||||
startLine: map1(b.startLine),
|
||||
endLine: map1(b.endLine),
|
||||
statements: b.statements?.map((s) => ({
|
||||
...s,
|
||||
line: map1(s.line),
|
||||
sites: s.sites?.map(mapSite),
|
||||
})),
|
||||
})),
|
||||
bindings: cfg.bindings?.map((bd) =>
|
||||
bd.declLine > 0 ? { ...bd, declLine: map1(bd.declLine) } : bd,
|
||||
),
|
||||
};
|
||||
}
|
||||
|
||||
export function collectFunctionCfgs(
|
||||
root: SyntaxNode,
|
||||
visitor: CfgVisitor<SyntaxNode>,
|
||||
filePath: string,
|
||||
maxFunctionLines = 0,
|
||||
lineOffset = 0,
|
||||
mapLine?: (row: number) => number,
|
||||
): CollectedCfgs {
|
||||
const cfgs: FunctionCfg[] = [];
|
||||
let tooManyLines = 0;
|
||||
|
|
@ -109,7 +135,9 @@ export function collectFunctionCfgs(
|
|||
// and silently drop every remaining function's CFG (#2195).
|
||||
try {
|
||||
const cfg = visitor.buildFunctionCfg(node, filePath);
|
||||
if (cfg) cfgs.push(shiftCfgLines(cfg, lineOffset));
|
||||
if (cfg) {
|
||||
cfgs.push(mapLine ? remapCfgLines(cfg, mapLine) : shiftCfgLines(cfg, lineOffset));
|
||||
}
|
||||
} catch (err) {
|
||||
if (err instanceof CfgNestingDepthError) tooDeeplyNested++;
|
||||
else buildError++;
|
||||
|
|
|
|||
697
gitnexus/src/core/ingestion/ipynb-extractor.ts
Normal file
697
gitnexus/src/core/ingestion/ipynb-extractor.ts
Normal file
|
|
@ -0,0 +1,697 @@
|
|||
/**
|
||||
* Jupyter notebook (.ipynb) Python extractor.
|
||||
*
|
||||
* Pulls code-cell source from nbformat JSON so the Python tree-sitter
|
||||
* grammar can parse it. Extraction helpers are I/O-free and worker-safe.
|
||||
* `extractNotebookPythonCached` is an optional process-local LRU for FTS.
|
||||
*
|
||||
* Graph coordinates stay 0-based lines in the on-disk JSON file. Extract
|
||||
* buffer lines map through {@link mapExtractLine}.
|
||||
*/
|
||||
|
||||
import { isNotebookFilename } from 'gitnexus-shared';
|
||||
import { buildLineIndex, lineFromOffset } from '../embeddings/line-index.js';
|
||||
|
||||
export interface NotebookLineSegment {
|
||||
readonly extractStartLine: number;
|
||||
readonly extractEndLine: number;
|
||||
readonly jsonStartLine: number;
|
||||
readonly jsonEndLine: number;
|
||||
}
|
||||
|
||||
export interface NotebookPythonExtraction {
|
||||
readonly pythonSource: string;
|
||||
readonly segments: readonly NotebookLineSegment[];
|
||||
}
|
||||
|
||||
const PYTHON_FAMILY = new Set([
|
||||
'python',
|
||||
'python2',
|
||||
'python3',
|
||||
'ipython',
|
||||
'sage',
|
||||
'sagemath',
|
||||
'micropython',
|
||||
'pyodide',
|
||||
'pypy',
|
||||
'pyspark',
|
||||
]);
|
||||
|
||||
export function isNotebookPath(filePath: string): boolean {
|
||||
return isNotebookFilename(filePath);
|
||||
}
|
||||
|
||||
export function isPythonFamilyLanguage(name: string | undefined | null): boolean {
|
||||
if (name === undefined || name === null) return false;
|
||||
const n = name.trim().toLowerCase();
|
||||
if (PYTHON_FAMILY.has(n)) return true;
|
||||
if (n.startsWith('ipython')) return true;
|
||||
return /^python(?:\d|\b)/.test(n);
|
||||
}
|
||||
|
||||
/** Kernel spec names that are a language id, not a conda env label. */
|
||||
const NON_PYTHON_KERNEL =
|
||||
/^(?:julia|ir|r|scala|rust|ruby|javascript|nodejs|node|bash|sh|sql|csharp|fsharp|go|java|kotlin|swift|php|perl|lua|octave|matlab|sas|stata|haskell|clojure|elixir|groovy|powershell|sos|cpp|cxx|dart|typescript|wolfram|scheme|racket|ocaml|fortran|gnuplot|sqlite|mysql|postgresql|tsql)(?:[-_][\w.-]+|\d[\w.-]*)?$/i;
|
||||
|
||||
function kernelNameLanguage(name: string): string | undefined {
|
||||
if (isPythonFamilyLanguage(name) || NON_PYTHON_KERNEL.test(name.trim())) return name;
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function extensionLanguage(ext: string): string | undefined {
|
||||
const e = ext.trim().toLowerCase();
|
||||
if (e === '.py' || e === '.pyi' || e === '.ipy') return 'python';
|
||||
if (
|
||||
e === '.r' ||
|
||||
e === '.jl' ||
|
||||
e === '.scala' ||
|
||||
e === '.js' ||
|
||||
e === '.java' ||
|
||||
e === '.go' ||
|
||||
e === '.rs' ||
|
||||
e === '.rb' ||
|
||||
e === '.php' ||
|
||||
e === '.kt' ||
|
||||
e === '.swift' ||
|
||||
e === '.cs'
|
||||
) {
|
||||
return ext;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function skipWs(content: string, i: number): number {
|
||||
while (i < content.length) {
|
||||
const c = content.charCodeAt(i);
|
||||
if (c === 32 || c === 9 || c === 10 || c === 13) i++;
|
||||
else break;
|
||||
}
|
||||
return i;
|
||||
}
|
||||
|
||||
/** Span of a JSON string or array value starting at the first `"` or `[`. */
|
||||
function jsonValueSpan(content: string, start: number): { start: number; end: number } | null {
|
||||
const i = skipWs(content, start);
|
||||
if (i >= content.length) return null;
|
||||
if (content[i] === '"') {
|
||||
let j = i + 1;
|
||||
while (j < content.length) {
|
||||
if (content[j] === '\\') {
|
||||
j += 2;
|
||||
continue;
|
||||
}
|
||||
if (content[j] === '"') return { start: i, end: j + 1 };
|
||||
j++;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
if (content[i] === '[') {
|
||||
let depth = 1;
|
||||
let j = i + 1;
|
||||
let inStr = false;
|
||||
while (j < content.length && depth > 0) {
|
||||
const ch = content[j];
|
||||
if (inStr) {
|
||||
if (ch === '\\') j += 2;
|
||||
else {
|
||||
if (ch === '"') inStr = false;
|
||||
j++;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (ch === '"') inStr = true;
|
||||
else if (ch === '[') depth++;
|
||||
else if (ch === ']') depth--;
|
||||
j++;
|
||||
}
|
||||
return { start: i, end: j };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function jsonBraceSpan(
|
||||
content: string,
|
||||
start: number,
|
||||
open: '{' | '[',
|
||||
close: '}' | ']',
|
||||
): { start: number; end: number } | null {
|
||||
const i = skipWs(content, start);
|
||||
if (i >= content.length || content[i] !== open) return null;
|
||||
let depth = 1;
|
||||
let j = i + 1;
|
||||
let inStr = false;
|
||||
while (j < content.length && depth > 0) {
|
||||
const ch = content[j];
|
||||
if (inStr) {
|
||||
if (ch === '\\') j += 2;
|
||||
else {
|
||||
if (ch === '"') inStr = false;
|
||||
j++;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (ch === '"') inStr = true;
|
||||
else if (ch === open) depth++;
|
||||
else if (ch === close) depth--;
|
||||
j++;
|
||||
}
|
||||
if (depth !== 0) return null;
|
||||
return { start: i, end: j };
|
||||
}
|
||||
|
||||
function findDepth1Key(
|
||||
content: string,
|
||||
objStart: number,
|
||||
objEnd: number,
|
||||
key: string,
|
||||
match: 'first' | 'last' = 'first',
|
||||
): number {
|
||||
let depth = 0;
|
||||
let inStr = false;
|
||||
let j = objStart;
|
||||
let found = -1;
|
||||
while (j < objEnd) {
|
||||
const ch = content[j];
|
||||
if (inStr) {
|
||||
if (ch === '\\') j += 2;
|
||||
else {
|
||||
if (ch === '"') inStr = false;
|
||||
j++;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (ch === '"') {
|
||||
if (depth === 1 && content.startsWith(key, j)) {
|
||||
if (match === 'first') return j;
|
||||
found = j;
|
||||
}
|
||||
inStr = true;
|
||||
j++;
|
||||
continue;
|
||||
}
|
||||
if (ch === '{' || ch === '[') depth++;
|
||||
else if (ch === '}' || ch === ']') depth--;
|
||||
j++;
|
||||
}
|
||||
return found;
|
||||
}
|
||||
|
||||
function arraySpanAfterKey(
|
||||
content: string,
|
||||
objStart: number,
|
||||
objEnd: number,
|
||||
key: '"cells"' | '"worksheets"',
|
||||
match: 'first' | 'last',
|
||||
): { start: number; end: number } | null {
|
||||
const keyAt = findDepth1Key(content, objStart, objEnd, key, match);
|
||||
if (keyAt < 0) return null;
|
||||
const colon = content.indexOf(':', keyAt + key.length);
|
||||
if (colon < 0 || colon >= objEnd) return null;
|
||||
return jsonBraceSpan(content, colon + 1, '[', ']');
|
||||
}
|
||||
|
||||
function codeCellArraySpans(content: string, origin: number): { start: number; end: number }[] {
|
||||
const root = jsonBraceSpan(content, origin, '{', '}');
|
||||
if (!root) return [];
|
||||
const cells = arraySpanAfterKey(content, root.start, root.end, '"cells"', 'last');
|
||||
if (cells) return [cells];
|
||||
const worksheets = arraySpanAfterKey(content, root.start, root.end, '"worksheets"', 'last');
|
||||
if (!worksheets) return [];
|
||||
const spans: { start: number; end: number }[] = [];
|
||||
let search = worksheets.start + 1;
|
||||
while (search < worksheets.end) {
|
||||
const brace = nextUnquotedChar(content, search, worksheets.end, '{');
|
||||
if (brace < 0) break;
|
||||
const obj = jsonBraceSpan(content, brace, '{', '}');
|
||||
if (!obj || obj.end > worksheets.end) {
|
||||
search = brace + 1;
|
||||
continue;
|
||||
}
|
||||
const nested = arraySpanAfterKey(content, obj.start, obj.end, '"cells"', 'first');
|
||||
if (nested) spans.push(nested);
|
||||
search = obj.end;
|
||||
}
|
||||
return spans;
|
||||
}
|
||||
|
||||
function readJsonString(content: string, start: number): string | null {
|
||||
const i = skipWs(content, start);
|
||||
if (content[i] !== '"') return null;
|
||||
let j = i + 1;
|
||||
let out = '';
|
||||
while (j < content.length) {
|
||||
const ch = content[j];
|
||||
if (ch === '\\') {
|
||||
out += content[j + 1] ?? '';
|
||||
j += 2;
|
||||
continue;
|
||||
}
|
||||
if (ch === '"') return out;
|
||||
out += ch;
|
||||
j++;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function nextUnquotedChar(content: string, from: number, until: number, needle: '{' | '"'): number {
|
||||
let inStr = false;
|
||||
let j = from;
|
||||
while (j < until) {
|
||||
const ch = content[j];
|
||||
if (inStr) {
|
||||
if (ch === '\\') j += 2;
|
||||
else {
|
||||
if (ch === '"') inStr = false;
|
||||
j++;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (ch === '"') {
|
||||
if (needle === '"') return j;
|
||||
inStr = true;
|
||||
j++;
|
||||
continue;
|
||||
}
|
||||
if (ch === needle) return j;
|
||||
j++;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
function firstQuoteInValue(content: string, span: { start: number; end: number }): number {
|
||||
const i = skipWs(content, span.start);
|
||||
if (i < span.end && content[i] === '"') return i;
|
||||
if (i < span.end && content[i] === '[') {
|
||||
const q = nextUnquotedChar(content, i + 1, span.end, '"');
|
||||
if (q >= 0) return q;
|
||||
}
|
||||
return span.start;
|
||||
}
|
||||
|
||||
function findNextCodeCellSourceSpan(
|
||||
content: string,
|
||||
from: number,
|
||||
cells: { start: number; end: number },
|
||||
): { span: { start: number; end: number } | null; nextFrom: number } | null {
|
||||
let search = Math.max(from, cells.start + 1);
|
||||
while (search < cells.end) {
|
||||
const brace = nextUnquotedChar(content, search, cells.end, '{');
|
||||
if (brace < 0) return null;
|
||||
const obj = jsonBraceSpan(content, brace, '{', '}');
|
||||
if (!obj || obj.end > cells.end) {
|
||||
search = brace + 1;
|
||||
continue;
|
||||
}
|
||||
const typeKey = findDepth1Key(content, obj.start, obj.end, '"cell_type"');
|
||||
if (typeKey < 0) {
|
||||
search = obj.end;
|
||||
continue;
|
||||
}
|
||||
const colon = content.indexOf(':', typeKey + 11);
|
||||
if (colon < 0 || colon >= obj.end) {
|
||||
search = obj.end;
|
||||
continue;
|
||||
}
|
||||
const valueStart = skipWs(content, colon + 1);
|
||||
const cellType = readJsonString(content, valueStart);
|
||||
if (cellType === null || cellType.toLowerCase() !== 'code') {
|
||||
search = obj.end;
|
||||
continue;
|
||||
}
|
||||
let sourceKey = findDepth1Key(content, obj.start, obj.end, '"source"');
|
||||
let keyLen = 8;
|
||||
if (sourceKey < 0) {
|
||||
sourceKey = findDepth1Key(content, obj.start, obj.end, '"input"');
|
||||
keyLen = 7;
|
||||
}
|
||||
if (sourceKey < 0) {
|
||||
return { span: null, nextFrom: obj.end };
|
||||
}
|
||||
const srcColon = content.indexOf(':', sourceKey + keyLen);
|
||||
if (srcColon < 0 || srcColon >= obj.end) {
|
||||
return { span: null, nextFrom: obj.end };
|
||||
}
|
||||
const span = jsonValueSpan(content, srcColon + 1);
|
||||
if (!span) {
|
||||
return { span: null, nextFrom: obj.end };
|
||||
}
|
||||
return { span, nextFrom: obj.end };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function flattenSource(source: unknown): string {
|
||||
if (typeof source === 'string') return source;
|
||||
if (Array.isArray(source)) {
|
||||
return source.map((part) => (typeof part === 'string' ? part : '')).join('');
|
||||
}
|
||||
return '';
|
||||
}
|
||||
|
||||
function cellLanguage(cell: Record<string, unknown>): string | undefined {
|
||||
if (typeof cell.language === 'string') return cell.language;
|
||||
const meta = cell.metadata;
|
||||
if (!meta || typeof meta !== 'object') return undefined;
|
||||
const m = meta as Record<string, unknown>;
|
||||
if (typeof m.language === 'string') return m.language;
|
||||
const vscode = m.vscode;
|
||||
if (vscode && typeof vscode === 'object') {
|
||||
const langId = (vscode as Record<string, unknown>).languageId;
|
||||
if (typeof langId === 'string') return langId;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function notebookLanguageFields(nb: Record<string, unknown>): {
|
||||
kernelspec?: string;
|
||||
languageInfo?: string;
|
||||
} {
|
||||
const metadata = nb.metadata;
|
||||
if (!metadata || typeof metadata !== 'object') return {};
|
||||
const md = metadata as Record<string, unknown>;
|
||||
const ks = md.kernelspec;
|
||||
const li = md.language_info;
|
||||
return {
|
||||
kernelspec:
|
||||
ks && typeof ks === 'object'
|
||||
? typeof (ks as Record<string, unknown>).language === 'string'
|
||||
? String((ks as Record<string, unknown>).language)
|
||||
: kernelNameLanguage(String((ks as Record<string, unknown>).name ?? ''))
|
||||
: undefined,
|
||||
languageInfo:
|
||||
li && typeof li === 'object'
|
||||
? typeof (li as Record<string, unknown>).name === 'string'
|
||||
? String((li as Record<string, unknown>).name)
|
||||
: extensionLanguage(String((li as Record<string, unknown>).file_extension ?? ''))
|
||||
: undefined,
|
||||
};
|
||||
}
|
||||
|
||||
function kernelShouldSkip(nb: Record<string, unknown>): boolean {
|
||||
const { kernelspec, languageInfo } = notebookLanguageFields(nb);
|
||||
const fields = [kernelspec, languageInfo].filter((x): x is string => x !== undefined);
|
||||
if (fields.length === 0) return false;
|
||||
return fields.some((f) => !isPythonFamilyLanguage(f));
|
||||
}
|
||||
|
||||
/** Cell magics whose body is still Python. Foreign %% cells are skipped. */
|
||||
const PYTHON_CELL_MAGICS = new Set([
|
||||
'time',
|
||||
'timeit',
|
||||
'capture',
|
||||
'prun',
|
||||
'lprun',
|
||||
'mprun',
|
||||
'debug',
|
||||
'px',
|
||||
'python',
|
||||
'python2',
|
||||
'python3',
|
||||
'ipython',
|
||||
'pypy',
|
||||
]);
|
||||
|
||||
function cellMagicName(line: string): string {
|
||||
const token = line.trimStart().slice(2).trim().split(/\s+/)[0] ?? '';
|
||||
return token.toLowerCase();
|
||||
}
|
||||
|
||||
function isIpythonHelpLine(line: string): boolean {
|
||||
const t = line.trim();
|
||||
if (t.length === 0 || t.startsWith('#')) return false;
|
||||
return /^\?\??\S/.test(t) || /^[A-Za-z_][\w.]*(?:\?\?|\?)\s*$/.test(t);
|
||||
}
|
||||
|
||||
function moduleSpecFromRunPath(raw: string): string | null {
|
||||
let path = raw
|
||||
.trim()
|
||||
.replace(/^['"]|['"]$/g, '')
|
||||
.replace(/\\/g, '/');
|
||||
if (path.startsWith('http://') || path.startsWith('https://') || path.endsWith('.ipynb')) {
|
||||
return null;
|
||||
}
|
||||
path = path.replace(/^\.\//, '');
|
||||
if (path.endsWith('.py')) path = path.slice(0, -3);
|
||||
const parts = path.split('/').filter((part) => part.length > 0);
|
||||
if (parts.length === 0 || parts.some((part) => part === '..' || !/^[A-Za-z_]\w*$/.test(part))) {
|
||||
return null;
|
||||
}
|
||||
return parts.join('.');
|
||||
}
|
||||
|
||||
function rewriteRunOrLoad(line: string): string | null {
|
||||
const trimmed = line.trimStart();
|
||||
const indent = line.slice(0, line.length - trimmed.length);
|
||||
const match = trimmed.match(/^%(?:run|load|loadpy)\s+(\S+)/);
|
||||
if (!match) return null;
|
||||
const spec = moduleSpecFromRunPath(match[1]);
|
||||
if (!spec) return null;
|
||||
return `${indent}import ${spec} # ${trimmed}`;
|
||||
}
|
||||
|
||||
function isBalancedPython(source: string): boolean {
|
||||
let quote: "'" | '"' | null = null;
|
||||
let triple = false;
|
||||
let paren = 0;
|
||||
let bracket = 0;
|
||||
let brace = 0;
|
||||
for (let i = 0; i < source.length; i++) {
|
||||
const ch = source[i];
|
||||
if (quote) {
|
||||
if (ch === '\\' && !triple) {
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
if (triple && source.startsWith(quote.repeat(3), i)) {
|
||||
quote = null;
|
||||
triple = false;
|
||||
i += 2;
|
||||
continue;
|
||||
}
|
||||
if (!triple && ch === quote) quote = null;
|
||||
continue;
|
||||
}
|
||||
if (ch === '#') {
|
||||
const nl = source.indexOf('\n', i);
|
||||
if (nl < 0) break;
|
||||
i = nl;
|
||||
continue;
|
||||
}
|
||||
if ((ch === '"' || ch === "'") && source.startsWith(ch.repeat(3), i)) {
|
||||
quote = ch;
|
||||
triple = true;
|
||||
i += 2;
|
||||
continue;
|
||||
}
|
||||
if (ch === '"' || ch === "'") {
|
||||
quote = ch;
|
||||
continue;
|
||||
}
|
||||
if (ch === '(') paren++;
|
||||
else if (ch === ')') paren--;
|
||||
else if (ch === '[') bracket++;
|
||||
else if (ch === ']') bracket--;
|
||||
else if (ch === '{') brace++;
|
||||
else if (ch === '}') brace--;
|
||||
if (paren < 0 || bracket < 0 || brace < 0) return false;
|
||||
}
|
||||
return quote === null && paren === 0 && bracket === 0 && brace === 0;
|
||||
}
|
||||
|
||||
function processCellLines(raw: string): { skipCell: boolean; lines: string[] } {
|
||||
const lines = raw.replace(/\r\n/g, '\n').replace(/\r/g, '\n').split('\n');
|
||||
const firstNonEmpty = lines.find((l) => l.trim().length > 0);
|
||||
const magic =
|
||||
firstNonEmpty !== undefined && firstNonEmpty.trimStart().startsWith('%%')
|
||||
? cellMagicName(firstNonEmpty)
|
||||
: undefined;
|
||||
if (magic !== undefined && !PYTHON_CELL_MAGICS.has(magic) && !isPythonFamilyLanguage(magic)) {
|
||||
return { skipCell: true, lines: [''] };
|
||||
}
|
||||
const rewritten = lines.map((line) => {
|
||||
const t = line.trimStart();
|
||||
if (isIpythonHelpLine(line) || t.startsWith('!')) return `# ${line}`;
|
||||
const runImport = rewriteRunOrLoad(line);
|
||||
if (runImport) return runImport;
|
||||
if (t.startsWith('%')) return `# ${line}`;
|
||||
return line;
|
||||
});
|
||||
if (isBalancedPython(rewritten.join('\n'))) {
|
||||
return { skipCell: false, lines: rewritten };
|
||||
}
|
||||
return {
|
||||
skipCell: false,
|
||||
lines: rewritten.map((line) => (line.trim().length === 0 ? line : `# ${line}`)),
|
||||
};
|
||||
}
|
||||
|
||||
function notebookCodeCells(nb: Record<string, unknown>): unknown[] | null {
|
||||
if (Array.isArray(nb.cells)) return nb.cells;
|
||||
if (!Array.isArray(nb.worksheets)) return null;
|
||||
const cells: unknown[] = [];
|
||||
for (const worksheet of nb.worksheets) {
|
||||
if (!worksheet || typeof worksheet !== 'object') continue;
|
||||
const nested = (worksheet as Record<string, unknown>).cells;
|
||||
if (Array.isArray(nested)) cells.push(...nested);
|
||||
}
|
||||
return cells.length > 0 ? cells : null;
|
||||
}
|
||||
|
||||
function cellSource(cell: Record<string, unknown>): unknown {
|
||||
return cell.source !== undefined ? cell.source : cell.input;
|
||||
}
|
||||
|
||||
export function extractNotebookPython(content: string): NotebookPythonExtraction | null {
|
||||
const origin = content.charCodeAt(0) === 0xfeff ? 1 : 0;
|
||||
let parsed: unknown;
|
||||
try {
|
||||
parsed = JSON.parse(origin === 0 ? content : content.slice(origin));
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
if (!parsed || typeof parsed !== 'object') return null;
|
||||
const nb = parsed as Record<string, unknown>;
|
||||
const codeCells = notebookCodeCells(nb);
|
||||
if (!codeCells) return null;
|
||||
if (kernelShouldSkip(nb)) return null;
|
||||
|
||||
const spans = codeCellArraySpans(content, origin);
|
||||
if (spans.length === 0) return null;
|
||||
|
||||
const lineStarts = buildLineIndex(content);
|
||||
const chunks: string[] = [];
|
||||
const segments: NotebookLineSegment[] = [];
|
||||
let spanIndex = 0;
|
||||
let searchFrom = 0;
|
||||
|
||||
for (const rawCell of codeCells) {
|
||||
if (!rawCell || typeof rawCell !== 'object') continue;
|
||||
const cell = rawCell as Record<string, unknown>;
|
||||
if (String(cell.cell_type).toLowerCase() !== 'code') continue;
|
||||
|
||||
let located: { span: { start: number; end: number } | null; nextFrom: number } | null = null;
|
||||
while (spanIndex < spans.length) {
|
||||
located = findNextCodeCellSourceSpan(content, searchFrom, spans[spanIndex]);
|
||||
if (located) break;
|
||||
spanIndex++;
|
||||
searchFrom = 0;
|
||||
}
|
||||
if (!located) return null;
|
||||
searchFrom = located.nextFrom;
|
||||
if (!located.span) continue;
|
||||
|
||||
const lang = cellLanguage(cell);
|
||||
if (lang !== undefined && !isPythonFamilyLanguage(lang)) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const { skipCell, lines } = processCellLines(flattenSource(cellSource(cell)));
|
||||
const jsonStartLine = lineFromOffset(lineStarts, firstQuoteInValue(content, located.span));
|
||||
const jsonEndLine = Math.max(jsonStartLine, lineFromOffset(lineStarts, located.span.end - 1));
|
||||
|
||||
if (skipCell) {
|
||||
continue;
|
||||
}
|
||||
|
||||
let text = lines.join('\n');
|
||||
const hadTrailingNewline = text.endsWith('\n');
|
||||
if (hadTrailingNewline) text = text.slice(0, -1);
|
||||
if (text.trim().length === 0) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (chunks.length > 0) {
|
||||
chunks.push('\n\n');
|
||||
}
|
||||
const lineCount = hadTrailingNewline ? Math.max(1, lines.length - 1) : lines.length;
|
||||
const extractStartLine =
|
||||
segments.length === 0 ? 0 : segments[segments.length - 1].extractEndLine + 2;
|
||||
chunks.push(text);
|
||||
segments.push({
|
||||
extractStartLine,
|
||||
extractEndLine: extractStartLine + lineCount - 1,
|
||||
jsonStartLine,
|
||||
jsonEndLine,
|
||||
});
|
||||
}
|
||||
|
||||
const pythonSource = chunks.join('');
|
||||
if (pythonSource.trim().length === 0) return null;
|
||||
return { pythonSource, segments };
|
||||
}
|
||||
|
||||
export function mapExtractLine(row: number, segments: readonly NotebookLineSegment[]): number {
|
||||
if (segments.length === 0) return row;
|
||||
for (const seg of segments) {
|
||||
if (row >= seg.extractStartLine && row <= seg.extractEndLine) {
|
||||
const delta = row - seg.extractStartLine;
|
||||
const jsonSpan = seg.jsonEndLine - seg.jsonStartLine;
|
||||
return seg.jsonStartLine + Math.min(delta, jsonSpan);
|
||||
}
|
||||
}
|
||||
if (row < segments[0].extractStartLine) return segments[0].jsonStartLine;
|
||||
const last = segments[segments.length - 1];
|
||||
return last.jsonEndLine;
|
||||
}
|
||||
|
||||
const extractCache = new Map<
|
||||
string,
|
||||
{ content: string; result: NotebookPythonExtraction | null }
|
||||
>();
|
||||
|
||||
/** Memoize extraction for FTS/CSV (same file, many symbols). */
|
||||
const EXTRACT_CACHE_LIMIT = 32;
|
||||
|
||||
export function extractNotebookPythonCached(
|
||||
filePath: string,
|
||||
content: string,
|
||||
): NotebookPythonExtraction | null {
|
||||
const hit = extractCache.get(filePath);
|
||||
if (hit && hit.content === content) {
|
||||
extractCache.delete(filePath);
|
||||
extractCache.set(filePath, hit);
|
||||
return hit.result;
|
||||
}
|
||||
const result = extractNotebookPython(content);
|
||||
if (extractCache.size >= EXTRACT_CACHE_LIMIT && !extractCache.has(filePath)) {
|
||||
const oldest = extractCache.keys().next().value;
|
||||
if (oldest !== undefined) extractCache.delete(oldest);
|
||||
}
|
||||
extractCache.set(filePath, { content, result });
|
||||
return result;
|
||||
}
|
||||
|
||||
/** Python snippet for a graph span stored in JSON file coordinates. */
|
||||
export function notebookPythonSnippetFromExtract(
|
||||
extracted: NotebookPythonExtraction,
|
||||
startLine: number,
|
||||
endLine: number,
|
||||
): string | null {
|
||||
const pyLines = extracted.pythonSource.split('\n');
|
||||
const out: string[] = [];
|
||||
for (const seg of extracted.segments) {
|
||||
if (seg.jsonEndLine < startLine || seg.jsonStartLine > endLine) continue;
|
||||
for (let extract = seg.extractStartLine; extract <= seg.extractEndLine; extract++) {
|
||||
const jsonSpan = seg.jsonEndLine - seg.jsonStartLine;
|
||||
const jsonLine = seg.jsonStartLine + Math.min(extract - seg.extractStartLine, jsonSpan);
|
||||
if (jsonLine >= startLine && jsonLine <= endLine) {
|
||||
out.push(pyLines[extract] ?? '');
|
||||
}
|
||||
}
|
||||
}
|
||||
if (out.length === 0) return null;
|
||||
return out.join('\n');
|
||||
}
|
||||
|
||||
export function notebookPythonSnippet(
|
||||
fileContent: string,
|
||||
startLine: number,
|
||||
endLine: number,
|
||||
filePath?: string,
|
||||
): string | null {
|
||||
const extracted = filePath
|
||||
? extractNotebookPythonCached(filePath, fileContent)
|
||||
: extractNotebookPython(fileContent);
|
||||
if (!extracted) return null;
|
||||
return notebookPythonSnippetFromExtract(extracted, startLine, endLine);
|
||||
}
|
||||
|
|
@ -26,6 +26,7 @@ import type {
|
|||
WorkspaceIndex,
|
||||
} from 'gitnexus-shared';
|
||||
import type { LanguageTypeConfig } from './type-extractors/types.js';
|
||||
import type { NotebookLineSegment } from './ipynb-extractor.js';
|
||||
import type { CallRouter } from './call-routing.js';
|
||||
import type { CallExtractor } from './call-types.js';
|
||||
import type { ClassExtractor } from './class-types.js';
|
||||
|
|
@ -828,6 +829,8 @@ interface LanguageProviderConfig {
|
|||
*/
|
||||
sourceMeta?: {
|
||||
readonly sourceKind?: 'full-file' | 'pre-extracted-script';
|
||||
/** Python `.ipynb` only: JSON line segments for the pre-extracted buffer. */
|
||||
readonly notebookSegments?: readonly NotebookLineSegment[];
|
||||
},
|
||||
) => readonly CaptureMatch[];
|
||||
|
||||
|
|
|
|||
|
|
@ -107,7 +107,7 @@ function normalizePythonStringLiteral(text: string): string | undefined {
|
|||
|
||||
export const pythonProvider = defineLanguage({
|
||||
id: SupportedLanguages.Python,
|
||||
extensions: ['.py'],
|
||||
extensions: ['.py', '.ipynb'],
|
||||
entryPointPatterns: [/^app$/, /^(get|post|put|delete|patch)_/i, /^api_/, /^view_/],
|
||||
astFrameworkPatterns: [
|
||||
{
|
||||
|
|
|
|||
|
|
@ -12,10 +12,10 @@
|
|||
* binding, and `__init__` assignments from annotated parameters emit
|
||||
* class-scoped instance-field bindings (see `receiver-binding.ts`).
|
||||
*
|
||||
* Pure given the input source text. No I/O, no globals consulted.
|
||||
* No I/O. A `.ipynb` path also depends on `filePath` and `sourceMeta`, not only the source text.
|
||||
*/
|
||||
|
||||
import type { Capture, CaptureMatch } from 'gitnexus-shared';
|
||||
import type { Capture, CaptureMatch, Range } from 'gitnexus-shared';
|
||||
import {
|
||||
nodeToCapture,
|
||||
syntheticCapture,
|
||||
|
|
@ -24,6 +24,12 @@ import {
|
|||
} from '../../utils/ast-helpers.js';
|
||||
import { splitImportStatement } from './import-decomposer.js';
|
||||
import { getPythonParser, getPythonScopeQuery } from './query.js';
|
||||
import {
|
||||
extractNotebookPython,
|
||||
isNotebookPath,
|
||||
mapExtractLine,
|
||||
type NotebookLineSegment,
|
||||
} from '../../ipynb-extractor.js';
|
||||
import {
|
||||
synthesizeConstructorFieldTypeBindings,
|
||||
synthesizeReceiverTypeBinding,
|
||||
|
|
@ -71,23 +77,36 @@ const PYTHON_CALLABLE_CAPTURE_OPTIONS = {
|
|||
|
||||
export function emitPythonScopeCaptures(
|
||||
sourceText: string,
|
||||
_filePath: string,
|
||||
filePath: string,
|
||||
cachedTree?: unknown,
|
||||
sourceMeta?: {
|
||||
sourceKind?: 'full-file' | 'pre-extracted-script';
|
||||
notebookSegments?: readonly NotebookLineSegment[];
|
||||
},
|
||||
): readonly CaptureMatch[] {
|
||||
let parseText = sourceText;
|
||||
let tree = cachedTree as ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined;
|
||||
let notebookSegments: readonly NotebookLineSegment[] | undefined;
|
||||
if (isNotebookPath(filePath)) {
|
||||
const resolved = resolveNotebookCaptureSource(sourceText, tree, sourceMeta);
|
||||
if (resolved === null) return [];
|
||||
parseText = resolved.parseText;
|
||||
tree = resolved.tree;
|
||||
notebookSegments = resolved.notebookSegments;
|
||||
}
|
||||
// Skip the parse when the caller (the scope-resolution orchestrator's
|
||||
// `treeCache`) already produced a Tree for this source — empty under
|
||||
// worker-pool runs, so cache miss = re-parse. The cachedTree parameter
|
||||
// is typed as `unknown` at the
|
||||
// contract layer (see `LanguageProvider.emitScopeCaptures`); cast
|
||||
// here at the use site.
|
||||
let tree = cachedTree as ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined;
|
||||
if (tree === undefined) {
|
||||
try {
|
||||
tree = parseSourceSafe(getPythonParser(), sourceText, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(sourceText),
|
||||
tree = parseSourceSafe(getPythonParser(), parseText, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(parseText),
|
||||
});
|
||||
} catch (err) {
|
||||
throw scopeExtractionError('parse', _filePath, err);
|
||||
throw scopeExtractionError('parse', filePath, err);
|
||||
}
|
||||
recordCacheMiss();
|
||||
} else {
|
||||
|
|
@ -98,7 +117,7 @@ export function emitPythonScopeCaptures(
|
|||
try {
|
||||
rawMatches = getPythonScopeQuery().matches(tree.rootNode);
|
||||
} catch (err) {
|
||||
throw scopeExtractionError('scope query', _filePath, err);
|
||||
throw scopeExtractionError('scope query', filePath, err);
|
||||
}
|
||||
|
||||
const out: CaptureMatch[] = [];
|
||||
|
|
@ -240,9 +259,64 @@ export function emitPythonScopeCaptures(
|
|||
out.push(...synthesizePythonInheritanceReferences(tree.rootNode));
|
||||
out.push(...synthesizeCallableFlowCaptures(tree.rootNode, PYTHON_CALLABLE_CAPTURE_OPTIONS));
|
||||
|
||||
if (notebookSegments !== undefined) {
|
||||
return out.map((match) => remapCaptureMatch(match, notebookSegments));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
function resolveNotebookCaptureSource(
|
||||
sourceText: string,
|
||||
cachedTree: ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined,
|
||||
sourceMeta?: {
|
||||
sourceKind?: 'full-file' | 'pre-extracted-script';
|
||||
notebookSegments?: readonly NotebookLineSegment[];
|
||||
},
|
||||
): {
|
||||
parseText: string;
|
||||
tree: ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined;
|
||||
notebookSegments?: readonly NotebookLineSegment[];
|
||||
} | null {
|
||||
if (sourceMeta?.notebookSegments) {
|
||||
return {
|
||||
parseText: sourceText,
|
||||
tree: cachedTree,
|
||||
notebookSegments: sourceMeta.notebookSegments,
|
||||
};
|
||||
}
|
||||
const extracted = extractNotebookPython(sourceText);
|
||||
if (extracted === null) {
|
||||
if (sourceMeta?.sourceKind === 'pre-extracted-script') {
|
||||
return { parseText: sourceText, tree: cachedTree };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
return {
|
||||
parseText: extracted.pythonSource,
|
||||
tree: sourceMeta?.sourceKind === 'pre-extracted-script' ? cachedTree : undefined,
|
||||
notebookSegments: extracted.segments,
|
||||
};
|
||||
}
|
||||
|
||||
function remapRange(range: Range, segments: readonly NotebookLineSegment[]): Range {
|
||||
return {
|
||||
...range,
|
||||
startLine: mapExtractLine(range.startLine - 1, segments) + 1,
|
||||
endLine: mapExtractLine(range.endLine - 1, segments) + 1,
|
||||
};
|
||||
}
|
||||
|
||||
function remapCaptureMatch(
|
||||
match: CaptureMatch,
|
||||
segments: readonly NotebookLineSegment[],
|
||||
): CaptureMatch {
|
||||
const next: Record<string, Capture> = {};
|
||||
for (const [key, cap] of Object.entries(match)) {
|
||||
next[key] = { ...cap, range: remapRange(cap.range, segments) };
|
||||
}
|
||||
return next;
|
||||
}
|
||||
|
||||
/**
|
||||
* Synthesize `@reference.inherits` captures from Python class superclass
|
||||
* lists so the registry-primary scope-resolution path emits EXTENDS edges
|
||||
|
|
|
|||
|
|
@ -27,6 +27,7 @@
|
|||
import type { ParsedFile } from 'gitnexus-shared';
|
||||
import { extract as extractScope } from './scope-extractor.js';
|
||||
import type { LanguageProvider } from './language-provider.js';
|
||||
import type { NotebookLineSegment } from './ipynb-extractor.js';
|
||||
|
||||
import { logger } from '../logger.js';
|
||||
/** Callback used to report scope-extraction warnings to the host (worker or direct). */
|
||||
|
|
@ -45,6 +46,7 @@ export function extractParsedFile(
|
|||
onWarn?: ScopeBridgeWarn,
|
||||
cachedTree?: unknown,
|
||||
sourceKind: ScopeCaptureSourceKind = 'full-file',
|
||||
notebookSegments?: readonly NotebookLineSegment[],
|
||||
): ParsedFile | undefined {
|
||||
if (provider.emitScopeCaptures === undefined) return undefined;
|
||||
if (sourceText.trim().length === 0) return undefined;
|
||||
|
|
@ -58,7 +60,10 @@ export function extractParsedFile(
|
|||
cachedTree === undefined
|
||||
? (provider.preprocessSource?.(sourceText, filePath) ?? sourceText)
|
||||
: sourceText;
|
||||
const captures = provider.emitScopeCaptures(parseText, filePath, cachedTree, { sourceKind });
|
||||
const captures = provider.emitScopeCaptures(parseText, filePath, cachedTree, {
|
||||
sourceKind,
|
||||
...(notebookSegments ? { notebookSegments } : {}),
|
||||
});
|
||||
return extractScope(captures, filePath, provider);
|
||||
} catch (err) {
|
||||
const message = `scope extraction failed for ${filePath}: ${
|
||||
|
|
|
|||
|
|
@ -135,6 +135,12 @@ import {
|
|||
extractTemplateComponents,
|
||||
isVueSetupTopLevel,
|
||||
} from '../vue-sfc-extractor.js';
|
||||
import {
|
||||
extractNotebookPython,
|
||||
isNotebookPath,
|
||||
mapExtractLine,
|
||||
type NotebookLineSegment,
|
||||
} from '../ipynb-extractor.js';
|
||||
import type { NodeLabel, ParameterTypeClass } from 'gitnexus-shared';
|
||||
import type { FieldInfo, FieldExtractorContext } from '../field-types.js';
|
||||
import type { MethodInfo, MethodExtractorContext } from '../method-types.js';
|
||||
|
|
@ -1597,6 +1603,9 @@ const processFileGroup = (
|
|||
let scopeSourceKind: ScopeCaptureSourceKind = 'full-file';
|
||||
let lineOffset = 0;
|
||||
let isVueSetup = false;
|
||||
let notebookSegments: readonly NotebookLineSegment[] | undefined;
|
||||
const mapRow = (row: number): number =>
|
||||
notebookSegments ? mapExtractLine(row, notebookSegments) : row + lineOffset;
|
||||
if (language === SupportedLanguages.Vue) {
|
||||
const extracted = extractVueScript(file.content);
|
||||
if (!extracted) continue; // skip .vue files with no script block
|
||||
|
|
@ -1604,6 +1613,12 @@ const processFileGroup = (
|
|||
scopeSourceKind = 'pre-extracted-script';
|
||||
lineOffset = extracted.lineOffset;
|
||||
isVueSetup = extracted.isSetup;
|
||||
} else if (language === SupportedLanguages.Python && isNotebookPath(file.path)) {
|
||||
const extracted = extractNotebookPython(file.content);
|
||||
if (!extracted) continue;
|
||||
parseContent = extracted.pythonSource;
|
||||
scopeSourceKind = 'pre-extracted-script';
|
||||
notebookSegments = extracted.segments;
|
||||
}
|
||||
|
||||
// Per-language source-text transform (e.g., UE macro stripping for C++).
|
||||
|
|
@ -1676,6 +1691,7 @@ const processFileGroup = (
|
|||
},
|
||||
tree,
|
||||
scopeSourceKind,
|
||||
notebookSegments,
|
||||
);
|
||||
if (scopeExtractionFailed) (result.scopeExtractionFailures ??= []).push(file.path);
|
||||
if (parsedFile !== undefined) {
|
||||
|
|
@ -1722,6 +1738,7 @@ const processFileGroup = (
|
|||
// `lineOffset` in the file — shift the CFG into file coordinates so
|
||||
// it joins its graph node and BasicBlock lines map to source.
|
||||
lineOffset,
|
||||
notebookSegments ? mapRow : undefined,
|
||||
);
|
||||
if (cfgs.length) withChannels = { ...withChannels, cfgSideChannel: cfgs };
|
||||
// Surface per-function CFG skips per-language (#2195): merged + logged
|
||||
|
|
@ -1878,7 +1895,7 @@ const processFileGroup = (
|
|||
sourceId: srcId,
|
||||
receiverText,
|
||||
propertyName,
|
||||
line: captureMap['assignment'].startPosition.row + 1,
|
||||
line: mapRow(captureMap['assignment'].startPosition.row) + 1,
|
||||
...(receiverTypeName ? { receiverTypeName } : {}),
|
||||
});
|
||||
}
|
||||
|
|
@ -1913,7 +1930,7 @@ const processFileGroup = (
|
|||
filePath: file.path,
|
||||
httpMethod,
|
||||
decoratorName,
|
||||
lineNumber: decoratorNode.startPosition.row + lineOffset,
|
||||
lineNumber: mapRow(decoratorNode.startPosition.row),
|
||||
...(decoratorReceiver ? { decoratorReceiver } : {}),
|
||||
...(handlerName ? { handlerName } : {}),
|
||||
};
|
||||
|
|
@ -2483,10 +2500,10 @@ const processFileGroup = (
|
|||
: definitionNode?.startPosition;
|
||||
const startLine =
|
||||
startPosition !== undefined
|
||||
? startPosition.row + lineOffset
|
||||
? mapRow(startPosition.row)
|
||||
: nameNode
|
||||
? nameNode.startPosition.row + lineOffset
|
||||
: lineOffset;
|
||||
? mapRow(nameNode.startPosition.row)
|
||||
: mapRow(0);
|
||||
const startColumn = startPosition?.column ?? nameNode?.startPosition.column ?? 0;
|
||||
|
||||
// Compute enclosing class BEFORE node ID — needed to qualify method IDs
|
||||
|
|
@ -2958,7 +2975,7 @@ const processFileGroup = (
|
|||
filePath: file.path,
|
||||
toolName: nodeName,
|
||||
description: (dec.arg || description || '').slice(0, 200),
|
||||
lineNumber: definitionNode.startPosition.row + lineOffset,
|
||||
lineNumber: mapRow(definitionNode.startPosition.row),
|
||||
handlerNodeId: nodeId,
|
||||
});
|
||||
}
|
||||
|
|
@ -3063,7 +3080,7 @@ const processFileGroup = (
|
|||
(objectLiteralBindingInfo?.ownerName || isArrayContainedObjectCallable)
|
||||
? { startColumn }
|
||||
: {}),
|
||||
endLine: definitionNode ? definitionNode.endPosition.row + lineOffset : startLine,
|
||||
endLine: definitionNode ? mapRow(definitionNode.endPosition.row) : startLine,
|
||||
language: language,
|
||||
isExported,
|
||||
...(qualifiedTypeName !== undefined ? { qualifiedName: qualifiedTypeName } : {}),
|
||||
|
|
|
|||
|
|
@ -23,6 +23,7 @@ import { VALID_NODE_TABLES, parseRelationSchemaPairs, RelPairRouter } from './re
|
|||
import { parseTruthyEnv } from '../ingestion/utils/env.js';
|
||||
import { SYMBOL_NODE_LABELS } from '../ingestion/utils/symbol-labels.js';
|
||||
import { applyCjkSegmentationIfEnabled } from '../search/cjk-segmentation.js';
|
||||
import { isNotebookPath, notebookPythonSnippet } from '../ingestion/ipynb-extractor.js';
|
||||
|
||||
/** Computed once — `RELATION_SCHEMA` is a static template literal. Exported so
|
||||
* the streamed sinks (`GraphEmitSink`, `PdgEmitSink`) share this parse
|
||||
|
|
@ -325,15 +326,24 @@ const extractContent = async (
|
|||
const endLine = node.properties.endLine;
|
||||
if (startLine === undefined || endLine === undefined) return '';
|
||||
|
||||
const MAX_SNIPPET = 5000;
|
||||
const capSnippet = (text: string): string =>
|
||||
text.length > MAX_SNIPPET ? text.slice(0, MAX_SNIPPET) + '\n... [truncated]' : text;
|
||||
|
||||
const notebookPath = String(filePath ?? '');
|
||||
if (isNotebookPath(notebookPath)) {
|
||||
const reconstructed = notebookPythonSnippet(content, startLine, endLine, notebookPath);
|
||||
if (reconstructed) {
|
||||
return normalizeFtsText(applyCjkSegmentationIfEnabled(capSnippet(reconstructed)));
|
||||
}
|
||||
}
|
||||
|
||||
const lines = prepared.lines;
|
||||
const exactSymbolContent = EXACT_SYMBOL_CONTENT_LABELS.has(node.label);
|
||||
const start = Math.max(0, exactSymbolContent ? startLine : startLine - 2);
|
||||
const end = Math.min(lines.length - 1, exactSymbolContent ? endLine : endLine + 2);
|
||||
const snippet = lines.slice(start, end + 1).join('\n');
|
||||
const MAX_SNIPPET = 5000;
|
||||
const capped =
|
||||
snippet.length > MAX_SNIPPET ? snippet.slice(0, MAX_SNIPPET) + '\n... [truncated]' : snippet;
|
||||
return normalizeFtsText(applyCjkSegmentationIfEnabled(capped));
|
||||
return normalizeFtsText(applyCjkSegmentationIfEnabled(capSnippet(snippet)));
|
||||
};
|
||||
|
||||
// ============================================================================
|
||||
|
|
|
|||
|
|
@ -785,7 +785,9 @@ import { copyV8CacheIfPresent, tryLoadV8Cache, writeV8CacheFile } from './v8-sid
|
|||
// Same v112: the `valueAlternatives` provider hook extends it to Kotlin
|
||||
// `?:`/`if`, Swift/Dart `??`/`?:`, and Python `x if c else y`, and keeps a
|
||||
// Ruby multi-statement `if` one opaque source.
|
||||
const SCHEMA_BUMP = 112;
|
||||
// v113 (#3371): `.ipynb` code cells are extracted to Python before parse.
|
||||
// Warm caches keyed on raw JSON would replay empty/failed Python parses.
|
||||
const SCHEMA_BUMP = 113;
|
||||
const GITNEXUS_PKG_VERSION = (() => {
|
||||
try {
|
||||
// package.json sits at gitnexus/package.json — two levels up from
|
||||
|
|
|
|||
|
|
@ -0,0 +1,93 @@
|
|||
/**
|
||||
* Jupyter notebooks are indexed as Python by extracting code cells.
|
||||
*/
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import path from 'path';
|
||||
import fs from 'node:fs';
|
||||
import os from 'node:os';
|
||||
import {
|
||||
getNodesByLabel,
|
||||
getNodesByLabelFull,
|
||||
getRelationships,
|
||||
runPipelineFromRepo,
|
||||
writeFixtureRepo,
|
||||
} from './helpers.js';
|
||||
import { extractNotebookPython } from '../../../src/core/ingestion/ipynb-extractor.js';
|
||||
|
||||
function pythonNotebook(cells: Array<{ source: string | string[]; language?: string }>): string {
|
||||
return JSON.stringify(
|
||||
{
|
||||
nbformat: 4,
|
||||
nbformat_minor: 5,
|
||||
metadata: {
|
||||
kernelspec: { display_name: 'Python 3', language: 'python', name: 'python3' },
|
||||
language_info: { name: 'python' },
|
||||
},
|
||||
cells: cells.map((c) => ({
|
||||
cell_type: 'code',
|
||||
metadata: c.language ? { language: c.language } : {},
|
||||
source: c.source,
|
||||
outputs: [],
|
||||
})),
|
||||
},
|
||||
null,
|
||||
2,
|
||||
);
|
||||
}
|
||||
|
||||
describe('Jupyter notebook Python pipeline', () => {
|
||||
it('indexes notebook functions, imports sibling py, and skip-fails closed', async () => {
|
||||
const root = fs.mkdtempSync(path.join(os.tmpdir(), 'gn-ipynb-'));
|
||||
try {
|
||||
writeFixtureRepo(root, {
|
||||
'lib.py': 'def helper():\n return 1\n',
|
||||
'analysis.ipynb': pythonNotebook([
|
||||
{ source: ['from lib import helper\n'] },
|
||||
{ source: ['def train():\n', ' return helper()\n'] },
|
||||
]),
|
||||
'broken.ipynb': '{not json',
|
||||
'julia.ipynb': JSON.stringify({
|
||||
nbformat: 4,
|
||||
nbformat_minor: 5,
|
||||
metadata: {
|
||||
kernelspec: { language: 'julia', name: 'julia', display_name: 'Julia' },
|
||||
},
|
||||
cells: [
|
||||
{ cell_type: 'code', metadata: {}, source: ['function train() end\n'], outputs: [] },
|
||||
],
|
||||
}),
|
||||
});
|
||||
const result = await runPipelineFromRepo(root, () => {}, { skipGraphPhases: true });
|
||||
const functions = getNodesByLabel(result, 'Function');
|
||||
expect(functions).toContain('train');
|
||||
expect(functions).toContain('helper');
|
||||
const train = getNodesByLabelFull(result, 'Function').find((n) => n.name === 'train');
|
||||
expect(train?.properties.filePath.replace(/\\/g, '/')).toMatch(/analysis\.ipynb$/);
|
||||
const extracted = extractNotebookPython(
|
||||
fs.readFileSync(path.join(root, 'analysis.ipynb'), 'utf8'),
|
||||
)!;
|
||||
expect(train?.properties.startLine).toBe(extracted.segments[1].jsonStartLine);
|
||||
const imports = getRelationships(result, 'IMPORTS');
|
||||
expect(
|
||||
imports.some(
|
||||
(e) =>
|
||||
e.sourceFilePath.replace(/\\/g, '/').endsWith('analysis.ipynb') &&
|
||||
(e.target === 'helper' || e.targetFilePath.replace(/\\/g, '/').endsWith('lib.py')),
|
||||
),
|
||||
).toBe(true);
|
||||
expect(
|
||||
getNodesByLabelFull(result, 'Function').filter(
|
||||
(n) =>
|
||||
n.properties.filePath.replace(/\\/g, '/').endsWith('julia.ipynb') && n.name === 'train',
|
||||
),
|
||||
).toHaveLength(0);
|
||||
expect(
|
||||
getNodesByLabelFull(result, 'Function').filter((n) =>
|
||||
n.properties.filePath.replace(/\\/g, '/').endsWith('broken.ipynb'),
|
||||
),
|
||||
).toHaveLength(0);
|
||||
} finally {
|
||||
fs.rmSync(root, { recursive: true, force: true });
|
||||
}
|
||||
}, 60000);
|
||||
});
|
||||
|
|
@ -123,4 +123,33 @@ describe('ensureAndParse', () => {
|
|||
expect(objcParse).toHaveBeenCalledTimes(2);
|
||||
expect(cppParse).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('parses extracted notebook Python and returns null when extraction fails', async () => {
|
||||
parseSourceSafeSpy.mockClear();
|
||||
const pyParse = vi.fn().mockReturnValue({ lang: 'py', rootNode: { type: 'module' } });
|
||||
createParserForLanguage.mockImplementation(async (language: string) => {
|
||||
if (language === 'python') return { parse: pyParse };
|
||||
throw new Error(`unexpected language ${language}`);
|
||||
});
|
||||
|
||||
const { ensureAndParse } = await import('../../src/core/embeddings/ast-utils.js');
|
||||
const nb = JSON.stringify({
|
||||
nbformat: 4,
|
||||
nbformat_minor: 5,
|
||||
metadata: { kernelspec: { language: 'python', name: 'python3', display_name: 'Python' } },
|
||||
cells: [
|
||||
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
|
||||
],
|
||||
});
|
||||
|
||||
await ensureAndParse(nb, 'analysis.ipynb');
|
||||
expect(parseSourceSafeSpy).toHaveBeenCalled();
|
||||
const parsedText = parseSourceSafeSpy.mock.calls.at(-1)?.[1] as string;
|
||||
expect(parsedText).toContain('def train');
|
||||
expect(parsedText).not.toContain('cell_type');
|
||||
|
||||
parseSourceSafeSpy.mockClear();
|
||||
expect(await ensureAndParse('{not json', 'broken.ipynb')).toBeNull();
|
||||
expect(parseSourceSafeSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -290,8 +290,9 @@ describe('PARSE_CACHE_VERSION', () => {
|
|||
// anonymous arrows, so both stores re-extract.
|
||||
// Moved 104 -> 112 for #3354: callable-value flow follows `??`/`||`/`?:`
|
||||
// branches. 105-111 are claimed by open PR #3326.
|
||||
it('pins SCHEMA_BUMP to 112 so concurrent bumps cannot silently collide (#2766, #3015, #3088, #2885, #3128, #2865, #3130, #1432, #3161, #3179, #3219, #3190, #3253, #3273, #3339, #3354)', () => {
|
||||
expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(112);
|
||||
// Moved 112 -> 113 for #3371: notebook code-cell extraction before Python parse.
|
||||
it('pins SCHEMA_BUMP to 113 so concurrent bumps cannot silently collide (#2766, #3015, #3088, #2885, #3128, #2865, #3130, #1432, #3161, #3179, #3219, #3190, #3253, #3273, #3339, #3354, #3371)', () => {
|
||||
expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).toBe(113);
|
||||
expect(PARSE_CACHE_BUCKET_COUNT).toBe(128);
|
||||
// The PREVIOUS version must fail the reuse gate, not merely differ from the
|
||||
// current one — a hardcoded number outside the conflict hunk rebases cleanly
|
||||
|
|
@ -300,7 +301,7 @@ describe('PARSE_CACHE_VERSION', () => {
|
|||
for (const taken of [
|
||||
59, 60, 61, 62, 63, 64, 65, 66, 67, 68, 69, 70, 71, 72, 73, 74, 75, 76, 77, 78, 79, 80, 81,
|
||||
82, 83, 84, 85, 86, 87, 88, 89, 90, 91, 92, 93, 94, 95, 96, 97, 98, 99, 100, 101, 102, 103,
|
||||
104, 105, 106, 107, 108, 109, 110, 111,
|
||||
104, 105, 106, 107, 108, 109, 110, 111, 112,
|
||||
]) {
|
||||
expect(Number(PARSE_CACHE_VERSION.split('+', 1)[0])).not.toBe(taken);
|
||||
}
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ import {
|
|||
getLanguageFromFilename,
|
||||
getSyntaxLanguageFromFilename,
|
||||
isBladeTemplateFilename,
|
||||
isNotebookFilename,
|
||||
SupportedLanguages,
|
||||
} from 'gitnexus-shared';
|
||||
import { getProvider, getProviderForFile } from '../../src/core/ingestion/languages/index.js';
|
||||
|
|
@ -50,6 +51,13 @@ describe('getLanguageFromFilename', () => {
|
|||
it('detects .py files', () => {
|
||||
expect(getLanguageFromFilename('main.py')).toBe(SupportedLanguages.Python);
|
||||
});
|
||||
|
||||
it('detects .ipynb files as Python', () => {
|
||||
expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python);
|
||||
expect(getProviderForFile('notebooks/analysis.ipynb')?.id).toBe(SupportedLanguages.Python);
|
||||
expect(isNotebookFilename('notebooks/analysis.ipynb')).toBe(true);
|
||||
expect(getSyntaxLanguageFromFilename('analysis.ipynb')).toBe('json');
|
||||
});
|
||||
});
|
||||
|
||||
describe('Java', () => {
|
||||
|
|
|
|||
428
gitnexus/test/unit/ipynb-extractor.test.ts
Normal file
428
gitnexus/test/unit/ipynb-extractor.test.ts
Normal file
|
|
@ -0,0 +1,428 @@
|
|||
import { describe, it, expect } from 'vitest';
|
||||
import {
|
||||
extractNotebookPython,
|
||||
mapExtractLine,
|
||||
notebookPythonSnippet,
|
||||
isPythonFamilyLanguage,
|
||||
} from '../../src/core/ingestion/ipynb-extractor.js';
|
||||
|
||||
function notebook(opts: {
|
||||
language?: string;
|
||||
languageInfo?: string;
|
||||
cells: Array<Record<string, unknown>>;
|
||||
}): string {
|
||||
const kernelspec =
|
||||
opts.language === undefined
|
||||
? undefined
|
||||
: { display_name: 'Python', language: opts.language, name: 'python' };
|
||||
const language_info = opts.languageInfo === undefined ? undefined : { name: opts.languageInfo };
|
||||
return JSON.stringify(
|
||||
{
|
||||
nbformat: 4,
|
||||
nbformat_minor: 5,
|
||||
metadata: {
|
||||
...(kernelspec ? { kernelspec } : {}),
|
||||
...(language_info ? { language_info } : {}),
|
||||
},
|
||||
cells: opts.cells,
|
||||
},
|
||||
null,
|
||||
2,
|
||||
);
|
||||
}
|
||||
|
||||
describe('isPythonFamilyLanguage', () => {
|
||||
it('accepts python3 and ipython', () => {
|
||||
expect(isPythonFamilyLanguage('python3')).toBe(true);
|
||||
expect(isPythonFamilyLanguage('IPython')).toBe(true);
|
||||
expect(isPythonFamilyLanguage('julia')).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('extractNotebookPython', () => {
|
||||
it('maps source that is serialized before cell_type', () => {
|
||||
const content = JSON.stringify({
|
||||
nbformat: 4,
|
||||
nbformat_minor: 5,
|
||||
metadata: { kernelspec: { language: 'python', name: 'python3', display_name: 'Python' } },
|
||||
cells: [
|
||||
{
|
||||
source: ['def train():\n', ' pass\n'],
|
||||
cell_type: 'code',
|
||||
metadata: {},
|
||||
outputs: [],
|
||||
},
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result?.pythonSource).toContain('def train');
|
||||
expect(result?.segments).toHaveLength(1);
|
||||
});
|
||||
|
||||
it('extracts def train from a Python v4 notebook', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{
|
||||
cell_type: 'code',
|
||||
metadata: {},
|
||||
source: ['def train():\n', ' pass\n'],
|
||||
outputs: [],
|
||||
},
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result).not.toBeNull();
|
||||
expect(result!.pythonSource).toContain('def train():');
|
||||
expect(result!.segments).toHaveLength(1);
|
||||
const defJsonLine = content.split('\n').findIndex((l) => l.includes('def train'));
|
||||
expect(result!.segments[0].jsonStartLine).toBe(defJsonLine);
|
||||
});
|
||||
|
||||
it('keeps two code cells in order with two segments', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{ cell_type: 'code', metadata: {}, source: ['x = 1\n'], outputs: [] },
|
||||
{
|
||||
cell_type: 'code',
|
||||
metadata: {},
|
||||
source: ['def train():\n', ' return x\n'],
|
||||
outputs: [],
|
||||
},
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result).not.toBeNull();
|
||||
expect(result!.pythonSource).toMatch(/x = 1\n+def train/);
|
||||
expect(result!.segments).toHaveLength(2);
|
||||
const defRow = result!.pythonSource.split('\n').findIndex((l) => l.startsWith('def train'));
|
||||
expect(defRow).toBe(result!.segments[1].extractStartLine);
|
||||
expect(mapExtractLine(defRow, result!.segments)).toBe(result!.segments[1].jsonStartLine);
|
||||
});
|
||||
|
||||
it('accepts source as a single string', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [{ cell_type: 'code', metadata: {}, source: 'y = 2\n', outputs: [] }],
|
||||
});
|
||||
expect(extractNotebookPython(content)?.pythonSource).toContain('y = 2');
|
||||
});
|
||||
|
||||
it('returns null for markdown-only notebooks', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [{ cell_type: 'markdown', metadata: {}, source: ['# hi\n'] }],
|
||||
});
|
||||
expect(extractNotebookPython(content)).toBeNull();
|
||||
});
|
||||
|
||||
it('returns null for invalid JSON', () => {
|
||||
expect(extractNotebookPython('{not json')).toBeNull();
|
||||
});
|
||||
|
||||
it('returns null for a Julia kernelspec', () => {
|
||||
const content = notebook({
|
||||
language: 'julia',
|
||||
cells: [{ cell_type: 'code', metadata: {}, source: ['1 + 1\n'], outputs: [] }],
|
||||
});
|
||||
expect(extractNotebookPython(content)).toBeNull();
|
||||
});
|
||||
|
||||
it('extracts python3 kernelspec', () => {
|
||||
const content = notebook({
|
||||
language: 'python3',
|
||||
cells: [{ cell_type: 'code', metadata: {}, source: ['a = 1\n'], outputs: [] }],
|
||||
});
|
||||
expect(extractNotebookPython(content)?.pythonSource).toContain('a = 1');
|
||||
});
|
||||
|
||||
it('comments line magics in place', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{
|
||||
cell_type: 'code',
|
||||
metadata: {},
|
||||
source: ['%time\n', 'x = 1\n', '!ls\n'],
|
||||
outputs: [],
|
||||
},
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result!.pythonSource.split('\n').filter((l) => l.length > 0)).toEqual([
|
||||
'# %time',
|
||||
'x = 1',
|
||||
'# !ls',
|
||||
]);
|
||||
});
|
||||
|
||||
it('skips a %%bash cell and still extracts later train', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{ cell_type: 'code', metadata: {}, source: ['%%bash\n', 'echo hi\n'], outputs: [] },
|
||||
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result!.pythonSource).toContain('def train');
|
||||
expect(result!.pythonSource).not.toContain('echo hi');
|
||||
});
|
||||
|
||||
it('maps identical duplicate cells to later JSON lines', () => {
|
||||
const src = ['print(1)\n'];
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{ cell_type: 'code', metadata: {}, source: src, outputs: [] },
|
||||
{ cell_type: 'code', metadata: {}, source: src, outputs: [] },
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result!.segments).toHaveLength(2);
|
||||
expect(result!.segments[1].jsonStartLine).toBeGreaterThan(result!.segments[0].jsonStartLine);
|
||||
});
|
||||
|
||||
it('skips an R-language code cell and keeps Python cells', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{
|
||||
cell_type: 'code',
|
||||
metadata: { language: 'R' },
|
||||
source: ['x <- 1\n'],
|
||||
outputs: [],
|
||||
},
|
||||
{ cell_type: 'code', metadata: {}, source: ['z = 3\n'], outputs: [] },
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result!.pythonSource).toContain('z = 3');
|
||||
expect(result!.pythonSource).not.toContain('x <- 1');
|
||||
});
|
||||
});
|
||||
|
||||
describe('mapExtractLine', () => {
|
||||
it('maps a second-cell extract row onto that cell JSON line', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{ cell_type: 'markdown', metadata: {}, source: ['# intro\n'] },
|
||||
{ cell_type: 'code', metadata: {}, source: ['a = 1\n'], outputs: [] },
|
||||
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
|
||||
],
|
||||
});
|
||||
const extracted = extractNotebookPython(content)!;
|
||||
const trainSeg = extracted.segments[1];
|
||||
const mapped = mapExtractLine(trainSeg.extractStartLine, extracted.segments);
|
||||
expect(mapped).toBe(trainSeg.jsonStartLine);
|
||||
expect(mapped).toBeGreaterThan(extracted.segments[0].jsonStartLine);
|
||||
});
|
||||
});
|
||||
|
||||
describe('notebookPythonSnippet', () => {
|
||||
it('returns Python def train not JSON cell_type', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
|
||||
],
|
||||
});
|
||||
const extracted = extractNotebookPython(content)!;
|
||||
const snippet = notebookPythonSnippet(
|
||||
content,
|
||||
extracted.segments[0].jsonStartLine,
|
||||
extracted.segments[0].jsonEndLine,
|
||||
);
|
||||
expect(snippet).toContain('def train');
|
||||
expect(snippet).not.toContain('cell_type');
|
||||
});
|
||||
});
|
||||
|
||||
describe('extractNotebookPython edge cases', () => {
|
||||
it('skips a code cell without source and keeps later Python', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{ cell_type: 'code', metadata: {}, source: ['def ok():\n', ' pass\n'], outputs: [] },
|
||||
{ cell_type: 'code', metadata: {}, outputs: [] },
|
||||
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result!.pythonSource).toContain('def ok');
|
||||
expect(result!.pythonSource).toContain('def train');
|
||||
});
|
||||
|
||||
it('returns null when language_info is julia without kernelspec', () => {
|
||||
const content = notebook({
|
||||
languageInfo: 'julia',
|
||||
cells: [{ cell_type: 'code', metadata: {}, source: ['1 + 1\n'], outputs: [] }],
|
||||
});
|
||||
expect(extractNotebookPython(content)).toBeNull();
|
||||
});
|
||||
|
||||
it('maps coordinates to the last cells array when the key is duplicated', () => {
|
||||
const decoy = JSON.stringify(
|
||||
[{ cell_type: 'code', metadata: {}, source: ['def decoy():\n', ' pass\n'], outputs: [] }],
|
||||
null,
|
||||
2,
|
||||
);
|
||||
const real = JSON.stringify(
|
||||
[{ cell_type: 'code', metadata: {}, source: ['def real():\n', ' pass\n'], outputs: [] }],
|
||||
null,
|
||||
2,
|
||||
);
|
||||
const content = `{
|
||||
"nbformat": 4,
|
||||
"nbformat_minor": 5,
|
||||
"metadata": { "kernelspec": { "language": "python", "name": "python3", "display_name": "Python" } },
|
||||
"cells": ${decoy},
|
||||
"cells": ${real}
|
||||
}`;
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result!.pythonSource).toContain('def real');
|
||||
expect(result!.pythonSource).not.toContain('decoy');
|
||||
const realLine = content.split('\n').findIndex((l) => l.includes('def real'));
|
||||
expect(result!.segments[0].jsonStartLine).toBe(realLine);
|
||||
});
|
||||
|
||||
it('keeps Python under %%time and skips %%bash', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{
|
||||
cell_type: 'code',
|
||||
metadata: {},
|
||||
source: ['%%time\n', 'def train():\n', ' pass\n'],
|
||||
outputs: [],
|
||||
},
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result!.pythonSource).toContain('def train');
|
||||
expect(result!.pythonSource).toContain('# %%time');
|
||||
});
|
||||
|
||||
it('comments IPython help lines', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{
|
||||
cell_type: 'code',
|
||||
metadata: {},
|
||||
source: ['train?\n', 'def train():\n', ' pass\n'],
|
||||
outputs: [],
|
||||
},
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result!.pythonSource.split('\n').filter((l) => l.length > 0)[0]).toBe('# train?');
|
||||
expect(result!.pythonSource).toContain('def train');
|
||||
});
|
||||
|
||||
it('parses a UTF-8 BOM notebook', () => {
|
||||
const content =
|
||||
'\uFEFF' +
|
||||
notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{
|
||||
cell_type: 'code',
|
||||
metadata: {},
|
||||
source: ['def train():\n', ' pass\n'],
|
||||
outputs: [],
|
||||
},
|
||||
],
|
||||
});
|
||||
expect(extractNotebookPython(content)?.pythonSource).toContain('def train');
|
||||
});
|
||||
|
||||
it('treats python and python3 metadata as the same family', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
languageInfo: 'python3',
|
||||
cells: [{ cell_type: 'code', metadata: {}, source: ['a = 1\n'], outputs: [] }],
|
||||
});
|
||||
expect(extractNotebookPython(content)?.pythonSource).toContain('a = 1');
|
||||
});
|
||||
|
||||
it('skips a notebook whose kernelspec name is R and has no language field', () => {
|
||||
const content = JSON.stringify({
|
||||
nbformat: 4,
|
||||
nbformat_minor: 5,
|
||||
metadata: { kernelspec: { name: 'ir', display_name: 'R' } },
|
||||
cells: [{ cell_type: 'code', metadata: {}, source: ['x <- 1\n'], outputs: [] }],
|
||||
});
|
||||
expect(extractNotebookPython(content)).toBeNull();
|
||||
});
|
||||
|
||||
it('extracts nbformat v3 worksheets via input', () => {
|
||||
const content = JSON.stringify(
|
||||
{
|
||||
nbformat: 3,
|
||||
nbformat_minor: 0,
|
||||
metadata: { name: 'legacy' },
|
||||
worksheets: [
|
||||
{
|
||||
cells: [
|
||||
{
|
||||
cell_type: 'code',
|
||||
language: 'python',
|
||||
input: ['def train():\n', ' pass\n'],
|
||||
outputs: [],
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
},
|
||||
null,
|
||||
2,
|
||||
);
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result!.pythonSource).toContain('def train');
|
||||
const line = content.split('\n').findIndex((l) => l.includes('def train'));
|
||||
expect(result!.segments[0].jsonStartLine).toBe(line);
|
||||
});
|
||||
|
||||
it('keeps later cells when an earlier cell has an unclosed string', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{ cell_type: 'code', metadata: {}, source: ['text = """unterminated\n'], outputs: [] },
|
||||
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result!.pythonSource).toContain('def train');
|
||||
expect(result!.pythonSource).not.toMatch(/^text = """/m);
|
||||
});
|
||||
|
||||
it('turns %run of a local module into an import', () => {
|
||||
const content = notebook({
|
||||
language: 'python',
|
||||
cells: [
|
||||
{
|
||||
cell_type: 'code',
|
||||
metadata: {},
|
||||
source: ['%run ./lib.py\n', 'def train():\n', ' pass\n'],
|
||||
outputs: [],
|
||||
},
|
||||
],
|
||||
});
|
||||
const result = extractNotebookPython(content);
|
||||
expect(result!.pythonSource).toContain('import lib # %run ./lib.py');
|
||||
expect(result!.pythonSource).toContain('def train');
|
||||
});
|
||||
|
||||
it('indexes a Sage kernel as Python', () => {
|
||||
const content = notebook({
|
||||
language: 'sage',
|
||||
cells: [
|
||||
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
|
||||
],
|
||||
});
|
||||
expect(extractNotebookPython(content)?.pythonSource).toContain('def train');
|
||||
});
|
||||
});
|
||||
52
gitnexus/test/unit/ipynb-python-captures.test.ts
Normal file
52
gitnexus/test/unit/ipynb-python-captures.test.ts
Normal file
|
|
@ -0,0 +1,52 @@
|
|||
import { describe, it, expect } from 'vitest';
|
||||
import { emitPythonScopeCaptures } from '../../src/core/ingestion/languages/python/captures.js';
|
||||
import { pythonProvider } from '../../src/core/ingestion/languages/python.js';
|
||||
import { extractParsedFile } from '../../src/core/ingestion/scope-extractor-bridge.js';
|
||||
import { extractNotebookPython, mapExtractLine } from '../../src/core/ingestion/ipynb-extractor.js';
|
||||
import { getLanguageFromFilename, SupportedLanguages } from 'gitnexus-shared';
|
||||
import { getProviderForFile } from '../../src/core/ingestion/languages/index.js';
|
||||
|
||||
const notebook = JSON.stringify(
|
||||
{
|
||||
nbformat: 4,
|
||||
nbformat_minor: 5,
|
||||
metadata: {
|
||||
kernelspec: { language: 'python', name: 'python3', display_name: 'Python' },
|
||||
},
|
||||
cells: [
|
||||
{ cell_type: 'code', metadata: {}, source: ['def train():\n', ' pass\n'], outputs: [] },
|
||||
],
|
||||
},
|
||||
null,
|
||||
2,
|
||||
);
|
||||
|
||||
describe('Python notebook scope captures', () => {
|
||||
it('detects .ipynb as Python', () => {
|
||||
expect(getLanguageFromFilename('analysis.ipynb')).toBe(SupportedLanguages.Python);
|
||||
expect(getProviderForFile('n.ipynb')?.id).toBe(SupportedLanguages.Python);
|
||||
});
|
||||
|
||||
it('does not parse raw JSON as Python on the full-file path', () => {
|
||||
const matches = emitPythonScopeCaptures(notebook, 'analysis.ipynb');
|
||||
const names = matches.flatMap((m) => Object.keys(m));
|
||||
expect(names.some((k) => k.includes('function'))).toBe(true);
|
||||
});
|
||||
|
||||
it('returns no captures for invalid notebook JSON', () => {
|
||||
expect(emitPythonScopeCaptures('{not json', 'broken.ipynb')).toEqual([]);
|
||||
});
|
||||
|
||||
it('extractParsedFile yields a Function for train', () => {
|
||||
const captured = emitPythonScopeCaptures(notebook, 'analysis.ipynb');
|
||||
const fnCapture = captured.find((m) => m['@scope.function'] !== undefined);
|
||||
const extracted = extractNotebookPython(notebook)!;
|
||||
const expectedJson = mapExtractLine(extracted.segments[0].extractStartLine, extracted.segments);
|
||||
expect(fnCapture?.['@scope.function']?.range.startLine).toBe(expectedJson + 1);
|
||||
const parsed = extractParsedFile(pythonProvider, notebook, 'analysis.ipynb');
|
||||
expect(parsed).toBeDefined();
|
||||
expect(parsed!.localDefs.some((d) => d.qualifiedName === 'train' || d.name === 'train')).toBe(
|
||||
true,
|
||||
);
|
||||
});
|
||||
});
|
||||
Loading…
Add table
Reference in a new issue