mirror of
https://github.com/abhigyanpatwari/GitNexus.git
synced 2026-10-07 02:58:02 +00:00
fix(group): detect function-local Python imports (#3254)
* fix(group): detect function-local Python imports * fix(group): parse Python imports structurally * fix(group): use safe Python parser wrapper * fix(group): isolate Python parse timeouts * fix(group): skip recovered indented Python from-imports Error-recovered trees can promote unclosed-docstring lookalikes into real import_from_statement nodes. Drop those indented imports, bound file reads, log parse timeouts, and add fingerprint floors for the tree-sitter scan. Co-authored-by: Cursor <cursoragent@cursor.com> * chore(autofix): apply prettier + eslint fixes via /autofix command --------- Co-authored-by: Gergő Magyar <gergomagyar@icloud.com> Co-authored-by: Gergo Magyar <gergomagyar0@gmail.com> Co-authored-by: Cursor <cursoragent@cursor.com> Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
This commit is contained in:
parent
32a5fc3c4f
commit
c9b391357e
5 changed files with 498 additions and 78 deletions
9
.github/workflows/ci-tests.yml
vendored
9
.github/workflows/ci-tests.yml
vendored
|
|
@ -599,6 +599,15 @@ jobs:
|
|||
run: node --import tsx bench/parse-dispatch-rounds/measure.mjs --check
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Python workspace import-scan guards (#3254)
|
||||
if: ${{ !cancelled() }}
|
||||
# Build-free: same baseline approach as parse-dispatch-rounds —
|
||||
# exact link/lookalike floors plus a fingerprint, then ratio timing
|
||||
# only (scan scaling and from-token prefilter advantage). See
|
||||
# bench/python-workspace-import-scan/measure.mjs.
|
||||
run: node --import tsx bench/python-workspace-import-scan/measure.mjs --check
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: C++ qualified-namespace resolution guards (#2788)
|
||||
if: ${{ !cancelled() }}
|
||||
# Build-free: asserts resolveCppQualifiedNamespaceMember resolves an
|
||||
|
|
|
|||
37
gitnexus/bench/python-workspace-import-scan/baselines.json
Normal file
37
gitnexus/bench/python-workspace-import-scan/baselines.json
Normal file
|
|
@ -0,0 +1,37 @@
|
|||
{
|
||||
"_what": "Baselines for bench/python-workspace-import-scan/measure.mjs --check. Guards Python workspace from-import scanning after #3254: function-local discovery, lookalike rejection, and parse-volume. Same approach as bench/parse-dispatch-rounds/baselines.json — exact floors and a fingerprint first; the only timing arms are ratios.",
|
||||
|
||||
"_triage": "READ THIS BEFORE RE-RUNNING. from_links, lookalike_links, no_from_links, from_files, no_from_files, lookalike_files and layout_fingerprint are DETERMINISTIC: a re-run never changes them, and none may be re-baselined to make CI green. scan_scaling_ratio is an UPPER timing arm; prefilter_advantage is a FLOOR timing arm. Runner contention dominates both, so re-run on an idle machine before investigating and read the reported `reps` first. If exactly one arm fails and it is a timing arm, suspect the machine.",
|
||||
|
||||
"from_files": 40,
|
||||
"no_from_files": 40,
|
||||
"lookalike_files": 8,
|
||||
"_shape_note": "THE FLOOR. Without these three, every arm below is a ceiling over nothing. from_links only asserts something while the corpus still walks many files. Shrink it to one happy-path module and from_links still reads 1 and still passes, asserting a property the corpus no longer has.",
|
||||
|
||||
"from_links": 1,
|
||||
"lookalike_links": 0,
|
||||
"no_from_links": 0,
|
||||
"_links_note": "Exact contract counts. from_links=1 is the deduped function-local Record import. lookalike_links=0 pins closed AND unclosed indented docstring examples. no_from_links=0 pins files that have no from-token.",
|
||||
|
||||
"layout_fingerprint": "02e51be915083909cf231688c935cfe11bb52c7c781a70532859ad4a0bb9b42e",
|
||||
"_layout_fingerprint_note": "sha256 over sorted from|to|contract rows on the from-corpus. A change here is a BEHAVIOUR change — the extractor returned a different contract set. Explain it, never re-baseline it alone.",
|
||||
|
||||
"scan_scaling_budget": 1.6,
|
||||
"_scan_scaling_note": "(t_4n / t_n) / 4 for extractPythonWorkspaceLinks over the mixed corpus; ~1.0 is linear. A RATIO rather than a millisecond ceiling, deliberately: wall-clock is runner-speed-dependent, and this repo has already been bitten by a fixed ms budget. Budget is 1.6, matching parse-dispatch-rounds' pack_scaling_budget. min-of-15 estimator.",
|
||||
|
||||
"prefilter_advantage_floor": 3,
|
||||
"_prefilter_note": "t_from / t_nofrom at the same file count and body size. The no-from corpus is large Python with no `from` token. A working prefilter keeps that arm cheap (~5x here). Removing the prefilter collapses the ratio toward 1 because both sides tree-sitter parse. Floor is 3 — about 1.6x below the measured minimum, same headroom philosophy as the 1.6 upper budget.",
|
||||
|
||||
"_measured": {
|
||||
"scan_scaling_ratio": 0.994,
|
||||
"scan_scaling_ratio_samples": [0.994, 0.963, 0.977, 0.986, 0.889],
|
||||
"prefilter_advantage": 5.411,
|
||||
"prefilter_advantage_samples": [5.197, 4.955, 5.057, 5.124, 5.411],
|
||||
"from_ms": 86.43,
|
||||
"no_from_ms": 16.35,
|
||||
"mixed_ms": 112.1,
|
||||
"large_ms_4x": 398.76,
|
||||
"reps": 15
|
||||
},
|
||||
"_measured_note": "Maxima (and sample lists) over 5 consecutive local runs using the min-of-15 estimator from parse-dispatch-rounds. Milliseconds are diagnostic context only — nothing gates on them."
|
||||
}
|
||||
270
gitnexus/bench/python-workspace-import-scan/measure.mjs
Normal file
270
gitnexus/bench/python-workspace-import-scan/measure.mjs
Normal file
|
|
@ -0,0 +1,270 @@
|
|||
/**
|
||||
* Build-free bench for Python workspace from-import scanning (#3254).
|
||||
*
|
||||
* WHY THIS EXISTS. `scanPythonImports` now tree-sitter-parses every `.py` file
|
||||
* that looks like it has a `from` import. Graph output does not show "how many
|
||||
* files we parsed" or "we skipped an unclosed-docstring lookalike" — a revert
|
||||
* to column-0 regex, a dropped `parseHadErrors` guard, or a removed `from`
|
||||
* prefilter can still emit the same one contract on a tiny fixture. This file
|
||||
* is the same shape as `bench/parse-dispatch-rounds`: exact floors first, one
|
||||
* ratio timing arm, never a millisecond ceiling.
|
||||
*
|
||||
* ARMS:
|
||||
*
|
||||
* - `from_links` / `lookalike_links` / `no_from_links` — EXACT. The
|
||||
* correctness floor. The from-corpus must still discover `datalib::Record`.
|
||||
* Closed + unclosed docstring lookalikes must stay at 0. Files with no
|
||||
* `from` token must stay at 0.
|
||||
*
|
||||
* - `from_files` / `no_from_files` / `lookalike_files` — EXACT, and they are
|
||||
* the FLOOR. `from_links === 1` only asserts something while the corpus
|
||||
* still has many files to walk. Shrink it to one happy-path file and the
|
||||
* link arm still passes, gating a property the corpus no longer has.
|
||||
*
|
||||
* - `layout_fingerprint` — EXACT. sha256 over sorted `from|to|contract` rows
|
||||
* on the from-corpus. Catches a contract-set change that leaves the count
|
||||
* intact.
|
||||
*
|
||||
* - `scan_scaling_ratio` — the only mixed-corpus timing arm, a RATIO not a
|
||||
* millisecond ceiling. `(t_4n / t_n) / 4` divides the machine out; ~1.0 is
|
||||
* linear. Superlinear AST work over file count lands here.
|
||||
*
|
||||
* - `prefilter_advantage` — `t_from / t_nofrom` at the same file count. The
|
||||
* no-from corpus is large Python with no `from` token. A working prefilter
|
||||
* keeps that arm cheap; removing it collapses the ratio toward 1 because
|
||||
* both sides parse.
|
||||
*
|
||||
* Usage:
|
||||
* node --import tsx bench/python-workspace-import-scan/measure.mjs
|
||||
* node --import tsx bench/python-workspace-import-scan/measure.mjs --check
|
||||
*/
|
||||
import { createHash } from 'node:crypto';
|
||||
import { mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import path from 'node:path';
|
||||
import { performance } from 'node:perf_hooks';
|
||||
import { extractPythonWorkspaceLinks } from '../../src/core/group/extractors/python-workspace-extractor.ts';
|
||||
|
||||
const baselines = JSON.parse(readFileSync(new URL('./baselines.json', import.meta.url), 'utf8'));
|
||||
|
||||
const REPS = 15;
|
||||
const FROM_FILES = 40;
|
||||
const NO_FROM_FILES = 40;
|
||||
const LOOKALIKE_FILES = 8;
|
||||
const NO_FROM_BODY = Array.from(
|
||||
{ length: 80 },
|
||||
(_, i) => `def helper_${i}(x):\n return x + ${i}\n`,
|
||||
).join('\n');
|
||||
|
||||
function writeWorkspace(root, { fromCount, noFromCount, lookalikeCount }) {
|
||||
mkdirSync(path.join(root, 'lib', 'datalib'), { recursive: true });
|
||||
writeFileSync(
|
||||
path.join(root, 'lib', 'pyproject.toml'),
|
||||
'[project]\nname = "datalib"\nversion = "0.1.0"\ndependencies = []\n',
|
||||
);
|
||||
writeFileSync(path.join(root, 'lib', 'datalib', 'models.py'), 'class Record: pass\n');
|
||||
|
||||
mkdirSync(path.join(root, 'app', 'myapp'), { recursive: true });
|
||||
writeFileSync(
|
||||
path.join(root, 'app', 'pyproject.toml'),
|
||||
'[project]\nname = "myapp"\nversion = "0.1.0"\ndependencies = ["datalib"]\n',
|
||||
);
|
||||
|
||||
for (let i = 0; i < fromCount; i++) {
|
||||
writeFileSync(
|
||||
path.join(root, 'app', 'myapp', `from_${i}.py`),
|
||||
`${NO_FROM_BODY}\ndef load_${i}():\n from datalib.models import Record\n return Record()\n`,
|
||||
);
|
||||
}
|
||||
for (let i = 0; i < noFromCount; i++) {
|
||||
writeFileSync(path.join(root, 'app', 'myapp', `plain_${i}.py`), NO_FROM_BODY);
|
||||
}
|
||||
for (let i = 0; i < lookalikeCount; i++) {
|
||||
const closed = i % 2 === 0;
|
||||
writeFileSync(
|
||||
path.join(root, 'app', 'myapp', `doc_${i}.py`),
|
||||
closed
|
||||
? `def describe_${i}():\n """Example:\n from datalib.models import Record\n """\n return None\n`
|
||||
: `def describe_${i}():\n """\n from datalib.models import Record\n`,
|
||||
);
|
||||
}
|
||||
|
||||
return {
|
||||
repos: { lib: 'datalib', app: 'myapp' },
|
||||
repoPaths: new Map([
|
||||
['lib', path.join(root, 'lib')],
|
||||
['app', path.join(root, 'app')],
|
||||
]),
|
||||
};
|
||||
}
|
||||
|
||||
async function scan(workspace) {
|
||||
return extractPythonWorkspaceLinks(workspace.repos, workspace.repoPaths);
|
||||
}
|
||||
|
||||
function fingerprint(links) {
|
||||
return createHash('sha256')
|
||||
.update(
|
||||
links
|
||||
.map((l) => `${l.from}|${l.to}|${l.contract}`)
|
||||
.sort()
|
||||
.join('\n'),
|
||||
)
|
||||
.digest('hex');
|
||||
}
|
||||
|
||||
async function fastest(fn, reps) {
|
||||
await fn();
|
||||
let best = Infinity;
|
||||
for (let r = 0; r < reps; r++) {
|
||||
const t0 = performance.now();
|
||||
await fn();
|
||||
best = Math.min(best, performance.now() - t0);
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
const roots = [];
|
||||
function workspace(opts) {
|
||||
const root = mkdtempSync(path.join(tmpdir(), 'gn-py-ws-bench-'));
|
||||
roots.push(root);
|
||||
return writeWorkspace(root, opts);
|
||||
}
|
||||
|
||||
try {
|
||||
const fromWs = workspace({
|
||||
fromCount: FROM_FILES,
|
||||
noFromCount: 0,
|
||||
lookalikeCount: 0,
|
||||
});
|
||||
const noFromWs = workspace({
|
||||
fromCount: 0,
|
||||
noFromCount: NO_FROM_FILES,
|
||||
lookalikeCount: 0,
|
||||
});
|
||||
const lookalikeWs = workspace({
|
||||
fromCount: 0,
|
||||
noFromCount: 0,
|
||||
lookalikeCount: LOOKALIKE_FILES,
|
||||
});
|
||||
const mixedWs = workspace({
|
||||
fromCount: FROM_FILES,
|
||||
noFromCount: NO_FROM_FILES,
|
||||
lookalikeCount: LOOKALIKE_FILES,
|
||||
});
|
||||
const mixed4x = workspace({
|
||||
fromCount: FROM_FILES * 4,
|
||||
noFromCount: NO_FROM_FILES * 4,
|
||||
lookalikeCount: LOOKALIKE_FILES * 4,
|
||||
});
|
||||
|
||||
const [fromResult, noFromResult, lookalikeResult] = await Promise.all([
|
||||
scan(fromWs),
|
||||
scan(noFromWs),
|
||||
scan(lookalikeWs),
|
||||
]);
|
||||
const layoutFingerprint = fingerprint(fromResult.links);
|
||||
|
||||
const fromMs = await fastest(() => scan(fromWs), REPS);
|
||||
const noFromMs = await fastest(() => scan(noFromWs), REPS);
|
||||
const smallMs = await fastest(() => scan(mixedWs), REPS);
|
||||
const largeMs = await fastest(() => scan(mixed4x), REPS);
|
||||
const scanScaling = largeMs / smallMs / 4;
|
||||
const prefilterAdvantage = noFromMs === 0 ? Infinity : fromMs / noFromMs;
|
||||
|
||||
console.log(`from_files : ${FROM_FILES} (expect ${baselines.from_files})`);
|
||||
console.log(`no_from_files : ${NO_FROM_FILES} (expect ${baselines.no_from_files})`);
|
||||
console.log(`lookalike_files : ${LOOKALIKE_FILES} (expect ${baselines.lookalike_files})`);
|
||||
console.log(
|
||||
`from_links : ${fromResult.links.length} (expect ${baselines.from_links})`,
|
||||
);
|
||||
console.log(
|
||||
`lookalike_links : ${lookalikeResult.links.length} (expect ${baselines.lookalike_links})`,
|
||||
);
|
||||
console.log(
|
||||
`no_from_links : ${noFromResult.links.length} (expect ${baselines.no_from_links})`,
|
||||
);
|
||||
console.log(`layout_fingerprint : ${layoutFingerprint}`);
|
||||
console.log(
|
||||
`scan_scaling_ratio : ${scanScaling.toFixed(3)} (budget <= ${baselines.scan_scaling_budget}; ~1.0 is linear)`,
|
||||
);
|
||||
console.log(
|
||||
`prefilter_advantage : ${prefilterAdvantage.toFixed(3)} (floor >= ${baselines.prefilter_advantage_floor})`,
|
||||
);
|
||||
console.log(
|
||||
`reps : ${REPS} from ${fromMs.toFixed(2)}ms / no_from ${noFromMs.toFixed(2)}ms / mixed ${smallMs.toFixed(2)}ms / 4x ${largeMs.toFixed(2)}ms`,
|
||||
);
|
||||
|
||||
if (process.argv.includes('--check')) {
|
||||
let failed = false;
|
||||
|
||||
if (layoutFingerprint !== baselines.layout_fingerprint) {
|
||||
failed = true;
|
||||
console.error(
|
||||
`\nFAIL layout_fingerprint: ${layoutFingerprint}\n` +
|
||||
` expected ${baselines.layout_fingerprint}\n` +
|
||||
` The from-corpus contract set moved. Explain it; do not re-baseline alone.`,
|
||||
);
|
||||
}
|
||||
|
||||
if (fromResult.links.length !== baselines.from_links) {
|
||||
failed = true;
|
||||
console.error(
|
||||
`\nFAIL from_links: ${fromResult.links.length}, expected exactly ${baselines.from_links}.`,
|
||||
);
|
||||
}
|
||||
if (lookalikeResult.links.length !== baselines.lookalike_links) {
|
||||
failed = true;
|
||||
console.error(
|
||||
`\nFAIL lookalike_links: ${lookalikeResult.links.length}, expected exactly ${baselines.lookalike_links}.\n` +
|
||||
` Closed or unclosed docstring lookalikes leaked a workspace contract.`,
|
||||
);
|
||||
}
|
||||
if (noFromResult.links.length !== baselines.no_from_links) {
|
||||
failed = true;
|
||||
console.error(
|
||||
`\nFAIL no_from_links: ${noFromResult.links.length}, expected exactly ${baselines.no_from_links}.`,
|
||||
);
|
||||
}
|
||||
|
||||
if (
|
||||
FROM_FILES !== baselines.from_files ||
|
||||
NO_FROM_FILES !== baselines.no_from_files ||
|
||||
LOOKALIKE_FILES !== baselines.lookalike_files
|
||||
) {
|
||||
failed = true;
|
||||
console.error(
|
||||
`\nFAIL shape: from_files ${FROM_FILES} (expected ${baselines.from_files}), ` +
|
||||
`no_from_files ${NO_FROM_FILES} (expected ${baselines.no_from_files}), ` +
|
||||
`lookalike_files ${LOOKALIKE_FILES} (expected ${baselines.lookalike_files}).\n` +
|
||||
` The corpus must stay large enough that the link arms still measure a walk.`,
|
||||
);
|
||||
}
|
||||
|
||||
if (scanScaling > baselines.scan_scaling_budget) {
|
||||
failed = true;
|
||||
console.error(
|
||||
`\nFAIL scan_scaling_ratio: ${scanScaling.toFixed(3)} exceeds ` +
|
||||
`${baselines.scan_scaling_budget} (~1.0 is linear).\n` +
|
||||
` Re-run on an idle machine before investigating, and check \`reps\` first.`,
|
||||
);
|
||||
}
|
||||
|
||||
if (prefilterAdvantage < baselines.prefilter_advantage_floor) {
|
||||
failed = true;
|
||||
console.error(
|
||||
`\nFAIL prefilter_advantage: ${prefilterAdvantage.toFixed(3)} below ` +
|
||||
`${baselines.prefilter_advantage_floor}.\n` +
|
||||
` The no-from corpus should stay cheaper than the from-corpus. A collapse\n` +
|
||||
` toward 1.0 usually means every file is being tree-sitter parsed again.`,
|
||||
);
|
||||
}
|
||||
|
||||
if (failed) process.exit(1);
|
||||
console.log('\nOK — within budget.');
|
||||
}
|
||||
} finally {
|
||||
for (const root of roots) {
|
||||
rmSync(root, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
|
|
@ -2,13 +2,22 @@ import fs from 'node:fs/promises';
|
|||
import path from 'node:path';
|
||||
import type { CypherExecutor } from '../contract-extractor.js';
|
||||
import type { GroupManifestLink, ContractRole } from '../types.js';
|
||||
import { getPythonParser } from '../../ingestion/languages/python/query.js';
|
||||
import { getMaxFileSizeBytes } from '../../ingestion/utils/max-file-size.js';
|
||||
import {
|
||||
ParseTimeoutError,
|
||||
parseHadErrors,
|
||||
parseSourceSafe,
|
||||
} from '../../tree-sitter/safe-parse.js';
|
||||
import {
|
||||
shouldIgnorePath,
|
||||
loadIgnoreRules,
|
||||
isHardcodedIgnoredDirectoryAtPath,
|
||||
} from '../../../config/ignore-service.js';
|
||||
import { readSafeBounded } from './fs-utils.js';
|
||||
|
||||
import { logger } from '../../logger.js';
|
||||
|
||||
interface PythonPackageMeta {
|
||||
name: string;
|
||||
importName: string;
|
||||
|
|
@ -18,9 +27,8 @@ interface PythonPackageMeta {
|
|||
}
|
||||
|
||||
interface ImportedSymbol {
|
||||
packageName: string;
|
||||
importName: string;
|
||||
symbolName: string;
|
||||
filePath: string;
|
||||
}
|
||||
|
||||
async function parsePythonManifest(
|
||||
|
|
@ -51,7 +59,7 @@ function parsePyproject(
|
|||
const nameMatch = content.match(/^\[project\]\s*\n(?:[^\n\[]*\n)*?name\s*=\s*"([^"]+)"/m);
|
||||
if (!nameMatch) return null;
|
||||
const name = nameMatch[1];
|
||||
const importName = name.replace(/-/g, '_');
|
||||
const importName = toPythonImportName(name);
|
||||
|
||||
const deps: string[] = [];
|
||||
const depsMatch = content.match(/^\[project\]\s*\n[\s\S]*?dependencies\s*=\s*\[([\s\S]*?)\]/m);
|
||||
|
|
@ -79,7 +87,7 @@ function parseSetupPy(
|
|||
const nameMatch = content.match(/name\s*=\s*['"]([^'"]+)['"]/);
|
||||
if (!nameMatch) return null;
|
||||
const name = nameMatch[1];
|
||||
const importName = name.replace(/-/g, '_');
|
||||
const importName = toPythonImportName(name);
|
||||
|
||||
const deps: string[] = [];
|
||||
const installMatch = content.match(/install_requires\s*=\s*\[([\s\S]*?)\]/);
|
||||
|
|
@ -97,51 +105,95 @@ function extractPepName(spec: string): string {
|
|||
return spec.split(/[><=!~;\[]/)[0].trim();
|
||||
}
|
||||
|
||||
function toPythonImportName(name: string): string {
|
||||
return name.replace(/-/g, '_');
|
||||
}
|
||||
|
||||
function collectFromImportNames(
|
||||
node: {
|
||||
namedChildCount: number;
|
||||
namedChild(index: number): {
|
||||
id: number;
|
||||
type: string;
|
||||
text: string;
|
||||
childForFieldName(name: string): { text: string } | null;
|
||||
} | null;
|
||||
},
|
||||
moduleNodeId: number,
|
||||
): string[] {
|
||||
const symbols: string[] = [];
|
||||
for (let i = 0; i < node.namedChildCount; i++) {
|
||||
const child = node.namedChild(i);
|
||||
if (!child || child.id === moduleNodeId) continue;
|
||||
if (child.type === 'dotted_name') {
|
||||
symbols.push(child.text);
|
||||
} else if (child.type === 'aliased_import') {
|
||||
const imported = child.childForFieldName('name');
|
||||
if (imported) symbols.push(imported.text);
|
||||
}
|
||||
}
|
||||
return symbols;
|
||||
}
|
||||
|
||||
/** Cheap skip before tree-sitter: real `from <mod>` tokens, not `fromage`. */
|
||||
function hasFromImportToken(content: string): boolean {
|
||||
return /(?:^|[\s;])from\s+\S/.test(content);
|
||||
}
|
||||
|
||||
async function scanPythonImports(
|
||||
repoPath: string,
|
||||
knownPackages: Map<string, string>,
|
||||
knownPackages: Set<string>,
|
||||
): Promise<ImportedSymbol[]> {
|
||||
const results: ImportedSymbol[] = [];
|
||||
const sourceFiles = await findPythonFiles(repoPath);
|
||||
const parser = getPythonParser();
|
||||
const maxFileSizeBytes = getMaxFileSizeBytes();
|
||||
|
||||
for (const relFile of sourceFiles) {
|
||||
const absPath = path.join(repoPath, relFile);
|
||||
let content: string;
|
||||
const content = await readSafeBounded(repoPath, relFile, maxFileSizeBytes);
|
||||
if (content == null || !hasFromImportToken(content)) continue;
|
||||
|
||||
let tree;
|
||||
try {
|
||||
content = await fs.readFile(absPath, 'utf-8');
|
||||
} catch {
|
||||
continue;
|
||||
tree = parseSourceSafe(parser, content, undefined, undefined, relFile);
|
||||
} catch (error) {
|
||||
if (error instanceof ParseTimeoutError) {
|
||||
logger.warn(
|
||||
{ file: relFile },
|
||||
'python-workspace-extractor: parse timed out, skipping file',
|
||||
);
|
||||
continue;
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
const degraded = parseHadErrors(tree);
|
||||
const visit = (node: (typeof tree)['rootNode']): void => {
|
||||
if (node.type !== 'import_from_statement') {
|
||||
for (let i = 0; i < node.namedChildCount; i++) {
|
||||
const child = node.namedChild(i);
|
||||
if (child) visit(child);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// from <pkg> import Foo, Bar
|
||||
// from <pkg>.module import Foo
|
||||
const fromImportRegex = /^from\s+(\w[\w.]*)\s+import\s+(.+)/gm;
|
||||
let match;
|
||||
while ((match = fromImportRegex.exec(content)) !== null) {
|
||||
const modulePath = match[1];
|
||||
const importClause = match[2];
|
||||
// Error recovery can promote unclosed-docstring lookalikes into real
|
||||
// import_from_statement nodes. Keep column-0 imports; drop indented ones
|
||||
// on a degraded tree so function-local discovery stays on clean parses.
|
||||
if (degraded && node.startPosition.column > 0) return;
|
||||
|
||||
const moduleNode = node.childForFieldName('module_name');
|
||||
const modulePath = moduleNode?.text;
|
||||
if (!modulePath) return;
|
||||
const rootModule = modulePath.split('.')[0];
|
||||
const originalName = knownPackages.get(rootModule);
|
||||
if (!originalName) continue;
|
||||
if (!knownPackages.has(rootModule)) return;
|
||||
|
||||
if (importClause.trim() === '(') continue;
|
||||
|
||||
const symbols = importClause
|
||||
.replace(/\(|\)/g, '')
|
||||
.split(',')
|
||||
.map((s) => {
|
||||
const trimmed = s.trim();
|
||||
const asMatch = trimmed.match(/^(\S+)\s+as\s+/);
|
||||
return asMatch ? asMatch[1] : trimmed;
|
||||
})
|
||||
.filter(Boolean);
|
||||
|
||||
for (const sym of symbols) {
|
||||
for (const sym of collectFromImportNames(node, moduleNode.id)) {
|
||||
if (isPascalCase(sym)) {
|
||||
results.push({ packageName: originalName, symbolName: sym, filePath: relFile });
|
||||
results.push({ importName: rootModule, symbolName: sym });
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
visit(tree.rootNode);
|
||||
}
|
||||
|
||||
return results;
|
||||
|
|
@ -224,21 +276,14 @@ export async function extractPythonWorkspaceLinks(
|
|||
const seen = new Set<string>();
|
||||
|
||||
for (const [, pkg] of packagesByGroupPath) {
|
||||
const normalizedDeps = pkg.workspaceDeps.map((d) => d.replace(/-/g, '_'));
|
||||
const normalizedDeps = pkg.workspaceDeps.map(toPythonImportName);
|
||||
const groupPkgDeps = normalizedDeps.filter((d) => packagesByImportName.has(d));
|
||||
if (groupPkgDeps.length === 0) continue;
|
||||
|
||||
const knownPackages = new Map<string, string>();
|
||||
for (const dep of groupPkgDeps) {
|
||||
const meta = packagesByImportName.get(dep);
|
||||
if (meta) knownPackages.set(dep, meta.name);
|
||||
}
|
||||
|
||||
const imports = await scanPythonImports(pkg.repoPath, knownPackages);
|
||||
const imports = await scanPythonImports(pkg.repoPath, new Set(groupPkgDeps));
|
||||
|
||||
for (const imp of imports) {
|
||||
const providerImportName = imp.packageName.replace(/-/g, '_');
|
||||
const providerPkg = packagesByImportName.get(providerImportName);
|
||||
const providerPkg = packagesByImportName.get(imp.importName);
|
||||
if (!providerPkg) continue;
|
||||
|
||||
const qualifiedContract = `${providerPkg.name}::${imp.symbolName}`;
|
||||
|
|
|
|||
|
|
@ -21,6 +21,44 @@ describe('PythonWorkspaceExtractor', () => {
|
|||
await fs.writeFile(absPath, content, 'utf-8');
|
||||
}
|
||||
|
||||
function twoRepos(aDir: string, aName: string, bDir: string, bName: string) {
|
||||
return {
|
||||
repos: { [aDir]: aName, [bDir]: bName },
|
||||
repoPaths: new Map([
|
||||
[aDir, path.join(tmpDir, aDir)],
|
||||
[bDir, path.join(tmpDir, bDir)],
|
||||
]),
|
||||
};
|
||||
}
|
||||
|
||||
async function writeDatalibMyapp(appMainPy: string) {
|
||||
await writeFile(
|
||||
'lib/pyproject.toml',
|
||||
'[project]\nname = "datalib"\nversion = "0.1.0"\ndependencies = []\n',
|
||||
);
|
||||
await writeFile('lib/datalib/models.py', 'class Record: pass\n');
|
||||
await writeFile(
|
||||
'app/pyproject.toml',
|
||||
'[project]\nname = "myapp"\nversion = "0.1.0"\ndependencies = ["datalib"]\n',
|
||||
);
|
||||
await writeFile('app/myapp/main.py', appMainPy);
|
||||
return twoRepos('lib', 'datalib', 'app', 'myapp');
|
||||
}
|
||||
|
||||
async function writeUtilsMyapp(appMainPy: string) {
|
||||
await writeFile(
|
||||
'lib/pyproject.toml',
|
||||
'[project]\nname = "utils"\nversion = "0.1.0"\ndependencies = []\n',
|
||||
);
|
||||
await writeFile('lib/utils/__init__.py', 'def helper(): pass\nclass Config: pass\n');
|
||||
await writeFile(
|
||||
'app/pyproject.toml',
|
||||
'[project]\nname = "myapp"\nversion = "0.1.0"\ndependencies = ["utils"]\n',
|
||||
);
|
||||
await writeFile('app/myapp/main.py', appMainPy);
|
||||
return twoRepos('lib', 'utils', 'app', 'myapp');
|
||||
}
|
||||
|
||||
it('discovers cross-package imports via pyproject.toml', async () => {
|
||||
await writeFile(
|
||||
'models/pyproject.toml',
|
||||
|
|
@ -128,23 +166,7 @@ describe('PythonWorkspaceExtractor', () => {
|
|||
});
|
||||
|
||||
it('handles submodule imports (from pkg.sub import Class)', async () => {
|
||||
await writeFile(
|
||||
'lib/pyproject.toml',
|
||||
'[project]\nname = "datalib"\nversion = "0.1.0"\ndependencies = []\n',
|
||||
);
|
||||
await writeFile('lib/datalib/models.py', 'class Record: pass\n');
|
||||
|
||||
await writeFile(
|
||||
'app/pyproject.toml',
|
||||
'[project]\nname = "myapp"\nversion = "0.1.0"\ndependencies = [\n "datalib",\n]\n',
|
||||
);
|
||||
await writeFile('app/myapp/main.py', 'from datalib.models import Record\n');
|
||||
|
||||
const repos = { lib: 'datalib', app: 'myapp' };
|
||||
const repoPaths = new Map([
|
||||
['lib', path.join(tmpDir, 'lib')],
|
||||
['app', path.join(tmpDir, 'app')],
|
||||
]);
|
||||
const { repos, repoPaths } = await writeDatalibMyapp('from datalib.models import Record\n');
|
||||
|
||||
const result = await extractPythonWorkspaceLinks(repos, repoPaths);
|
||||
|
||||
|
|
@ -152,24 +174,61 @@ describe('PythonWorkspaceExtractor', () => {
|
|||
expect(result.links[0].contract).toBe('datalib::Record');
|
||||
});
|
||||
|
||||
it('discovers function-local imports', async () => {
|
||||
const { repos, repoPaths } = await writeDatalibMyapp(
|
||||
'def load_record():\n from datalib.models import Record\n return Record()\n',
|
||||
);
|
||||
|
||||
const result = await extractPythonWorkspaceLinks(repos, repoPaths);
|
||||
|
||||
expect(result.links).toHaveLength(1);
|
||||
expect(result.links[0].contract).toBe('datalib::Record');
|
||||
});
|
||||
|
||||
it('ignores import-shaped text inside indented docstrings', async () => {
|
||||
const { repos, repoPaths } = await writeDatalibMyapp(
|
||||
'def describe():\n """Example:\n from datalib.models import Record\n """\n return None\n',
|
||||
);
|
||||
|
||||
const result = await extractPythonWorkspaceLinks(repos, repoPaths);
|
||||
|
||||
expect(result.links).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('ignores import-shaped text inside an unclosed indented docstring', async () => {
|
||||
const { repos, repoPaths } = await writeDatalibMyapp(
|
||||
'def describe():\n """\n from datalib.models import Record\n',
|
||||
);
|
||||
|
||||
const result = await extractPythonWorkspaceLinks(repos, repoPaths);
|
||||
|
||||
expect(result.links).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('discovers parenthesized function-local imports', async () => {
|
||||
const { repos, repoPaths } = await writeDatalibMyapp(
|
||||
'def load_record():\n from datalib.models import (\n Record,\n )\n return Record()\n',
|
||||
);
|
||||
|
||||
const result = await extractPythonWorkspaceLinks(repos, repoPaths);
|
||||
|
||||
expect(result.links).toHaveLength(1);
|
||||
expect(result.links[0].contract).toBe('datalib::Record');
|
||||
});
|
||||
|
||||
it('keeps PascalCase from a function-local mixed import', async () => {
|
||||
const { repos, repoPaths } = await writeUtilsMyapp(
|
||||
'def load():\n from utils import helper, Config\n return Config()\n',
|
||||
);
|
||||
|
||||
const result = await extractPythonWorkspaceLinks(repos, repoPaths);
|
||||
|
||||
expect(result.links).toHaveLength(1);
|
||||
expect(result.links[0].contract).toBe('utils::Config');
|
||||
});
|
||||
|
||||
it('ignores snake_case imports (functions, not types)', async () => {
|
||||
await writeFile(
|
||||
'lib/pyproject.toml',
|
||||
'[project]\nname = "utils"\nversion = "0.1.0"\ndependencies = []\n',
|
||||
);
|
||||
await writeFile('lib/utils/__init__.py', 'def helper(): pass\nclass Config: pass\n');
|
||||
|
||||
await writeFile(
|
||||
'app/pyproject.toml',
|
||||
'[project]\nname = "myapp"\nversion = "0.1.0"\ndependencies = [\n "utils",\n]\n',
|
||||
);
|
||||
await writeFile('app/myapp/main.py', 'from utils import helper, Config\n');
|
||||
|
||||
const repos = { lib: 'utils', app: 'myapp' };
|
||||
const repoPaths = new Map([
|
||||
['lib', path.join(tmpDir, 'lib')],
|
||||
['app', path.join(tmpDir, 'app')],
|
||||
]);
|
||||
const { repos, repoPaths } = await writeUtilsMyapp('from utils import helper, Config\n');
|
||||
|
||||
const result = await extractPythonWorkspaceLinks(repos, repoPaths);
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue