diff --git a/.github/workflows/ci-tests.yml b/.github/workflows/ci-tests.yml index 20657d9b2..386fd9e1d 100644 --- a/.github/workflows/ci-tests.yml +++ b/.github/workflows/ci-tests.yml @@ -599,6 +599,15 @@ jobs: run: node --import tsx bench/parse-dispatch-rounds/measure.mjs --check working-directory: gitnexus + - name: Python workspace import-scan guards (#3254) + if: ${{ !cancelled() }} + # Build-free: same baseline approach as parse-dispatch-rounds — + # exact link/lookalike floors plus a fingerprint, then ratio timing + # only (scan scaling and from-token prefilter advantage). See + # bench/python-workspace-import-scan/measure.mjs. + run: node --import tsx bench/python-workspace-import-scan/measure.mjs --check + working-directory: gitnexus + - name: C++ qualified-namespace resolution guards (#2788) if: ${{ !cancelled() }} # Build-free: asserts resolveCppQualifiedNamespaceMember resolves an diff --git a/gitnexus/bench/python-workspace-import-scan/baselines.json b/gitnexus/bench/python-workspace-import-scan/baselines.json new file mode 100644 index 000000000..0213561e0 --- /dev/null +++ b/gitnexus/bench/python-workspace-import-scan/baselines.json @@ -0,0 +1,37 @@ +{ + "_what": "Baselines for bench/python-workspace-import-scan/measure.mjs --check. Guards Python workspace from-import scanning after #3254: function-local discovery, lookalike rejection, and parse-volume. Same approach as bench/parse-dispatch-rounds/baselines.json — exact floors and a fingerprint first; the only timing arms are ratios.", + + "_triage": "READ THIS BEFORE RE-RUNNING. from_links, lookalike_links, no_from_links, from_files, no_from_files, lookalike_files and layout_fingerprint are DETERMINISTIC: a re-run never changes them, and none may be re-baselined to make CI green. scan_scaling_ratio is an UPPER timing arm; prefilter_advantage is a FLOOR timing arm. Runner contention dominates both, so re-run on an idle machine before investigating and read the reported `reps` first. If exactly one arm fails and it is a timing arm, suspect the machine.", + + "from_files": 40, + "no_from_files": 40, + "lookalike_files": 8, + "_shape_note": "THE FLOOR. Without these three, every arm below is a ceiling over nothing. from_links only asserts something while the corpus still walks many files. Shrink it to one happy-path module and from_links still reads 1 and still passes, asserting a property the corpus no longer has.", + + "from_links": 1, + "lookalike_links": 0, + "no_from_links": 0, + "_links_note": "Exact contract counts. from_links=1 is the deduped function-local Record import. lookalike_links=0 pins closed AND unclosed indented docstring examples. no_from_links=0 pins files that have no from-token.", + + "layout_fingerprint": "02e51be915083909cf231688c935cfe11bb52c7c781a70532859ad4a0bb9b42e", + "_layout_fingerprint_note": "sha256 over sorted from|to|contract rows on the from-corpus. A change here is a BEHAVIOUR change — the extractor returned a different contract set. Explain it, never re-baseline it alone.", + + "scan_scaling_budget": 1.6, + "_scan_scaling_note": "(t_4n / t_n) / 4 for extractPythonWorkspaceLinks over the mixed corpus; ~1.0 is linear. A RATIO rather than a millisecond ceiling, deliberately: wall-clock is runner-speed-dependent, and this repo has already been bitten by a fixed ms budget. Budget is 1.6, matching parse-dispatch-rounds' pack_scaling_budget. min-of-15 estimator.", + + "prefilter_advantage_floor": 3, + "_prefilter_note": "t_from / t_nofrom at the same file count and body size. The no-from corpus is large Python with no `from` token. A working prefilter keeps that arm cheap (~5x here). Removing the prefilter collapses the ratio toward 1 because both sides tree-sitter parse. Floor is 3 — about 1.6x below the measured minimum, same headroom philosophy as the 1.6 upper budget.", + + "_measured": { + "scan_scaling_ratio": 0.994, + "scan_scaling_ratio_samples": [0.994, 0.963, 0.977, 0.986, 0.889], + "prefilter_advantage": 5.411, + "prefilter_advantage_samples": [5.197, 4.955, 5.057, 5.124, 5.411], + "from_ms": 86.43, + "no_from_ms": 16.35, + "mixed_ms": 112.1, + "large_ms_4x": 398.76, + "reps": 15 + }, + "_measured_note": "Maxima (and sample lists) over 5 consecutive local runs using the min-of-15 estimator from parse-dispatch-rounds. Milliseconds are diagnostic context only — nothing gates on them." +} diff --git a/gitnexus/bench/python-workspace-import-scan/measure.mjs b/gitnexus/bench/python-workspace-import-scan/measure.mjs new file mode 100644 index 000000000..fbd351f4a --- /dev/null +++ b/gitnexus/bench/python-workspace-import-scan/measure.mjs @@ -0,0 +1,270 @@ +/** + * Build-free bench for Python workspace from-import scanning (#3254). + * + * WHY THIS EXISTS. `scanPythonImports` now tree-sitter-parses every `.py` file + * that looks like it has a `from` import. Graph output does not show "how many + * files we parsed" or "we skipped an unclosed-docstring lookalike" — a revert + * to column-0 regex, a dropped `parseHadErrors` guard, or a removed `from` + * prefilter can still emit the same one contract on a tiny fixture. This file + * is the same shape as `bench/parse-dispatch-rounds`: exact floors first, one + * ratio timing arm, never a millisecond ceiling. + * + * ARMS: + * + * - `from_links` / `lookalike_links` / `no_from_links` — EXACT. The + * correctness floor. The from-corpus must still discover `datalib::Record`. + * Closed + unclosed docstring lookalikes must stay at 0. Files with no + * `from` token must stay at 0. + * + * - `from_files` / `no_from_files` / `lookalike_files` — EXACT, and they are + * the FLOOR. `from_links === 1` only asserts something while the corpus + * still has many files to walk. Shrink it to one happy-path file and the + * link arm still passes, gating a property the corpus no longer has. + * + * - `layout_fingerprint` — EXACT. sha256 over sorted `from|to|contract` rows + * on the from-corpus. Catches a contract-set change that leaves the count + * intact. + * + * - `scan_scaling_ratio` — the only mixed-corpus timing arm, a RATIO not a + * millisecond ceiling. `(t_4n / t_n) / 4` divides the machine out; ~1.0 is + * linear. Superlinear AST work over file count lands here. + * + * - `prefilter_advantage` — `t_from / t_nofrom` at the same file count. The + * no-from corpus is large Python with no `from` token. A working prefilter + * keeps that arm cheap; removing it collapses the ratio toward 1 because + * both sides parse. + * + * Usage: + * node --import tsx bench/python-workspace-import-scan/measure.mjs + * node --import tsx bench/python-workspace-import-scan/measure.mjs --check + */ +import { createHash } from 'node:crypto'; +import { mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; +import { tmpdir } from 'node:os'; +import path from 'node:path'; +import { performance } from 'node:perf_hooks'; +import { extractPythonWorkspaceLinks } from '../../src/core/group/extractors/python-workspace-extractor.ts'; + +const baselines = JSON.parse(readFileSync(new URL('./baselines.json', import.meta.url), 'utf8')); + +const REPS = 15; +const FROM_FILES = 40; +const NO_FROM_FILES = 40; +const LOOKALIKE_FILES = 8; +const NO_FROM_BODY = Array.from( + { length: 80 }, + (_, i) => `def helper_${i}(x):\n return x + ${i}\n`, +).join('\n'); + +function writeWorkspace(root, { fromCount, noFromCount, lookalikeCount }) { + mkdirSync(path.join(root, 'lib', 'datalib'), { recursive: true }); + writeFileSync( + path.join(root, 'lib', 'pyproject.toml'), + '[project]\nname = "datalib"\nversion = "0.1.0"\ndependencies = []\n', + ); + writeFileSync(path.join(root, 'lib', 'datalib', 'models.py'), 'class Record: pass\n'); + + mkdirSync(path.join(root, 'app', 'myapp'), { recursive: true }); + writeFileSync( + path.join(root, 'app', 'pyproject.toml'), + '[project]\nname = "myapp"\nversion = "0.1.0"\ndependencies = ["datalib"]\n', + ); + + for (let i = 0; i < fromCount; i++) { + writeFileSync( + path.join(root, 'app', 'myapp', `from_${i}.py`), + `${NO_FROM_BODY}\ndef load_${i}():\n from datalib.models import Record\n return Record()\n`, + ); + } + for (let i = 0; i < noFromCount; i++) { + writeFileSync(path.join(root, 'app', 'myapp', `plain_${i}.py`), NO_FROM_BODY); + } + for (let i = 0; i < lookalikeCount; i++) { + const closed = i % 2 === 0; + writeFileSync( + path.join(root, 'app', 'myapp', `doc_${i}.py`), + closed + ? `def describe_${i}():\n """Example:\n from datalib.models import Record\n """\n return None\n` + : `def describe_${i}():\n """\n from datalib.models import Record\n`, + ); + } + + return { + repos: { lib: 'datalib', app: 'myapp' }, + repoPaths: new Map([ + ['lib', path.join(root, 'lib')], + ['app', path.join(root, 'app')], + ]), + }; +} + +async function scan(workspace) { + return extractPythonWorkspaceLinks(workspace.repos, workspace.repoPaths); +} + +function fingerprint(links) { + return createHash('sha256') + .update( + links + .map((l) => `${l.from}|${l.to}|${l.contract}`) + .sort() + .join('\n'), + ) + .digest('hex'); +} + +async function fastest(fn, reps) { + await fn(); + let best = Infinity; + for (let r = 0; r < reps; r++) { + const t0 = performance.now(); + await fn(); + best = Math.min(best, performance.now() - t0); + } + return best; +} + +const roots = []; +function workspace(opts) { + const root = mkdtempSync(path.join(tmpdir(), 'gn-py-ws-bench-')); + roots.push(root); + return writeWorkspace(root, opts); +} + +try { + const fromWs = workspace({ + fromCount: FROM_FILES, + noFromCount: 0, + lookalikeCount: 0, + }); + const noFromWs = workspace({ + fromCount: 0, + noFromCount: NO_FROM_FILES, + lookalikeCount: 0, + }); + const lookalikeWs = workspace({ + fromCount: 0, + noFromCount: 0, + lookalikeCount: LOOKALIKE_FILES, + }); + const mixedWs = workspace({ + fromCount: FROM_FILES, + noFromCount: NO_FROM_FILES, + lookalikeCount: LOOKALIKE_FILES, + }); + const mixed4x = workspace({ + fromCount: FROM_FILES * 4, + noFromCount: NO_FROM_FILES * 4, + lookalikeCount: LOOKALIKE_FILES * 4, + }); + + const [fromResult, noFromResult, lookalikeResult] = await Promise.all([ + scan(fromWs), + scan(noFromWs), + scan(lookalikeWs), + ]); + const layoutFingerprint = fingerprint(fromResult.links); + + const fromMs = await fastest(() => scan(fromWs), REPS); + const noFromMs = await fastest(() => scan(noFromWs), REPS); + const smallMs = await fastest(() => scan(mixedWs), REPS); + const largeMs = await fastest(() => scan(mixed4x), REPS); + const scanScaling = largeMs / smallMs / 4; + const prefilterAdvantage = noFromMs === 0 ? Infinity : fromMs / noFromMs; + + console.log(`from_files : ${FROM_FILES} (expect ${baselines.from_files})`); + console.log(`no_from_files : ${NO_FROM_FILES} (expect ${baselines.no_from_files})`); + console.log(`lookalike_files : ${LOOKALIKE_FILES} (expect ${baselines.lookalike_files})`); + console.log( + `from_links : ${fromResult.links.length} (expect ${baselines.from_links})`, + ); + console.log( + `lookalike_links : ${lookalikeResult.links.length} (expect ${baselines.lookalike_links})`, + ); + console.log( + `no_from_links : ${noFromResult.links.length} (expect ${baselines.no_from_links})`, + ); + console.log(`layout_fingerprint : ${layoutFingerprint}`); + console.log( + `scan_scaling_ratio : ${scanScaling.toFixed(3)} (budget <= ${baselines.scan_scaling_budget}; ~1.0 is linear)`, + ); + console.log( + `prefilter_advantage : ${prefilterAdvantage.toFixed(3)} (floor >= ${baselines.prefilter_advantage_floor})`, + ); + console.log( + `reps : ${REPS} from ${fromMs.toFixed(2)}ms / no_from ${noFromMs.toFixed(2)}ms / mixed ${smallMs.toFixed(2)}ms / 4x ${largeMs.toFixed(2)}ms`, + ); + + if (process.argv.includes('--check')) { + let failed = false; + + if (layoutFingerprint !== baselines.layout_fingerprint) { + failed = true; + console.error( + `\nFAIL layout_fingerprint: ${layoutFingerprint}\n` + + ` expected ${baselines.layout_fingerprint}\n` + + ` The from-corpus contract set moved. Explain it; do not re-baseline alone.`, + ); + } + + if (fromResult.links.length !== baselines.from_links) { + failed = true; + console.error( + `\nFAIL from_links: ${fromResult.links.length}, expected exactly ${baselines.from_links}.`, + ); + } + if (lookalikeResult.links.length !== baselines.lookalike_links) { + failed = true; + console.error( + `\nFAIL lookalike_links: ${lookalikeResult.links.length}, expected exactly ${baselines.lookalike_links}.\n` + + ` Closed or unclosed docstring lookalikes leaked a workspace contract.`, + ); + } + if (noFromResult.links.length !== baselines.no_from_links) { + failed = true; + console.error( + `\nFAIL no_from_links: ${noFromResult.links.length}, expected exactly ${baselines.no_from_links}.`, + ); + } + + if ( + FROM_FILES !== baselines.from_files || + NO_FROM_FILES !== baselines.no_from_files || + LOOKALIKE_FILES !== baselines.lookalike_files + ) { + failed = true; + console.error( + `\nFAIL shape: from_files ${FROM_FILES} (expected ${baselines.from_files}), ` + + `no_from_files ${NO_FROM_FILES} (expected ${baselines.no_from_files}), ` + + `lookalike_files ${LOOKALIKE_FILES} (expected ${baselines.lookalike_files}).\n` + + ` The corpus must stay large enough that the link arms still measure a walk.`, + ); + } + + if (scanScaling > baselines.scan_scaling_budget) { + failed = true; + console.error( + `\nFAIL scan_scaling_ratio: ${scanScaling.toFixed(3)} exceeds ` + + `${baselines.scan_scaling_budget} (~1.0 is linear).\n` + + ` Re-run on an idle machine before investigating, and check \`reps\` first.`, + ); + } + + if (prefilterAdvantage < baselines.prefilter_advantage_floor) { + failed = true; + console.error( + `\nFAIL prefilter_advantage: ${prefilterAdvantage.toFixed(3)} below ` + + `${baselines.prefilter_advantage_floor}.\n` + + ` The no-from corpus should stay cheaper than the from-corpus. A collapse\n` + + ` toward 1.0 usually means every file is being tree-sitter parsed again.`, + ); + } + + if (failed) process.exit(1); + console.log('\nOK — within budget.'); + } +} finally { + for (const root of roots) { + rmSync(root, { recursive: true, force: true }); + } +} diff --git a/gitnexus/src/core/group/extractors/python-workspace-extractor.ts b/gitnexus/src/core/group/extractors/python-workspace-extractor.ts index c07e5d8cc..47157e0b2 100644 --- a/gitnexus/src/core/group/extractors/python-workspace-extractor.ts +++ b/gitnexus/src/core/group/extractors/python-workspace-extractor.ts @@ -2,13 +2,22 @@ import fs from 'node:fs/promises'; import path from 'node:path'; import type { CypherExecutor } from '../contract-extractor.js'; import type { GroupManifestLink, ContractRole } from '../types.js'; +import { getPythonParser } from '../../ingestion/languages/python/query.js'; +import { getMaxFileSizeBytes } from '../../ingestion/utils/max-file-size.js'; +import { + ParseTimeoutError, + parseHadErrors, + parseSourceSafe, +} from '../../tree-sitter/safe-parse.js'; import { shouldIgnorePath, loadIgnoreRules, isHardcodedIgnoredDirectoryAtPath, } from '../../../config/ignore-service.js'; +import { readSafeBounded } from './fs-utils.js'; import { logger } from '../../logger.js'; + interface PythonPackageMeta { name: string; importName: string; @@ -18,9 +27,8 @@ interface PythonPackageMeta { } interface ImportedSymbol { - packageName: string; + importName: string; symbolName: string; - filePath: string; } async function parsePythonManifest( @@ -51,7 +59,7 @@ function parsePyproject( const nameMatch = content.match(/^\[project\]\s*\n(?:[^\n\[]*\n)*?name\s*=\s*"([^"]+)"/m); if (!nameMatch) return null; const name = nameMatch[1]; - const importName = name.replace(/-/g, '_'); + const importName = toPythonImportName(name); const deps: string[] = []; const depsMatch = content.match(/^\[project\]\s*\n[\s\S]*?dependencies\s*=\s*\[([\s\S]*?)\]/m); @@ -79,7 +87,7 @@ function parseSetupPy( const nameMatch = content.match(/name\s*=\s*['"]([^'"]+)['"]/); if (!nameMatch) return null; const name = nameMatch[1]; - const importName = name.replace(/-/g, '_'); + const importName = toPythonImportName(name); const deps: string[] = []; const installMatch = content.match(/install_requires\s*=\s*\[([\s\S]*?)\]/); @@ -97,51 +105,95 @@ function extractPepName(spec: string): string { return spec.split(/[><=!~;\[]/)[0].trim(); } +function toPythonImportName(name: string): string { + return name.replace(/-/g, '_'); +} + +function collectFromImportNames( + node: { + namedChildCount: number; + namedChild(index: number): { + id: number; + type: string; + text: string; + childForFieldName(name: string): { text: string } | null; + } | null; + }, + moduleNodeId: number, +): string[] { + const symbols: string[] = []; + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (!child || child.id === moduleNodeId) continue; + if (child.type === 'dotted_name') { + symbols.push(child.text); + } else if (child.type === 'aliased_import') { + const imported = child.childForFieldName('name'); + if (imported) symbols.push(imported.text); + } + } + return symbols; +} + +/** Cheap skip before tree-sitter: real `from ` tokens, not `fromage`. */ +function hasFromImportToken(content: string): boolean { + return /(?:^|[\s;])from\s+\S/.test(content); +} + async function scanPythonImports( repoPath: string, - knownPackages: Map, + knownPackages: Set, ): Promise { const results: ImportedSymbol[] = []; const sourceFiles = await findPythonFiles(repoPath); + const parser = getPythonParser(); + const maxFileSizeBytes = getMaxFileSizeBytes(); for (const relFile of sourceFiles) { - const absPath = path.join(repoPath, relFile); - let content: string; + const content = await readSafeBounded(repoPath, relFile, maxFileSizeBytes); + if (content == null || !hasFromImportToken(content)) continue; + + let tree; try { - content = await fs.readFile(absPath, 'utf-8'); - } catch { - continue; + tree = parseSourceSafe(parser, content, undefined, undefined, relFile); + } catch (error) { + if (error instanceof ParseTimeoutError) { + logger.warn( + { file: relFile }, + 'python-workspace-extractor: parse timed out, skipping file', + ); + continue; + } + throw error; } + const degraded = parseHadErrors(tree); + const visit = (node: (typeof tree)['rootNode']): void => { + if (node.type !== 'import_from_statement') { + for (let i = 0; i < node.namedChildCount; i++) { + const child = node.namedChild(i); + if (child) visit(child); + } + return; + } - // from import Foo, Bar - // from .module import Foo - const fromImportRegex = /^from\s+(\w[\w.]*)\s+import\s+(.+)/gm; - let match; - while ((match = fromImportRegex.exec(content)) !== null) { - const modulePath = match[1]; - const importClause = match[2]; + // Error recovery can promote unclosed-docstring lookalikes into real + // import_from_statement nodes. Keep column-0 imports; drop indented ones + // on a degraded tree so function-local discovery stays on clean parses. + if (degraded && node.startPosition.column > 0) return; + + const moduleNode = node.childForFieldName('module_name'); + const modulePath = moduleNode?.text; + if (!modulePath) return; const rootModule = modulePath.split('.')[0]; - const originalName = knownPackages.get(rootModule); - if (!originalName) continue; + if (!knownPackages.has(rootModule)) return; - if (importClause.trim() === '(') continue; - - const symbols = importClause - .replace(/\(|\)/g, '') - .split(',') - .map((s) => { - const trimmed = s.trim(); - const asMatch = trimmed.match(/^(\S+)\s+as\s+/); - return asMatch ? asMatch[1] : trimmed; - }) - .filter(Boolean); - - for (const sym of symbols) { + for (const sym of collectFromImportNames(node, moduleNode.id)) { if (isPascalCase(sym)) { - results.push({ packageName: originalName, symbolName: sym, filePath: relFile }); + results.push({ importName: rootModule, symbolName: sym }); } } - } + }; + visit(tree.rootNode); } return results; @@ -224,21 +276,14 @@ export async function extractPythonWorkspaceLinks( const seen = new Set(); for (const [, pkg] of packagesByGroupPath) { - const normalizedDeps = pkg.workspaceDeps.map((d) => d.replace(/-/g, '_')); + const normalizedDeps = pkg.workspaceDeps.map(toPythonImportName); const groupPkgDeps = normalizedDeps.filter((d) => packagesByImportName.has(d)); if (groupPkgDeps.length === 0) continue; - const knownPackages = new Map(); - for (const dep of groupPkgDeps) { - const meta = packagesByImportName.get(dep); - if (meta) knownPackages.set(dep, meta.name); - } - - const imports = await scanPythonImports(pkg.repoPath, knownPackages); + const imports = await scanPythonImports(pkg.repoPath, new Set(groupPkgDeps)); for (const imp of imports) { - const providerImportName = imp.packageName.replace(/-/g, '_'); - const providerPkg = packagesByImportName.get(providerImportName); + const providerPkg = packagesByImportName.get(imp.importName); if (!providerPkg) continue; const qualifiedContract = `${providerPkg.name}::${imp.symbolName}`; diff --git a/gitnexus/test/unit/group/python-workspace-extractor.test.ts b/gitnexus/test/unit/group/python-workspace-extractor.test.ts index 87b9ffcb3..995f8481d 100644 --- a/gitnexus/test/unit/group/python-workspace-extractor.test.ts +++ b/gitnexus/test/unit/group/python-workspace-extractor.test.ts @@ -21,6 +21,44 @@ describe('PythonWorkspaceExtractor', () => { await fs.writeFile(absPath, content, 'utf-8'); } + function twoRepos(aDir: string, aName: string, bDir: string, bName: string) { + return { + repos: { [aDir]: aName, [bDir]: bName }, + repoPaths: new Map([ + [aDir, path.join(tmpDir, aDir)], + [bDir, path.join(tmpDir, bDir)], + ]), + }; + } + + async function writeDatalibMyapp(appMainPy: string) { + await writeFile( + 'lib/pyproject.toml', + '[project]\nname = "datalib"\nversion = "0.1.0"\ndependencies = []\n', + ); + await writeFile('lib/datalib/models.py', 'class Record: pass\n'); + await writeFile( + 'app/pyproject.toml', + '[project]\nname = "myapp"\nversion = "0.1.0"\ndependencies = ["datalib"]\n', + ); + await writeFile('app/myapp/main.py', appMainPy); + return twoRepos('lib', 'datalib', 'app', 'myapp'); + } + + async function writeUtilsMyapp(appMainPy: string) { + await writeFile( + 'lib/pyproject.toml', + '[project]\nname = "utils"\nversion = "0.1.0"\ndependencies = []\n', + ); + await writeFile('lib/utils/__init__.py', 'def helper(): pass\nclass Config: pass\n'); + await writeFile( + 'app/pyproject.toml', + '[project]\nname = "myapp"\nversion = "0.1.0"\ndependencies = ["utils"]\n', + ); + await writeFile('app/myapp/main.py', appMainPy); + return twoRepos('lib', 'utils', 'app', 'myapp'); + } + it('discovers cross-package imports via pyproject.toml', async () => { await writeFile( 'models/pyproject.toml', @@ -128,23 +166,7 @@ describe('PythonWorkspaceExtractor', () => { }); it('handles submodule imports (from pkg.sub import Class)', async () => { - await writeFile( - 'lib/pyproject.toml', - '[project]\nname = "datalib"\nversion = "0.1.0"\ndependencies = []\n', - ); - await writeFile('lib/datalib/models.py', 'class Record: pass\n'); - - await writeFile( - 'app/pyproject.toml', - '[project]\nname = "myapp"\nversion = "0.1.0"\ndependencies = [\n "datalib",\n]\n', - ); - await writeFile('app/myapp/main.py', 'from datalib.models import Record\n'); - - const repos = { lib: 'datalib', app: 'myapp' }; - const repoPaths = new Map([ - ['lib', path.join(tmpDir, 'lib')], - ['app', path.join(tmpDir, 'app')], - ]); + const { repos, repoPaths } = await writeDatalibMyapp('from datalib.models import Record\n'); const result = await extractPythonWorkspaceLinks(repos, repoPaths); @@ -152,24 +174,61 @@ describe('PythonWorkspaceExtractor', () => { expect(result.links[0].contract).toBe('datalib::Record'); }); + it('discovers function-local imports', async () => { + const { repos, repoPaths } = await writeDatalibMyapp( + 'def load_record():\n from datalib.models import Record\n return Record()\n', + ); + + const result = await extractPythonWorkspaceLinks(repos, repoPaths); + + expect(result.links).toHaveLength(1); + expect(result.links[0].contract).toBe('datalib::Record'); + }); + + it('ignores import-shaped text inside indented docstrings', async () => { + const { repos, repoPaths } = await writeDatalibMyapp( + 'def describe():\n """Example:\n from datalib.models import Record\n """\n return None\n', + ); + + const result = await extractPythonWorkspaceLinks(repos, repoPaths); + + expect(result.links).toHaveLength(0); + }); + + it('ignores import-shaped text inside an unclosed indented docstring', async () => { + const { repos, repoPaths } = await writeDatalibMyapp( + 'def describe():\n """\n from datalib.models import Record\n', + ); + + const result = await extractPythonWorkspaceLinks(repos, repoPaths); + + expect(result.links).toHaveLength(0); + }); + + it('discovers parenthesized function-local imports', async () => { + const { repos, repoPaths } = await writeDatalibMyapp( + 'def load_record():\n from datalib.models import (\n Record,\n )\n return Record()\n', + ); + + const result = await extractPythonWorkspaceLinks(repos, repoPaths); + + expect(result.links).toHaveLength(1); + expect(result.links[0].contract).toBe('datalib::Record'); + }); + + it('keeps PascalCase from a function-local mixed import', async () => { + const { repos, repoPaths } = await writeUtilsMyapp( + 'def load():\n from utils import helper, Config\n return Config()\n', + ); + + const result = await extractPythonWorkspaceLinks(repos, repoPaths); + + expect(result.links).toHaveLength(1); + expect(result.links[0].contract).toBe('utils::Config'); + }); + it('ignores snake_case imports (functions, not types)', async () => { - await writeFile( - 'lib/pyproject.toml', - '[project]\nname = "utils"\nversion = "0.1.0"\ndependencies = []\n', - ); - await writeFile('lib/utils/__init__.py', 'def helper(): pass\nclass Config: pass\n'); - - await writeFile( - 'app/pyproject.toml', - '[project]\nname = "myapp"\nversion = "0.1.0"\ndependencies = [\n "utils",\n]\n', - ); - await writeFile('app/myapp/main.py', 'from utils import helper, Config\n'); - - const repos = { lib: 'utils', app: 'myapp' }; - const repoPaths = new Map([ - ['lib', path.join(tmpDir, 'lib')], - ['app', path.join(tmpDir, 'app')], - ]); + const { repos, repoPaths } = await writeUtilsMyapp('from utils import helper, Config\n'); const result = await extractPythonWorkspaceLinks(repos, repoPaths);