From 4f7697c43b1aff0662eae528fc8a1bc01db6a284 Mon Sep 17 00:00:00 2001 From: Sparsh <73558748+prajapatisparsh@users.noreply.github.com> Date: Tue, 2 Jun 2026 01:38:07 +0530 Subject: [PATCH] =?UTF-8?q?fix:=20COBOL=20parsing-layer=20coverage=20gaps?= =?UTF-8?q?=20=E2=80=94=20F17-F23=20(#1925)=20(#1959)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- gitnexus/bench/scope-capture/baselines.json | 4 +- .../src/core/ingestion/cobol-processor.ts | 59 ++- .../ingestion/cobol/cobol-preprocessor.ts | 335 ++++++++++++++---- .../ingestion/languages/cobol/captures.ts | 26 +- .../arithmetic-verbs.cbl | 18 + .../cobol-parsing-coverage/digit-leading.cbl | 17 + .../fixed-format-offset.cbl | 9 + .../cobol-parsing-coverage/free-format.cbl | 10 + .../cobol-parsing-coverage/move-subscript.cbl | 16 + .../multi-table-sql.cbl | 23 ++ .../cobol-parsing-coverage/perform-times.cbl | 26 ++ .../resolvers/cobol-parsing-coverage.test.ts | 222 ++++++++++++ .../test/integration/resolvers/cobol.test.ts | 8 +- 13 files changed, 691 insertions(+), 82 deletions(-) create mode 100644 gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/arithmetic-verbs.cbl create mode 100644 gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/digit-leading.cbl create mode 100644 gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/fixed-format-offset.cbl create mode 100644 gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/free-format.cbl create mode 100644 gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/move-subscript.cbl create mode 100644 gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/multi-table-sql.cbl create mode 100644 gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/perform-times.cbl create mode 100644 gitnexus/test/integration/resolvers/cobol-parsing-coverage.test.ts diff --git a/gitnexus/bench/scope-capture/baselines.json b/gitnexus/bench/scope-capture/baselines.json index d28a998b8..17e5d168a 100644 --- a/gitnexus/bench/scope-capture/baselines.json +++ b/gitnexus/bench/scope-capture/baselines.json @@ -6,9 +6,9 @@ "_rebaselined": "#1956 synth-widening: + go-qualified-base fixture; synthesizeGoInheritanceReferences now emits embeds for qualified_type (pkg.Base), generic_type (Box[T]), pointer, AND interface_type embeds (matching the #1940 legacy leg), reduced to bare names at parity. go-ambiguous gains an embed inherits capture. Linear (~1.01). (Earlier #1956: heritage-bearing scale source so the synth is gated at scale.)" }, "cobol": { - "fingerprint": "575016f329c0be29eb90db974f750d02a21b4a12515f7029bda312df713b27b0", + "fingerprint": "68ee0e95eb9f86f2d92ca35f730f4c2d4d83abc1b5241ae767ff3437780ec8d1", "scaling_budget": 1.5, - "_note": "COBOL has no inheritance construct, so its scale source stays flat; unchanged." + "_note": "Updated for F17-F23 fixes (P2: TIMES guard, ADD GIVING, SQL AS alias). See PR #1959." }, "c": { "fingerprint": "0de009bdbfe095f530fa87eb32bce6ab83092c904f26b3c8fe8d8ab587cf6dc9", diff --git a/gitnexus/src/core/ingestion/cobol-processor.ts b/gitnexus/src/core/ingestion/cobol-processor.ts index b7f3835b1..e7aa064de 100644 --- a/gitnexus/src/core/ingestion/cobol-processor.ts +++ b/gitnexus/src/core/ingestion/cobol-processor.ts @@ -25,6 +25,8 @@ import { import { expandCopies } from './cobol/cobol-copy-expander.js'; import { processJclFiles } from './cobol/jcl-processor.js'; +import { logger } from '../logger.js'; + // --------------------------------------------------------------------------- // File detection // --------------------------------------------------------------------------- @@ -60,6 +62,7 @@ export interface CobolProcessResult { sets: number; inspects: number; initializes: number; + arithmeticOps: number; } /** Returns true if the file is a COBOL or copybook file. */ @@ -114,6 +117,7 @@ export const processCobol = ( sets: 0, inspects: 0, initializes: 0, + arithmeticOps: 0, }; // ── 1. Separate programs, copybooks, and JCL ─────────────────────── @@ -174,7 +178,16 @@ export const processCobol = ( const moduleNodeIds = new Map(); // uppercase program name -> node id // ── 3. Process each COBOL program ────────────────────────────────── + const raw = parseInt(process.env.GITNEXUS_MAX_COBOL_FILE_SIZE_BYTES ?? '', 10); + const MAX_COBOL_FILE_SIZE = Number.isFinite(raw) && raw > 0 ? raw : 5 * 1024 * 1024; for (const file of programs) { + // File-size guard: skip excessively large files to prevent OOM + if (file.content.length > MAX_COBOL_FILE_SIZE) { + logger.warn( + `[cobol-processor] Skipping oversized file (${(file.content.length / 1024 / 1024).toFixed(1)}MB > ${(MAX_COBOL_FILE_SIZE / 1024 / 1024).toFixed(0)}MB): ${file.path}`, + ); + continue; + } const fileNodeId = generateId('File', file.path); // Skip if file node doesn't exist (structure-processor creates it) if (!graph.getNode(fileNodeId)) continue; @@ -214,6 +227,7 @@ export const processCobol = ( result.sets += extracted.sets.length; result.inspects += extracted.inspects.length; result.initializes += extracted.initializes.length; + result.arithmeticOps += extracted.arithmeticOps.length; } // ── 4. Second pass: resolve cross-program CALL targets ───────────── @@ -1226,7 +1240,9 @@ function mapToGraph( // ── MOVE data flow -> ACCESSES edges (read/write) ────────────── for (const move of extracted.moves) { - const fromPropId = dataItemMap.get(move.from.toUpperCase()); + // Strip any subscript from the source name for data item lookup + const fromBase = stripMoveSubscript(move.from); + const fromPropId = dataItemMap.get(fromBase.toUpperCase()); const callerId = scopedCallerLookup(move.caller, move.line); // One read edge per MOVE (regardless of number of targets) @@ -1243,7 +1259,8 @@ function mapToGraph( // One write edge per target for (const target of move.targets) { - const toPropId = dataItemMap.get(target.toUpperCase()); + const toBase = stripMoveSubscript(target); + const toPropId = dataItemMap.get(toBase.toUpperCase()); if (toPropId) { graph.addRelationship({ id: generateId('ACCESSES', `${callerId}->write->${target}:L${move.line}`), @@ -1257,6 +1274,39 @@ function mapToGraph( } } + // ── Arithmetic operations -> ACCESSES edges ────────────────── + for (const arith of extracted.arithmeticOps) { + const callerId = scopedCallerLookup(arith.caller, arith.line); + // Write edge to target variable + const targetBase = stripMoveSubscript(arith.target); + const targetPropId = dataItemMap.get(targetBase.toUpperCase()); + if (targetPropId) { + graph.addRelationship({ + id: generateId('ACCESSES', `${callerId}->arith-write->${arith.target}:L${arith.line}`), + type: 'ACCESSES', + sourceId: callerId, + targetId: targetPropId, + confidence: 0.9, + reason: 'cobol-arithmetic-write', + }); + } + // Read edge for each source operand + for (const src of arith.sources) { + const srcBase = stripMoveSubscript(src); + const srcPropId = dataItemMap.get(srcBase.toUpperCase()); + if (srcPropId) { + graph.addRelationship({ + id: generateId('ACCESSES', `${callerId}->arith-read->${src}:L${arith.line}`), + type: 'ACCESSES', + sourceId: callerId, + targetId: srcPropId, + confidence: 0.9, + reason: 'cobol-arithmetic-read', + }); + } + } + } + // ── File declarations -> Record nodes ────────────────────────── for (const fd of extracted.fileDeclarations) { const fdId = generateId('Record', `${filePath}:${fd.selectName}`); @@ -1398,6 +1448,11 @@ function mapToGraph( // Helpers // --------------------------------------------------------------------------- +/** Strip parenthesized subscript/reference-modification suffixes */ +function stripMoveSubscript(name: string): string { + return name.replace(/\([^)]*\)/g, '').trim(); +} + /** Find the enclosing program name for a given line number (innermost wins). */ function findOwningProgramName( lineNum: number, diff --git a/gitnexus/src/core/ingestion/cobol/cobol-preprocessor.ts b/gitnexus/src/core/ingestion/cobol/cobol-preprocessor.ts index 34be6bc03..4e28b6d2a 100644 --- a/gitnexus/src/core/ingestion/cobol/cobol-preprocessor.ts +++ b/gitnexus/src/core/ingestion/cobol/cobol-preprocessor.ts @@ -183,6 +183,19 @@ export interface CobolRegexResults { // Phase 4.1: INITIALIZE initializes: Array<{ target: string; line: number; caller: string | null }>; + + // Phase 4.2: Arithmetic operations (COMPUTE, ADD, SUBTRACT, MULTIPLY, DIVIDE) + arithmeticOps: Array<{ + verb: 'COMPUTE' | 'ADD' | 'SUBTRACT' | 'MULTIPLY' | 'DIVIDE'; + /** Target variable (written to) */ + target: string; + /** Source operand variables (read from) */ + sources: string[]; + line: number; + caller: string | null; + /** For ADD/SUBTRACT/MULTIPLY/DIVIDE with GIVING: the GIVING target */ + givingTarget?: string; + }>; } // --------------------------------------------------------------------------- @@ -306,36 +319,37 @@ const RE_SECTION = /\b(WORKING-STORAGE|LINKAGE|FILE|LOCAL-STORAGE|SCREEN|INPUT-OUTPUT|CONFIGURATION)\s+SECTION\b/i; // IDENTIFICATION DIVISION -const RE_PROGRAM_ID = /\bPROGRAM-ID\.\s*([A-Z][A-Z0-9-]*)(?:\s+IS\s+COMMON)?/i; -const RE_END_PROGRAM = /\bEND\s+PROGRAM\s+([A-Z][A-Z0-9-]*)\s*\./i; +const RE_PROGRAM_ID = /\bPROGRAM-ID\.\s*([A-Z0-9][A-Z0-9-]*)(?:\s+IS\s+COMMON)?/i; +const RE_END_PROGRAM = /\bEND\s+PROGRAM\s+([A-Z0-9][A-Z0-9-]*)\s*\./i; const RE_AUTHOR = /^\s+AUTHOR\.\s*(.+)/i; const RE_DATE_WRITTEN = /^\s+DATE-WRITTEN\.\s*(.+)/i; const RE_DATE_COMPILED = /^\s+DATE-COMPILED\.\s*(.+)/i; const RE_INSTALLATION = /^\s+INSTALLATION\.\s*(.+)/i; // ENVIRONMENT DIVISION — SELECT -const RE_SELECT_START = /\bSELECT\s+(?:OPTIONAL\s+)?([A-Z][A-Z0-9-]+)/i; +const RE_SELECT_START = /\bSELECT\s+(?:OPTIONAL\s+)?([A-Z0-9][A-Z0-9-]+)/i; // DATA DIVISION // ^\s* (not ^\s+) to support both fixed-format (indented) and free-format (trimmed) -const RE_FD = /^\s*(?:FD|SD|RD)\s+([A-Z][A-Z0-9-]+)/i; -const RE_DATA_ITEM = /^\s*(\d{1,2})\s+([A-Z][A-Z0-9-]+)\s*(.*)/i; -const RE_ANONYMOUS_REDEFINES = /^\s*(\d{1,2})\s+REDEFINES\s+([A-Z][A-Z0-9-]+)/i; -const RE_88_LEVEL = /^\s*88\s+([A-Z][A-Z0-9-]+)\s+VALUES?\s+(?:ARE\s+)?(.+)/i; +const RE_FD = /^\s*(?:FD|SD|RD)\s+([A-Z0-9][A-Z0-9-]+)/i; +const RE_DATA_ITEM = /^\s*(\d{1,2})\s+([A-Z0-9][A-Z0-9-]+)\s*(.*)/i; +const RE_ANONYMOUS_REDEFINES = /^\s*(\d{1,2})\s+REDEFINES\s+([A-Z0-9][A-Z0-9-]+)/i; +const RE_88_LEVEL = /^\s*88\s+([A-Z0-9][A-Z0-9-]+)\s+VALUES?\s+(?:ARE\s+)?(.+)/i; // PROCEDURE DIVISION // These patterns support both fixed-format (7 leading spaces) and free-format (any indentation) -const RE_PROC_SECTION = /^\s*([A-Z][A-Z0-9-]+)\s+SECTION(?:\s+\d+)?\.\s*$/i; -const RE_PROC_PARAGRAPH = /^\s*([A-Z][A-Z0-9-]+)\.\s*$/i; -const RE_PERFORM = /\bPERFORM\s+([A-Z][A-Z0-9-]+)(?:\s+(?:THRU|THROUGH)\s+([A-Z][A-Z0-9-]+))?/gi; +const RE_PROC_SECTION = /^\s*([A-Z0-9][A-Z0-9-]+)\s+SECTION(?:\s+\d+)?\.\s*$/i; +const RE_PROC_PARAGRAPH = /^\s*([A-Z0-9][A-Z0-9-]+)\.\s*$/i; +const RE_PERFORM = + /\bPERFORM\s+([A-Z0-9][A-Z0-9-]+)(?:\s+(?:THRU|THROUGH)\s+([A-Z0-9][A-Z0-9-]+))?/gi; // ALL DIVISIONS // Both double-quoted ("PROG") and single-quoted ('PROG') targets are valid COBOL. // Use separate alternation groups so quotes must match (prevents "PROG' false-matches). const RE_CALL = /\bCALL\s+(?:"([^"]+)"|'([^']+)')/gi; // Dynamic CALL via data item (no quotes): CALL WS-PROGRAM-NAME -const RE_CALL_DYNAMIC = /(? - const redefMatch = text.match(/\bREDEFINES\s+([A-Z][A-Z0-9-]+)/i); + const redefMatch = text.match(/\bREDEFINES\s+([A-Z0-9][A-Z0-9-]+)/i); if (redefMatch) { result.redefines = redefMatch[1]; } // OCCURS [TO ] [TIMES] [DEPENDING ON ] const occursMatch = text.match( - /\bOCCURS\s+(\d+)(?:\s+TO\s+(\d+))?\s*(?:TIMES\s*)?(?:DEPENDING\s+ON\s+([A-Z][A-Z0-9-]+(?:\s*\([^)]*\))?))?/i, + /\bOCCURS\s+(\d+)(?:\s+TO\s+(\d+))?\s*(?:TIMES\s*)?(?:DEPENDING\s+ON\s+([A-Z0-9][A-Z0-9-]+(?:\s*\([^)]*\))?))?/i, ); if (occursMatch) { result.occurs = parseInt(occursMatch[1], 10); @@ -653,7 +668,7 @@ function parseDataItemClauses(rest: string): { result.value = numMatch[1]; } else { // Try figurative constant or identifier - const identMatch = afterValue.match(/^([A-Z][A-Z0-9-]*)/i); + const identMatch = afterValue.match(/^([A-Z0-9][A-Z0-9-]*)/i); if (identMatch) result.value = identMatch[1].toUpperCase(); } } @@ -721,7 +736,7 @@ function parseSelectStatement(stmt: string, startLine: number): FileDeclaration // Normalize whitespace const text = stmt.replace(/\s+/g, ' ').trim(); - const nameMatch = text.match(/^SELECT\s+(?:OPTIONAL\s+)?([A-Z][A-Z0-9-]+)/i); + const nameMatch = text.match(/^SELECT\s+(?:OPTIONAL\s+)?([A-Z0-9][A-Z0-9-]+)/i); if (!nameMatch) return null; const result: FileDeclaration = { @@ -730,7 +745,7 @@ function parseSelectStatement(stmt: string, startLine: number): FileDeclaration line: startLine, }; - const assignMatch = text.match(/\bASSIGN\s+(?:TO\s+)?("([^"]+)"|([A-Z][A-Z0-9-]*))/i); + const assignMatch = text.match(/\bASSIGN\s+(?:TO\s+)?("([^"]+)"|([A-Z0-9][A-Z0-9-]*))/i); if (assignMatch) { result.assignTo = assignMatch[2] || assignMatch[3] || ''; } @@ -747,19 +762,21 @@ function parseSelectStatement(stmt: string, startLine: number): FileDeclaration result.access = accessMatch[1].toUpperCase(); } - const keyMatch = text.match(/\bRECORD\s+KEY\s+(?:IS\s+)?([A-Z][A-Z0-9-]+)/i); + const keyMatch = text.match(/\bRECORD\s+KEY\s+(?:IS\s+)?([A-Z0-9][A-Z0-9-]+)/i); if (keyMatch) { result.recordKey = keyMatch[1]; } // ALTERNATE RECORD KEY - const altKeyMatches = text.matchAll(/\bALTERNATE\s+RECORD\s+KEY\s+(?:IS\s+)?([A-Z][A-Z0-9-]+)/gi); + const altKeyMatches = text.matchAll( + /\bALTERNATE\s+RECORD\s+KEY\s+(?:IS\s+)?([A-Z0-9][A-Z0-9-]+)/gi, + ); const alternateKeys: string[] = []; for (const m of altKeyMatches) alternateKeys.push(m[1]); if (alternateKeys.length > 0) result.alternateKeys = alternateKeys; // FILE STATUS IS / STATUS IS - const statusMatch = text.match(/\b(?:FILE\s+)?STATUS\s+(?:IS\s+)?([A-Z][A-Z0-9-]+)/i); + const statusMatch = text.match(/\b(?:FILE\s+)?STATUS\s+(?:IS\s+)?([A-Z0-9][A-Z0-9-]+)/i); if (statusMatch) { result.fileStatus = statusMatch[1]; } @@ -789,11 +806,12 @@ function parseExecSqlBlock( block: string, line: number, ): CobolRegexResults['execSqlBlocks'][number] { - // Strip EXEC SQL ... END-EXEC wrapper + // Strip EXEC SQL ... END-EXEC wrapper and trailing period const body = block .replace(/\bEXEC\s+SQL\b/i, '') .replace(/\bEND-EXEC\b/i, '') .replace(/\s+/g, ' ') + .replace(/\.\s*$/, '') .trim(); // Determine operation from first SQL keyword @@ -823,18 +841,24 @@ function parseExecSqlBlock( // Extract table names from FROM, INTO (INSERT), UPDATE, DELETE FROM, JOIN const tables: string[] = []; const tablePatterns = [ - /\bFROM\s+([A-Z][A-Z0-9_]+)/gi, - /\bINSERT\s+INTO\s+([A-Z][A-Z0-9_]+)/gi, - /\bUPDATE\s+([A-Z][A-Z0-9_]+)/gi, - /\bJOIN\s+([A-Z][A-Z0-9_]+)/gi, + // FROM table1 [AS alias], table2 [AS alias] … — handle comma-separated + // lists with optional AS keyword, terminated by SQL clause keywords + // (WHERE, JOIN, GROUP, ON, ORDER, HAVING, UNION, SET, INTO, VALUES, FETCH, FOR, LIMIT, OFFSET, WITH). + /\bFROM\s+([A-Z0-9][A-Z0-9_]+(?:\s+(?:AS\s+)?[A-Z0-9][A-Z0-9_]*)?(?:\s*,\s*[A-Z0-9][A-Z0-9_]+(?:\s+(?:AS\s+)?[A-Z0-9][A-Z0-9_]*)?)*)(?:\s+(?:WHERE|JOIN|GROUP|ON|ORDER|HAVING|UNION|SET|INTO|VALUES|FETCH|FOR|LIMIT|OFFSET|WITH)\b|$)/gi, + /\bINSERT\s+INTO\s+([A-Z0-9][A-Z0-9_]+)/gi, + /\bUPDATE\s+([A-Z0-9][A-Z0-9_]+)/gi, + /\bJOIN\s+([A-Z0-9][A-Z0-9_]+)/gi, ]; for (const re of tablePatterns) { let m: RegExpExecArray | null; while ((m = re.exec(body)) !== null) { - const name = m[1].toUpperCase(); - // Skip host variables and SQL keywords - if (!name.startsWith(':') && !tables.includes(name)) { - tables.push(name); + // Split comma-separated table list and strip aliases + const names = m[1].split(',').map((n) => n.trim().split(/\s+/)[0].toUpperCase()); + for (const name of names) { + // Skip host variables and SQL keywords + if (!name.startsWith(':') && !tables.includes(name)) { + tables.push(name); + } } } } @@ -849,7 +873,7 @@ function parseExecSqlBlock( // Extract host variables: :VARIABLE-NAME (strip the colon) const hostVariables: string[] = []; - const hostRe = /:([A-Z][A-Z0-9-]+)/gi; + const hostRe = /:([A-Z0-9][A-Z0-9-]+)/gi; let hm: RegExpExecArray | null; while ((hm = hostRe.exec(body)) !== null) { const name = hm[1]; @@ -910,24 +934,24 @@ function parseExecCicsBlock( const result: CobolRegexResults['execCicsBlocks'][number] = { line, command }; // MAP name: MAP('name') or MAP("name") or MAP(IDENTIFIER) - const mapMatch = body.match(/\bMAP\s*\(\s*(?:['"]([^'"]+)['"]|([A-Z][A-Z0-9-]+))\s*\)/i); + const mapMatch = body.match(/\bMAP\s*\(\s*(?:['"]([^'"]+)['"]|([A-Z0-9][A-Z0-9-]+))\s*\)/i); if (mapMatch) result.mapName = mapMatch[1] ?? mapMatch[2]; // PROGRAM name: PROGRAM('name') or PROGRAM("name") or PROGRAM(VARIABLE) - const progMatch = body.match(/\bPROGRAM\s*\(\s*(?:['"]([^'"]+)['"]|([A-Z][A-Z0-9-]+))\s*\)/i); + const progMatch = body.match(/\bPROGRAM\s*\(\s*(?:['"]([^'"]+)['"]|([A-Z0-9][A-Z0-9-]+))\s*\)/i); if (progMatch) { result.programName = progMatch[1] ?? progMatch[2]; result.programIsLiteral = !!progMatch[1]; } // TRANSID: TRANSID('name') or TRANSID("name") or TRANSID(VARIABLE) - const transMatch = body.match(/\bTRANSID\s*\(\s*(?:['"]([^'"]+)['"]|([A-Z][A-Z0-9-]+))\s*\)/i); + const transMatch = body.match(/\bTRANSID\s*\(\s*(?:['"]([^'"]+)['"]|([A-Z0-9][A-Z0-9-]+))\s*\)/i); if (transMatch) result.transId = transMatch[1] ?? transMatch[2]; // FILE/DATASET: FILE('name') or DATASET('name') or FILE(VARIABLE) // Used in CICS READ, WRITE, REWRITE, DELETE, STARTBR, READNEXT, READPREV, ENDBR const fileMatch = body.match( - /\b(?:FILE|DATASET)\s*\(\s*(?:['"]([^'"]+)['"]|([A-Z][A-Z0-9-]+))\s*\)/i, + /\b(?:FILE|DATASET)\s*\(\s*(?:['"]([^'"]+)['"]|([A-Z0-9][A-Z0-9-]+))\s*\)/i, ); if (fileMatch) { result.fileName = fileMatch[1] ?? fileMatch[2]; @@ -935,19 +959,19 @@ function parseExecCicsBlock( } // QUEUE: QUEUE('name') — used in WRITEQ/READQ TS/TD - const queueMatch = body.match(/\bQUEUE\s*\(\s*(?:['"]([^'"]+)['"]|([A-Z][A-Z0-9-]+))\s*\)/i); + const queueMatch = body.match(/\bQUEUE\s*\(\s*(?:['"]([^'"]+)['"]|([A-Z0-9][A-Z0-9-]+))\s*\)/i); if (queueMatch) result.queueName = queueMatch[1] ?? queueMatch[2]; // HANDLE ABEND LABEL(paragraph-name) — error handler target - const labelMatch = body.match(/\bLABEL\s*\(\s*([A-Z][A-Z0-9-]+)\s*\)/i); + const labelMatch = body.match(/\bLABEL\s*\(\s*([A-Z0-9][A-Z0-9-]+)\s*\)/i); if (labelMatch) result.labelName = labelMatch[1]; // INTO(data-area) — data target (READ INTO, RECEIVE INTO, RETRIEVE INTO, READQ INTO) - const intoMatch = body.match(/\bINTO\s*\(\s*([A-Z][A-Z0-9-]+)\s*\)/i); + const intoMatch = body.match(/\bINTO\s*\(\s*([A-Z0-9][A-Z0-9-]+)\s*\)/i); if (intoMatch) result.intoField = intoMatch[1]; // FROM(data-area) — data source (WRITE FROM, SEND FROM, WRITEQ FROM, START FROM) - const fromMatch = body.match(/\bFROM\s*\(\s*([A-Z][A-Z0-9-]+)\s*\)/i); + const fromMatch = body.match(/\bFROM\s*\(\s*([A-Z0-9][A-Z0-9-]+)\s*\)/i); if (fromMatch) result.fromField = fromMatch[1]; return result; @@ -972,16 +996,16 @@ function parseExecDliBlock( const pcbMatch = body.match(/\bUSING\s+PCB\s*\(\s*(\d+)\s*\)/i); if (pcbMatch) result.pcbNumber = parseInt(pcbMatch[1], 10); - const segMatch = body.match(/\bSEGMENT\s*\(\s*([A-Z][A-Z0-9-]*)\s*\)/i); + const segMatch = body.match(/\bSEGMENT\s*\(\s*([A-Z0-9][A-Z0-9-]*)\s*\)/i); if (segMatch) result.segmentName = segMatch[1]; - const intoMatch = body.match(/\bINTO\s*\(\s*([A-Z][A-Z0-9-]+)\s*\)/i); + const intoMatch = body.match(/\bINTO\s*\(\s*([A-Z0-9][A-Z0-9-]+)\s*\)/i); if (intoMatch) result.intoField = intoMatch[1]; - const fromMatch = body.match(/\bFROM\s*\(\s*([A-Z][A-Z0-9-]+)\s*\)/i); + const fromMatch = body.match(/\bFROM\s*\(\s*([A-Z0-9][A-Z0-9-]+)\s*\)/i); if (fromMatch) result.fromField = fromMatch[1]; - const psbMatch = body.match(/\bPSB\s*\(\s*([A-Z][A-Z0-9-]+)\s*\)/i); + const psbMatch = body.match(/\bPSB\s*\(\s*([A-Z0-9][A-Z0-9-]+)\s*\)/i); if (psbMatch) result.psbName = psbMatch[1]; return result; @@ -1028,6 +1052,7 @@ export function extractCobolSymbolsWithRegex( sets: [], inspects: [], initializes: [], + arithmeticOps: [], }; // --- State --- @@ -1451,7 +1476,7 @@ export function extractCobolSymbolsWithRegex( } } else if ( currentDivision === 'procedure' && - /(? f.replace(/\.$/, '')) - .filter((f) => /^[A-Z][A-Z0-9-]+$/i.test(f) && !SORT_CLAUSE_NOISE.has(f.toUpperCase())), + .filter( + (f) => /^[A-Z0-9][A-Z0-9-]+$/i.test(f) && !SORT_CLAUSE_NOISE.has(f.toUpperCase()), + ), ); } if (givingIdx >= 0) { @@ -1597,16 +1624,18 @@ export function extractCobolSymbolsWithRegex( .trim() .split(/\s+/) .map((f) => f.replace(/\.$/, '')) - .filter((f) => /^[A-Z][A-Z0-9-]+$/i.test(f) && !SORT_CLAUSE_NOISE.has(f.toUpperCase())), + .filter( + (f) => /^[A-Z0-9][A-Z0-9-]+$/i.test(f) && !SORT_CLAUSE_NOISE.has(f.toUpperCase()), + ), ); } // INPUT PROCEDURE IS / OUTPUT PROCEDURE IS → control-flow targets (like PERFORM) // Supports optional THRU/THROUGH range: INPUT PROCEDURE IS proc-start THRU proc-end const inputProcMatch = fullSort.match( - /\bINPUT\s+PROCEDURE\s+(?:IS\s+)?([A-Z][A-Z0-9-]+)(?:\s+(?:THRU|THROUGH)\s+([A-Z][A-Z0-9-]+))?/i, + /\bINPUT\s+PROCEDURE\s+(?:IS\s+)?([A-Z0-9][A-Z0-9-]+)(?:\s+(?:THRU|THROUGH)\s+([A-Z0-9][A-Z0-9-]+))?/i, ); const outputProcMatch = fullSort.match( - /\bOUTPUT\s+PROCEDURE\s+(?:IS\s+)?([A-Z][A-Z0-9-]+)(?:\s+(?:THRU|THROUGH)\s+([A-Z][A-Z0-9-]+))?/i, + /\bOUTPUT\s+PROCEDURE\s+(?:IS\s+)?([A-Z0-9][A-Z0-9-]+)(?:\s+(?:THRU|THROUGH)\s+([A-Z0-9][A-Z0-9-]+))?/i, ); if (inputProcMatch) { result.performs.push({ @@ -1632,7 +1661,7 @@ export function extractCobolSymbolsWithRegex( function flushInspect(): void { if (inspectAccum === null) return; const text = inspectAccum; - const fieldMatch = text.match(/\bINSPECT\s+([A-Z][A-Z0-9-]+)/i); + const fieldMatch = text.match(/\bINSPECT\s+([A-Z0-9][A-Z0-9-]+)/i); if (!fieldMatch) { inspectAccum = null; return; @@ -1643,7 +1672,7 @@ export function extractCobolSymbolsWithRegex( /\bTALLYING\b([\s\S]+?)(?:\bREPLACING\b|\bCONVERTING\b|\.\s*$)/i, ); if (tallySection) { - const counterRe = /([A-Z][A-Z0-9-]+)\s+FOR\b/gi; + const counterRe = /([A-Z0-9][A-Z0-9-]+)\s+FOR\b/gi; let cm: RegExpExecArray | null; while ((cm = counterRe.exec(tallySection[1])) !== null) { counters.push(cm[1]); @@ -1693,10 +1722,10 @@ export function extractCobolSymbolsWithRegex( (s) => s.length > 0 && !CALL_USING_FILTER.has(s.toUpperCase()) && - /^[A-Z][A-Z0-9-]+$/i.test(s), + /^[A-Z0-9][A-Z0-9-]+$/i.test(s), ) : undefined; - const retMatch = afterCall.match(/\bRETURNING\s+([A-Z][A-Z0-9-]+)/i); + const retMatch = afterCall.match(/\bRETURNING\s+([A-Z0-9][A-Z0-9-]+)/i); const returning = retMatch ? retMatch[1] : undefined; result.calls.push({ target: callTarget, @@ -1720,10 +1749,10 @@ export function extractCobolSymbolsWithRegex( (s) => s.length > 0 && !CALL_USING_FILTER.has(s.toUpperCase()) && - /^[A-Z][A-Z0-9-]+$/i.test(s), + /^[A-Z0-9][A-Z0-9-]+$/i.test(s), ) : undefined; - const dynRetMatch = afterDynCall.match(/\bRETURNING\s+([A-Z][A-Z0-9-]+)/i); + const dynRetMatch = afterDynCall.match(/\bRETURNING\s+([A-Z0-9][A-Z0-9-]+)/i); const dynReturning = dynRetMatch ? dynRetMatch[1] : undefined; result.calls.push({ target: dynCallMatch[1], @@ -1933,11 +1962,14 @@ export function extractCobolSymbolsWithRegex( const target = perfMatch[1]; // Skip COBOL inline-perform keywords that are not paragraph names if (!PERFORM_KEYWORD_SKIP.has(target.toUpperCase())) { - // Also check for "PERFORM identifier TIMES" — the identifier is a - // data item count, not a paragraph name (fundamental regex ambiguity). const matchEnd = perfMatch.index! + perfMatch[0].length; const afterTarget = line.substring(matchEnd).trim(); - if (!/^TIMES\b/i.test(afterTarget)) { + // Check for inline PERFORM ... TIMES pattern where the target IS + // the counter variable itself (e.g., PERFORM WS-COUNT TIMES). + // Out-of-line PERFORM target count TIMES (e.g., PERFORM 2000-PROCESS 3 TIMES) + // IS a real paragraph call — do NOT suppress it. + const hasTimesClause = /^\s*TIMES\b/i.test(afterTarget); + if (!hasTimesClause) { result.performs.push({ caller: currentParagraph, target, @@ -1976,7 +2008,7 @@ export function extractCobolSymbolsWithRegex( // MOVE CORRESPONDING is always single-target per COBOL standard const targets = isCorresponding ? [moveMatch[3].replace(/\..*$/, '').trim().split(/\s+/)[0]].filter((t) => - /^[A-Z][A-Z0-9-]+$/i.test(t), + /^[A-Z0-9][A-Z0-9-]+$/i.test(t), ) : extractMoveTargets(moveMatch[3]); @@ -1992,13 +2024,169 @@ export function extractCobolSymbolsWithRegex( } } + // Arithmetic statements — COMPUTE, ADD, SUBTRACT, MULTIPLY, DIVIDE + // All extract target (written) and source operands (read) for ACCESSES edges + // Mask quoted strings before matching to avoid false positives from + // arithmetic keywords inside string literals (e.g., DISPLAY "COMPUTE"). + const lineForArith = line.replace(/"[^"]*"/g, ' ').replace(/'[^']*'/g, ' '); + const arithMatch = lineForArith.match(/\b(COMPUTE|ADD|SUBTRACT|MULTIPLY|DIVIDE)\s+(.+)/i); + if (arithMatch) { + const verb = arithMatch[1].toUpperCase() as + | 'COMPUTE' + | 'ADD' + | 'SUBTRACT' + | 'MULTIPLY' + | 'DIVIDE'; + const rest = arithMatch[2].replace(/\..*$/, '').trim(); + let target = ''; + const sources: string[] = []; + let givingTarget: string | undefined; + + switch (verb) { + case 'COMPUTE': { + // COMPUTE target = expression + const eqIdx = rest.indexOf('='); + if (eqIdx > 0) { + target = rest.substring(0, eqIdx).trim().split(/\s+/)[0] || ''; + const expr = rest.substring(eqIdx + 1).trim(); + // Extract identifiers from expression (skip literals and operators) + const idRe = /[A-Z0-9][A-Z0-9-]*/gi; + let idMatch: RegExpExecArray | null; + while ((idMatch = idRe.exec(expr)) !== null) { + const name = idMatch[0]; + if ( + !/^(?:AND|OR|NOT|IN|OF|BY|TO|FROM|DIVIDED|INTO|GIVING|TIMES|PLUS|MINUS|MULTIPLIED)$/i.test( + name, + ) + ) { + if (!sources.includes(name)) sources.push(name); + } + } + } + break; + } + case 'ADD': { + // ADD a TO b [GIVING c] — target is after TO or GIVING. + // If no TO, try GIVING directly (ADD a GIVING b). + const addGiving = rest.match( + /\bTO\s+([A-Z0-9][A-Z0-9-]+)(?:\s+GIVING\s+([A-Z0-9][A-Z0-9-]+))?/i, + ); + if (addGiving) { + target = addGiving[2] ?? addGiving[1]; + if (addGiving[2]) givingTarget = addGiving[2]; + // Everything before TO is sources + const beforeTo = rest.substring(0, rest.toUpperCase().indexOf(' TO ')); + beforeTo.replace(/\b([A-Z0-9][A-Z0-9-]+)\b/gi, (m: string) => { + if (!/^(?:ADD|CORRESPONDING|CORR)$/i.test(m) && !sources.includes(m)) { + sources.push(m); + } + return m; + }); + // Non-GIVING ADD A TO B: the TO operand (B) is both read and written + // (the existing value is read, added, then stored back). Add B as a + // source so both ACCESSES edges are created. + if (!addGiving[2]) { + if (!sources.includes(target)) sources.push(target); + } + } else { + // No TO — try GIVING directly: ADD a GIVING b + const addOnlyGiving = rest.match(/\bGIVING\s+([A-Z0-9][A-Z0-9-]+)/i); + if (addOnlyGiving) { + target = addOnlyGiving[1]; + givingTarget = addOnlyGiving[1]; + // Everything before GIVING is sources + const beforeGiving = rest.substring(0, rest.toUpperCase().indexOf(' GIVING ')); + beforeGiving.replace(/\b([A-Z0-9][A-Z0-9-]+)\b/gi, (m: string) => { + if (!/^(?:ADD|CORRESPONDING|CORR)$/i.test(m) && !sources.includes(m)) { + sources.push(m); + } + return m; + }); + } + } + break; + } + case 'SUBTRACT': { + // SUBTRACT a FROM b [GIVING c] — target is after FROM or GIVING + const subGiving = rest.match( + /\bFROM\s+([A-Z0-9][A-Z0-9-]+)(?:\s+GIVING\s+([A-Z0-9][A-Z0-9-]+))?/i, + ); + if (subGiving) { + target = subGiving[2] ?? subGiving[1]; + if (subGiving[2]) givingTarget = subGiving[2]; + const beforeFrom = rest.substring(0, rest.toUpperCase().indexOf(' FROM ')); + beforeFrom.replace(/\b([A-Z0-9][A-Z0-9-]+)\b/gi, (m: string) => { + if (!/^(?:SUBTRACT|CORRESPONDING|CORR)$/i.test(m) && !sources.includes(m)) { + sources.push(m); + } + return m; + }); + } + break; + } + case 'MULTIPLY': { + // MULTIPLY a BY b [GIVING c] — target is after BY or GIVING + const mulGiving = rest.match( + /\bBY\s+([A-Z0-9][A-Z0-9-]+)(?:\s+GIVING\s+([A-Z0-9][A-Z0-9-]+))?/i, + ); + if (mulGiving) { + target = mulGiving[2] ?? mulGiving[1]; + if (mulGiving[2]) givingTarget = mulGiving[2]; + const beforeBy = rest.substring(0, rest.toUpperCase().indexOf(' BY ')); + beforeBy.replace(/\b([A-Z0-9][A-Z0-9-]+)\b/gi, (m: string) => { + if (!/^(?:MULTIPLY|CORRESPONDING|CORR)$/i.test(m) && !sources.includes(m)) { + sources.push(m); + } + return m; + }); + } + break; + } + case 'DIVIDE': { + // DIVIDE a INTO b [GIVING c] or DIVIDE a BY b [GIVING c] + const divInto = rest.match( + /\bINTO\s+([A-Z0-9][A-Z0-9-]+)(?:\s+GIVING\s+([A-Z0-9][A-Z0-9-]+))?/i, + ); + const divBy = !divInto + ? rest.match(/\bBY\s+([A-Z0-9][A-Z0-9-]+)(?:\s+GIVING\s+([A-Z0-9][A-Z0-9-]+))?/i) + : null; + const divMatch = divInto ?? divBy; + if (divMatch) { + target = divMatch[2] ?? divMatch[1]; + if (divMatch[2]) givingTarget = divMatch[2]; + const beforeKeyword = divInto + ? rest.substring(0, rest.toUpperCase().indexOf(' INTO ')) + : rest.substring(0, rest.toUpperCase().indexOf(' BY ')); + beforeKeyword.replace(/\b([A-Z0-9][A-Z0-9-]+)\b/gi, (m: string) => { + if (!/^(?:DIVIDE|CORRESPONDING|CORR)$/i.test(m) && !sources.includes(m)) { + sources.push(m); + } + return m; + }); + } + break; + } + } + + if (target) { + result.arithmeticOps.push({ + verb, + target, + sources, + line: lineNum, + caller: currentParagraph, + givingTarget, + }); + } + } + // GO TO — control flow transfer (handles GO TO p1 p2 p3 DEPENDING ON x) const gotoMatch = line.match(RE_GOTO); if (gotoMatch) { const targets = gotoMatch[1] .trim() .split(/\s+/) - .filter((t) => /^[A-Z][A-Z0-9-]+$/i.test(t)); + .filter((t) => /^[A-Z0-9][A-Z0-9-]+$/i.test(t)); for (const target of targets) { result.gotos.push({ caller: currentParagraph, target, line: lineNum }); } @@ -2046,7 +2234,7 @@ export function extractCobolSymbolsWithRegex( } } } - const inspectMatch = line.match(/\bINSPECT\s+([A-Z][A-Z0-9-]+)/i); + const inspectMatch = line.match(/\bINSPECT\s+([A-Z0-9][A-Z0-9-]+)/i); if (inspectMatch && inspectAccum === null) { inspectAccum = line; inspectStartLine = lineNum; @@ -2079,7 +2267,7 @@ export function extractCobolSymbolsWithRegex( const targets = setTrueMatch[1] .trim() .split(/\s+/) - .filter((t) => /^[A-Z][A-Z0-9-]+$/i.test(t) && t.toUpperCase() !== 'OF'); + .filter((t) => /^[A-Z0-9][A-Z0-9-]+$/i.test(t) && t.toUpperCase() !== 'OF'); if (targets.length > 0) { result.sets.push({ targets, form: 'to-true', line: lineNum, caller: currentParagraph }); } @@ -2089,7 +2277,7 @@ export function extractCobolSymbolsWithRegex( const targets = setIdxMatch[1] .trim() .split(/\s+/) - .filter((t) => /^[A-Z][A-Z0-9-]+$/i.test(t)); + .filter((t) => /^[A-Z0-9][A-Z0-9-]+$/i.test(t)); const mode = setIdxMatch[2].toUpperCase(); const form = mode === 'TO' @@ -2114,7 +2302,8 @@ export function extractCobolSymbolsWithRegex( .trim() .split(/\s+/) .filter( - (t) => /^[A-Z][A-Z0-9-]+$/i.test(t) && !INITIALIZE_CLAUSE_KEYWORDS.has(t.toUpperCase()), + (t) => + /^[A-Z0-9][A-Z0-9-]+$/i.test(t) && !INITIALIZE_CLAUSE_KEYWORDS.has(t.toUpperCase()), ); for (const target of targets) { result.initializes.push({ target, line: lineNum, caller: currentParagraph }); diff --git a/gitnexus/src/core/ingestion/languages/cobol/captures.ts b/gitnexus/src/core/ingestion/languages/cobol/captures.ts index 04906f7a5..3f9c2620f 100644 --- a/gitnexus/src/core/ingestion/languages/cobol/captures.ts +++ b/gitnexus/src/core/ingestion/languages/cobol/captures.ts @@ -74,9 +74,13 @@ export function emitCobolScopeCaptures( const endCol = endColFrom(lines[Math.min(endLine, lines.length) - 1] ?? ''); const progIdLine = findProgramIdLine(cleaned, name); + // Determine PROGRAM-ID name column: free-format has no fixed column; + // fixed-format uses column 7 (after 6-char sequence area replaced by preprocessing) + const isFreeFormat = />>SOURCE\s+(?:FORMAT\s+(?:IS\s+)?)?FREE/i.test(cleaned); + const nameCol = isFreeFormat ? findProgramIdNameColumn(lines, progIdLine) : 7; const nameRange = progIdLine !== -1 - ? rangeOf(progIdLine, 7, progIdLine, lines[progIdLine - 1]?.length ?? endCol) + ? rangeOf(progIdLine, nameCol, progIdLine, lines[progIdLine - 1]?.length ?? endCol) : rangeOf(startLine, startCol, endLine, endCol); const grouped: Record = { @@ -116,9 +120,11 @@ export function emitCobolScopeCaptures( const endCol = endColFrom(lines[Math.min(endLine, lines.length) - 1] ?? ''); const progIdLine = findProgramIdLine(cleaned, prog.name); + const isFreeFormatNested = />>SOURCE\s+(?:FORMAT\s+(?:IS\s+)?)?FREE/i.test(cleaned); + const nameColNested = isFreeFormatNested ? findProgramIdNameColumn(lines, progIdLine) : 7; const nameRange = progIdLine !== -1 - ? rangeOf(progIdLine, 7, progIdLine, lines[progIdLine - 1]?.length ?? endCol) + ? rangeOf(progIdLine, nameColNested, progIdLine, lines[progIdLine - 1]?.length ?? endCol) : rangeOf(startLine, startCol, endLine, endCol); const grouped: Record = { @@ -297,3 +303,19 @@ function findProgramIdLine(cleanedSource: string, programName: string): number { function escapeRegex(s: string): string { return s.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); } + +/** + * Find the column position of the program name on the PROGRAM-ID line. + * Searches for `PROGRAM-ID. name` and returns the column where name starts. + * Returns 0 as fallback if the line can't be parsed (the range will be + * from column 0 which is still valid for capture bounds). + */ +function findProgramIdNameColumn(lines: string[], lineNum: number): number { + if (lineNum < 1 || lineNum > lines.length) return 0; + const line = lines[lineNum - 1]; + const m = line.match(/\bPROGRAM-ID\.\s+([A-Z0-9][A-Z0-9-]*)/i); + if (!m || m.index === undefined) return 0; + // Column = index of start of capture group 1 + const nameStart = m.index + m[0].length - m[1].length; + return nameStart; +} diff --git a/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/arithmetic-verbs.cbl b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/arithmetic-verbs.cbl new file mode 100644 index 000000000..d6a1379cd --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/arithmetic-verbs.cbl @@ -0,0 +1,18 @@ + IDENTIFICATION DIVISION. + PROGRAM-ID. ARITHOPS. + DATA DIVISION. + WORKING-STORAGE SECTION. + 01 WS-A PIC 9(03) VALUE 10. + 01 WS-B PIC 9(03) VALUE 20. + 01 WS-C PIC 9(03) VALUE 30. + 01 WS-D PIC 9(03) VALUE 40. + 01 WS-RESULT PIC 9(05). + PROCEDURE DIVISION. + 0000-MAIN. + ADD WS-A TO WS-B. + SUBTRACT WS-A FROM WS-C. + MULTIPLY WS-A BY WS-B. + DIVIDE WS-A INTO WS-D. + COMPUTE WS-RESULT = WS-A + WS-B. + STOP RUN. + END PROGRAM ARITHOPS. diff --git a/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/digit-leading.cbl b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/digit-leading.cbl new file mode 100644 index 000000000..d64e36353 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/digit-leading.cbl @@ -0,0 +1,17 @@ + IDENTIFICATION DIVISION. + PROGRAM-ID. DIGITLEAD. + DATA DIVISION. + WORKING-STORAGE SECTION. + 01 WS-DUMMY PIC X(01). + PROCEDURE DIVISION. + 1000-MAIN. + PERFORM 2000-READ-FILE. + PERFORM 2000-READ-FILE THRU 2100-PROCESS. + GO TO 9000-EXIT. + 2000-READ-FILE. + MOVE 'R' TO WS-DUMMY. + 2100-PROCESS. + MOVE 'P' TO WS-DUMMY. + 9000-EXIT. + STOP RUN. + END PROGRAM DIGITLEAD. diff --git a/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/fixed-format-offset.cbl b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/fixed-format-offset.cbl new file mode 100644 index 000000000..7a027118a --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/fixed-format-offset.cbl @@ -0,0 +1,9 @@ + IDENTIFICATION DIVISION. + PROGRAM-ID. OFFSETPGM. + DATA DIVISION. + WORKING-STORAGE SECTION. + 01 WS-FLAG PIC X(01). + PROCEDURE DIVISION. + MOVE 'X' TO WS-FLAG. + STOP RUN. + END PROGRAM OFFSETPGM. diff --git a/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/free-format.cbl b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/free-format.cbl new file mode 100644 index 000000000..7ddb52877 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/free-format.cbl @@ -0,0 +1,10 @@ +>>SOURCE FORMAT FREE +IDENTIFICATION DIVISION. +PROGRAM-ID. FREEFMT. +DATA DIVISION. +WORKING-STORAGE SECTION. +01 WS-NAME PIC X(20). +PROCEDURE DIVISION. + DISPLAY "hello". + STOP RUN. +END PROGRAM FREEFMT. diff --git a/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/move-subscript.cbl b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/move-subscript.cbl new file mode 100644 index 000000000..4cac6478b --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/move-subscript.cbl @@ -0,0 +1,16 @@ + IDENTIFICATION DIVISION. + PROGRAM-ID. MOVESUBS. + DATA DIVISION. + WORKING-STORAGE SECTION. + 01 WS-NAME PIC X(50). + 01 WS-CUSTOMER PIC X(50). + 01 CUSTOMER-NAME PIC X(50). + 01 CUSTOMER-TABLE PIC X(100). + 01 CUSTOMER-TABLE-IDX PIC 9(03). + PROCEDURE DIVISION. + 0000-MAIN. + MOVE CUSTOMER-NAME(1:10) TO WS-NAME. + MOVE CUSTOMER-TABLE(CUSTOMER-TABLE-IDX) TO WS-CUSTOMER. + MOVE CUSTOMER-TABLE(CUSTOMER-TABLE-IDX)(1:5) TO WS-NAME. + STOP RUN. + END PROGRAM MOVESUBS. diff --git a/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/multi-table-sql.cbl b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/multi-table-sql.cbl new file mode 100644 index 000000000..6f4ef515f --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/multi-table-sql.cbl @@ -0,0 +1,23 @@ + IDENTIFICATION DIVISION. + PROGRAM-ID. MULTISQL. + DATA DIVISION. + WORKING-STORAGE SECTION. + 01 WS-DATA PIC X(100). + PROCEDURE DIVISION. + EXEC SQL + SELECT A.CUST_NAME, B.ACCT_BAL + FROM CUSTOMER C, ACCOUNT A + WHERE C.CUST_ID = A.CUST_ID + END-EXEC. + EXEC SQL + SELECT * + FROM CUSTOMER, ACCOUNT + WHERE CUSTOMER.ID = ACCOUNT.CUST_ID + END-EXEC. + EXEC SQL + SELECT * + FROM INVENTORY + WHERE QTY > 0 + END-EXEC. + STOP RUN. + END PROGRAM MULTISQL. diff --git a/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/perform-times.cbl b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/perform-times.cbl new file mode 100644 index 000000000..192c02424 --- /dev/null +++ b/gitnexus/test/fixtures/lang-resolution/cobol-parsing-coverage/perform-times.cbl @@ -0,0 +1,26 @@ + IDENTIFICATION DIVISION. + PROGRAM-ID. PERFTIMS. + DATA DIVISION. + WORKING-STORAGE SECTION. + 01 WS-COUNT PIC 9(03) VALUE 5. + 01 WS-DONE PIC X(01). + PROCEDURE DIVISION. + 1000-START. + PERFORM 2000-PROCESS. + PERFORM 2000-PROCESS THRU 2100-CLEANUP. + * PERFORM TIMES with inline count — not a paragraph call: + PERFORM 2000-PROCESS 3 TIMES. + * PERFORM TIMES with identifier — not a paragraph call: + PERFORM 2000-PROCESS WS-COUNT TIMES. + * PERFORM VARYING — not a paragraph call: + PERFORM VARYING WS-COUNT FROM 1 BY 1 UNTIL WS-COUNT > 10 + CONTINUE + END-PERFORM. + GO TO 9000-END. + 2000-PROCESS. + MOVE 'X' TO WS-DONE. + 2100-CLEANUP. + MOVE 'Y' TO WS-DONE. + 9000-END. + STOP RUN. + END PROGRAM PERFTIMS. diff --git a/gitnexus/test/integration/resolvers/cobol-parsing-coverage.test.ts b/gitnexus/test/integration/resolvers/cobol-parsing-coverage.test.ts new file mode 100644 index 000000000..7b17c711a --- /dev/null +++ b/gitnexus/test/integration/resolvers/cobol-parsing-coverage.test.ts @@ -0,0 +1,222 @@ +/** + * COBOL parsing-coverage regression tests (F17-F23 from issue #1925). + * + * Each finding has its own fixture file and exact-count assertions. + * These tests must FAIL on main and PASS on the fix branch. + */ +import { describe, it, expect, beforeAll } from 'vitest'; +import path from 'path'; +import { + FIXTURES, + getRelationships, + getNodesByLabel, + runPipelineFromRepo, + type PipelineResult, +} from './helpers.js'; + +const COVERAGE_FIXTURE = path.join(FIXTURES, 'cobol-parsing-coverage'); + +describe('COBOL parsing coverage (F17-F23)', () => { + let result: PipelineResult; + + beforeAll(async () => { + result = await runPipelineFromRepo(COVERAGE_FIXTURE, () => {}, { + skipGraphPhases: true, + }); + }, 60000); + + // ===================================================================== + // F17: Digit-leading paragraph names + // ===================================================================== + describe('F17 — digit-leading paragraph names', () => { + const DIGITLEAD_FUNCS = ['1000-MAIN', '2000-READ-FILE', '2100-PROCESS', '9000-EXIT']; + + it('captures digit-leading paragraphs as Function nodes', () => { + const funcs = getNodesByLabel(result, 'Function'); + for (const name of DIGITLEAD_FUNCS) { + expect(funcs).toContain(name); + } + }); + + it('PERFORM 2000-READ-FILE resolves to correct paragraph', () => { + const perfCalls = getRelationships(result, 'CALLS').filter( + (e) => e.rel.reason === 'cobol-perform', + ); + const targets = perfCalls.map((e) => e.target); + // DIGITLEAD: PERFORM 2000-READ-FILE (twice: bare + THRU first-target) + expect(targets.filter((t) => t === '2000-READ-FILE').length).toBeGreaterThanOrEqual(1); + }); + + it('PERFORM THRU with digit-leading targets emits both edges', () => { + const perfThruEdges = getRelationships(result, 'CALLS').filter( + (e) => e.rel.reason === 'cobol-perform-thru', + ); + const thruTargets = perfThruEdges.map((e) => e.target); + // DIGITLEAD: PERFORM 2000-READ-FILE THRU 2100-PROCESS + expect(thruTargets).toContain('2100-PROCESS'); + // PERFTIMS: PERFORM 2000-PROCESS THRU 2100-CLEANUP + expect(thruTargets).toContain('2100-CLEANUP'); + }); + + it('GO TO 9000-EXIT resolves to digit-leading target', () => { + const gotoCalls = getRelationships(result, 'CALLS').filter( + (e) => e.rel.reason === 'cobol-goto', + ); + const gotoTargets = gotoCalls.map((e) => e.target); + expect(gotoTargets).toContain('9000-EXIT'); + expect(gotoTargets).toContain('9000-END'); + }); + + // INPUT/OUTPUT PROCEDURE (SORT/MERGE): The regex patterns at + // L1605-1610 use the same [A-Z0-9] character class as all other + // F17 patterns — verified via code audit that the capture groups + // accept digit-leading paragraph names identically to RE_PROC_PARAGRAPH. + // Dedicated SORT/INPUT PROCEDURE fixture requires file declarations + // (SELECT/ASSIGN) which adds complexity beyond this findings scope. + // Deferred to future work; existing regex coverage is confirmed. + }); + + // ===================================================================== + // F18: MOVE source subscript + // ===================================================================== + describe('F18 — MOVE with subscripts', () => { + it('emits cobol-move-read ACCESSES edges', () => { + const readEdges = getRelationships(result, 'ACCESSES').filter( + (e) => e.rel.reason === 'cobol-move-read', + ); + expect(readEdges.length).toBeGreaterThan(0); + }); + + it('emits cobol-move-write ACCESSES edges', () => { + const writeEdges = getRelationships(result, 'ACCESSES').filter( + (e) => e.rel.reason === 'cobol-move-write', + ); + expect(writeEdges.length).toBeGreaterThan(0); + }); + }); + + // ===================================================================== + // F19: Multi-table SQL FROM + // ===================================================================== + describe('F19 — multi-table SQL FROM', () => { + it('captures all tables from comma-separated FROM clauses', () => { + const sqlAccesses = getRelationships(result, 'ACCESSES').filter( + (e) => e.rel.reason === 'sql-select', + ); + // MULTISQL has: + // FROM CUSTOMER C, ACCOUNT A → CUSTOMER, ACCOUNT (2 tables) + // FROM CUSTOMER, ACCOUNT → CUSTOMER, ACCOUNT (2 tables) + // FROM INVENTORY → INVENTORY (1 table) + // Total: 5 table references across 3 SQL blocks + expect(sqlAccesses.length).toBe(5); + const targets = sqlAccesses.map((e) => e.target); + expect(targets).toContain('Record::CUSTOMER'); + expect(targets).toContain('Record::ACCOUNT'); + expect(targets).toContain('Record::INVENTORY'); + }); + }); + + // ===================================================================== + // F20: All 5 arithmetic verbs + // ===================================================================== + describe('F20 — arithmetic verb ACCESSES edges', () => { + it('ADD A TO B produces both edges', () => { + const arithRead = getRelationships(result, 'ACCESSES').filter( + (e) => e.rel.reason === 'cobol-arithmetic-read', + ); + const arithWrite = getRelationships(result, 'ACCESSES').filter( + (e) => e.rel.reason === 'cobol-arithmetic-write', + ); + // ARITHOPS fixture: + // ADD WS-A TO WS-B → read(WS-A) write(WS-B) + // SUBTRACT WS-A FROM WS-C → read(WS-A) write(WS-C) + // MULTIPLY WS-A BY WS-B → read(WS-A) write(WS-B) + // DIVIDE WS-A INTO WS-D → read(WS-A) write(WS-D) + // COMPUTE WS-RESULT = ... → read(WS-A, WS-B) write(WS-RESULT) + // Total: 6+ reads, 5 writes + expect(arithRead.length).toBeGreaterThanOrEqual(6); + expect(arithWrite.length).toBeGreaterThanOrEqual(5); + }); + }); + + // ===================================================================== + // F21: Free-format column detection + // ===================================================================== + describe('F21 — free-format PROGRAM-ID column', () => { + it('produces Module nodes for both fixtures', () => { + const modules = getNodesByLabel(result, 'Module'); + expect(modules).toContain('FREEFMT'); + expect(modules).toContain('OFFSETPGM'); + }); + }); + + // ===================================================================== + // F22: File-size guard — edge case verification + // ===================================================================== + describe('F22 — file-size guard', () => { + // When threshold is below file sizes, files are skipped (no Module nodes). + // When threshold is above, files process normally. + let skipResult: PipelineResult; + + beforeAll(async () => { + process.env.GITNEXUS_MAX_COBOL_FILE_SIZE_BYTES = '100'; + skipResult = await runPipelineFromRepo(COVERAGE_FIXTURE, () => {}, { + skipGraphPhases: true, + }); + delete process.env.GITNEXUS_MAX_COBOL_FILE_SIZE_BYTES; + }, 60000); + + it('file above threshold is skipped — zero Module nodes', () => { + // With threshold=100, all fixture files (203-906 bytes) are over the limit. + // The guard calls logger.warn with the file path and size — visible in test stderr. + const modules = getNodesByLabel(skipResult, 'Module'); + expect(modules.length).toBe(0); + }); + + it('file near threshold (below limit) processes normally', async () => { + // Set threshold to 10MB — well above all fixture file sizes + process.env.GITNEXUS_MAX_COBOL_FILE_SIZE_BYTES = String(10 * 1024 * 1024); + const norResult = await runPipelineFromRepo(COVERAGE_FIXTURE, () => {}, { + skipGraphPhases: true, + }); + delete process.env.GITNEXUS_MAX_COBOL_FILE_SIZE_BYTES; + const modules = getNodesByLabel(norResult, 'Module'); + expect(modules.length).toBeGreaterThan(0); + }); + }); + + // ===================================================================== + // F23: PERFORM TIMES + THRU on digit-leading targets + // ===================================================================== + describe('F23 — PERFORM TIMES and THRU', () => { + it('PERFORM 2000-PROCESS THRU 2100-CLEANUP resolves THRU target', () => { + const perfThruEdges = getRelationships(result, 'CALLS').filter( + (e) => e.rel.reason === 'cobol-perform-thru', + ); + const thruTargets = perfThruEdges.map((e) => e.target); + expect(thruTargets).toContain('2100-CLEANUP'); + }); + + it('PERFORM with VARYING does NOT create spurious CALLS edge', () => { + // PERFTIMS has PERFORM VARYING WS-COUNT FROM 1 BY 1... + // VARYING is in PERFORM_KEYWORD_SKIP, so it should be skipped. + // Verify there are no spurious perform edges with target "VARYING" + const perfEdges = getRelationships(result, 'CALLS').filter( + (e) => e.rel.reason === 'cobol-perform', + ); + const targets = perfEdges.map((e) => e.target); + expect(targets).not.toContain('VARYING'); + expect(targets).not.toContain('WS-COUNT'); + }); + + it('total CALLS count stays reasonable (TIMES with count is real)', () => { + // PERFTIMS has: 2 perform (2000-PROCESS + 2000-PROCESS THRU first-target) + // + 2 perform for count TIMES (2000-PROCESS 3 TIMES + WS-COUNT TIMES) + // + 1 perform-thru + 1 goto = 6 CALLS + // DIGITLEAD has: 2 perform + 1 perform-thru + 1 goto = 4 CALLS + // Total: 10 across both fixtures + const calls = getRelationships(result, 'CALLS'); + expect(calls.length).toBe(10); + }); + }); +}); diff --git a/gitnexus/test/integration/resolvers/cobol.test.ts b/gitnexus/test/integration/resolvers/cobol.test.ts index 44111b00b..e2ec7f85a 100644 --- a/gitnexus/test/integration/resolvers/cobol.test.ts +++ b/gitnexus/test/integration/resolvers/cobol.test.ts @@ -713,12 +713,14 @@ describe('COBOL full system extraction', () => { expect(getRelationships(result, 'IMPORTS').length).toBe(2); }); - it('produces exactly 25 total ACCESSES edges', () => { - // 4 move-read + 5 move-write + 1 move-corresponding-read + 1 move-corresponding-write + it('produces exactly 28 total ACCESSES edges', () => { + // Original: 4 move-read + 5 move-write + 1 move-corresponding-read + 1 move-corresponding-write // + 1 file-read + 1 map + 1 queue-write // + 1 receive-into + 2 send-from + 1 search + 1 sort-using + 1 sort-giving // + 2 procedure-using + 1 sql-select + 2 call-using - expect(getRelationships(result, 'ACCESSES').length).toBe(25); + // plus arithmetic: +1 arithmetic-read (WS-AMOUNT) + 1 arithmetic-write (CUST-BALANCE) + // plus ADD TO read+write: +1 arithmetic-read for CUST-BALANCE (TO operand is both read+written) + expect(getRelationships(result, 'ACCESSES').length).toBe(28); }); });