diff --git a/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs b/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs index e68aca1de..564384f83 100644 --- a/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs +++ b/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs @@ -85,40 +85,241 @@ function findGitNexusDir(startDir) { return null; } +function tokenizeShellWords(command) { + const tokens = []; + let current = ''; + let quote = null; + let escaped = false; + let hasToken = false; + + for (let index = 0; index < command.length; index += 1) { + const char = command[index]; + if (escaped) { + current += char; + escaped = false; + hasToken = true; + continue; + } + + if (quote === "'") { + if (char === "'") quote = null; + else current += char; + hasToken = true; + continue; + } + + if (quote === '"') { + if (char === '"') { + quote = null; + } else if (char === '\\') { + const next = command[index + 1]; + if (next === '$' || next === '`' || next === '"' || next === '\\') { + escaped = true; + } else { + current += '\\'; + } + } else { + current += char; + } + hasToken = true; + continue; + } + + if (char === '\\') { + const next = command[index + 1]; + if (next === undefined || /\s/.test(next) || next === "'" || next === '"' || next === '\\') { + escaped = true; + } else { + current += '\\' + next; + index += 1; + } + hasToken = true; + } else if (char === "'" || char === '"') { + quote = char; + hasToken = true; + } else if (/\s/.test(char)) { + if (hasToken) tokens.push(current); + current = ''; + hasToken = false; + } else if (char === ';' || char === '|' || char === '&') { + if (hasToken) tokens.push(current); + current = ''; + hasToken = false; + const next = command[index + 1]; + if ((char === '|' || char === '&') && next === char) { + tokens.push(char + char); + index += 1; + } else { + tokens.push(char); + } + } else { + current += char; + hasToken = true; + } + } + + if (escaped) current += '\\'; + if (hasToken) tokens.push(current); + return tokens; +} + function parseRgGrepPattern(cmd) { - const tokens = cmd.split(/\s+/); + const tokens = tokenizeShellWords(cmd); let foundCmd = false; let skipNext = false; + let skipNextAsPattern = false; + let endOfOptions = false; + let explicitPatternSeen = false; + let patternFileSeen = false; const flagsWithValues = new Set([ '-e', '-f', + '--file', '-m', + '--max-count', '-A', '-B', '-C', '-g', '--glob', + '--iglob', '-t', '--type', '--include', '--exclude', + '--encoding', + '--path', ]); + const rgValueFlags = new Set(['-r', '--replace']); + const patternFlags = new Set(['-e', '--regexp']); + const connectors = new Set(['&&', '||', ';', '|', '&']); + const wrappers = new Set([ + 'npx', + 'bunx', + 'pnpm', + 'yarn', + 'npm', + 'sudo', + 'env', + 'command', + 'time', + 'nice', + 'xargs', + 'dlx', + 'exec', + 'run', + 'git', + ]); + const wrapperFlagsWithValues = new Set([ + '--package', + '-p', + '--call', + '--prefix', + '--shell', + '--filter', + '--workspace', + '--dir', + '--cwd', + ]); + const basename = (token) => + token + .split(/[\\/]/) + .pop() + ?.replace(/\.(exe|cmd|bat)$/i, ''); + let previousToken; + let seenWrapper = false; + let searchCommand = null; for (const token of tokens) { if (skipNext) { skipNext = false; + if (skipNextAsPattern) { + skipNextAsPattern = false; + if (token.length >= 3) return token; + } + previousToken = token; continue; } if (!foundCmd) { - if (/\brg$|\bgrep$/.test(token)) foundCmd = true; + if (connectors.has(token)) { + seenWrapper = false; + previousToken = token; + continue; + } + const commandName = basename(token); + if (wrappers.has(commandName)) { + seenWrapper = true; + previousToken = token; + continue; + } + if (seenWrapper && token.startsWith('-')) { + const flagName = token.split('=', 1)[0]; + if (!token.includes('=') && wrapperFlagsWithValues.has(flagName)) skipNext = true; + previousToken = token; + continue; + } + if (seenWrapper && /^[A-Za-z_][A-Za-z0-9_]*=/.test(token)) { + previousToken = token; + continue; + } + const atCommandPosition = + previousToken === undefined || + connectors.has(previousToken) || + wrappers.has(basename(previousToken)) || + seenWrapper; + if (atCommandPosition && (commandName === 'rg' || commandName === 'grep')) { + foundCmd = true; + searchCommand = commandName; + } else if (seenWrapper) { + seenWrapper = false; + } + previousToken = token; + continue; + } + previousToken = token; + if (endOfOptions) { + if (explicitPatternSeen || patternFileSeen) continue; + return token.length >= 3 ? token : null; + } + if (token === '--') { + endOfOptions = true; continue; } if (token.startsWith('-')) { - if (flagsWithValues.has(token)) skipNext = true; + if (token === '-f' || token === '--file') { + skipNext = true; + patternFileSeen = true; + continue; + } + if (token.startsWith('--file=')) { + patternFileSeen = true; + continue; + } + if (token.startsWith('--regexp=')) { + explicitPatternSeen = true; + const value = token.slice('--regexp='.length); + if (value.length >= 3) return value; + continue; + } + const attachedPattern = token.match(/^-e(.+)$/); + if (attachedPattern) { + explicitPatternSeen = true; + if (attachedPattern[1].length >= 3) return attachedPattern[1]; + continue; + } + if ( + flagsWithValues.has(token) || + patternFlags.has(token) || + (searchCommand === 'rg' && rgValueFlags.has(token)) + ) { + skipNext = true; + skipNextAsPattern = patternFlags.has(token); + if (skipNextAsPattern) explicitPatternSeen = true; + } continue; } - const cleaned = token.replace(/['"]/g, ''); - return cleaned.length >= 3 ? cleaned : null; + if (explicitPatternSeen || patternFileSeen) continue; + return token.length >= 3 ? token : null; } return null; } @@ -179,12 +380,6 @@ function extractPattern(toolName, toolInput) { if (t === 'shell') { const cmd = toolInput.command || ''; if (!/\brg\b|\bgrep\b/.test(cmd)) return null; - // NOTE: parseRgGrepPattern uses split(/\s+/) and cannot handle shell - // quoting. `rg "User Service" src/` returns "User" (the first token - // after the rg/grep arg, with surrounding quotes stripped) — the - // multi-word pattern is intentionally not reconstructed since BM25 is - // already token-tolerant. Quoted single tokens (`rg "validateUser"`) - // work fine. return parseRgGrepPattern(cmd); } @@ -282,4 +477,6 @@ function main() { } } -main(); +if (require.main === module) main(); + +module.exports = { parseRgGrepPattern, tokenizeShellWords }; diff --git a/gitnexus/test/unit/cursor-hook.test.ts b/gitnexus/test/unit/cursor-hook.test.ts index 54b5c1f84..dc9bc423b 100644 --- a/gitnexus/test/unit/cursor-hook.test.ts +++ b/gitnexus/test/unit/cursor-hook.test.ts @@ -19,6 +19,7 @@ */ import { describe, it, expect, beforeAll, afterAll } from 'vitest'; import { spawnSync } from 'child_process'; +import { createRequire } from 'module'; import fs from 'fs'; import path from 'path'; import os from 'os'; @@ -55,6 +56,12 @@ const CURSOR_HOOKS_JSON = path.resolve( 'hooks.json', ); +const require = createRequire(import.meta.url); +const { parseRgGrepPattern, tokenizeShellWords } = require(CURSOR_HOOK) as { + parseRgGrepPattern: (command: string) => string | null; + tokenizeShellWords: (command: string) => string[]; +}; + // ─── Cursor-specific output parser ────────────────────────────────── // Cursor postToolUse output shape: { "additional_context": "..." } @@ -549,37 +556,77 @@ describe('Cursor hook concurrency guard (integration)', () => { }); }); -// ─── Documented contract behavior (extractPattern via the live hook) ─ +// ─── Shell pattern parsing ────────────────────────────────────────── -describe('Shell quoted-pattern parser limitations (documented)', () => { - // The Shell parser cannot reconstruct shell quoting. These tests pin the - // current behavior so a future "fix" doesn't silently change extraction - // — and so users diagnosing a noisy/missed pattern can find the behavior - // documented in tests. - // - // We can't observe the extracted pattern directly without an indexed - // repo, but we *can* confirm the hook reaches the augment-call path - // (vs. early-exiting) by checking exit status + clean stdout for cases - // where parseRgGrepPattern would yield a >=3-char token. - - it('quoted multi-word `rg "User Service"` extracts the first word only', () => { - const result = runHook(CURSOR_HOOK, { - tool_name: 'Shell', - tool_input: { command: 'rg "User Service" src/' }, - cwd: tmpDir, // no .gitnexus → exits early after extract - }); - expect(result.status).toBe(0); - expect(result.stdout.trim()).toBe(''); +describe('Shell quoted-pattern parser', () => { + it.each([ + ['rg "User Service" src/', 'User Service'], + ["grep 'error boundary' -- src/", 'error boundary'], + ['rg User\\ Service src/', 'User Service'], + [String.raw`rg "C:\Users" src/`, String.raw`C:\Users`], + ['rg -e "User Service" src/', 'User Service'], + ['rg --regexp=UserService src/', 'UserService'], + ['grep -eUserService src/', 'UserService'], + ['rg -e x -e LongPattern src/', 'LongPattern'], + ['rg -ex -eLongPattern src/', 'LongPattern'], + ['rg --regexp=x --regexp=LongPattern src/', 'LongPattern'], + ['/usr/bin/rg -- "User Service" src/', 'User Service'], + ['rg -- -error src/', '-error'], + [String.raw`C:\Users\me\bin\rg.exe UserService src/`, 'UserService'], + ['rg.exe "validateUser" src/', 'validateUser'], + ['grep.cmd -e LongPattern src/', 'LongPattern'], + ['cd grep && rg LongPattern src/', 'LongPattern'], + ['npx rg "User Service" src/', 'User Service'], + ['npx --yes rg UserService src/', 'UserService'], + ['npx --package rg grep LongPattern src/', 'LongPattern'], + ['rg UserService; echo done', 'UserService'], + ['rg UserService&& echo done', 'UserService'], + ['rg --max-count 100 UserService src/', 'UserService'], + ['grep -r UserService src/', 'UserService'], + ['rg --replace x UserService src/', 'UserService'], + ['git grep UserService src/', 'UserService'], + ])('extracts %j from %j', (command, expected) => { + expect(parseRgGrepPattern(command)).toBe(expected); }); - it('single-token quoted `rg "validateUser"` works as expected', () => { - const result = runHook(CURSOR_HOOK, { - tool_name: 'Shell', - tool_input: { command: 'rg "validateUser"' }, - cwd: tmpDir, - }); - expect(result.status).toBe(0); - expect(result.stdout.trim()).toBe(''); + it.each([ + ['rg --regexp= src/'], + ['rg --regexp="" src/'], + ['rg -e x -- LongPattern src/'], + ['rg -f patterns.txt src/'], + ['rg --file=patterns.txt src/'], + ['rg -eab src/'], + ['sudo echo rg UserService src/'], + ])('extracts no pattern from %j', (command) => { + expect(parseRgGrepPattern(command)).toBeNull(); + }); + + it('does not treat a path after a short explicit pattern as the pattern', () => { + expect(parseRgGrepPattern('rg -e x src/')).toBeNull(); + }); + + it('keeps single-token quoted patterns intact', () => { + expect(parseRgGrepPattern('rg "validateUser"')).toBe('validateUser'); + }); + + it('keeps backslashes in unquoted Windows paths but honours escaped spaces', () => { + expect(tokenizeShellWords(String.raw`C:\foo\bar`)).toEqual([String.raw`C:\foo\bar`]); + expect(tokenizeShellWords('User\\ Service')).toEqual(['User Service']); + expect(tokenizeShellWords('trailing\\')).toEqual(['trailing\\']); + }); + + it('splits unquoted shell operators from adjacent arguments', () => { + expect(tokenizeShellWords('rg UserService; echo done')).toEqual([ + 'rg', + 'UserService', + ';', + 'echo', + 'done', + ]); + expect(tokenizeShellWords("rg 'UserService; echo done'")).toEqual([ + 'rg', + 'UserService; echo done', + ]); }); });