import { spawnSync } from 'node:child_process'; import { existsSync, mkdirSync, mkdtempSync, readdirSync, readFileSync, rmSync, writeFileSync, } from 'node:fs'; import { createRequire } from 'node:module'; import { tmpdir } from 'node:os'; import path from 'node:path'; import { load } from 'js-yaml'; import { describe, expect, it, vi } from 'vitest'; const WORKFLOW_PATH = path.resolve( __dirname, '../../../.github/workflows/gitnexus-review-agent.yml', ); const RUNTIME_PACKAGE_PATH = path.resolve( __dirname, '../../../.github/gitnexus-review-runtime/package.json', ); const RUNTIME_LOCK_PATH = path.resolve( __dirname, '../../../.github/gitnexus-review-runtime/package-lock.json', ); const CLAUDE_RUNTIME_PACKAGE_PATH = path.resolve( __dirname, '../../../.github/claude-canary-runtime/package.json', ); const CLAUDE_RUNTIME_LOCK_PATH = path.resolve( __dirname, '../../../.github/claude-canary-runtime/package-lock.json', ); const workflow = readFileSync(WORKFLOW_PATH, 'utf8'); // Both pinned installs run through this helper, so the flags that keep them // inert and lock-bound are asserted against it rather than the step bodies. const npmCiHelper = readFileSync( path.resolve(__dirname, '../../../.github/scripts/npm-ci-retry.sh'), 'utf8', ); const runtimePackage = JSON.parse(readFileSync(RUNTIME_PACKAGE_PATH, 'utf8')) as { dependencies?: Record; engines?: Record; }; const runtimeLock = JSON.parse(readFileSync(RUNTIME_LOCK_PATH, 'utf8')) as { packages?: Record; }; const claudeRuntimePackage = JSON.parse(readFileSync(CLAUDE_RUNTIME_PACKAGE_PATH, 'utf8')) as { dependencies?: Record; engines?: Record; }; const claudeRuntimeLock = JSON.parse(readFileSync(CLAUDE_RUNTIME_LOCK_PATH, 'utf8')) as { lockfileVersion?: number; packages?: Record< string, { dependencies?: Record; engines?: Record; version?: string; integrity?: string; } >; }; const requireCjs = createRequire(import.meta.url); const workflowDocument = load(workflow) as { jobs?: Record< string, { steps?: Array<{ name?: string; env?: Record; run?: unknown; with?: Record & { script?: unknown }; }>; } >; }; const PR_NUMBER = 2431; const CONTROL_SHA = '1'.repeat(40); const HEAD_SHA = '2'.repeat(40); const BASE_SHA = '3'.repeat(40); const CHANGED_PATH = 'gitnexus/src/cli/status.ts'; // Long enough to clear the assembler's MIN_BODY_CHARS floor, which exists so a // stub like the literal string "placeholder" can never be published. const ACCEPTED_BODY = `**APPROVE.** ${'Accepted graph-backed review of the changed surface. '.repeat(5)}`; function jobBlock(name: string): string { const match = workflow.match( new RegExp(`\\n ${name}:\\n[\\s\\S]*?(?=\\n [a-zA-Z0-9_-]+:\\n|$)`), ); return match?.[0] ?? ''; } function jobScript(job: string, stepName: string): string { const step = workflowDocument.jobs?.[job]?.steps?.find(({ name }) => name === stepName); return typeof step?.with?.script === 'string' ? step.with.script : ''; } function jobRun(job: string, stepName: string): string { const step = workflowDocument.jobs?.[job]?.steps?.find(({ name }) => name === stepName); return typeof step?.run === 'string' ? step.run : ''; } function embeddedNodeScript(job: string, stepName: string): string { const run = jobRun(job, stepName); const marker = "node <<'NODE'\n"; const start = run.indexOf(marker); const end = run.lastIndexOf('\nNODE'); if (start < 0 || end <= start) throw new Error(`${stepName} Node heredoc not found`); return run.slice(start + marker.length, end); } function runGit(cwd: string, arguments_: string[]): void { const result = spawnSync('git', arguments_, { cwd, encoding: 'utf8' }); if (result.status !== 0) { throw new Error(`git ${arguments_.join(' ')} failed: ${result.stderr}`); } } type ContextScenario = { controlSha?: string; dispatchPr?: string; eventName?: 'issue_comment' | 'workflow_dispatch'; eventPr?: string; permission?: string; permissionError?: Error; pull?: Record; pullError?: Error; existingComments?: Array<{ user?: { login: string }; body?: string }>; }; async function runContextScenario({ controlSha = CONTROL_SHA, dispatchPr = String(PR_NUMBER), eventName = 'workflow_dispatch', eventPr = String(PR_NUMBER), permission = 'write', permissionError, pull, pullError, existingComments = [], }: ContextScenario = {}) { const contextScript = jobScript('analyze', 'Normalize and authorize the request'); if (!contextScript) throw new Error('context github-script block not found'); const resolvedPull = pull ?? ({ state: 'open', head: { sha: HEAD_SHA, repo: { full_name: 'fork/repo' } }, base: { sha: BASE_SHA, repo: { full_name: 'owner/repo' } }, } as Record); const getPermission = permissionError ? vi.fn().mockRejectedValue(permissionError) : vi.fn().mockResolvedValue({ data: { permission } }); const getPull = pullError ? vi.fn().mockRejectedValue(pullError) : vi.fn().mockResolvedValue({ data: resolvedPull }); const listComments = vi.fn(); const github = { rest: { repos: { getCollaboratorPermissionLevel: getPermission }, pulls: { get: getPull }, issues: { listComments }, }, paginate: { iterator: vi.fn(function* iterate() { yield { data: existingComments }; }), }, }; const outputs = new Map(); const core = { debug: vi.fn(), notice: vi.fn(), setOutput: vi.fn((name: string, value: string) => outputs.set(name, value)), }; const context = { actor: 'trusted-maintainer', eventName, repo: { owner: 'owner', repo: 'repo' }, }; try { vi.stubEnv('CONTROL_SHA', controlSha); vi.stubEnv('DISPATCH_PR', dispatchPr); vi.stubEnv('EVENT_PR', eventPr); const AsyncFunction = Object.getPrototypeOf(async function () {}).constructor as new ( ...arguments_: string[] ) => (...arguments_: unknown[]) => Promise; const execute = new AsyncFunction('github', 'context', 'core', contextScript); await execute(github, context, core); } finally { vi.unstubAllEnvs(); } return { core, getPermission, getPull, outputs }; } function runChangedPathManifest(rawNameStatus: string | Uint8Array) { const script = embeddedNodeScript('analyze', 'Prepare exact merge-base review inputs'); const inputDirectory = mkdtempSync(path.join(tmpdir(), 'gitnexus-review-name-status-')); writeFileSync(path.join(inputDirectory, 'changed-name-status.bin'), rawNameStatus); try { const result = spawnSync(process.execPath, ['-'], { encoding: 'utf8', env: { ...process.env, PR_NUMBER: String(PR_NUMBER), HEAD_SHA, BASE_SHA, MERGE_BASE: CONTROL_SHA, INPUT_DIR: inputDirectory, }, input: script, }); const manifestPath = path.join(inputDirectory, 'changed-paths.json'); return { result, manifest: existsSync(manifestPath) ? (JSON.parse(readFileSync(manifestPath, 'utf8')) as Record) : undefined, }; } finally { rmSync(inputDirectory, { recursive: true, force: true }); } } type PublisherScenario = { artifactStatus: 'success' | 'failure'; artifactOverrides?: Record; rawArtifact?: string | Uint8Array; comments?: Array<{ id: number; body: string; user: { login: string }; }>; commentPages?: Array< Array<{ id: number; body: string; user: { login: string }; }> >; currentBase?: string; currentHead?: string; finalBase?: string; finalHead?: string; finalState?: string; }; type ArtifactScenario = { basePaths?: string[]; basePrescanPaths?: string[]; changedPaths?: string[]; entries?: Array>; executionFileOutput?: string; repairStructuredOutput?: string; repairOutcome?: string; noIndexableChangedSymbols?: boolean; rawTranscript?: string | Uint8Array | ((runnerTemp: string) => string | Uint8Array); structuredOutput?: string; }; function contextResultContent(filePath = CHANGED_PATH): string { return `${JSON.stringify({ status: 'found', symbol: { uid: 'Function:gitnexus/src/cli/status.ts:statusCommand', name: 'statusCommand', kind: 'Function', filePath, startLine: 1, endLine: 20, }, })}\n\n---\n**Next:** use impact() for blast radius.`; } function reviewTranscript({ toolName = 'mcp__gitnexus__context', // No file_path: the gate is call-argument-agnostic, and a default that // carried one would imply the opposite. toolInput = { name: 'statusCommand' }, toolResultContent = contextResultContent(), resultIsError = false, toolUseId = 'tool-1', parentToolUseId = null, }: { toolName?: string; toolInput?: Record; toolResultContent?: unknown; resultIsError?: boolean | undefined; toolUseId?: string; parentToolUseId?: string | null; } = {}): Array> { const toolResult: Record = { type: 'tool_result', tool_use_id: toolUseId, content: toolResultContent, }; if (resultIsError !== undefined) toolResult.is_error = resultIsError; return [ { type: 'system', subtype: 'init', session_id: 'session-1', uuid: '11111111-1111-4111-8111-111111111111', }, { type: 'assistant', parent_tool_use_id: parentToolUseId, session_id: 'session-1', uuid: '22222222-2222-4222-8222-222222222222', message: { role: 'assistant', content: [ { type: 'tool_use', id: toolUseId, name: toolName, input: toolInput, }, ], }, }, { type: 'user', parent_tool_use_id: parentToolUseId, session_id: 'session-1', uuid: '33333333-3333-4333-8333-333333333333', message: { role: 'user', content: [toolResult], }, }, { type: 'result', subtype: 'success', is_error: false, num_turns: 25, total_cost_usd: 3.8797, session_id: 'session-1', uuid: '44444444-4444-4444-8444-444444444444', }, ]; } // Two orchestrator context calls in one turn: the first proves the evidence, // the second is an ordinary exploratory call whose result may be junk. function twoCallTranscript({ firstResult, secondResult, }: { firstResult: string; secondResult: string; }): Array> { return [ { type: 'system', subtype: 'init', session_id: 'session-1', uuid: '11111111-1111-4111-8111-111111111111', }, { type: 'assistant', parent_tool_use_id: null, session_id: 'session-1', uuid: '22222222-2222-4222-8222-222222222222', message: { role: 'assistant', content: [ { type: 'tool_use', id: 'tool-1', name: 'mcp__gitnexus__context', input: { name: 'statusCommand' }, }, { type: 'tool_use', id: 'tool-2', name: 'mcp__gitnexus__context', input: { name: 'bigHotSymbol' }, }, ], }, }, { type: 'user', parent_tool_use_id: null, session_id: 'session-1', uuid: '33333333-3333-4333-8333-333333333333', message: { role: 'user', content: [ { type: 'tool_result', tool_use_id: 'tool-1', is_error: false, content: firstResult }, { type: 'tool_result', tool_use_id: 'tool-2', is_error: false, content: secondResult }, ], }, }, { type: 'result', subtype: 'success', is_error: false, session_id: 'session-1', uuid: '44444444-4444-4444-8444-444444444444', }, ]; } function reviewTranscriptWithoutTools(): Array> { return [ { type: 'system', subtype: 'init', session_id: 'session-1', uuid: '11111111-1111-4111-8111-111111111111', }, { type: 'assistant', parent_tool_use_id: null, session_id: 'session-1', uuid: '22222222-2222-4222-8222-222222222222', message: { role: 'assistant', content: [] }, }, { type: 'result', subtype: 'success', is_error: false, session_id: 'session-1', uuid: '44444444-4444-4444-8444-444444444444', }, ]; } function runArtifactScenario({ basePaths = [], basePrescanPaths = basePaths, changedPaths = [CHANGED_PATH], entries = [ ...changedPaths.map((headPath) => ({ status: 'A', head_path: headPath })), ...basePaths.map((basePath) => ({ status: 'D', base_path: basePath })), ], executionFileOutput, repairStructuredOutput, repairOutcome = repairStructuredOutput ? 'success' : 'skipped', noIndexableChangedSymbols = false, rawTranscript = JSON.stringify(reviewTranscript()), structuredOutput = JSON.stringify({ body: ACCEPTED_BODY, complete: true }), }: ArtifactScenario = {}) { const script = embeddedNodeScript('analyze', 'Assemble bounded review artifact'); const runnerTemp = mkdtempSync(path.join(tmpdir(), 'gitnexus-review-artifact-')); const inputDirectory = path.join(runnerTemp, 'gitnexus-review-control', 'review-input'); const transcriptPath = path.join(runnerTemp, 'claude-execution-output.json'); const githubOutput = path.join(runnerTemp, 'github-output'); mkdirSync(inputDirectory, { recursive: true }); writeFileSync( path.join(inputDirectory, 'changed-paths.json'), `${JSON.stringify({ schema: 'gitnexus.changed-paths/v2', entries, head_paths: changedPaths, base_paths: basePaths, base_prescan_paths: basePrescanPaths, prescan: { head_has_indexable_symbol: !noIndexableChangedSymbols && changedPaths.length > 0, base_has_indexable_symbol: !noIndexableChangedSymbols && basePrescanPaths.length > 0, no_indexable_changed_symbols: noIndexableChangedSymbols, }, })}\n`, ); writeFileSync( transcriptPath, typeof rawTranscript === 'function' ? rawTranscript(runnerTemp) : rawTranscript, ); writeFileSync(githubOutput, ''); const workspace = path.join(runnerTemp, 'workspace'); const headCheckout = path.join(workspace, 'pr-target'); const baseCheckout = path.join(runnerTemp, 'gitnexus-review-merge-base'); mkdirSync(path.join(workspace, '.github', 'scripts'), { recursive: true }); writeFileSync( path.join(workspace, '.github', 'scripts', 'review-citations.cjs'), readFileSync(path.resolve(__dirname, '../../../.github/scripts/review-citations.cjs'), 'utf8'), ); // 40 real lines per changed file so a citation can resolve or overrun. for (const [checkout, files] of [ [headCheckout, changedPaths], [baseCheckout, basePaths], ] as const) { for (const filePath of files) { const absolute = path.join(checkout, filePath); mkdirSync(path.dirname(absolute), { recursive: true }); writeFileSync( absolute, Array.from({ length: 40 }, (_unused, i) => `line ${i + 1}`).join('\n'), ); } } const environment = { ...process.env, RUNNER_TEMP: runnerTemp, GITHUB_REPOSITORY: 'owner/repo', MERGE_BASE_SHA: BASE_SHA, GITHUB_WORKSPACE: workspace, GITHUB_OUTPUT: githubOutput, PR_NUMBER: String(PR_NUMBER), CONTROL_SHA, HEAD_SHA, BASE_SHA, CONTEXT_READY: 'true', FAILURE_CODE: 'none', CONTROL_OUTCOME: 'success', HEAD_OUTCOME: 'success', VALIDATE_OUTCOME: 'success', SETUP_NODE_OUTCOME: 'success', ISOLATION_OUTCOME: 'success', CLAUDE_RUNTIME_OUTCOME: 'success', RUNTIME_OUTCOME: 'success', INDEX_OUTCOME: 'success', INPUTS_OUTCOME: 'success', MERGE_BASE_SOURCE_OUTCOME: 'success', GRAPH_PRESCAN_OUTCOME: 'success', CLAUDE_RECHECK_OUTCOME: 'success', CLAUDE_OUTCOME: 'success', EXECUTION_FILE: executionFileOutput ?? transcriptPath, STRUCTURED_OUTPUT: structuredOutput, REPAIR_OUTCOME: repairOutcome, REPAIR_STRUCTURED_OUTPUT: repairStructuredOutput ?? '', REPAIR_EXECUTION_FILE: repairStructuredOutput ? transcriptPath : '', }; try { const result = spawnSync(process.execPath, ['-'], { encoding: 'utf8', env: environment, input: script, maxBuffer: 20_000_000, }); if (result.status !== 0) { throw new Error(`artifact assembler failed: ${result.stderr || result.stdout}`); } const artifact = JSON.parse( readFileSync(path.join(runnerTemp, 'gitnexus-review-artifact', 'review.json'), 'utf8'), ) as { body: string; failure_code: string | null; graph_evidence: { base_has_indexable_symbol: boolean; head_has_indexable_symbol: boolean; mode: 'context' | 'no_indexable_changed_symbols'; } | null; status: 'success' | 'failure'; }; return { artifact, stderr: result.stderr, stdout: result.stdout }; } finally { rmSync(runnerTemp, { recursive: true, force: true }); } } async function runPublisherScenario({ artifactStatus, artifactOverrides = {}, rawArtifact, comments = [], commentPages, currentBase = BASE_SHA, currentHead = HEAD_SHA, finalBase = currentBase, finalHead = currentHead, finalState = 'open', }: PublisherScenario) { const publisherScript = jobScript( 'publish', 'Validate freshness and upsert an accepted same-SHA comment', ); if (!publisherScript) throw new Error('publisher github-script block not found'); const tempDir = mkdtempSync(path.join(tmpdir(), 'gitnexus-review-publisher-')); const artifactPath = path.join(tempDir, 'review.json'); const artifact = { schema: 'gitnexus.review/v2', pr_number: PR_NUMBER, control_sha: CONTROL_SHA, head_sha: HEAD_SHA, base_sha: BASE_SHA, status: artifactStatus, body: artifactStatus === 'success' ? 'Accepted review body' : 'Model failed safely.', failure_code: artifactStatus === 'success' ? null : 'model_failed', graph_evidence: artifactStatus === 'success' ? { mode: 'context', head_has_indexable_symbol: true, base_has_indexable_symbol: false, } : null, ...artifactOverrides, }; writeFileSync(artifactPath, rawArtifact ?? `${JSON.stringify(artifact)}\n`); const updateComment = vi.fn().mockResolvedValue({ data: {} }); const createComment = vi.fn().mockResolvedValue({ data: { id: 99 } }); const paginateIterator = vi.fn(() => { const pages = commentPages ?? [comments]; return (async function* () { for (const page of pages) yield { data: page }; })(); }); let pullRead = 0; const getPull = vi.fn().mockImplementation(async () => { const initial = pullRead++ === 0; return { data: { state: initial ? 'open' : finalState, head: { sha: initial ? currentHead : finalHead, repo: { full_name: 'fork/repo' }, }, base: { sha: initial ? currentBase : finalBase, repo: { full_name: 'owner/repo' }, }, }, }; }); const github = { paginate: Object.assign(vi.fn(), { iterator: paginateIterator }), rest: { pulls: { get: getPull, }, issues: { listComments: vi.fn(), updateComment, createComment, }, }, }; const core = { info: vi.fn(), notice: vi.fn(), setFailed: vi.fn(), warning: vi.fn(), }; const context = { repo: { owner: 'owner', repo: 'repo' } }; const environment = { ARTIFACT_PATH: artifactPath, DOWNLOAD_OUTCOME: 'success', PR_NUMBER: String(PR_NUMBER), CONTROL_SHA, HEAD_SHA, BASE_SHA, }; try { for (const [key, value] of Object.entries(environment)) vi.stubEnv(key, value); const AsyncFunction = Object.getPrototypeOf(async function () {}).constructor as new ( ...arguments_: string[] ) => (...arguments_: unknown[]) => Promise; const execute = new AsyncFunction('github', 'context', 'core', 'require', publisherScript); await execute(github, context, core, requireCjs); } finally { vi.unstubAllEnvs(); rmSync(tempDir, { recursive: true, force: true }); } return { core, createComment, getPull, paginate: paginateIterator, updateComment }; } describe('gitnexus review-agent workflow security contract', () => { it('ships default-off activation and rollback instructions with the workflow', () => { expect(workflow).toContain('Activation checklist (the comment-trigger lane is OFF by default)'); expect(workflow).toContain('Configure the repository secret CLAUDE_CODE_OAUTH_TOKEN'); expect(workflow).toContain( 'Run workflow_dispatch against a disposable same-repo PR and a fork PR', ); expect(workflow).toContain('GITNEXUS_REVIEW_COMMENT_ENABLED=true'); expect(workflow).toContain('Roll back immediately by setting that variable to false'); }); it('pins every third-party action and the GitNexus analyzer exactly', () => { const expectedPins = [ 'actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0', 'actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3', 'actions/setup-node@820762786026740c76f36085b0efc47a31fe5020', 'actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a', 'actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c', 'anthropics/claude-code-action/base-action@3553f84341b92da26052e28acf1aa898f9511f32', ]; for (const pin of expectedPins) { expect(workflow).toContain(pin); } const uses = [...workflow.matchAll(/^\s*-?\s*uses:\s*([^\s#]+)/gm)].map((match) => match[1]); expect(uses.length).toBeGreaterThan(0); for (const action of uses) { expect(action, `${action} must use a full immutable SHA`).toMatch(/@[0-9a-f]{40}$/); } expect(runtimePackage.dependencies?.gitnexus).toBe('1.6.9'); expect(runtimePackage.engines?.node).toBe('22.18.0'); expect(runtimeLock.packages?.['node_modules/gitnexus']?.version).toBe('1.6.9'); expect(runtimeLock.packages?.['node_modules/gitnexus']?.integrity).toMatch(/^sha512-/); expect(workflow).not.toMatch(/gitnexus@(latest|next|beta)/); expect(workflow).toContain("node-version: '22.18.0'"); expect(workflow).toContain('test "$(node --version)" = \'v22.18.0\''); expect(workflow).toContain('.github/scripts/npm-ci-retry.sh'); expect(workflow).not.toContain('--package-lock=false'); expect(workflow).toContain( 'actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0', ); expect(workflow).toContain( 'actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0', ); }); it('installs inert lock payloads, then activates and preflights them offline', () => { const runtimeStep = workflowDocument.jobs?.analyze?.steps?.find( ({ name }) => name === 'Prepare exact GitNexus runtime and strict MCP config', ); expect(runtimeStep?.env).toEqual({ NPM_CONFIG_IGNORE_SCRIPTS: 'true', ONNXRUNTIME_NODE_INSTALL: 'skip', SCARF_ANALYTICS: 'false', DO_NOT_TRACK: '1', }); const script = typeof runtimeStep?.run === 'string' ? runtimeStep.run : ''; expect(script).toContain('.github/scripts/npm-ci-retry.sh'); expect(npmCiHelper).toContain('--ignore-scripts=true'); expect(script).toContain('"${npm_path}" rebuild'); expect(script).toContain('NPM_CONFIG_OFFLINE=true'); expect(script).toContain('--offline'); expect(script).toContain('--unshare-net'); expect(script).toContain('NPM_CONFIG_IGNORE_SCRIPTS=false'); expect(script).toContain('"${runtime_dir}/node_modules/.bin/gitnexus" analyze'); expect(script).toContain('"${runtime_dir}/node_modules/.bin/gitnexus" status'); // A registry ECONNRESET during this install burned a whole review run; the // lock is pinned, so a bounded retry can only refetch the identical tree. // Both installs call one shared helper, exercised behaviourally below. expect(script).toContain('/.github/scripts/npm-ci-retry.sh'); expect(script).toContain("'analyzer runtime'"); expect(script.indexOf('npm ci')).toBeLessThan(script.indexOf('"${npm_path}" rebuild')); expect(script.indexOf('"${npm_path}" rebuild')).toBeLessThan( script.indexOf('"${runtime_dir}/node_modules/.bin/gitnexus" analyze'), ); }); it('runs the hostile-derived MCP database reader in a separate bounded sandbox', () => { const script = jobRun('analyze', 'Prepare exact GitNexus runtime and strict MCP config'); const start = script.indexOf('# The index is derived from hostile parser input.'); const end = script.indexOf('chmod 0755 "${mcp_wrapper}"', start); expect(start).toBeGreaterThan(-1); expect(end).toBeGreaterThan(start); const mcpSandbox = script.slice(start, end); expect(script).toContain('mcp_wrapper="${RUNNER_TEMP}/gitnexus-review-mcp"'); expect(script).toContain('MCP_WRAPPER="${mcp_wrapper}" MCP_CONFIG="${mcp_config}"'); expect(script).toContain('command: process.env.MCP_WRAPPER'); expect(script).toContain('args: []'); expect(script).not.toContain('command: process.env.WRAPPER'); expect(mcpSandbox).toContain('--unshare-user'); expect(mcpSandbox).toContain('--unshare-pid'); expect(mcpSandbox).toContain('--unshare-net'); expect(mcpSandbox).toContain('--die-with-parent'); expect(mcpSandbox).toContain('--new-session'); expect(mcpSandbox).toContain('--ro-bind / /'); expect(mcpSandbox).toContain('--ro-bind "${source_dir}" "${source_dir}"'); expect(mcpSandbox).toContain('--ro-bind "${base_source_dir}" "${base_source_dir}"'); expect(mcpSandbox).toContain('--ro-bind "${runtime_dir}" "${runtime_dir}"'); expect(mcpSandbox).toContain('--ro-bind "${claude_runtime_dir}" "${claude_runtime_dir}"'); expect(mcpSandbox).toContain('--bind "${storage_dir}" "${storage_dir}"'); expect(mcpSandbox).toContain('--bind "${base_storage_dir}" "${base_storage_dir}"'); expect(mcpSandbox).toContain('--bind "${index_home}" "${index_home}"'); expect(mcpSandbox).toContain('--bind "${mcp_home}" "${mcp_home}"'); expect(mcpSandbox).toContain('--bind "${mcp_tmp}" "${mcp_tmp}"'); expect(mcpSandbox).toContain('/usr/bin/env -i'); expect(mcpSandbox).toContain('GITNEXUS_MCP_READ_ONLY=1'); expect(mcpSandbox).toContain('GITNEXUS_MCP_ALLOWED_REPOS=${source_dir},${base_source_dir}'); expect(mcpSandbox).toContain('requested_command=(mcp)'); expect(mcpSandbox).toContain( '"${runtime_dir}/node_modules/.bin/gitnexus" "${requested_command[@]}"', ); expect(mcpSandbox).not.toContain('--bind "${source_dir}" "${source_dir}"'); expect(mcpSandbox).not.toContain('CLAUDE_CODE_OAUTH_TOKEN'); }); it('installs the exact secret-consuming Claude executable before the action runs', () => { const runtimeStep = workflowDocument.jobs?.analyze?.steps?.find( ({ name }) => name === 'Prepare exact Claude Code executable', ); const claudeStep = workflowDocument.jobs?.analyze?.steps?.find( ({ name }) => name === 'Run read-only graph-backed review', ); const script = typeof runtimeStep?.run === 'string' ? runtimeStep.run : ''; const expectedExecutable = '${{ runner.temp }}/gitnexus-review-claude-runtime/node_modules/@anthropic-ai/claude-code/bin/claude.exe'; expect(claudeRuntimePackage.dependencies?.['@anthropic-ai/claude-code']).toBe('2.1.214'); expect(claudeRuntimePackage.engines?.node).toBe('22.18.0'); expect(claudeRuntimeLock.lockfileVersion).toBe(3); expect(claudeRuntimeLock.packages?.['node_modules/@anthropic-ai/claude-code']).toMatchObject({ version: '2.1.214', integrity: 'sha512-Gf8XbPHBacVqBlxx8sMnKWPEU6AvRNUcjD0FS6zhD44fCgCHcpbpxwSoTbHlLTqKsr/0S7wdfhjjOIq8WlYbng==', }); expect( claudeRuntimeLock.packages?.['node_modules/@anthropic-ai/claude-code-linux-x64'], ).toMatchObject({ version: '2.1.214', integrity: 'sha512-NSQjXX8QjjjYdDlYbPvlse5yQ3UwsmV2vuPNR3eFaXnGVv7ymFHvDSMIkTFRLXQlmPjp+tvAN5fbH3e1C38SOw==', }); expect(runtimeStep?.env).toEqual({ NPM_CONFIG_IGNORE_SCRIPTS: 'true', DO_NOT_TRACK: '1', }); expect(script).toContain('.github/claude-canary-runtime/package-lock.json'); expect(script).toContain('.github/scripts/npm-ci-retry.sh'); expect(script).toContain('/.github/scripts/npm-ci-retry.sh'); expect(script).toContain("'Claude runtime'"); expect(npmCiHelper).toContain('--ignore-scripts=true'); expect(script).toContain('--unshare-net'); expect(script).toContain('NPM_CONFIG_OFFLINE=true'); expect(script).toContain('@anthropic-ai/claude-code/install.cjs'); expect(script).toContain('cmp --silent'); expect(script).toContain('3c029136f7c81f54ed4a38e9d52e655aad536433dbbde50519c8c31bb646ad14'); expect(script).toContain("'2.1.214 (Claude Code)'"); expect(claudeStep?.with?.path_to_claude_code_executable).toBe(expectedExecutable); expect(workflow).not.toContain('https://claude.ai/install.sh'); expect(workflow.indexOf('id: claude-runtime')).toBeLessThan( workflow.indexOf('claude_code_oauth_token:'), ); expect(workflow).toContain('CLAUDE_RUNTIME_OUTCOME: ${{ steps.claude-runtime.outcome }}'); }); it('gates comment triggers exactly and checks API write authority before model spend', () => { expect(workflow).toContain("github.event.comment.body == '@gitnexus review'"); expect(workflow).not.toContain("contains(github.event.comment.body, '@gitnexus review')"); expect(workflow).toContain("vars.GITNEXUS_REVIEW_COMMENT_ENABLED == 'true'"); expect(workflow).toContain("github.event.comment.author_association == 'OWNER'"); expect(workflow).toContain("github.event.comment.author_association == 'MEMBER'"); expect(workflow).toContain("github.event.comment.author_association == 'COLLABORATOR'"); expect(workflow).not.toContain("github.event.comment.author_association == 'CONTRIBUTOR'"); const permissionCheck = workflow.indexOf('getCollaboratorPermissionLevel'); const modelInvocation = workflow.indexOf('anthropics/claude-code-action/base-action@'); expect(permissionCheck).toBeGreaterThan(-1); expect(modelInvocation).toBeGreaterThan(permissionCheck); expect(workflow).toContain("['admin', 'maintain', 'write'].includes(permission)"); expect(workflow).toContain("steps.context.outputs.authorized == 'true'"); }); it('normalizes both events through the API and rejects unsafe PR metadata', () => { expect(workflow).toContain("github.event_name == 'workflow_dispatch'"); expect(workflow).toContain('github.rest.pulls.get'); expect(workflow).toContain("pull.state !== 'open'"); expect(workflow).toContain('pull.base.repo.full_name'); expect(workflow).toContain('pull.head.repo'); expect(workflow).toContain('pull.head.sha'); expect(workflow).toContain('pull.base.sha'); expect(workflow).toMatch(/\^\\d\+\$|\^\[1-9\]\\d\*\$/); expect(workflow).toContain('/^[0-9a-f]{40}$/'); expect(workflow).toContain('context.repo.owner'); expect(workflow).toContain('context.repo.repo'); expect(workflow).toContain('Number.isSafeInteger(prNumber)'); }); it('executes normalization for dispatch and comment events before enabling the model path', async () => { const dispatch = await runContextScenario({ eventName: 'workflow_dispatch', dispatchPr: String(PR_NUMBER), eventPr: 'not-used', permission: 'admin', }); expect(dispatch.getPermission).toHaveBeenCalledWith({ owner: 'owner', repo: 'repo', username: 'trusted-maintainer', }); expect(dispatch.getPull).toHaveBeenCalledWith({ owner: 'owner', repo: 'repo', pull_number: PR_NUMBER, }); expect(Object.fromEntries(dispatch.outputs)).toMatchObject({ authorized: 'true', ready: 'true', pr_number: String(PR_NUMBER), control_sha: CONTROL_SHA, head_repo: 'fork/repo', head_sha: HEAD_SHA, base_sha: BASE_SHA, failure_code: 'none', }); const comment = await runContextScenario({ eventName: 'issue_comment', dispatchPr: 'hostile-not-used', eventPr: String(PR_NUMBER), permission: 'maintain', }); expect(comment.outputs.get('ready')).toBe('true'); expect(comment.outputs.get('failure_code')).toBe('none'); }); it('executes the authorization boundary and fails hostile request metadata closed', async () => { for (const rawPr of ['0', '01', '-1', '1e3', '2431 trailing', '9007199254740992']) { const invalid = await runContextScenario({ dispatchPr: rawPr }); expect(invalid.outputs.get('authorized')).toBe('false'); expect(invalid.outputs.get('ready')).toBe('false'); expect(invalid.outputs.get('failure_code')).toBe('invalid_pr_number'); expect(invalid.getPermission).not.toHaveBeenCalled(); expect(invalid.getPull).not.toHaveBeenCalled(); } const invalidControl = await runContextScenario({ controlSha: `g${CONTROL_SHA.slice(1)}` }); expect(invalidControl.outputs.get('failure_code')).toBe('invalid_control_sha'); expect(invalidControl.getPermission).not.toHaveBeenCalled(); const reader = await runContextScenario({ permission: 'read' }); expect(reader.outputs.get('authorized')).toBe('false'); expect(reader.outputs.get('ready')).toBe('false'); expect(reader.outputs.get('failure_code')).toBe('actor_not_authorized'); expect(reader.getPull).not.toHaveBeenCalled(); }); it('executes PR tuple validation and rejects hostile API responses', async () => { const validHead = { sha: HEAD_SHA, repo: { full_name: 'fork/repo' } }; const validBase = { sha: BASE_SHA, repo: { full_name: 'owner/repo' } }; const cases: Array<{ failure: string; pull: Record }> = [ { failure: 'invalid_pr_sha', pull: { state: 'open', head: { ...validHead, sha: 'not-a-sha' }, base: validBase }, }, { failure: 'pr_not_open', pull: { state: 'closed', head: validHead, base: validBase }, }, { failure: 'wrong_base_repository', pull: { state: 'open', head: validHead, base: { ...validBase, repo: { full_name: 'attacker/repo' } }, }, }, { failure: 'head_repository_deleted', pull: { state: 'open', head: { sha: HEAD_SHA, repo: null }, base: validBase }, }, { failure: 'invalid_head_repository', pull: { state: 'open', head: { sha: HEAD_SHA, repo: { full_name: 'attacker/repo/extra' } }, base: validBase, }, }, ]; for (const scenario of cases) { const result = await runContextScenario({ pull: scenario.pull }); expect(result.outputs.get('ready')).toBe('false'); expect(result.outputs.get('failure_code')).toBe(scenario.failure); } const unavailable = await runContextScenario({ permissionError: new Error('API unavailable'), }); expect(unavailable.outputs.get('authorized')).toBe('false'); expect(unavailable.outputs.get('ready')).toBe('false'); expect(unavailable.outputs.get('failure_code')).toBe('metadata_unavailable'); expect(unavailable.core.debug).toHaveBeenCalled(); }); it('checks out trusted control code at the workflow SHA and treats fork code as passive data', () => { const analyze = jobBlock('analyze'); expect(analyze).not.toBe(''); expect(analyze).toContain('repository: ${{ github.repository }}'); expect(analyze).toContain('ref: ${{ steps.context.outputs.control_sha }}'); expect(analyze).toContain('repository: ${{ steps.context.outputs.head_repo }}'); expect(analyze).toContain('ref: ${{ steps.context.outputs.head_sha }}'); expect(analyze).toContain('path: pr-target'); expect(analyze.match(/fetch-depth: 0/g)?.length).toBeGreaterThanOrEqual(2); expect(analyze.match(/persist-credentials: false/g)?.length).toBeGreaterThanOrEqual(2); expect(analyze.match(/submodules: false/g)?.length).toBeGreaterThanOrEqual(2); expect(analyze.match(/lfs: false/g)?.length).toBeGreaterThanOrEqual(2); // Fetch the base commit as an object without checking it out. A non-shallow // fetch is intentional because the trusted diff step must compute the true // merge-base even when the base branch advanced after the fork point. expect(analyze).toContain('git fetch --no-tags origin "${BASE_SHA}"'); expect(analyze).toContain('git cat-file -e "${BASE_SHA}^{commit}"'); expect(analyze).toContain('git rev-parse HEAD'); expect(analyze).toContain('git -C pr-target rev-parse HEAD'); expect(analyze).toContain('find pr-target -path pr-target/.gitnexus -prune -o -type l -print0'); expect(analyze).not.toContain('find pr-target -type l -print0'); expect(analyze).toContain('realpath -m'); expect(analyze).toContain('Escaping symlink'); expect(analyze).toContain('gitnexus-review-hostile-dot-gitnexus'); // The trusted job installs the lock-resolved analyzer into RUNNER_TEMP, // but it must never invoke package scripts from the PR checkout. expect(analyze).not.toMatch(/cd[^\n]*pr-target[\s\S]{0,200}npm\s+(ci|install|run)\b/); expect(analyze).not.toContain('npm install'); expect(analyze).toContain('.github/scripts/npm-ci-retry.sh'); expect(npmCiHelper).toContain('--prefix "${runtime_dir}"'); expect(analyze).not.toContain('pr-target/.mcp.json'); expect(analyze).not.toContain('pr-target/.claude'); expect(analyze).not.toContain('node pr-target/.gitnexus/run.cjs'); expect(analyze).toContain("GITNEXUS_NO_GITIGNORE: '1'"); expect(analyze).toContain('for config in .gitnexusrc .gitnexusignore'); expect(analyze).toContain('trap restore_target_config EXIT'); }); it('contains the real hostile index build and exposes only dedicated writable stores', () => { const script = jobRun('analyze', 'Build the exact-head graph index'); const readOnlySource = '--ro-bind "${GITHUB_WORKSPACE}/pr-target" "${GITHUB_WORKSPACE}/pr-target"'; const writableStorage = '--bind "${storage_dir}" "${storage_dir}"'; const analyzeInvocation = '"${wrapper}" analyze --force --pdg --index-only --no-stats'; expect(script).toContain('storage_dir="${GITHUB_WORKSPACE}/pr-target/.gitnexus"'); expect(script).toContain('storage_quarantine='); expect(script).toContain('mv -- "${storage_dir}" "${storage_quarantine}"'); expect(script).toContain('test ! -e "${storage_dir}" && test ! -L "${storage_dir}"'); expect(script).toContain('install -d -m 0700 "${storage_dir}"'); expect(script).toContain("stat -c '%u'"); expect(script).toContain("stat -c '%a'"); expect(script).toContain('--unshare-user'); expect(script).toContain('--unshare-pid'); expect(script).toContain('--unshare-net'); expect(script).toContain('--die-with-parent'); expect(script).toContain('--new-session'); expect(script).toContain('--ro-bind / /'); expect(script).toContain(readOnlySource); expect(script).toContain(writableStorage); expect(script).toContain('--bind "${index_home}" "${index_home}"'); expect(script).toContain('--bind "${sandbox_home}" "${sandbox_home}"'); expect(script).toContain('--bind "${sandbox_tmp}" "${sandbox_tmp}"'); expect(script).toContain('/usr/bin/env -i'); expect(script).toContain(analyzeInvocation); expect(script.indexOf(readOnlySource)).toBeLessThan(script.indexOf(writableStorage)); expect(script.indexOf(writableStorage)).toBeLessThan(script.indexOf(analyzeInvocation)); expect(script).not.toContain( '--bind "${GITHUB_WORKSPACE}/pr-target" "${GITHUB_WORKSPACE}/pr-target"', ); expect(script).not.toContain('"${RUNNER_TEMP}/gitnexus-review" analyze'); }); it('builds exact head and merge-base graphs from trusted name-status topology', () => { const runtime = jobRun('analyze', 'Prepare exact GitNexus runtime and strict MCP config'); const mergeBase = jobRun('analyze', 'Materialize the exact merge-base graph source'); const index = jobRun('analyze', 'Build the exact-head graph index'); const inputs = jobRun('analyze', 'Prepare exact merge-base review inputs'); const prescan = jobRun('analyze', 'Prescan exact changed-symbol graph evidence'); expect(mergeBase).toContain('git -C pr-target merge-base'); expect(mergeBase).toContain('checkout --quiet --detach "${merge_base}"'); expect(mergeBase).toContain('write-tree'); expect(mergeBase).toContain('Escaping merge-base symlink'); expect(index).toContain('gitnexus-review-merge-base'); expect(index).toContain('gitnexus-review-hostile-base-dot-gitnexus'); expect(index.match(/analyze --force --pdg --index-only --no-stats/g)).toHaveLength(2); expect(runtime).toContain('GITNEXUS_MCP_ALLOWED_REPOS=${source_dir},${base_source_dir}'); expect(runtime).toContain('--ro-bind "${base_source_dir}" "${base_source_dir}"'); expect(inputs).toContain('--name-status'); expect(inputs).toContain("schema: 'gitnexus.changed-paths/v2'"); expect(inputs).toContain("status.startsWith('R')"); expect(inputs).toContain('basePaths.add(oldPath)'); expect(inputs).toContain('basePrescanPaths.add(oldPath)'); expect(inputs).toContain('headPaths.add(newPath)'); expect(prescan).toContain('manifest.base_prescan_paths'); expect(prescan).toContain("['cypher', statement, '--repo', repo, '--limit', '1']"); expect(prescan).toContain("AND NOT n.id STARTS WITH 'BasicBlock:'"); expect(prescan).toContain('no_indexable_changed_symbols'); }); it('parses deletion and rename-old paths into the merge-base evidence set', () => { const deletedPath = 'src/deleted.ts'; const oldPath = 'src/old-name.ts'; const newPath = 'src/new-name.ts'; const copySource = 'src/copy-source.ts'; const copyTarget = 'src/copy-target.ts'; const addedPath = 'src/added.ts'; const modifiedPath = 'src/modified.ts'; const { manifest, result } = runChangedPathManifest( `D\0${deletedPath}\0R077\0${oldPath}\0${newPath}\0C050\0${copySource}\0${copyTarget}\0A\0${addedPath}\0M\0${modifiedPath}\0`, ); expect(result.status).toBe(0); expect(manifest).toEqual({ schema: 'gitnexus.changed-paths/v2', entries: [ { status: 'D', base_path: deletedPath }, { status: 'R077', base_path: oldPath, head_path: newPath }, { status: 'C050', head_path: copyTarget, copy_source: copySource }, { status: 'A', head_path: addedPath }, { status: 'M', base_prescan_path: modifiedPath, head_path: modifiedPath, }, ], head_paths: [newPath, copyTarget, addedPath, modifiedPath], base_paths: [deletedPath, oldPath], base_prescan_paths: [deletedPath, oldPath, modifiedPath], prescan: null, }); const hostile = runChangedPathManifest('D\0../escape.ts\0'); expect(hostile.result.status).not.toBe(0); expect(hostile.manifest).toBeUndefined(); expect(hostile.result.stderr).toContain('invalid path'); }); it('accepts zero-padded rename and copy scores at the artifact boundary', () => { const renamedFrom = 'src/renamed-from.ts'; const renamedTo = 'src/renamed-to.ts'; const copiedFrom = 'src/copied-from.ts'; const copiedTo = 'src/copied-to.ts'; const { artifact } = runArtifactScenario({ basePaths: [renamedFrom], changedPaths: [renamedTo, copiedTo], entries: [ { status: 'R077', base_path: renamedFrom, head_path: renamedTo }, { status: 'C050', copy_source: copiedFrom, head_path: copiedTo }, ], rawTranscript: JSON.stringify( reviewTranscript({ toolInput: { name: 'renamedCommand', file_path: renamedTo }, toolResultContent: contextResultContent(renamedTo), }), ), }); expect(artifact.status).toBe('success'); }); it('copies the indexed HEAD tree below the passive review root without prefix collisions', () => { const analyze = jobBlock('analyze'); const prefixTemplate = analyze.match(/checkout-index --all --force --prefix="([^"]+)"/)?.[1]; expect(prefixTemplate).toBe('${review_dir}/'); const tempDir = mkdtempSync(path.join(tmpdir(), 'gitnexus-review-checkout-index-')); const repository = path.join(tempDir, 'source'); const reviewDirectory = path.join(tempDir, 'control', 'pr-target'); try { mkdirSync(path.join(repository, 'nested'), { recursive: true }); mkdirSync(reviewDirectory, { recursive: true }); writeFileSync(path.join(repository, 'root.txt'), 'root\n'); writeFileSync(path.join(repository, 'nested', 'child.txt'), 'child\n'); runGit(repository, ['init', '--quiet']); runGit(repository, ['add', 'root.txt', 'nested/child.txt']); const expandedPrefix = prefixTemplate?.replace('${review_dir}', reviewDirectory); expect(expandedPrefix).toBe(`${reviewDirectory}/`); runGit(repository, ['checkout-index', '--all', '--force', `--prefix=${expandedPrefix}`]); expect(existsSync(path.join(reviewDirectory, 'root.txt'))).toBe(true); expect(existsSync(path.join(reviewDirectory, 'nested', 'child.txt'))).toBe(true); expect(existsSync(`${reviewDirectory}root.txt`)).toBe(false); } finally { rmSync(tempDir, { recursive: true, force: true }); } }); it('keeps the model job read-only and the publisher secretless and checkout-free', () => { const analyze = jobBlock('analyze'); const publish = jobBlock('publish'); expect(workflow).toMatch(/^permissions:\s*\{\}\s*$/m); expect(analyze).toContain('contents: read'); expect(analyze).toContain('pull-requests: read'); expect(analyze).not.toContain('issues: write'); expect(publish).toContain('issues: write'); expect(publish).toContain('pull-requests: write'); expect(publish).not.toContain('contents: write'); expect(publish).not.toContain('actions/checkout@'); expect(publish).not.toContain('ANTHROPIC_API_KEY'); expect(publish).not.toContain('CLAUDE_CODE_OAUTH_TOKEN'); expect(publish).not.toContain('secrets.'); expect(publish).not.toContain('anthropics/claude-code-action/'); }); it('preflights the exact Bubblewrap primitive before exposing the model secret', () => { const analyze = jobBlock('analyze'); const install = analyze.indexOf( 'sudo apt-get install --yes --no-install-recommends bubblewrap', ); const canary = analyze.indexOf('--unshare-user'); const model = analyze.indexOf('claude_code_oauth_token:'); expect(install).toBeGreaterThan(-1); expect(canary).toBeGreaterThan(install); expect(model).toBeGreaterThan(canary); expect(analyze).toContain('kernel.apparmor_restrict_unprivileged_userns=0'); expect(analyze).toContain('--unshare-pid'); expect(analyze).toContain('--die-with-parent'); expect(analyze).toContain('--new-session'); expect(analyze).toContain('--ro-bind / /'); expect(analyze).toContain('CLAUDE_CODE_SUBPROCESS_ENV_SCRUB'); }); it('rechecks the pinned Claude executable immediately before exposing the secret', () => { const recheck = jobRun('analyze', 'Reverify exact Claude executable at secret boundary'); const inputs = workflow.indexOf('- name: Prepare exact merge-base review inputs'); const recheckStep = workflow.indexOf( '- name: Reverify exact Claude executable at secret boundary', ); const modelStep = workflow.indexOf('- name: Run read-only graph-backed review'); const token = workflow.indexOf('claude_code_oauth_token:'); expect(recheck).toContain('cmp --silent -- "${native_binary}" "${claude_binary}"'); expect(recheck).toContain('3c029136f7c81f54ed4a38e9d52e655aad536433dbbde50519c8c31bb646ad14'); expect(recheck).toContain("'2.1.214 (Claude Code)'"); expect(inputs).toBeGreaterThan(-1); expect(recheckStep).toBeGreaterThan(inputs); expect(modelStep).toBeGreaterThan(recheckStep); expect(token).toBeGreaterThan(modelStep); expect(workflow).toContain("steps.claude-recheck.outcome == 'success'"); expect(workflow).toContain('CLAUDE_RECHECK_OUTCOME: ${{ steps.claude-recheck.outcome }}'); expect(workflow).toContain("process.env.CLAUDE_RECHECK_OUTCOME !== 'success'"); }); it('uses only trusted agent configuration and a strict, exact analyzer MCP', () => { const analyze = jobBlock('analyze'); expect(analyze).toContain('anthropics/claude-code-action/base-action@'); expect(analyze).not.toContain('uses: anthropics/claude-code-action@'); expect(analyze).not.toContain('github_token:'); expect(analyze).toContain('--add-dir "${{ runner.temp }}/gitnexus-review-pr-target"'); expect(analyze).toContain('Read(${{ runner.temp }}/gitnexus-review-pr-target/**)'); expect(analyze).toContain('review_dir="${RUNNER_TEMP}/gitnexus-review-pr-target"'); expect(analyze).not.toContain('review_dir="${control_dir}/pr-target"'); expect(analyze).toContain( 'CLAUDE_CONFIG_DIR: ${{ runner.temp }}/gitnexus-review-claude-config', ); expect(analyze).toContain('CLAUDE_WORKING_DIR: ${{ runner.temp }}/gitnexus-review-control'); expect(analyze).toContain("NODE_VERSION: '22.18.0'"); expect(analyze).toContain('checkout-index --all --force'); expect(analyze).toContain('find "${review_dir}" -type l -print0'); expect(analyze).toContain('Escaping copied review symlink'); expect(analyze).toContain( 'cp -a -- .claude/skills/gitnexus-review/. "${control_dir}/trusted-skill/"', ); expect(analyze).toContain('trusted-skill/SKILL.md'); expect(analyze).toContain('"disableAllHooks":true'); expect(analyze).toContain('"disableSkillShellExecution":true'); expect(analyze).toContain('--setting-sources user'); expect(analyze).not.toContain('--setting-sources ""'); expect(analyze).toContain('--strict-mcp-config'); expect(analyze).toContain('--mcp-config'); expect(analyze).toContain('--disable-slash-commands'); expect(analyze).toContain('.github/gitnexus-review-runtime/package-lock.json'); expect(analyze).toContain('GITNEXUS_MCP_READ_ONLY=1'); expect(analyze).toContain('GITNEXUS_MCP_ALLOWED_REPOS'); expect(analyze).toContain('GITNEXUS_MCP_DEFAULT_REPO'); expect(analyze).toContain('NPM_CONFIG_IGNORE_SCRIPTS'); expect(analyze).toContain("CLAUDE_CODE_SUBPROCESS_ENV_SCRUB: '1'"); expect(analyze).toContain("CLAUDE_CODE_ADDITIONAL_DIRECTORIES_CLAUDE_MD: '0'"); expect(analyze).toContain('--disallowedTools'); expect(analyze).toContain('Bash'); expect(analyze).toContain('Write'); expect(analyze).toContain('Edit'); const allowedTools = analyze.match(/--allowedTools "([^"]+)"/)?.[1] ?? ''; const allowedToolRules = allowedTools.split(','); expect(allowedToolRules).toContain('Read(./**)'); expect(allowedToolRules).not.toContain('Read'); // Glob/Grep are neither enabled nor allow-listed: bare Glob/Grep are separate // tools that the Read()-scoped path denies below (/proc, github.workspace, // ...) do not cover, so allow-listing them would open an undenied read path // to the raw checkouts and host paths. Leaving them in --tools while denying // them by omission only burned model turns on denied calls, so they are off // the tool set entirely; lanes read via the scoped Read() rules and the MCP. expect(allowedToolRules).not.toContain('Glob'); expect(allowedToolRules).not.toContain('Grep'); // The merge-base source checkout is readable so lanes can inspect deleted or // rename-old source; a Read() allow rule grants access without triggering // --add-dir agent discovery. expect(allowedTools).toContain('Read(${{ runner.temp }}/gitnexus-review-merge-base/**)'); expect(allowedTools).toContain('mcp__gitnexus__impact'); expect(allowedTools).not.toContain('mcp__gitnexus__detect_changes'); expect(allowedTools).not.toContain('mcp__gitnexus__rename'); expect(allowedTools).not.toContain('mcp__gitnexus__cypher'); // Swarm posture: the orchestrator dispatches subagents via the Agent tool // (renamed from Task in Claude Code 2.1.63), scoped to the six trusted // control-SHA personas; Agent is not bare-denied (deny beats allow), and // lane calls cannot satisfy the evidence gate. expect(analyze).toContain('--tools "Read,Agent"'); // One rule per persona, never a grouped Agent(a,b,c): the pinned base // action parses allowedTools with `.flatMap((v) => v.split(","))`, which // would shatter a grouped rule into `Agent(ci-correctness-lens`, bare // names, and `ci-critic-lens)` before the SDK ever sees it. expect(allowedToolRules).toContain('Agent(ci-correctness-lens)'); expect(allowedToolRules).toContain('Agent(ci-critic-lens)'); expect(allowedTools).not.toMatch(/Agent\([^)]*,/); expect(allowedTools).not.toContain('Task'); const disallowedTools = analyze.match(/--disallowedTools "([^"]+)"/)?.[1] ?? ''; const disallowedToolRules = disallowedTools.split(','); expect(disallowedToolRules).toContain('Bash'); expect(disallowedToolRules).not.toContain('Agent'); expect(disallowedTools).not.toContain('Task'); expect(analyze).toContain( 'cp -a -- .claude/skills/gitnexus-review/ci-personas/. "${claude_config}/agents/"', ); // The passive add-dir tree is scanned for agent definitions; drop any // PR-controlled ones at any depth so only the trusted control-SHA personas // can be dispatched. Skills under the copy are NOT pruned (a skill-editing // PR must stay reviewable). expect(analyze).toContain( `find "\${review_dir}" -type d -path '*/.claude/agents' -prune -exec rm -rf -- {} +`, ); expect(analyze).not.toContain(".claude/skills' -prune"); expect(analyze).toContain("satisfy the publisher's context-evidence gate"); // The orchestrator's own evidence call is a precondition of dispatch, so a // fully-delegated run cannot leave the gate unsatisfied. expect(analyze).toContain('dispatching any lane'); expect(analyze).toContain('Read(/proc/**)'); expect(analyze).toContain('Read(${{ github.workspace }}/**)'); expect(analyze).toContain( 'mcp__gitnexus__detect_changes,mcp__gitnexus__rename,mcp__gitnexus__cypher', ); expect(analyze).toContain('# shellcheck disable=SC2016'); expect(analyze).toContain('HEAD_SHA: ${{ steps.context.outputs.head_sha }}'); expect(analyze).toContain('test "$(git -C pr-target rev-parse HEAD)" = "${HEAD_SHA}"'); expect(analyze).not.toContain( 'test "$(git -C pr-target rev-parse HEAD)" = "${{ steps.context.outputs.head_sha }}"', ); }); it('scopes Agent dispatch to exactly the installed ci-personas', () => { // Real dispatch cannot be proven without a model turn (print mode silently // ignores invalid settings and does not validate permission-rule content at // parse time), so the canary is the acceptance gate for that. What a unit // test CAN pin is that the scoped allowlist, the persona filenames, and each // persona's frontmatter name are the same set — catching a rename or typo in // any of the three without auth. Names are read one-per-rule because the // pinned action splits allowedTools on commas. const analyze = jobBlock('analyze'); const allowed = analyze.match(/--allowedTools "([^"]+)"/)?.[1] ?? ''; const allowlistNames = [...allowed.matchAll(/Agent\(([^),]+)\)/g)] .map((match) => match[1].trim()) .sort(); const personasDir = path.resolve( __dirname, '../../../.claude/skills/gitnexus-review/ci-personas', ); const personaStems = readdirSync(personasDir) .filter((file) => file.endsWith('.md')) .map((file) => file.replace(/\.md$/, '')) .sort(); expect(allowlistNames).toEqual(personaStems); const frontmatterNames = personaStems.map((stem) => { const body = readFileSync(path.join(personasDir, `${stem}.md`), 'utf8'); return body.match(/^name:\s*(\S+)\s*$/m)?.[1] ?? ''; }); expect(frontmatterNames).toEqual(personaStems); // The install source the workflow copies matches the directory the // allowlist scopes to, so the six names above are the six spawnable agents. expect(analyze).toContain('cp -a -- .claude/skills/gitnexus-review/ci-personas/.'); }); it('bounds swarm transcript volume with per-persona maxTurns that fit the caps', () => { const analyze = jobBlock('analyze'); const orchestratorTurns = Number(analyze.match(/--max-turns (\d+)/)?.[1] ?? '0'); const maxMessages = Number( (analyze.match(/MAX_TRANSCRIPT_MESSAGES = ([\d_]+)/)?.[1] ?? '0').replace(/_/g, ''), ); expect(orchestratorTurns).toBeGreaterThan(0); expect(maxMessages).toBeGreaterThan(0); const personasDir = path.resolve( __dirname, '../../../.claude/skills/gitnexus-review/ci-personas', ); const laneTurns = readdirSync(personasDir) .filter((file) => file.endsWith('.md')) .map((file) => { const body = readFileSync(path.join(personasDir, file), 'utf8'); const value = Number(body.match(/^maxTurns:\s*(\d+)\s*$/m)?.[1] ?? '0'); // Every lane declares a positive-integer turn budget so the transcript // is deterministically bounded (the runtime rejects non-positive values). expect(value).toBeGreaterThan(0); return { file, value }; }); expect(laneTurns).toHaveLength(6); const criticTurns = laneTurns.find((lane) => lane.file === 'ci-critic-lens.md')?.value ?? 0; const totalLaneTurns = laneTurns.reduce((sum, lane) => sum + lane.value, 0); // Worst case: the orchestrator, every lane once, and a second critic pass, // each turn yielding at most an assistant + a user(tool_result) message. The // bound must stay under the transcript cap so a full swarm run never bricks a // valid review; this fails if maxTurns is bumped without revisiting the cap. const worstCaseMessages = 2 * (orchestratorTurns + totalLaneTurns + criticTurns); expect(worstCaseMessages).toBeLessThan(maxMessages); }); it('marks the review in progress from a write-scoped job without weakening analyze', () => { const acknowledge = jobBlock('acknowledge'); // A dedicated, write-scoped job posts the in-progress marker under the same // authorization gate as analyze, so the model-facing analyze job stays // secretless and read-only. expect(acknowledge).toContain('pull-requests: write'); expect(acknowledge).toContain("github.event.comment.body == '@gitnexus review'"); expect(acknowledge).toContain("author_association == 'OWNER'"); expect(acknowledge).toContain('\nAccepted earlier review`, }; const { createComment, updateComment } = await runPublisherScenario({ artifactStatus: 'failure', comments: [existing], }); expect(updateComment).not.toHaveBeenCalled(); expect(createComment).not.toHaveBeenCalled(); }); it('replaces a same-tuple failure with a later accepted review', async () => { const existing = { id: 18, user: { login: 'github-actions[bot]' }, body: `\nEarlier failure`, }; const { createComment, updateComment } = await runPublisherScenario({ artifactStatus: 'success', comments: [existing], }); expect(createComment).not.toHaveBeenCalled(); expect(updateComment).toHaveBeenCalledOnce(); expect(updateComment.mock.calls[0]?.[0]).toMatchObject({ comment_id: existing.id, }); expect(updateComment.mock.calls[0]?.[0].body).toContain('Accepted review body'); }); it('re-fetches the PR tuple immediately before update or create and rejects a late move', async () => { const movedHead = '7'.repeat(40); const result = await runPublisherScenario({ artifactStatus: 'success', finalHead: movedHead, }); expect(result.getPull).toHaveBeenCalledTimes(2); expect(result.paginate).toHaveBeenCalledOnce(); expect(result.updateComment).not.toHaveBeenCalled(); expect(result.createComment).not.toHaveBeenCalled(); expect(result.core.setFailed).toHaveBeenCalledWith( expect.stringContaining('changed immediately before publication'), ); }); it('publishes an initial failure fallback but fails every stale tuple', async () => { const failed = await runPublisherScenario({ artifactStatus: 'failure' }); expect(failed.updateComment).not.toHaveBeenCalled(); expect(failed.createComment).toHaveBeenCalledOnce(); expect(failed.createComment.mock.calls[0]?.[0].body).toContain( 'GitNexus review — unable to complete', ); const stale = await runPublisherScenario({ artifactStatus: 'success', currentHead: '4'.repeat(40), }); expect(stale.updateComment).not.toHaveBeenCalled(); expect(stale.createComment).not.toHaveBeenCalled(); expect(stale.paginate).not.toHaveBeenCalled(); expect(stale.core.setFailed).toHaveBeenCalledWith(expect.stringContaining('stale output')); const staleBase = await runPublisherScenario({ artifactStatus: 'success', currentBase: '5'.repeat(40), }); expect(staleBase.createComment).not.toHaveBeenCalled(); expect(staleBase.paginate).not.toHaveBeenCalled(); expect(staleBase.core.setFailed).toHaveBeenCalledWith(expect.stringContaining('stale output')); }); it('rejects malformed and mismatched artifacts at the publisher boundary', async () => { const scenarios: PublisherScenario[] = [ { artifactStatus: 'success', rawArtifact: '{not-json' }, { artifactStatus: 'success', rawArtifact: Uint8Array.from([0xff, 0xfe]) }, { artifactStatus: 'success', artifactOverrides: { unexpected: true } }, { artifactStatus: 'success', artifactOverrides: { base_sha: '6'.repeat(40) } }, { artifactStatus: 'success', artifactOverrides: { body: 'x'.repeat(54_001) } }, { artifactStatus: 'success', artifactOverrides: { failure_code: 'model_failed' } }, { artifactStatus: 'success', artifactOverrides: { graph_evidence: { mode: 'no_indexable_changed_symbols', head_has_indexable_symbol: true, base_has_indexable_symbol: false, }, }, }, ]; for (const scenario of scenarios) { const result = await runPublisherScenario(scenario); expect(result.updateComment).not.toHaveBeenCalled(); expect(result.createComment).toHaveBeenCalledOnce(); expect(result.createComment.mock.calls[0]?.[0].body).toContain('failed safely'); expect(result.core.warning).toHaveBeenCalledOnce(); } }); it('streams bounded comment pages and keeps only the latest matching marker', async () => { const first = { id: 20, user: { login: 'github-actions[bot]' }, body: `\nFirst`, }; const latest = { ...first, id: 21, body: `${first.body}\nLatest` }; const bounded = await runPublisherScenario({ artifactStatus: 'success', commentPages: [[first], [latest]], }); expect(bounded.paginate).toHaveBeenCalledOnce(); expect(bounded.updateComment).toHaveBeenCalledOnce(); expect(bounded.updateComment.mock.calls[0]?.[0]).toMatchObject({ comment_id: latest.id }); const overCap = await runPublisherScenario({ artifactStatus: 'success', commentPages: Array.from({ length: 21 }, () => []), }); expect(overCap.createComment).not.toHaveBeenCalled(); expect(overCap.updateComment).not.toHaveBeenCalled(); expect(overCap.core.setFailed).toHaveBeenCalledWith( expect.stringContaining('exceeded the bounded publication scan'), ); }); it('retries a failing pinned install, then gives up, and stops on first success', () => { // The string assertions above cannot tell a working retry from a loop whose // `npm ci` was moved outside it, so drive the real helper with a stub npm. const script = path.resolve(__dirname, '../../../.github/scripts/npm-ci-retry.sh'); const runHelper = (failures: number) => { const dir = mkdtempSync(path.join(tmpdir(), 'npm-ci-retry-')); const counter = path.join(dir, 'attempts'); writeFileSync(counter, ''); writeFileSync( path.join(dir, 'npm'), `#!/usr/bin/env bash\nprintf 'x' >> ${counter}\nattempts=$(wc -c < ${counter})\n` + `if [ "$attempts" -le ${failures} ]; then exit 1; fi\nexit 0\n`, { mode: 0o755 }, ); writeFileSync(path.join(dir, 'sleep'), '#!/usr/bin/env bash\nexit 0\n', { mode: 0o755 }); const result = spawnSync('bash', [script, 'test runtime', dir, path.join(dir, '.npmrc')], { encoding: 'utf8', env: { ...process.env, PATH: `${dir}:${process.env.PATH ?? ''}` }, }); const attempts = readFileSync(counter, 'utf8').length; rmSync(dir, { recursive: true, force: true }); return { status: result.status, stderr: result.stderr, attempts }; }; expect(runHelper(0)).toMatchObject({ status: 0, attempts: 1 }); const recovered = runHelper(2); expect(recovered).toMatchObject({ status: 0, attempts: 3 }); expect(recovered.stderr).toContain('retrying (1/3)'); const exhausted = runHelper(3); expect(exhausted).toMatchObject({ status: 1, attempts: 3 }); expect(exhausted.stderr).toContain('failed after 3 attempts'); }); it('ships the install helper executable, since the workflow invokes it directly', () => { // Committed as 100644 it would fail on the runner with permission denied, // and no other check in this suite would notice. const mode = spawnSync('git', ['ls-files', '-s', '.github/scripts/npm-ci-retry.sh'], { cwd: path.resolve(__dirname, '../../..'), encoding: 'utf8', }).stdout; expect(mode.startsWith('100755')).toBe(true); }); it('bounds every install attempt so a hung registry cannot eat the job budget', () => { const helper = readFileSync( path.resolve(__dirname, '../../../.github/scripts/npm-ci-retry.sh'), 'utf8', ); expect(helper).toContain('timeout "${attempt_timeout}" npm ci'); expect(helper).toContain('NPM_CI_ATTEMPT_TIMEOUT_SECONDS:-600'); expect(helper).toContain('set -euo pipefail'); }); it('rejects a context result whose filePath is not a string', () => { // The transcript is treated as hostile-adjacent data: a non-string filePath // must be a clean reject, never an uncaught type error. const nonString = runArtifactScenario({ rawTranscript: JSON.stringify( reviewTranscript({ toolResultContent: JSON.stringify({ status: 'found', symbol: { uid: 'u', name: 'n', filePath: 42, startLine: 1, endLine: 2 }, }), }), ), }); expect(nonString.artifact.failure_code).toBe('missing_graph_evidence'); expect(nonString.stderr).toContain('results outside the changed paths: 1'); }); it('sanitizes an adversarial resolved path before it reaches the job log', () => { const hostile = 'src/\u001b[31m\n::set-output name=x::y\u202egnp.js'; const sanitized = runArtifactScenario({ rawTranscript: JSON.stringify( reviewTranscript({ toolResultContent: JSON.stringify({ status: 'found', symbol: { uid: 'u', name: 'n', filePath: hostile, startLine: 1, endLine: 2 }, }), }), ), }); expect(sanitized.artifact.failure_code).toBe('missing_graph_evidence'); expect(sanitized.stderr).not.toContain('::set-output'); expect(sanitized.stderr).not.toContain('\u001b'); expect(sanitized.stderr).not.toContain('\u202e'); expect(sanitized.stderr).toContain('src/?'); }); it('names the transcript shape when the envelope is rejected', () => { const wrongFirstMessage = runArtifactScenario({ rawTranscript: JSON.stringify([ { type: 'assistant', message: { role: 'assistant', content: [] } }, { type: 'result', subtype: 'success', is_error: false }, ]), }); expect(wrongFirstMessage.artifact.failure_code).toBe('invalid_execution_transcript'); expect(wrongFirstMessage.stderr).toContain('2 messages, first assistant/undefined'); }); it('counts an in-scope call whose result never arrives', () => { const noResult = runArtifactScenario({ rawTranscript: JSON.stringify([ { type: 'system', subtype: 'init', session_id: 's', uuid: '1' }, { type: 'assistant', parent_tool_use_id: null, session_id: 's', uuid: '2', message: { role: 'assistant', content: [ { type: 'tool_use', id: 'tool-1', name: 'mcp__gitnexus__context', input: { name: 'statusCommand' }, }, ], }, }, { type: 'result', subtype: 'success', is_error: false, session_id: 's', uuid: '3' }, ]), }); expect(noResult.artifact.failure_code).toBe('missing_graph_evidence'); expect(noResult.stderr).toContain('in-scope calls with no usable result: 1'); }); it('always clears the in-progress marker, even when analysis was never authorized', () => { // The acknowledge job posts the marker from the event alone, so gating the // whole publish job on authorization stranded it on the PR forever. const publish = jobBlock('publish'); expect(publish).toContain('if: always()'); expect(publish).toContain('- name: Remove the in-progress marker'); const removalIndex = publish.indexOf('- name: Remove the in-progress marker'); const gatedIndex = publish.indexOf( 'Validate freshness and upsert an accepted same-SHA comment', ); expect(gatedIndex).toBeGreaterThan(-1); expect(removalIndex).toBeGreaterThan(gatedIndex); // Publication itself stays authorization-gated. expect(publish).toContain("needs.analyze.outputs.authorized == 'true'"); }); it('refuses a stub body even when the model admits it is incomplete', () => { // A production run returned {"body":"placeholder","complete":false}: the // gate had already been satisfied, so this published a maintainer-visible // comment whose entire content was that word. const stub = runArtifactScenario({ structuredOutput: JSON.stringify({ body: 'placeholder', complete: false }), }); expect(stub.artifact).toMatchObject({ status: 'failure', failure_code: 'invalid_model_output', }); expect(stub.artifact.body).not.toContain('placeholder'); const stubButComplete = runArtifactScenario({ structuredOutput: JSON.stringify({ body: 'LGTM', complete: true }), }); expect(stubButComplete.artifact.failure_code).toBe('invalid_model_output'); // A real partial review still publishes, labelled incomplete. const realPartial = runArtifactScenario({ structuredOutput: JSON.stringify({ body: `**REQUEST CHANGES.** ${'This is a genuine partial review of the diff. '.repeat(6)}`, complete: false, }), }); expect(realPartial.artifact.failure_code).toBe('incomplete_analysis'); expect(realPartial.artifact.body).toContain('genuine partial review'); }); it('reports swarm dispatch from the transcript on every run', () => { const laneTranscript = (lanes: number) => { const messages: Array> = [ { type: 'system', subtype: 'init', session_id: 's', uuid: '1' }, { type: 'assistant', parent_tool_use_id: null, session_id: 's', uuid: '2', message: { role: 'assistant', content: [ ...Array.from({ length: lanes }, (_unused, index) => ({ type: 'tool_use', id: `lane-${index}`, name: 'Agent', input: { subagent_type: 'ci-correctness-lens' }, })), { type: 'tool_use', id: 'tool-1', name: 'mcp__gitnexus__context', input: { name: 'statusCommand' }, }, ], }, }, ...Array.from({ length: lanes }, (_unused, index) => ({ type: 'assistant', parent_tool_use_id: `lane-${index}`, session_id: 's', uuid: `lane-turn-${index}`, message: { role: 'assistant', content: [] }, })), { type: 'user', parent_tool_use_id: null, session_id: 's', uuid: '3', message: { role: 'user', content: [ { type: 'tool_result', tool_use_id: 'tool-1', is_error: false, content: contextResultContent(), }, ], }, }, { type: 'result', subtype: 'success', is_error: false, session_id: 's', uuid: '4' }, ]; return JSON.stringify(messages); }; const dispatched = runArtifactScenario({ rawTranscript: laneTranscript(6) }); expect(dispatched.artifact.failure_code).toBeNull(); expect(dispatched.stdout).toContain( 'Swarm dispatch: lane dispatches requested: 6; lanes that produced transcript turns: 6', ); // The failure this exists to expose: dispatch attempted, nothing came back. const silent = runArtifactScenario({ rawTranscript: laneTranscript(0) }); expect(silent.stdout).toContain( 'Swarm dispatch: lane dispatches requested: 0; lanes that produced transcript turns: 0', ); }); it('rejects a review that cites a location which does not exist', () => { const cite = (sha: string, file: string, line: string) => `https://github.com/owner/repo/blob/${sha}/${file}#L${line}`; const withBody = (link: string) => runArtifactScenario({ structuredOutput: JSON.stringify({ body: `**APPROVE.** ${'Reviewed the changed surface in detail. '.repeat(5)} See [here](${link}).`, complete: true, }), }); const real = withBody(cite(HEAD_SHA, CHANGED_PATH, '12-L20')); expect(real.artifact).toMatchObject({ status: 'success', failure_code: null }); expect(real.stdout).toContain('1 checked, 1 resolve, 1 land in the diff, 0 unverifiable'); const pastEof = withBody(cite(HEAD_SHA, CHANGED_PATH, '900')); expect(pastEof.artifact.failure_code).toBe('unverifiable_citations'); expect(pastEof.stderr).toContain('cites line 900 of a 40-line file'); const missingFile = withBody(cite(HEAD_SHA, 'gitnexus/src/cli/invented.ts', '3')); expect(missingFile.artifact.failure_code).toBe('unverifiable_citations'); expect(missingFile.stderr).toContain('cites a path that does not exist at that commit'); const foreignSha = withBody(cite('f'.repeat(40), CHANGED_PATH, '3')); expect(foreignSha.artifact.failure_code).toBe('unverifiable_citations'); expect(foreignSha.stderr).toContain('cites a commit that was not analyzed'); // The published body never carries the unverifiable text. expect(missingFile.artifact.body).toContain('do not exist at the analyzed commits'); expect(missingFile.artifact.body).not.toContain('invented.ts'); }); it('allows citing an unchanged file, and reports grounding without enforcing it', () => { // A caller the change breaks lives outside the diff; citing it is correct // review work, so existence is enforced and diff-membership is only logged. const unchanged = 'gitnexus/src/cli/untouched.ts'; const result = runArtifactScenario({ changedPaths: [CHANGED_PATH, unchanged], structuredOutput: JSON.stringify({ body: `**APPROVE.** ${'Reviewed the changed surface. '.repeat(6)} See [caller](https://github.com/owner/repo/blob/${HEAD_SHA}/${unchanged}#L5).`, complete: true, }), }); expect(result.artifact).toMatchObject({ status: 'success', failure_code: null }); expect(result.stdout).toContain('1 checked, 1 resolve, 1 land in the diff'); const noCitations = runArtifactScenario(); expect(noCitations.artifact.failure_code).toBeNull(); expect(noCitations.stdout).toContain('0 checked, 0 resolve, 0 land in the diff'); }); it('hands the rejection reason back to the model and publishes the repaired review', () => { // Before this, every rejection was terminal: the gate runs after the // transcript closes, so the model never learned why it failed. const analyze = jobBlock('analyze'); expect(analyze).toContain('- name: Check the model result before the transcript closes'); expect(analyze).toContain('review-precheck.cjs'); expect(analyze).toContain("steps.precheck.outputs.repair_reason != ''"); // The binary is re-verified before the secret is exposed a second time. const recheckIndex = analyze.indexOf( '- name: Reverify exact Claude executable before the repair', ); const repairIndex = analyze.indexOf( '- name: Repair the review once when the first result is unpublishable', ); expect(recheckIndex).toBeGreaterThan(-1); expect(repairIndex).toBeGreaterThan(recheckIndex); expect(analyze).toContain("steps.repair-recheck.outcome == 'success'"); // The repair is bounded well below the first attempt. const turnCaps = [...analyze.matchAll(/--max-turns (\d+)/g)].map((match) => Number(match[1])); expect(turnCaps).toEqual([150, 60]); const repaired = runArtifactScenario({ structuredOutput: JSON.stringify({ body: 'placeholder', complete: false }), repairStructuredOutput: JSON.stringify({ body: `**APPROVE.** ${'The repaired review covers the changed surface. '.repeat(5)}`, complete: true, }), }); expect(repaired.artifact).toMatchObject({ status: 'success', failure_code: null }); expect(repaired.artifact.body).toContain('repaired review covers'); expect(repaired.stdout).toContain('Publishing the repaired review'); // A repair that itself fails must not rescue the rejected first result. const repairFailed = runArtifactScenario({ structuredOutput: JSON.stringify({ body: 'placeholder', complete: false }), repairOutcome: 'failure', }); expect(repairFailed.artifact.failure_code).toBe('invalid_model_output'); }); it('skips a request whose exact head and base already carry an accepted review', async () => { // Real PRs took two and three full runs each; a repeat request reproduced a // comment that was already on the page, at full model cost. const marker = ``; const accepted = await runContextScenario({ permission: 'write', existingComments: [ { user: { login: 'github-actions[bot]' }, body: `${marker}\n**APPROVE.** Looks good.` }, ], }); expect(accepted.outputs.get('ready')).toBe('false'); expect(accepted.outputs.get('failure_code')).toBe('already_reviewed'); // A previous FAILURE at the same tuple must still be retryable. const previouslyFailed = await runContextScenario({ permission: 'write', existingComments: [ { user: { login: 'github-actions[bot]' }, body: `${marker}\n### GitNexus review — unable to complete\n\nnope`, }, ], }); expect(previouslyFailed.outputs.get('ready')).toBe('true'); // A review of a different commit does not suppress this one. const otherCommit = await runContextScenario({ permission: 'write', existingComments: [ { user: { login: 'github-actions[bot]' }, body: `\nold`, }, ], }); expect(otherCommit.outputs.get('ready')).toBe('true'); // And a human comment quoting the marker cannot suppress a review. const impostor = await runContextScenario({ permission: 'write', existingComments: [{ user: { login: 'someone' }, body: `${marker}\nlooks fine to me` }], }); expect(impostor.outputs.get('ready')).toBe('true'); }); it('stops before the model when the pull request moved during preparation', () => { const analyze = jobBlock('analyze'); expect(analyze).toContain( '- name: Confirm the pull request has not moved before spending the model', ); expect(analyze).toContain("steps.freshness.outcome == 'success'"); const freshnessIndex = analyze.indexOf('id: freshness'); const modelIndex = analyze.indexOf('- name: Run read-only graph-backed review'); expect(freshnessIndex).toBeGreaterThan(-1); expect(modelIndex).toBeGreaterThan(freshnessIndex); const freshnessScript = jobScript( 'analyze', 'Confirm the pull request has not moved before spending the model', ); expect(freshnessScript).toContain('core.setFailed'); expect(freshnessScript).toContain('stopping before the model runs'); }); it('records what each run spent so waste is visible without grepping logs', () => { const spent = runArtifactScenario(); expect(spent.stdout).toContain('Model spend: turns: 25; cost: $3.88.'); // A transcript without the fields must not break the run. const unknown = runArtifactScenario({ rawTranscript: JSON.stringify(reviewTranscriptWithoutTools()), changedPaths: ['docs/x.md'], noIndexableChangedSymbols: true, }); expect(unknown.stdout).toContain('Model spend: turns: unknown; cost: unknown.'); }); });