From b118464d402ae561f8ba71a91ce631af07cb27f7 Mon Sep 17 00:00:00 2001 From: cemiboou Date: Thu, 7 May 2026 16:15:40 +0800 Subject: [PATCH] feat(llm-client): add anthropic provider for Claude Messages API Adds 'anthropic' to LLMProvider so users can point gitnexus at api.anthropic.com (or any Anthropic-protocol proxy / gateway) without running an OpenAI-shim like LiteLLM. Changes in src/core/wiki/llm-client.ts: - LLMProvider union gains 'anthropic'. - Adds DEFAULT_ANTHROPIC_VERSION constant ('2023-06-01') and LLMConfig.anthropicVersion override. - buildRequestUrl: routes to '/messages' when provider === 'anthropic'. - callLLM: branches headers (x-api-key + anthropic-version), request body (top-level system, max_tokens, content blocks), and response parsing (content[].text, usage.input_tokens / output_tokens). - New readAnthropicSSEStream parser handles message_start / content_block_delta / message_delta / error events. Changes in src/storage/repo-manager.ts: - CLIConfig provider union gains 'anthropic'. - New optional CLIConfig.anthropicVersion field. Existing openai / openrouter / azure / cursor / custom paths are untouched. Resolves usage of native Anthropic endpoints from the wiki CLI. refactor(repo-manager): import LLMProvider instead of inlining the union Removes the duplicated 'openai' | 'openrouter' | 'azure' | 'anthropic' | 'custom' | 'cursor' literal in CLIConfig and re-uses the single source of truth in llm-client.ts via a type-only import. Future provider additions only need one edit. --- README.md | 2 +- gitnexus/src/cli/index.ts | 7 +- gitnexus/src/cli/wiki.ts | 54 ++++- gitnexus/src/core/wiki/llm-client.ts | 249 ++++++++++++++++----- gitnexus/src/storage/repo-manager.ts | 5 +- gitnexus/test/unit/wiki-llm-client.test.ts | 231 +++++++++++++++++++ 6 files changed, 488 insertions(+), 60 deletions(-) diff --git a/README.md b/README.md index 2dad7f2ef..a290d1182 100644 --- a/README.md +++ b/README.md @@ -707,7 +707,7 @@ gitnexus wiki # Use a custom model or provider gitnexus wiki --model gpt-4o -gitnexus wiki --base-url https://api.anthropic.com/v1 +gitnexus wiki --provider anthropic --base-url https://api.anthropic.com --model claude-sonnet-4-5 # Force full regeneration gitnexus wiki --force diff --git a/gitnexus/src/cli/index.ts b/gitnexus/src/cli/index.ts index b89b40db0..bde0a25d6 100644 --- a/gitnexus/src/cli/index.ts +++ b/gitnexus/src/cli/index.ts @@ -133,11 +133,14 @@ program .command('wiki [path]') .description('Generate repository wiki from knowledge graph') .option('-f, --force', 'Force full regeneration even if up to date') - .option('--provider ', 'LLM provider: openai or cursor (default: openai)') + .option( + '--provider ', + 'LLM provider: openai, openrouter, azure, anthropic, custom, or cursor (default: openai)', + ) .option('--model ', 'LLM model or Azure deployment name (default: minimax/minimax-m2.5)') .option( '--base-url ', - 'LLM API base URL. Azure v1: https://{resource}.openai.azure.com/openai/v1', + 'LLM API base URL. Azure v1: https://{resource}.openai.azure.com/openai/v1. Anthropic: https://api.anthropic.com', ) .option('--api-key ', 'LLM API key or Azure api-key (saved to ~/.gitnexus/config.json)') .option( diff --git a/gitnexus/src/cli/wiki.ts b/gitnexus/src/cli/wiki.ts index ccd1cae4e..277cad2e4 100644 --- a/gitnexus/src/cli/wiki.ts +++ b/gitnexus/src/cli/wiki.ts @@ -192,7 +192,7 @@ export const wikiCommand = async (inputPath?: string, options?: WikiCommandOptio } else { console.log(" No LLM configured. Let's set it up.\n"); console.log( - ' Supports OpenAI, OpenRouter, Azure, any OpenAI-compatible API, or Cursor CLI.\n', + ' Supports OpenAI, OpenRouter, Azure, Anthropic, any OpenAI-compatible API, or Cursor CLI.\n', ); // Check if Cursor CLI is available @@ -203,12 +203,14 @@ export const wikiCommand = async (inputPath?: string, options?: WikiCommandOptio console.log(' [2] OpenRouter (openrouter.ai)'); console.log(' [3] Azure OpenAI'); console.log(' [4] Custom endpoint'); + const anthropicChoice = hasCursor ? '6' : '5'; if (hasCursor) { console.log(' [5] Cursor CLI (local, uses your Cursor subscription)'); } + console.log(` [${anthropicChoice}] Anthropic (api.anthropic.com)`); console.log(''); - const maxChoice = hasCursor ? '5' : '4'; + const maxChoice = anthropicChoice; const choice = await prompt(` Select provider (1/${maxChoice}): `); let baseUrl: string; @@ -292,6 +294,54 @@ export const wikiCommand = async (inputPath?: string, options?: WikiCommandOptio model: deploymentName, provider: 'azure', }; + } else if (choice === anthropicChoice) { + // Anthropic — fixed base URL, prompt for model + key + const anthropicBaseUrl = 'https://api.anthropic.com'; + const defaultAnthropicModel = 'claude-sonnet-4-5'; + + const modelInput = await prompt(` Model (default: ${defaultAnthropicModel}): `); + const model = modelInput || defaultAnthropicModel; + + // API key — prefer ANTHROPIC_API_KEY, fall back to generic GitNexus/OpenAI vars + const envKey = + process.env.ANTHROPIC_API_KEY || + process.env.GITNEXUS_API_KEY || + process.env.OPENAI_API_KEY || + ''; + let anthropicKey: string; + if (envKey) { + const masked = envKey.slice(0, 6) + '...' + envKey.slice(-4); + const useEnv = await prompt(` Use existing env key (${masked})? (Y/n): `); + if (!useEnv || useEnv.toLowerCase() === 'y' || useEnv.toLowerCase() === 'yes') { + anthropicKey = envKey; + } else { + anthropicKey = await prompt(' API key: ', true); + } + } else { + anthropicKey = await prompt(' API key: ', true); + } + + if (!anthropicKey) { + console.log('\n No key provided. Aborting.\n'); + process.exitCode = 1; + return; + } + + await saveCLIConfig({ + apiKey: anthropicKey, + baseUrl: anthropicBaseUrl, + model, + provider: 'anthropic', + }); + console.log(' Config saved to ~/.gitnexus/config.json\n'); + + llmConfig = { + ...llmConfig, + apiKey: anthropicKey, + baseUrl: anthropicBaseUrl, + model, + provider: 'anthropic', + }; } else { // OpenAI-compatible provider (OpenAI, OpenRouter, Custom) if (choice === '2') { diff --git a/gitnexus/src/core/wiki/llm-client.ts b/gitnexus/src/core/wiki/llm-client.ts index 6f446be15..6f2f4e69b 100644 --- a/gitnexus/src/core/wiki/llm-client.ts +++ b/gitnexus/src/core/wiki/llm-client.ts @@ -7,7 +7,10 @@ * Config priority: CLI flags > env vars > defaults */ -export type LLMProvider = 'openai' | 'openrouter' | 'azure' | 'custom' | 'cursor'; +export type LLMProvider = 'openai' | 'openrouter' | 'azure' | 'anthropic' | 'custom' | 'cursor'; + +/** Anthropic API version sent in the `anthropic-version` header. */ +export const DEFAULT_ANTHROPIC_VERSION = '2023-06-01'; export interface LLMConfig { apiKey: string; @@ -16,9 +19,11 @@ export interface LLMConfig { maxTokens: number; temperature: number; /** Provider type — controls auth header behaviour */ - provider?: 'openai' | 'openrouter' | 'azure' | 'custom' | 'cursor'; + provider?: LLMProvider; /** Azure api-version query param (e.g. '2024-10-21'). Appended to URL when set. */ apiVersion?: string; + /** Anthropic API version (e.g. '2023-06-01'). Sent as `anthropic-version` header when provider is 'anthropic'. */ + anthropicVersion?: string; /** When true, strips sampling params and uses max_completion_tokens instead of max_tokens */ isReasoningModel?: boolean; } @@ -64,6 +69,10 @@ export async function resolveLLMConfig(overrides?: Partial): Promise< provider: overrides?.provider ?? savedConfig.provider ?? 'openai', apiVersion: overrides?.apiVersion || process.env.GITNEXUS_AZURE_API_VERSION || savedConfig.apiVersion, + anthropicVersion: + overrides?.anthropicVersion || + process.env.GITNEXUS_ANTHROPIC_VERSION || + savedConfig.anthropicVersion, isReasoningModel: overrides?.isReasoningModel ?? savedConfig.isReasoningModel, }; } @@ -102,10 +111,25 @@ export function isReasoningModel(model: string, override?: boolean): boolean { } /** - * Build the full chat completions URL, appending ?api-version when provided. + * Build the full request URL. + * + * - For Anthropic: `${baseUrl}/v1/messages`. The `/v1` segment is auto-prepended when + * the base URL lacks any `/vN` version segment, so a user can configure + * `https://api.anthropic.com` and still hit the correct endpoint. + * - For OpenAI-compatible providers: `${baseUrl}/chat/completions`, with optional Azure + * `?api-version=` query param when provided. */ -export function buildRequestUrl(baseUrl: string, apiVersion: string | undefined): string { - const base = `${baseUrl.replace(/\/+$/, '')}/chat/completions`; +export function buildRequestUrl( + baseUrl: string, + apiVersion: string | undefined, + provider?: LLMProvider, +): string { + const trimmed = baseUrl.replace(/\/+$/, ''); + if (provider === 'anthropic') { + const versioned = /\/v\d+$/.test(trimmed) ? trimmed : `${trimmed}/v1`; + return `${versioned}/messages`; + } + const base = `${trimmed}/chat/completions`; return apiVersion ? `${base}?api-version=${encodeURIComponent(apiVersion)}` : base; } @@ -124,14 +148,9 @@ export async function callLLM( systemPrompt?: string, options?: CallLLMOptions, ): Promise { - const messages: Array<{ role: string; content: string }> = []; - if (systemPrompt) { - messages.push({ role: 'system', content: systemPrompt }); - } - messages.push({ role: 'user', content: prompt }); - - // Detect Azure endpoint (by provider field or URL pattern) - const azure = config.provider === 'azure' || isAzureProvider(config.baseUrl); + const anthropic = config.provider === 'anthropic'; + // Detect Azure endpoint (by provider field or URL pattern). Anthropic short-circuits Azure. + const azure = !anthropic && (config.provider === 'azure' || isAzureProvider(config.baseUrl)); // Warn when using Azure legacy deployment URL without api-version if (azure && !config.apiVersion && config.baseUrl.includes('/deployments/')) { @@ -143,29 +162,56 @@ export async function callLLM( // Detect reasoning model (o1, o3, o4-mini etc.) or explicit override const reasoning = isReasoningModel(config.model, config.isReasoningModel); - const url = buildRequestUrl(config.baseUrl, azure ? config.apiVersion : undefined); + const url = buildRequestUrl( + config.baseUrl, + azure ? config.apiVersion : undefined, + config.provider, + ); const useStream = !!options?.onChunk; - // Build request body — reasoning models reject temperature and use max_completion_tokens - const body: Record = { - model: config.model, - messages, - }; + // Build request body — Anthropic uses a different shape than OpenAI-compatible APIs. + let body: Record; + if (anthropic) { + body = { + model: config.model, + max_tokens: config.maxTokens, + messages: [{ role: 'user', content: prompt }], + }; + if (systemPrompt) body.system = systemPrompt; + if (config.temperature !== undefined) body.temperature = config.temperature; + if (useStream) body.stream = true; + } else { + const messages: Array<{ role: string; content: string }> = []; + if (systemPrompt) { + messages.push({ role: 'system', content: systemPrompt }); + } + messages.push({ role: 'user', content: prompt }); - // max_tokens is deprecated; use max_completion_tokens for all models - body.max_completion_tokens = config.maxTokens; + body = { + model: config.model, + messages, + }; - // Only send temperature for non-Azure providers — some Azure models reject non-default values - if (!reasoning && !azure && config.temperature !== undefined) { - body.temperature = config.temperature; + // max_tokens is deprecated; use max_completion_tokens for all OpenAI-compatible models + body.max_completion_tokens = config.maxTokens; + + // Only send temperature for non-Azure providers — some Azure models reject non-default values + if (!reasoning && !azure && config.temperature !== undefined) { + body.temperature = config.temperature; + } + + if (useStream) body.stream = true; } - if (useStream) body.stream = true; - - // Build auth headers — Azure uses api-key header, everyone else uses Authorization: Bearer - const authHeaders: Record = azure - ? { 'api-key': config.apiKey } - : { Authorization: `Bearer ${config.apiKey}` }; + // Build auth headers — provider determines header style. + const authHeaders: Record = anthropic + ? { + 'x-api-key': config.apiKey, + 'anthropic-version': config.anthropicVersion || DEFAULT_ANTHROPIC_VERSION, + } + : azure + ? { 'api-key': config.apiKey } + : { Authorization: `Bearer ${config.apiKey}` }; const MAX_RETRIES = 3; let lastError: Error | null = null; @@ -213,13 +259,32 @@ export async function callLLM( throw new Error(`LLM API error (${response.status}): ${errorText.slice(0, 500)}`); } - // Streaming path + // Streaming path — same reader, provider-specific event parser if (useStream && response.body) { - return await readSSEStream(response.body, options!.onChunk!); + const parse = anthropic ? parseAnthropicSSEEvent : parseOpenAISSEEvent; + return await readSSEStream(response.body, options!.onChunk!, parse); } // Non-streaming path const json = (await response.json()) as any; + + if (anthropic) { + const text = Array.isArray(json.content) + ? json.content + .filter((b: { type: string; text?: string }) => b.type === 'text' && b.text) + .map((b: { text: string }) => b.text) + .join('') + : ''; + if (!text) { + throw new Error('LLM returned empty response'); + } + return { + content: text, + promptTokens: json.usage?.input_tokens, + completionTokens: json.usage?.output_tokens, + }; + } + const choice = json.choices?.[0]; if (!choice?.message?.content) { throw new Error('LLM returned empty response'); @@ -250,17 +315,90 @@ export async function callLLM( } /** - * Read an SSE stream from an OpenAI-compatible streaming response. + * Outcome a provider-specific parser may report for one SSE event. + * + * The shared {@link readSSEStream} loop accumulates `delta` text, tracks token + * counts, throws immediately on `error`, and throws after stream end if a + * `refusalReason` was set — independent of the wire format. + */ +interface SSEEventResult { + /** Text fragment to append to the accumulated content. */ + delta?: string; + /** Prompt/input token count, when reported in this event. */ + inputTokens?: number; + /** Completion/output token count, when reported in this event. */ + outputTokens?: number; + /** + * If set, the stream is treated as refused/blocked: any `delta` on this event + * is dropped and the reader throws this message after the stream finishes. + */ + refusalReason?: string; + /** If set, the reader throws this message immediately. */ + error?: string; +} + +type SSEEventParser = (event: any) => SSEEventResult; + +/** Parse one OpenAI-compatible streaming event (`choices[].delta.content`). */ +function parseOpenAISSEEvent(event: any): SSEEventResult { + const choice = event?.choices?.[0]; + if (choice?.finish_reason === 'content_filter') { + return { + refusalReason: + 'content filter triggered mid-stream. The generated content was blocked by content policy. Adjust your prompt and retry.', + }; + } + const delta = choice?.delta?.content; + return delta ? { delta } : {}; +} + +/** Parse one Anthropic Messages streaming event (`message_start`/`content_block_delta`/`message_delta`/`error`). */ +function parseAnthropicSSEEvent(event: any): SSEEventResult { + switch (event?.type) { + case 'message_start': + return { inputTokens: event.message?.usage?.input_tokens }; + case 'content_block_delta': + if (event.delta?.type === 'text_delta' && typeof event.delta.text === 'string') { + return { delta: event.delta.text }; + } + return {}; + case 'message_delta': { + const result: SSEEventResult = {}; + if (event.usage?.output_tokens !== undefined) { + result.outputTokens = event.usage.output_tokens; + } + if (event.delta?.stop_reason === 'refusal') { + result.refusalReason = + 'Anthropic refused to generate content for this prompt. Adjust your prompt and retry.'; + } + return result; + } + case 'error': + return { + error: `Anthropic streaming error: ${event.error?.message || JSON.stringify(event.error)}`, + }; + default: + return {}; + } +} + +/** + * Read an SSE stream and accumulate `delta` text returned by the provider-specific + * `parse` function. Provider differences live entirely in `parse`; the loop, buffer + * handling, refusal/error semantics, and final response shape are shared. */ async function readSSEStream( body: ReadableStream, onChunk: (charsReceived: number) => void, + parse: SSEEventParser, ): Promise { const decoder = new TextDecoder(); const reader = body.getReader(); let content = ''; let buffer = ''; - let contentFilterTriggered = false; + let inputTokens: number | undefined; + let outputTokens: number | undefined; + let refusalReason: string | undefined; while (true) { const { done, value } = await reader.read(); @@ -276,38 +414,41 @@ async function readSSEStream( const data = trimmed.slice(6); if (data === '[DONE]') continue; + let event: unknown; try { - const parsed = JSON.parse(data); - const choice = parsed.choices?.[0]; - - // Detect content filter finish reason — skip delta from this chunk - if (choice?.finish_reason === 'content_filter') { - contentFilterTriggered = true; - continue; - } - - const delta = choice?.delta?.content; - if (delta) { - content += delta; - onChunk(content.length); - } + event = JSON.parse(data); } catch { // Skip malformed SSE chunks + continue; + } + + const result = parse(event); + + if (result.error) throw new Error(result.error); + + if (result.refusalReason) { + // Latch the first refusal; drop any delta on this event + refusalReason = refusalReason || result.refusalReason; + continue; + } + + if (result.inputTokens !== undefined) inputTokens = result.inputTokens; + if (result.outputTokens !== undefined) outputTokens = result.outputTokens; + + if (result.delta) { + content += result.delta; + onChunk(content.length); } } } - if (contentFilterTriggered) { - throw new Error( - 'content filter triggered mid-stream. The generated content was blocked by content policy. Adjust your prompt and retry.', - ); - } + if (refusalReason) throw new Error(refusalReason); if (!content) { throw new Error('LLM returned empty streaming response'); } - return { content }; + return { content, promptTokens: inputTokens, completionTokens: outputTokens }; } function sleep(ms: number): Promise { diff --git a/gitnexus/src/storage/repo-manager.ts b/gitnexus/src/storage/repo-manager.ts index 8c0bda95f..54c95b2d8 100644 --- a/gitnexus/src/storage/repo-manager.ts +++ b/gitnexus/src/storage/repo-manager.ts @@ -11,6 +11,7 @@ import { realpathSync } from 'fs'; import path from 'path'; import os from 'os'; import { getInferredRepoName, resolveRepoIdentityRoot } from './git.js'; +import type { LLMProvider } from '../core/wiki/llm-client.js'; /** * Normalise a repo path for registry comparison across platforms @@ -849,10 +850,12 @@ export interface CLIConfig { apiKey?: string; model?: string; baseUrl?: string; - provider?: 'openai' | 'openrouter' | 'azure' | 'custom' | 'cursor'; + provider?: LLMProvider; cursorModel?: string; /** Azure api-version query param (e.g. '2024-10-21'). Only used when provider is 'azure'. */ apiVersion?: string; + /** Anthropic API version (e.g. '2023-06-01'). Only used when provider is 'anthropic'. */ + anthropicVersion?: string; /** Set true when the deployment is a reasoning model (o1, o3, o4-mini). Auto-detected for OpenAI; must be set for Azure deployments. */ isReasoningModel?: boolean; } diff --git a/gitnexus/test/unit/wiki-llm-client.test.ts b/gitnexus/test/unit/wiki-llm-client.test.ts index c9d429904..af9a5c149 100644 --- a/gitnexus/test/unit/wiki-llm-client.test.ts +++ b/gitnexus/test/unit/wiki-llm-client.test.ts @@ -88,6 +88,27 @@ describe('buildRequestUrl', () => { 'https://myres.openai.azure.com/openai/v1/chat/completions', ); }); + + it('auto-prepends /v1 for Anthropic when base URL has no version segment', () => { + expect(buildRequestUrl('https://api.anthropic.com', undefined, 'anthropic')).toBe( + 'https://api.anthropic.com/v1/messages', + ); + }); + + it('strips trailing slash and auto-prepends /v1 for Anthropic', () => { + expect(buildRequestUrl('https://api.anthropic.com/', undefined, 'anthropic')).toBe( + 'https://api.anthropic.com/v1/messages', + ); + }); + + it('keeps existing /v1 segment for Anthropic without doubling', () => { + expect(buildRequestUrl('https://api.anthropic.com/v1', undefined, 'anthropic')).toBe( + 'https://api.anthropic.com/v1/messages', + ); + expect(buildRequestUrl('https://api.anthropic.com/v1/', undefined, 'anthropic')).toBe( + 'https://api.anthropic.com/v1/messages', + ); + }); }); describe('callLLM — auth header', () => { @@ -330,3 +351,213 @@ describe('readSSEStream — content_filter handling', () => { ).rejects.toThrow('content filter'); }); }); + +describe('callLLM — Anthropic provider', () => { + afterEach(() => vi.unstubAllGlobals()); + + it('uses x-api-key + anthropic-version headers and hits /v1/messages', async () => { + const fetchSpy = vi.fn().mockResolvedValue( + new Response( + JSON.stringify({ + content: [{ type: 'text', text: 'hi from claude' }], + usage: { input_tokens: 10, output_tokens: 5 }, + }), + { status: 200, headers: { 'Content-Type': 'application/json' } }, + ), + ); + vi.stubGlobal('fetch', fetchSpy); + + const { callLLM } = await import('../../src/core/wiki/llm-client.js'); + const res = await callLLM( + 'hello', + { + apiKey: 'sk-ant-test', + baseUrl: 'https://api.anthropic.com', + model: 'claude-sonnet-4-5', + maxTokens: 256, + temperature: 0, + provider: 'anthropic', + }, + 'be helpful', + ); + + const [url, init] = fetchSpy.mock.calls[0] as [ + string, + RequestInit & { headers: Record }, + ]; + + expect(url).toBe('https://api.anthropic.com/v1/messages'); + expect((init.headers as any)['x-api-key']).toBe('sk-ant-test'); + expect((init.headers as any)['anthropic-version']).toBe('2023-06-01'); + expect(init.headers['Authorization']).toBeUndefined(); + + const body = JSON.parse(init.body as string); + expect(body.model).toBe('claude-sonnet-4-5'); + expect(body.max_tokens).toBe(256); + expect(body.max_completion_tokens).toBeUndefined(); + expect(body.system).toBe('be helpful'); + expect(body.messages).toEqual([{ role: 'user', content: 'hello' }]); + + expect(res.content).toBe('hi from claude'); + expect(res.promptTokens).toBe(10); + expect(res.completionTokens).toBe(5); + }); + + it('honours configured anthropicVersion override', async () => { + const fetchSpy = vi.fn().mockResolvedValue( + new Response(JSON.stringify({ content: [{ type: 'text', text: 'ok' }], usage: {} }), { + status: 200, + headers: { 'Content-Type': 'application/json' }, + }), + ); + vi.stubGlobal('fetch', fetchSpy); + + const { callLLM } = await import('../../src/core/wiki/llm-client.js'); + await callLLM('test', { + apiKey: 'sk-ant-test', + baseUrl: 'https://api.anthropic.com/v1', + model: 'claude-sonnet-4-5', + maxTokens: 100, + temperature: 0, + provider: 'anthropic', + anthropicVersion: '2024-10-22', + }); + + const [url, init] = fetchSpy.mock.calls[0] as [ + string, + RequestInit & { headers: Record }, + ]; + expect(url).toBe('https://api.anthropic.com/v1/messages'); + expect((init.headers as any)['anthropic-version']).toBe('2024-10-22'); + }); + + it('streams Anthropic typed events and accumulates text deltas', async () => { + const streamContent = [ + 'data: {"type":"message_start","message":{"usage":{"input_tokens":7}}}\n\n', + 'data: {"type":"content_block_delta","delta":{"type":"text_delta","text":"Hello"}}\n\n', + 'data: {"type":"content_block_delta","delta":{"type":"text_delta","text":" world"}}\n\n', + 'data: {"type":"message_delta","usage":{"output_tokens":3},"delta":{"stop_reason":"end_turn"}}\n\n', + ].join(''); + + const encoder = new TextEncoder(); + const stream = new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(streamContent)); + controller.close(); + }, + }); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue( + new Response(stream, { + status: 200, + headers: { 'Content-Type': 'text/event-stream' }, + }), + ), + ); + + const chunks: number[] = []; + const { callLLM } = await import('../../src/core/wiki/llm-client.js'); + const res = await callLLM( + 'hi', + { + apiKey: 'sk-ant-test', + baseUrl: 'https://api.anthropic.com', + model: 'claude-sonnet-4-5', + maxTokens: 100, + temperature: 0, + provider: 'anthropic', + }, + undefined, + { onChunk: (n) => chunks.push(n) }, + ); + + expect(res.content).toBe('Hello world'); + expect(res.promptTokens).toBe(7); + expect(res.completionTokens).toBe(3); + expect(chunks).toEqual([5, 11]); + }); + + it('throws when Anthropic stream stop_reason is refusal', async () => { + const streamContent = [ + 'data: {"type":"message_start","message":{"usage":{"input_tokens":4}}}\n\n', + 'data: {"type":"content_block_delta","delta":{"type":"text_delta","text":"partial"}}\n\n', + 'data: {"type":"message_delta","usage":{"output_tokens":1},"delta":{"stop_reason":"refusal"}}\n\n', + ].join(''); + + const encoder = new TextEncoder(); + const stream = new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(streamContent)); + controller.close(); + }, + }); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue( + new Response(stream, { + status: 200, + headers: { 'Content-Type': 'text/event-stream' }, + }), + ), + ); + + const { callLLM } = await import('../../src/core/wiki/llm-client.js'); + await expect( + callLLM( + 'test', + { + apiKey: 'sk-ant-test', + baseUrl: 'https://api.anthropic.com', + model: 'claude-sonnet-4-5', + maxTokens: 100, + temperature: 0, + provider: 'anthropic', + }, + undefined, + { onChunk: () => {} }, + ), + ).rejects.toThrow('Anthropic refused'); + }); + + it('throws immediately on Anthropic error event', async () => { + const streamContent = [ + 'data: {"type":"message_start","message":{"usage":{"input_tokens":4}}}\n\n', + 'data: {"type":"error","error":{"type":"overloaded_error","message":"Service busy"}}\n\n', + ].join(''); + + const encoder = new TextEncoder(); + const stream = new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode(streamContent)); + controller.close(); + }, + }); + vi.stubGlobal( + 'fetch', + vi.fn().mockResolvedValue( + new Response(stream, { + status: 200, + headers: { 'Content-Type': 'text/event-stream' }, + }), + ), + ); + + const { callLLM } = await import('../../src/core/wiki/llm-client.js'); + await expect( + callLLM( + 'test', + { + apiKey: 'sk-ant-test', + baseUrl: 'https://api.anthropic.com', + model: 'claude-sonnet-4-5', + maxTokens: 100, + temperature: 0, + provider: 'anthropic', + }, + undefined, + { onChunk: () => {} }, + ), + ).rejects.toThrow('Anthropic streaming error: Service busy'); + }); +});