diff --git a/src/api/providers/__tests__/anthropic-error-handling.spec.ts b/src/api/providers/__tests__/anthropic-error-handling.spec.ts new file mode 100644 index 0000000000..3fbe50df38 --- /dev/null +++ b/src/api/providers/__tests__/anthropic-error-handling.spec.ts @@ -0,0 +1,218 @@ +import { describe, it, expect, vi, beforeEach } from "vitest" +import { AnthropicHandler } from "../anthropic" +import { Anthropic } from "@anthropic-ai/sdk" + +// Mock the Anthropic SDK +vi.mock("@anthropic-ai/sdk", () => { + const mockAnthropicConstructor = vi.fn().mockImplementation(() => ({ + messages: { + create: vi.fn(), + }, + })) + + return { + Anthropic: mockAnthropicConstructor, + } +}) + +describe("AnthropicHandler Error Handling", () => { + let handler: AnthropicHandler + let mockClient: any + + beforeEach(() => { + vi.clearAllMocks() + handler = new AnthropicHandler({ + apiKey: "test-api-key", + apiModelId: "claude-opus-4-1-20250805", + }) + mockClient = (handler as any).client + }) + + describe("createMessage error handling", () => { + it("should handle false positive 'Claude AI usage limit reached' error with future timestamp", async () => { + // Create a timestamp that's far in the future (year 2030+) + const currentTime = Math.floor(Date.now() / 1000) + const futureTimestamp = currentTime + 10 * 365 * 24 * 60 * 60 // 10 years from now + const errorMessage = `Claude AI usage limit reached|${futureTimestamp}` + + mockClient.messages.create.mockRejectedValue(new Error(errorMessage)) + + const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }]) + + await expect(async () => { + const results = [] + for await (const chunk of generator) { + results.push(chunk) + } + }).rejects.toThrow(/API error for model.*The API returned an unexpected error format/) + }) + + it("should handle legitimate rate limit errors with status 429", async () => { + const error = new Error("Rate limit exceeded") + ;(error as any).status = 429 + + mockClient.messages.create.mockRejectedValue(error) + + const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }]) + + await expect(async () => { + const results = [] + for await (const chunk of generator) { + results.push(chunk) + } + }).rejects.toThrow(/Rate limit exceeded for model.*Please wait before making more requests/) + }) + + it("should handle rate_limit_error in message", async () => { + const error = new Error("rate_limit_error: Too many requests") + + mockClient.messages.create.mockRejectedValue(error) + + const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }]) + + await expect(async () => { + const results = [] + for await (const chunk of generator) { + results.push(chunk) + } + }).rejects.toThrow(/Rate limit exceeded for model.*Please wait before making more requests/) + }) + + it("should pass through other errors unchanged", async () => { + const error = new Error("Some other API error") + + mockClient.messages.create.mockRejectedValue(error) + + const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }]) + + await expect(async () => { + const results = [] + for await (const chunk of generator) { + results.push(chunk) + } + }).rejects.toThrow("Some other API error") + }) + + it("should handle 'Claude AI usage limit reached' with valid timestamp correctly", async () => { + // Use a timestamp that's not in the future + const currentTimestamp = Math.floor(Date.now() / 1000) + const errorMessage = `Claude AI usage limit reached|${currentTimestamp}` + + mockClient.messages.create.mockRejectedValue(new Error(errorMessage)) + + const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }]) + + // Should pass through the original error since timestamp is valid + await expect(async () => { + const results = [] + for await (const chunk of generator) { + results.push(chunk) + } + }).rejects.toThrow(errorMessage) + }) + }) + + describe("completePrompt error handling", () => { + it("should handle false positive 'Claude AI usage limit reached' error with future timestamp", async () => { + // Create a timestamp that's far in the future (year 2030+) + const currentTime = Math.floor(Date.now() / 1000) + const futureTimestamp = currentTime + 10 * 365 * 24 * 60 * 60 // 10 years from now + const errorMessage = `Claude AI usage limit reached|${futureTimestamp}` + + mockClient.messages.create.mockRejectedValue(new Error(errorMessage)) + + await expect(handler.completePrompt("Test prompt")).rejects.toThrow( + /API error for model.*The API returned an unexpected error format/, + ) + }) + + it("should handle legitimate rate limit errors with status 429", async () => { + const error = new Error("Rate limit exceeded") + ;(error as any).status = 429 + + mockClient.messages.create.mockRejectedValue(error) + + await expect(handler.completePrompt("Test prompt")).rejects.toThrow( + /Rate limit exceeded for model.*Please wait before making more requests/, + ) + }) + + it("should handle rate_limit_error in message", async () => { + const error = new Error("rate_limit_error: Too many requests") + + mockClient.messages.create.mockRejectedValue(error) + + await expect(handler.completePrompt("Test prompt")).rejects.toThrow( + /Rate limit exceeded for model.*Please wait before making more requests/, + ) + }) + + it("should pass through other errors unchanged", async () => { + const error = new Error("Some other API error") + + mockClient.messages.create.mockRejectedValue(error) + + await expect(handler.completePrompt("Test prompt")).rejects.toThrow("Some other API error") + }) + + it("should handle 'Claude AI usage limit reached' with valid timestamp correctly", async () => { + // Use a timestamp that's not in the future + const currentTimestamp = Math.floor(Date.now() / 1000) + const errorMessage = `Claude AI usage limit reached|${currentTimestamp}` + + mockClient.messages.create.mockRejectedValue(new Error(errorMessage)) + + // Should pass through the original error since timestamp is valid + await expect(handler.completePrompt("Test prompt")).rejects.toThrow(errorMessage) + }) + }) + + describe("edge cases", () => { + it("should handle error without message property", async () => { + const error = { status: 500 } + + mockClient.messages.create.mockRejectedValue(error) + + const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }]) + + await expect(async () => { + const results = [] + for await (const chunk of generator) { + results.push(chunk) + } + }).rejects.toEqual(error) + }) + + it("should handle error with non-string message", async () => { + const error = new Error() + ;(error as any).message = { code: "ERROR_CODE" } + + mockClient.messages.create.mockRejectedValue(error) + + const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }]) + + await expect(async () => { + const results = [] + for await (const chunk of generator) { + results.push(chunk) + } + }).rejects.toEqual(error) + }) + + it("should handle 'Claude AI usage limit reached' without timestamp", async () => { + const errorMessage = "Claude AI usage limit reached" + + mockClient.messages.create.mockRejectedValue(new Error(errorMessage)) + + const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }]) + + // Should pass through the original error since no timestamp to validate + await expect(async () => { + const results = [] + for await (const chunk of generator) { + results.push(chunk) + } + }).rejects.toThrow(errorMessage) + }) + }) +}) diff --git a/src/api/providers/anthropic.ts b/src/api/providers/anthropic.ts index 3fb60c0e4f..f6b4ff50df 100644 --- a/src/api/providers/anthropic.ts +++ b/src/api/providers/anthropic.ts @@ -41,205 +41,246 @@ export class AnthropicHandler extends BaseProvider implements SingleCompletionHa messages: Anthropic.Messages.MessageParam[], metadata?: ApiHandlerCreateMessageMetadata, ): ApiStream { - let stream: AnthropicStream - const cacheControl: CacheControlEphemeral = { type: "ephemeral" } - let { id: modelId, betas = [], maxTokens, temperature, reasoning: thinking } = this.getModel() + try { + let stream: AnthropicStream + const cacheControl: CacheControlEphemeral = { type: "ephemeral" } + let { id: modelId, betas = [], maxTokens, temperature, reasoning: thinking } = this.getModel() - // Add 1M context beta flag if enabled for Claude Sonnet 4 and 4.5 - if ( - (modelId === "claude-sonnet-4-20250514" || modelId === "claude-sonnet-4-5") && - this.options.anthropicBeta1MContext - ) { - betas.push("context-1m-2025-08-07") - } + // Add 1M context beta flag if enabled for Claude Sonnet 4 and 4.5 + if ( + (modelId === "claude-sonnet-4-20250514" || modelId === "claude-sonnet-4-5") && + this.options.anthropicBeta1MContext + ) { + betas.push("context-1m-2025-08-07") + } - switch (modelId) { - case "claude-sonnet-4-5": - case "claude-sonnet-4-20250514": - case "claude-opus-4-1-20250805": - case "claude-opus-4-20250514": - case "claude-3-7-sonnet-20250219": - case "claude-3-5-sonnet-20241022": - case "claude-3-5-haiku-20241022": - case "claude-3-opus-20240229": - case "claude-3-haiku-20240307": { - /** - * The latest message will be the new user message, one before - * will be the assistant message from a previous request, and - * the user message before that will be a previously cached user - * message. So we need to mark the latest user message as - * ephemeral to cache it for the next request, and mark the - * second to last user message as ephemeral to let the server - * know the last message to retrieve from the cache for the - * current request. - */ - const userMsgIndices = messages.reduce( - (acc, msg, index) => (msg.role === "user" ? [...acc, index] : acc), - [] as number[], - ) + switch (modelId) { + case "claude-sonnet-4-5": + case "claude-sonnet-4-20250514": + case "claude-opus-4-1-20250805": + case "claude-opus-4-20250514": + case "claude-3-7-sonnet-20250219": + case "claude-3-5-sonnet-20241022": + case "claude-3-5-haiku-20241022": + case "claude-3-opus-20240229": + case "claude-3-haiku-20240307": { + /** + * The latest message will be the new user message, one before + * will be the assistant message from a previous request, and + * the user message before that will be a previously cached user + * message. So we need to mark the latest user message as + * ephemeral to cache it for the next request, and mark the + * second to last user message as ephemeral to let the server + * know the last message to retrieve from the cache for the + * current request. + */ + const userMsgIndices = messages.reduce( + (acc, msg, index) => (msg.role === "user" ? [...acc, index] : acc), + [] as number[], + ) - const lastUserMsgIndex = userMsgIndices[userMsgIndices.length - 1] ?? -1 - const secondLastMsgUserIndex = userMsgIndices[userMsgIndices.length - 2] ?? -1 + const lastUserMsgIndex = userMsgIndices[userMsgIndices.length - 1] ?? -1 + const secondLastMsgUserIndex = userMsgIndices[userMsgIndices.length - 2] ?? -1 - stream = await this.client.messages.create( - { + stream = await this.client.messages.create( + { + model: modelId, + max_tokens: maxTokens ?? ANTHROPIC_DEFAULT_MAX_TOKENS, + temperature, + thinking, + // Setting cache breakpoint for system prompt so new tasks can reuse it. + system: [{ text: systemPrompt, type: "text", cache_control: cacheControl }], + messages: messages.map((message, index) => { + if (index === lastUserMsgIndex || index === secondLastMsgUserIndex) { + return { + ...message, + content: + typeof message.content === "string" + ? [{ type: "text", text: message.content, cache_control: cacheControl }] + : message.content.map((content, contentIndex) => + contentIndex === message.content.length - 1 + ? { ...content, cache_control: cacheControl } + : content, + ), + } + } + return message + }), + stream: true, + }, + (() => { + // prompt caching: https://x.com/alexalbert__/status/1823751995901272068 + // https://github.com/anthropics/anthropic-sdk-typescript?tab=readme-ov-file#default-headers + // https://github.com/anthropics/anthropic-sdk-typescript/commit/c920b77fc67bd839bfeb6716ceab9d7c9bbe7393 + + // Then check for models that support prompt caching + switch (modelId) { + case "claude-sonnet-4-5": + case "claude-sonnet-4-20250514": + case "claude-opus-4-1-20250805": + case "claude-opus-4-20250514": + case "claude-3-7-sonnet-20250219": + case "claude-3-5-sonnet-20241022": + case "claude-3-5-haiku-20241022": + case "claude-3-opus-20240229": + case "claude-3-haiku-20240307": + betas.push("prompt-caching-2024-07-31") + return { headers: { "anthropic-beta": betas.join(",") } } + default: + return undefined + } + })(), + ) + break + } + default: { + stream = (await this.client.messages.create({ model: modelId, max_tokens: maxTokens ?? ANTHROPIC_DEFAULT_MAX_TOKENS, temperature, - thinking, - // Setting cache breakpoint for system prompt so new tasks can reuse it. - system: [{ text: systemPrompt, type: "text", cache_control: cacheControl }], - messages: messages.map((message, index) => { - if (index === lastUserMsgIndex || index === secondLastMsgUserIndex) { - return { - ...message, - content: - typeof message.content === "string" - ? [{ type: "text", text: message.content, cache_control: cacheControl }] - : message.content.map((content, contentIndex) => - contentIndex === message.content.length - 1 - ? { ...content, cache_control: cacheControl } - : content, - ), - } - } - return message - }), + system: [{ text: systemPrompt, type: "text" }], + messages, stream: true, - }, - (() => { - // prompt caching: https://x.com/alexalbert__/status/1823751995901272068 - // https://github.com/anthropics/anthropic-sdk-typescript?tab=readme-ov-file#default-headers - // https://github.com/anthropics/anthropic-sdk-typescript/commit/c920b77fc67bd839bfeb6716ceab9d7c9bbe7393 - - // Then check for models that support prompt caching - switch (modelId) { - case "claude-sonnet-4-5": - case "claude-sonnet-4-20250514": - case "claude-opus-4-1-20250805": - case "claude-opus-4-20250514": - case "claude-3-7-sonnet-20250219": - case "claude-3-5-sonnet-20241022": - case "claude-3-5-haiku-20241022": - case "claude-3-opus-20240229": - case "claude-3-haiku-20240307": - betas.push("prompt-caching-2024-07-31") - return { headers: { "anthropic-beta": betas.join(",") } } - default: - return undefined - } - })(), - ) - break - } - default: { - stream = (await this.client.messages.create({ - model: modelId, - max_tokens: maxTokens ?? ANTHROPIC_DEFAULT_MAX_TOKENS, - temperature, - system: [{ text: systemPrompt, type: "text" }], - messages, - stream: true, - })) as any - break - } - } - - let inputTokens = 0 - let outputTokens = 0 - let cacheWriteTokens = 0 - let cacheReadTokens = 0 - - for await (const chunk of stream) { - switch (chunk.type) { - case "message_start": { - // Tells us cache reads/writes/input/output. - const { - input_tokens = 0, - output_tokens = 0, - cache_creation_input_tokens, - cache_read_input_tokens, - } = chunk.message.usage - - yield { - type: "usage", - inputTokens: input_tokens, - outputTokens: output_tokens, - cacheWriteTokens: cache_creation_input_tokens || undefined, - cacheReadTokens: cache_read_input_tokens || undefined, - } - - inputTokens += input_tokens - outputTokens += output_tokens - cacheWriteTokens += cache_creation_input_tokens || 0 - cacheReadTokens += cache_read_input_tokens || 0 - + })) as any break } - case "message_delta": - // Tells us stop_reason, stop_sequence, and output tokens - // along the way and at the end of the message. - yield { - type: "usage", - inputTokens: 0, - outputTokens: chunk.usage.output_tokens || 0, - } - - break - case "message_stop": - // No usage data, just an indicator that the message is done. - break - case "content_block_start": - switch (chunk.content_block.type) { - case "thinking": - // We may receive multiple text blocks, in which - // case just insert a line break between them. - if (chunk.index > 0) { - yield { type: "reasoning", text: "\n" } - } - - yield { type: "reasoning", text: chunk.content_block.thinking } - break - case "text": - // We may receive multiple text blocks, in which - // case just insert a line break between them. - if (chunk.index > 0) { - yield { type: "text", text: "\n" } - } - - yield { type: "text", text: chunk.content_block.text } - break - } - break - case "content_block_delta": - switch (chunk.delta.type) { - case "thinking_delta": - yield { type: "reasoning", text: chunk.delta.thinking } - break - case "text_delta": - yield { type: "text", text: chunk.delta.text } - break - } - - break - case "content_block_stop": - break } - } - if (inputTokens > 0 || outputTokens > 0 || cacheWriteTokens > 0 || cacheReadTokens > 0) { - yield { - type: "usage", - inputTokens: 0, - outputTokens: 0, - totalCost: calculateApiCostAnthropic( - this.getModel().info, - inputTokens, - outputTokens, - cacheWriteTokens, - cacheReadTokens, - ), + let inputTokens = 0 + let outputTokens = 0 + let cacheWriteTokens = 0 + let cacheReadTokens = 0 + + for await (const chunk of stream) { + switch (chunk.type) { + case "message_start": { + // Tells us cache reads/writes/input/output. + const { + input_tokens = 0, + output_tokens = 0, + cache_creation_input_tokens, + cache_read_input_tokens, + } = chunk.message.usage + + yield { + type: "usage", + inputTokens: input_tokens, + outputTokens: output_tokens, + cacheWriteTokens: cache_creation_input_tokens || undefined, + cacheReadTokens: cache_read_input_tokens || undefined, + } + + inputTokens += input_tokens + outputTokens += output_tokens + cacheWriteTokens += cache_creation_input_tokens || 0 + cacheReadTokens += cache_read_input_tokens || 0 + + break + } + case "message_delta": + // Tells us stop_reason, stop_sequence, and output tokens + // along the way and at the end of the message. + yield { + type: "usage", + inputTokens: 0, + outputTokens: chunk.usage.output_tokens || 0, + } + + break + case "message_stop": + // No usage data, just an indicator that the message is done. + break + case "content_block_start": + switch (chunk.content_block.type) { + case "thinking": + // We may receive multiple text blocks, in which + // case just insert a line break between them. + if (chunk.index > 0) { + yield { type: "reasoning", text: "\n" } + } + + yield { type: "reasoning", text: chunk.content_block.thinking } + break + case "text": + // We may receive multiple text blocks, in which + // case just insert a line break between them. + if (chunk.index > 0) { + yield { type: "text", text: "\n" } + } + + yield { type: "text", text: chunk.content_block.text } + break + } + break + case "content_block_delta": + switch (chunk.delta.type) { + case "thinking_delta": + yield { type: "reasoning", text: chunk.delta.thinking } + break + case "text_delta": + yield { type: "text", text: chunk.delta.text } + break + } + + break + case "content_block_stop": + break + } } + + if (inputTokens > 0 || outputTokens > 0 || cacheWriteTokens > 0 || cacheReadTokens > 0) { + yield { + type: "usage", + inputTokens: 0, + outputTokens: 0, + totalCost: calculateApiCostAnthropic( + this.getModel().info, + inputTokens, + outputTokens, + cacheWriteTokens, + cacheReadTokens, + ), + } + } + } catch (error: any) { + // Handle specific error messages that might be incorrectly formatted + if (error.message && typeof error.message === "string") { + // Check for the specific malformed error message pattern + if (error.message.includes("Claude AI usage limit reached")) { + // Parse the timestamp if present + const timestampMatch = error.message.match(/\|(\d+)/) + const timestamp = timestampMatch ? parseInt(timestampMatch[1]) : null + + // Check if this is likely a false positive (timestamp in the future or unrealistic) + const currentTime = Math.floor(Date.now() / 1000) + const oneYearFromNow = currentTime + 365 * 24 * 60 * 60 + + if (timestamp && timestamp > oneYearFromNow) { + // This is likely a false positive error + console.error( + `Detected potentially false rate limit error for model ${this.getModel().id}. Original error: ${error.message}`, + ) + + // Throw a more informative error + throw new Error( + `API error for model ${this.getModel().id}: The API returned an unexpected error format. ` + + `This may be a temporary issue. Please try again or check your API configuration. ` + + `(Original message: ${error.message})`, + ) + } + } + + // Check for actual rate limit errors from Anthropic API + if (error.status === 429 || error.message.includes("rate_limit_error")) { + throw new Error( + `Rate limit exceeded for model ${this.getModel().id}. ` + + `Please wait before making more requests or consider upgrading your API plan.`, + ) + } + } + + // Re-throw the original error if it doesn't match our patterns + throw error } } @@ -284,19 +325,60 @@ export class AnthropicHandler extends BaseProvider implements SingleCompletionHa } async completePrompt(prompt: string) { - let { id: model, temperature } = this.getModel() + try { + let { id: model, temperature } = this.getModel() - const message = await this.client.messages.create({ - model, - max_tokens: ANTHROPIC_DEFAULT_MAX_TOKENS, - thinking: undefined, - temperature, - messages: [{ role: "user", content: prompt }], - stream: false, - }) + const message = await this.client.messages.create({ + model, + max_tokens: ANTHROPIC_DEFAULT_MAX_TOKENS, + thinking: undefined, + temperature, + messages: [{ role: "user", content: prompt }], + stream: false, + }) - const content = message.content.find(({ type }) => type === "text") - return content?.type === "text" ? content.text : "" + const content = message.content.find(({ type }) => type === "text") + return content?.type === "text" ? content.text : "" + } catch (error: any) { + // Handle specific error messages that might be incorrectly formatted + if (error.message && typeof error.message === "string") { + // Check for the specific malformed error message pattern + if (error.message.includes("Claude AI usage limit reached")) { + // Parse the timestamp if present + const timestampMatch = error.message.match(/\|(\d+)/) + const timestamp = timestampMatch ? parseInt(timestampMatch[1]) : null + + // Check if this is likely a false positive (timestamp in the future or unrealistic) + const currentTime = Math.floor(Date.now() / 1000) + const oneYearFromNow = currentTime + 365 * 24 * 60 * 60 + + if (timestamp && timestamp > oneYearFromNow) { + // This is likely a false positive error + console.error( + `Detected potentially false rate limit error for model ${this.getModel().id}. Original error: ${error.message}`, + ) + + // Throw a more informative error + throw new Error( + `API error for model ${this.getModel().id}: The API returned an unexpected error format. ` + + `This may be a temporary issue. Please try again or check your API configuration. ` + + `(Original message: ${error.message})`, + ) + } + } + + // Check for actual rate limit errors from Anthropic API + if (error.status === 429 || error.message.includes("rate_limit_error")) { + throw new Error( + `Rate limit exceeded for model ${this.getModel().id}. ` + + `Please wait before making more requests or consider upgrading your API plan.`, + ) + } + } + + // Re-throw the original error if it doesn't match our patterns + throw error + } } /** diff --git a/tmp/Roo-Code b/tmp/Roo-Code new file mode 160000 index 0000000000..8111da66bd --- /dev/null +++ b/tmp/Roo-Code @@ -0,0 +1 @@ +Subproject commit 8111da66bd59ca8d500e5eae23b24a0419ed7345 diff --git a/tmp/Roo-Code-rc b/tmp/Roo-Code-rc new file mode 160000 index 0000000000..7b7bb49572 --- /dev/null +++ b/tmp/Roo-Code-rc @@ -0,0 +1 @@ +Subproject commit 7b7bb49572975c4aeff2381a0ebea99b3aa4542c