mirror of
https://github.com/RooVetGit/Roo-Code.git
synced 2026-10-06 02:47:56 +00:00
fix: handle false positive Claude AI usage limit errors
- Add error handling to detect and properly handle malformed rate limit errors - Check for unrealistic future timestamps in error messages - Provide clearer error messages for both real and false rate limits - Add comprehensive test coverage for error handling scenarios Fixes #8404
This commit is contained in:
parent
702b269a1b
commit
1466cc9c6e
4 changed files with 499 additions and 197 deletions
218
src/api/providers/__tests__/anthropic-error-handling.spec.ts
Normal file
218
src/api/providers/__tests__/anthropic-error-handling.spec.ts
Normal file
|
|
@ -0,0 +1,218 @@
|
|||
import { describe, it, expect, vi, beforeEach } from "vitest"
|
||||
import { AnthropicHandler } from "../anthropic"
|
||||
import { Anthropic } from "@anthropic-ai/sdk"
|
||||
|
||||
// Mock the Anthropic SDK
|
||||
vi.mock("@anthropic-ai/sdk", () => {
|
||||
const mockAnthropicConstructor = vi.fn().mockImplementation(() => ({
|
||||
messages: {
|
||||
create: vi.fn(),
|
||||
},
|
||||
}))
|
||||
|
||||
return {
|
||||
Anthropic: mockAnthropicConstructor,
|
||||
}
|
||||
})
|
||||
|
||||
describe("AnthropicHandler Error Handling", () => {
|
||||
let handler: AnthropicHandler
|
||||
let mockClient: any
|
||||
|
||||
beforeEach(() => {
|
||||
vi.clearAllMocks()
|
||||
handler = new AnthropicHandler({
|
||||
apiKey: "test-api-key",
|
||||
apiModelId: "claude-opus-4-1-20250805",
|
||||
})
|
||||
mockClient = (handler as any).client
|
||||
})
|
||||
|
||||
describe("createMessage error handling", () => {
|
||||
it("should handle false positive 'Claude AI usage limit reached' error with future timestamp", async () => {
|
||||
// Create a timestamp that's far in the future (year 2030+)
|
||||
const currentTime = Math.floor(Date.now() / 1000)
|
||||
const futureTimestamp = currentTime + 10 * 365 * 24 * 60 * 60 // 10 years from now
|
||||
const errorMessage = `Claude AI usage limit reached|${futureTimestamp}`
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(new Error(errorMessage))
|
||||
|
||||
const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }])
|
||||
|
||||
await expect(async () => {
|
||||
const results = []
|
||||
for await (const chunk of generator) {
|
||||
results.push(chunk)
|
||||
}
|
||||
}).rejects.toThrow(/API error for model.*The API returned an unexpected error format/)
|
||||
})
|
||||
|
||||
it("should handle legitimate rate limit errors with status 429", async () => {
|
||||
const error = new Error("Rate limit exceeded")
|
||||
;(error as any).status = 429
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(error)
|
||||
|
||||
const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }])
|
||||
|
||||
await expect(async () => {
|
||||
const results = []
|
||||
for await (const chunk of generator) {
|
||||
results.push(chunk)
|
||||
}
|
||||
}).rejects.toThrow(/Rate limit exceeded for model.*Please wait before making more requests/)
|
||||
})
|
||||
|
||||
it("should handle rate_limit_error in message", async () => {
|
||||
const error = new Error("rate_limit_error: Too many requests")
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(error)
|
||||
|
||||
const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }])
|
||||
|
||||
await expect(async () => {
|
||||
const results = []
|
||||
for await (const chunk of generator) {
|
||||
results.push(chunk)
|
||||
}
|
||||
}).rejects.toThrow(/Rate limit exceeded for model.*Please wait before making more requests/)
|
||||
})
|
||||
|
||||
it("should pass through other errors unchanged", async () => {
|
||||
const error = new Error("Some other API error")
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(error)
|
||||
|
||||
const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }])
|
||||
|
||||
await expect(async () => {
|
||||
const results = []
|
||||
for await (const chunk of generator) {
|
||||
results.push(chunk)
|
||||
}
|
||||
}).rejects.toThrow("Some other API error")
|
||||
})
|
||||
|
||||
it("should handle 'Claude AI usage limit reached' with valid timestamp correctly", async () => {
|
||||
// Use a timestamp that's not in the future
|
||||
const currentTimestamp = Math.floor(Date.now() / 1000)
|
||||
const errorMessage = `Claude AI usage limit reached|${currentTimestamp}`
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(new Error(errorMessage))
|
||||
|
||||
const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }])
|
||||
|
||||
// Should pass through the original error since timestamp is valid
|
||||
await expect(async () => {
|
||||
const results = []
|
||||
for await (const chunk of generator) {
|
||||
results.push(chunk)
|
||||
}
|
||||
}).rejects.toThrow(errorMessage)
|
||||
})
|
||||
})
|
||||
|
||||
describe("completePrompt error handling", () => {
|
||||
it("should handle false positive 'Claude AI usage limit reached' error with future timestamp", async () => {
|
||||
// Create a timestamp that's far in the future (year 2030+)
|
||||
const currentTime = Math.floor(Date.now() / 1000)
|
||||
const futureTimestamp = currentTime + 10 * 365 * 24 * 60 * 60 // 10 years from now
|
||||
const errorMessage = `Claude AI usage limit reached|${futureTimestamp}`
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(new Error(errorMessage))
|
||||
|
||||
await expect(handler.completePrompt("Test prompt")).rejects.toThrow(
|
||||
/API error for model.*The API returned an unexpected error format/,
|
||||
)
|
||||
})
|
||||
|
||||
it("should handle legitimate rate limit errors with status 429", async () => {
|
||||
const error = new Error("Rate limit exceeded")
|
||||
;(error as any).status = 429
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(error)
|
||||
|
||||
await expect(handler.completePrompt("Test prompt")).rejects.toThrow(
|
||||
/Rate limit exceeded for model.*Please wait before making more requests/,
|
||||
)
|
||||
})
|
||||
|
||||
it("should handle rate_limit_error in message", async () => {
|
||||
const error = new Error("rate_limit_error: Too many requests")
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(error)
|
||||
|
||||
await expect(handler.completePrompt("Test prompt")).rejects.toThrow(
|
||||
/Rate limit exceeded for model.*Please wait before making more requests/,
|
||||
)
|
||||
})
|
||||
|
||||
it("should pass through other errors unchanged", async () => {
|
||||
const error = new Error("Some other API error")
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(error)
|
||||
|
||||
await expect(handler.completePrompt("Test prompt")).rejects.toThrow("Some other API error")
|
||||
})
|
||||
|
||||
it("should handle 'Claude AI usage limit reached' with valid timestamp correctly", async () => {
|
||||
// Use a timestamp that's not in the future
|
||||
const currentTimestamp = Math.floor(Date.now() / 1000)
|
||||
const errorMessage = `Claude AI usage limit reached|${currentTimestamp}`
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(new Error(errorMessage))
|
||||
|
||||
// Should pass through the original error since timestamp is valid
|
||||
await expect(handler.completePrompt("Test prompt")).rejects.toThrow(errorMessage)
|
||||
})
|
||||
})
|
||||
|
||||
describe("edge cases", () => {
|
||||
it("should handle error without message property", async () => {
|
||||
const error = { status: 500 }
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(error)
|
||||
|
||||
const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }])
|
||||
|
||||
await expect(async () => {
|
||||
const results = []
|
||||
for await (const chunk of generator) {
|
||||
results.push(chunk)
|
||||
}
|
||||
}).rejects.toEqual(error)
|
||||
})
|
||||
|
||||
it("should handle error with non-string message", async () => {
|
||||
const error = new Error()
|
||||
;(error as any).message = { code: "ERROR_CODE" }
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(error)
|
||||
|
||||
const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }])
|
||||
|
||||
await expect(async () => {
|
||||
const results = []
|
||||
for await (const chunk of generator) {
|
||||
results.push(chunk)
|
||||
}
|
||||
}).rejects.toEqual(error)
|
||||
})
|
||||
|
||||
it("should handle 'Claude AI usage limit reached' without timestamp", async () => {
|
||||
const errorMessage = "Claude AI usage limit reached"
|
||||
|
||||
mockClient.messages.create.mockRejectedValue(new Error(errorMessage))
|
||||
|
||||
const generator = handler.createMessage("System prompt", [{ role: "user", content: "Test message" }])
|
||||
|
||||
// Should pass through the original error since no timestamp to validate
|
||||
await expect(async () => {
|
||||
const results = []
|
||||
for await (const chunk of generator) {
|
||||
results.push(chunk)
|
||||
}
|
||||
}).rejects.toThrow(errorMessage)
|
||||
})
|
||||
})
|
||||
})
|
||||
|
|
@ -41,205 +41,246 @@ export class AnthropicHandler extends BaseProvider implements SingleCompletionHa
|
|||
messages: Anthropic.Messages.MessageParam[],
|
||||
metadata?: ApiHandlerCreateMessageMetadata,
|
||||
): ApiStream {
|
||||
let stream: AnthropicStream<Anthropic.Messages.RawMessageStreamEvent>
|
||||
const cacheControl: CacheControlEphemeral = { type: "ephemeral" }
|
||||
let { id: modelId, betas = [], maxTokens, temperature, reasoning: thinking } = this.getModel()
|
||||
try {
|
||||
let stream: AnthropicStream<Anthropic.Messages.RawMessageStreamEvent>
|
||||
const cacheControl: CacheControlEphemeral = { type: "ephemeral" }
|
||||
let { id: modelId, betas = [], maxTokens, temperature, reasoning: thinking } = this.getModel()
|
||||
|
||||
// Add 1M context beta flag if enabled for Claude Sonnet 4 and 4.5
|
||||
if (
|
||||
(modelId === "claude-sonnet-4-20250514" || modelId === "claude-sonnet-4-5") &&
|
||||
this.options.anthropicBeta1MContext
|
||||
) {
|
||||
betas.push("context-1m-2025-08-07")
|
||||
}
|
||||
// Add 1M context beta flag if enabled for Claude Sonnet 4 and 4.5
|
||||
if (
|
||||
(modelId === "claude-sonnet-4-20250514" || modelId === "claude-sonnet-4-5") &&
|
||||
this.options.anthropicBeta1MContext
|
||||
) {
|
||||
betas.push("context-1m-2025-08-07")
|
||||
}
|
||||
|
||||
switch (modelId) {
|
||||
case "claude-sonnet-4-5":
|
||||
case "claude-sonnet-4-20250514":
|
||||
case "claude-opus-4-1-20250805":
|
||||
case "claude-opus-4-20250514":
|
||||
case "claude-3-7-sonnet-20250219":
|
||||
case "claude-3-5-sonnet-20241022":
|
||||
case "claude-3-5-haiku-20241022":
|
||||
case "claude-3-opus-20240229":
|
||||
case "claude-3-haiku-20240307": {
|
||||
/**
|
||||
* The latest message will be the new user message, one before
|
||||
* will be the assistant message from a previous request, and
|
||||
* the user message before that will be a previously cached user
|
||||
* message. So we need to mark the latest user message as
|
||||
* ephemeral to cache it for the next request, and mark the
|
||||
* second to last user message as ephemeral to let the server
|
||||
* know the last message to retrieve from the cache for the
|
||||
* current request.
|
||||
*/
|
||||
const userMsgIndices = messages.reduce(
|
||||
(acc, msg, index) => (msg.role === "user" ? [...acc, index] : acc),
|
||||
[] as number[],
|
||||
)
|
||||
switch (modelId) {
|
||||
case "claude-sonnet-4-5":
|
||||
case "claude-sonnet-4-20250514":
|
||||
case "claude-opus-4-1-20250805":
|
||||
case "claude-opus-4-20250514":
|
||||
case "claude-3-7-sonnet-20250219":
|
||||
case "claude-3-5-sonnet-20241022":
|
||||
case "claude-3-5-haiku-20241022":
|
||||
case "claude-3-opus-20240229":
|
||||
case "claude-3-haiku-20240307": {
|
||||
/**
|
||||
* The latest message will be the new user message, one before
|
||||
* will be the assistant message from a previous request, and
|
||||
* the user message before that will be a previously cached user
|
||||
* message. So we need to mark the latest user message as
|
||||
* ephemeral to cache it for the next request, and mark the
|
||||
* second to last user message as ephemeral to let the server
|
||||
* know the last message to retrieve from the cache for the
|
||||
* current request.
|
||||
*/
|
||||
const userMsgIndices = messages.reduce(
|
||||
(acc, msg, index) => (msg.role === "user" ? [...acc, index] : acc),
|
||||
[] as number[],
|
||||
)
|
||||
|
||||
const lastUserMsgIndex = userMsgIndices[userMsgIndices.length - 1] ?? -1
|
||||
const secondLastMsgUserIndex = userMsgIndices[userMsgIndices.length - 2] ?? -1
|
||||
const lastUserMsgIndex = userMsgIndices[userMsgIndices.length - 1] ?? -1
|
||||
const secondLastMsgUserIndex = userMsgIndices[userMsgIndices.length - 2] ?? -1
|
||||
|
||||
stream = await this.client.messages.create(
|
||||
{
|
||||
stream = await this.client.messages.create(
|
||||
{
|
||||
model: modelId,
|
||||
max_tokens: maxTokens ?? ANTHROPIC_DEFAULT_MAX_TOKENS,
|
||||
temperature,
|
||||
thinking,
|
||||
// Setting cache breakpoint for system prompt so new tasks can reuse it.
|
||||
system: [{ text: systemPrompt, type: "text", cache_control: cacheControl }],
|
||||
messages: messages.map((message, index) => {
|
||||
if (index === lastUserMsgIndex || index === secondLastMsgUserIndex) {
|
||||
return {
|
||||
...message,
|
||||
content:
|
||||
typeof message.content === "string"
|
||||
? [{ type: "text", text: message.content, cache_control: cacheControl }]
|
||||
: message.content.map((content, contentIndex) =>
|
||||
contentIndex === message.content.length - 1
|
||||
? { ...content, cache_control: cacheControl }
|
||||
: content,
|
||||
),
|
||||
}
|
||||
}
|
||||
return message
|
||||
}),
|
||||
stream: true,
|
||||
},
|
||||
(() => {
|
||||
// prompt caching: https://x.com/alexalbert__/status/1823751995901272068
|
||||
// https://github.com/anthropics/anthropic-sdk-typescript?tab=readme-ov-file#default-headers
|
||||
// https://github.com/anthropics/anthropic-sdk-typescript/commit/c920b77fc67bd839bfeb6716ceab9d7c9bbe7393
|
||||
|
||||
// Then check for models that support prompt caching
|
||||
switch (modelId) {
|
||||
case "claude-sonnet-4-5":
|
||||
case "claude-sonnet-4-20250514":
|
||||
case "claude-opus-4-1-20250805":
|
||||
case "claude-opus-4-20250514":
|
||||
case "claude-3-7-sonnet-20250219":
|
||||
case "claude-3-5-sonnet-20241022":
|
||||
case "claude-3-5-haiku-20241022":
|
||||
case "claude-3-opus-20240229":
|
||||
case "claude-3-haiku-20240307":
|
||||
betas.push("prompt-caching-2024-07-31")
|
||||
return { headers: { "anthropic-beta": betas.join(",") } }
|
||||
default:
|
||||
return undefined
|
||||
}
|
||||
})(),
|
||||
)
|
||||
break
|
||||
}
|
||||
default: {
|
||||
stream = (await this.client.messages.create({
|
||||
model: modelId,
|
||||
max_tokens: maxTokens ?? ANTHROPIC_DEFAULT_MAX_TOKENS,
|
||||
temperature,
|
||||
thinking,
|
||||
// Setting cache breakpoint for system prompt so new tasks can reuse it.
|
||||
system: [{ text: systemPrompt, type: "text", cache_control: cacheControl }],
|
||||
messages: messages.map((message, index) => {
|
||||
if (index === lastUserMsgIndex || index === secondLastMsgUserIndex) {
|
||||
return {
|
||||
...message,
|
||||
content:
|
||||
typeof message.content === "string"
|
||||
? [{ type: "text", text: message.content, cache_control: cacheControl }]
|
||||
: message.content.map((content, contentIndex) =>
|
||||
contentIndex === message.content.length - 1
|
||||
? { ...content, cache_control: cacheControl }
|
||||
: content,
|
||||
),
|
||||
}
|
||||
}
|
||||
return message
|
||||
}),
|
||||
system: [{ text: systemPrompt, type: "text" }],
|
||||
messages,
|
||||
stream: true,
|
||||
},
|
||||
(() => {
|
||||
// prompt caching: https://x.com/alexalbert__/status/1823751995901272068
|
||||
// https://github.com/anthropics/anthropic-sdk-typescript?tab=readme-ov-file#default-headers
|
||||
// https://github.com/anthropics/anthropic-sdk-typescript/commit/c920b77fc67bd839bfeb6716ceab9d7c9bbe7393
|
||||
|
||||
// Then check for models that support prompt caching
|
||||
switch (modelId) {
|
||||
case "claude-sonnet-4-5":
|
||||
case "claude-sonnet-4-20250514":
|
||||
case "claude-opus-4-1-20250805":
|
||||
case "claude-opus-4-20250514":
|
||||
case "claude-3-7-sonnet-20250219":
|
||||
case "claude-3-5-sonnet-20241022":
|
||||
case "claude-3-5-haiku-20241022":
|
||||
case "claude-3-opus-20240229":
|
||||
case "claude-3-haiku-20240307":
|
||||
betas.push("prompt-caching-2024-07-31")
|
||||
return { headers: { "anthropic-beta": betas.join(",") } }
|
||||
default:
|
||||
return undefined
|
||||
}
|
||||
})(),
|
||||
)
|
||||
break
|
||||
}
|
||||
default: {
|
||||
stream = (await this.client.messages.create({
|
||||
model: modelId,
|
||||
max_tokens: maxTokens ?? ANTHROPIC_DEFAULT_MAX_TOKENS,
|
||||
temperature,
|
||||
system: [{ text: systemPrompt, type: "text" }],
|
||||
messages,
|
||||
stream: true,
|
||||
})) as any
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
let inputTokens = 0
|
||||
let outputTokens = 0
|
||||
let cacheWriteTokens = 0
|
||||
let cacheReadTokens = 0
|
||||
|
||||
for await (const chunk of stream) {
|
||||
switch (chunk.type) {
|
||||
case "message_start": {
|
||||
// Tells us cache reads/writes/input/output.
|
||||
const {
|
||||
input_tokens = 0,
|
||||
output_tokens = 0,
|
||||
cache_creation_input_tokens,
|
||||
cache_read_input_tokens,
|
||||
} = chunk.message.usage
|
||||
|
||||
yield {
|
||||
type: "usage",
|
||||
inputTokens: input_tokens,
|
||||
outputTokens: output_tokens,
|
||||
cacheWriteTokens: cache_creation_input_tokens || undefined,
|
||||
cacheReadTokens: cache_read_input_tokens || undefined,
|
||||
}
|
||||
|
||||
inputTokens += input_tokens
|
||||
outputTokens += output_tokens
|
||||
cacheWriteTokens += cache_creation_input_tokens || 0
|
||||
cacheReadTokens += cache_read_input_tokens || 0
|
||||
|
||||
})) as any
|
||||
break
|
||||
}
|
||||
case "message_delta":
|
||||
// Tells us stop_reason, stop_sequence, and output tokens
|
||||
// along the way and at the end of the message.
|
||||
yield {
|
||||
type: "usage",
|
||||
inputTokens: 0,
|
||||
outputTokens: chunk.usage.output_tokens || 0,
|
||||
}
|
||||
|
||||
break
|
||||
case "message_stop":
|
||||
// No usage data, just an indicator that the message is done.
|
||||
break
|
||||
case "content_block_start":
|
||||
switch (chunk.content_block.type) {
|
||||
case "thinking":
|
||||
// We may receive multiple text blocks, in which
|
||||
// case just insert a line break between them.
|
||||
if (chunk.index > 0) {
|
||||
yield { type: "reasoning", text: "\n" }
|
||||
}
|
||||
|
||||
yield { type: "reasoning", text: chunk.content_block.thinking }
|
||||
break
|
||||
case "text":
|
||||
// We may receive multiple text blocks, in which
|
||||
// case just insert a line break between them.
|
||||
if (chunk.index > 0) {
|
||||
yield { type: "text", text: "\n" }
|
||||
}
|
||||
|
||||
yield { type: "text", text: chunk.content_block.text }
|
||||
break
|
||||
}
|
||||
break
|
||||
case "content_block_delta":
|
||||
switch (chunk.delta.type) {
|
||||
case "thinking_delta":
|
||||
yield { type: "reasoning", text: chunk.delta.thinking }
|
||||
break
|
||||
case "text_delta":
|
||||
yield { type: "text", text: chunk.delta.text }
|
||||
break
|
||||
}
|
||||
|
||||
break
|
||||
case "content_block_stop":
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
if (inputTokens > 0 || outputTokens > 0 || cacheWriteTokens > 0 || cacheReadTokens > 0) {
|
||||
yield {
|
||||
type: "usage",
|
||||
inputTokens: 0,
|
||||
outputTokens: 0,
|
||||
totalCost: calculateApiCostAnthropic(
|
||||
this.getModel().info,
|
||||
inputTokens,
|
||||
outputTokens,
|
||||
cacheWriteTokens,
|
||||
cacheReadTokens,
|
||||
),
|
||||
let inputTokens = 0
|
||||
let outputTokens = 0
|
||||
let cacheWriteTokens = 0
|
||||
let cacheReadTokens = 0
|
||||
|
||||
for await (const chunk of stream) {
|
||||
switch (chunk.type) {
|
||||
case "message_start": {
|
||||
// Tells us cache reads/writes/input/output.
|
||||
const {
|
||||
input_tokens = 0,
|
||||
output_tokens = 0,
|
||||
cache_creation_input_tokens,
|
||||
cache_read_input_tokens,
|
||||
} = chunk.message.usage
|
||||
|
||||
yield {
|
||||
type: "usage",
|
||||
inputTokens: input_tokens,
|
||||
outputTokens: output_tokens,
|
||||
cacheWriteTokens: cache_creation_input_tokens || undefined,
|
||||
cacheReadTokens: cache_read_input_tokens || undefined,
|
||||
}
|
||||
|
||||
inputTokens += input_tokens
|
||||
outputTokens += output_tokens
|
||||
cacheWriteTokens += cache_creation_input_tokens || 0
|
||||
cacheReadTokens += cache_read_input_tokens || 0
|
||||
|
||||
break
|
||||
}
|
||||
case "message_delta":
|
||||
// Tells us stop_reason, stop_sequence, and output tokens
|
||||
// along the way and at the end of the message.
|
||||
yield {
|
||||
type: "usage",
|
||||
inputTokens: 0,
|
||||
outputTokens: chunk.usage.output_tokens || 0,
|
||||
}
|
||||
|
||||
break
|
||||
case "message_stop":
|
||||
// No usage data, just an indicator that the message is done.
|
||||
break
|
||||
case "content_block_start":
|
||||
switch (chunk.content_block.type) {
|
||||
case "thinking":
|
||||
// We may receive multiple text blocks, in which
|
||||
// case just insert a line break between them.
|
||||
if (chunk.index > 0) {
|
||||
yield { type: "reasoning", text: "\n" }
|
||||
}
|
||||
|
||||
yield { type: "reasoning", text: chunk.content_block.thinking }
|
||||
break
|
||||
case "text":
|
||||
// We may receive multiple text blocks, in which
|
||||
// case just insert a line break between them.
|
||||
if (chunk.index > 0) {
|
||||
yield { type: "text", text: "\n" }
|
||||
}
|
||||
|
||||
yield { type: "text", text: chunk.content_block.text }
|
||||
break
|
||||
}
|
||||
break
|
||||
case "content_block_delta":
|
||||
switch (chunk.delta.type) {
|
||||
case "thinking_delta":
|
||||
yield { type: "reasoning", text: chunk.delta.thinking }
|
||||
break
|
||||
case "text_delta":
|
||||
yield { type: "text", text: chunk.delta.text }
|
||||
break
|
||||
}
|
||||
|
||||
break
|
||||
case "content_block_stop":
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
if (inputTokens > 0 || outputTokens > 0 || cacheWriteTokens > 0 || cacheReadTokens > 0) {
|
||||
yield {
|
||||
type: "usage",
|
||||
inputTokens: 0,
|
||||
outputTokens: 0,
|
||||
totalCost: calculateApiCostAnthropic(
|
||||
this.getModel().info,
|
||||
inputTokens,
|
||||
outputTokens,
|
||||
cacheWriteTokens,
|
||||
cacheReadTokens,
|
||||
),
|
||||
}
|
||||
}
|
||||
} catch (error: any) {
|
||||
// Handle specific error messages that might be incorrectly formatted
|
||||
if (error.message && typeof error.message === "string") {
|
||||
// Check for the specific malformed error message pattern
|
||||
if (error.message.includes("Claude AI usage limit reached")) {
|
||||
// Parse the timestamp if present
|
||||
const timestampMatch = error.message.match(/\|(\d+)/)
|
||||
const timestamp = timestampMatch ? parseInt(timestampMatch[1]) : null
|
||||
|
||||
// Check if this is likely a false positive (timestamp in the future or unrealistic)
|
||||
const currentTime = Math.floor(Date.now() / 1000)
|
||||
const oneYearFromNow = currentTime + 365 * 24 * 60 * 60
|
||||
|
||||
if (timestamp && timestamp > oneYearFromNow) {
|
||||
// This is likely a false positive error
|
||||
console.error(
|
||||
`Detected potentially false rate limit error for model ${this.getModel().id}. Original error: ${error.message}`,
|
||||
)
|
||||
|
||||
// Throw a more informative error
|
||||
throw new Error(
|
||||
`API error for model ${this.getModel().id}: The API returned an unexpected error format. ` +
|
||||
`This may be a temporary issue. Please try again or check your API configuration. ` +
|
||||
`(Original message: ${error.message})`,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// Check for actual rate limit errors from Anthropic API
|
||||
if (error.status === 429 || error.message.includes("rate_limit_error")) {
|
||||
throw new Error(
|
||||
`Rate limit exceeded for model ${this.getModel().id}. ` +
|
||||
`Please wait before making more requests or consider upgrading your API plan.`,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// Re-throw the original error if it doesn't match our patterns
|
||||
throw error
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -284,19 +325,60 @@ export class AnthropicHandler extends BaseProvider implements SingleCompletionHa
|
|||
}
|
||||
|
||||
async completePrompt(prompt: string) {
|
||||
let { id: model, temperature } = this.getModel()
|
||||
try {
|
||||
let { id: model, temperature } = this.getModel()
|
||||
|
||||
const message = await this.client.messages.create({
|
||||
model,
|
||||
max_tokens: ANTHROPIC_DEFAULT_MAX_TOKENS,
|
||||
thinking: undefined,
|
||||
temperature,
|
||||
messages: [{ role: "user", content: prompt }],
|
||||
stream: false,
|
||||
})
|
||||
const message = await this.client.messages.create({
|
||||
model,
|
||||
max_tokens: ANTHROPIC_DEFAULT_MAX_TOKENS,
|
||||
thinking: undefined,
|
||||
temperature,
|
||||
messages: [{ role: "user", content: prompt }],
|
||||
stream: false,
|
||||
})
|
||||
|
||||
const content = message.content.find(({ type }) => type === "text")
|
||||
return content?.type === "text" ? content.text : ""
|
||||
const content = message.content.find(({ type }) => type === "text")
|
||||
return content?.type === "text" ? content.text : ""
|
||||
} catch (error: any) {
|
||||
// Handle specific error messages that might be incorrectly formatted
|
||||
if (error.message && typeof error.message === "string") {
|
||||
// Check for the specific malformed error message pattern
|
||||
if (error.message.includes("Claude AI usage limit reached")) {
|
||||
// Parse the timestamp if present
|
||||
const timestampMatch = error.message.match(/\|(\d+)/)
|
||||
const timestamp = timestampMatch ? parseInt(timestampMatch[1]) : null
|
||||
|
||||
// Check if this is likely a false positive (timestamp in the future or unrealistic)
|
||||
const currentTime = Math.floor(Date.now() / 1000)
|
||||
const oneYearFromNow = currentTime + 365 * 24 * 60 * 60
|
||||
|
||||
if (timestamp && timestamp > oneYearFromNow) {
|
||||
// This is likely a false positive error
|
||||
console.error(
|
||||
`Detected potentially false rate limit error for model ${this.getModel().id}. Original error: ${error.message}`,
|
||||
)
|
||||
|
||||
// Throw a more informative error
|
||||
throw new Error(
|
||||
`API error for model ${this.getModel().id}: The API returned an unexpected error format. ` +
|
||||
`This may be a temporary issue. Please try again or check your API configuration. ` +
|
||||
`(Original message: ${error.message})`,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// Check for actual rate limit errors from Anthropic API
|
||||
if (error.status === 429 || error.message.includes("rate_limit_error")) {
|
||||
throw new Error(
|
||||
`Rate limit exceeded for model ${this.getModel().id}. ` +
|
||||
`Please wait before making more requests or consider upgrading your API plan.`,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
// Re-throw the original error if it doesn't match our patterns
|
||||
throw error
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
|
|||
1
tmp/Roo-Code
Submodule
1
tmp/Roo-Code
Submodule
|
|
@ -0,0 +1 @@
|
|||
Subproject commit 8111da66bd59ca8d500e5eae23b24a0419ed7345
|
||||
1
tmp/Roo-Code-rc
Submodule
1
tmp/Roo-Code-rc
Submodule
|
|
@ -0,0 +1 @@
|
|||
Subproject commit 7b7bb49572975c4aeff2381a0ebea99b3aa4542c
|
||||
Loading…
Add table
Reference in a new issue