mirror of
https://github.com/RooVetGit/Roo-Code.git
synced 2026-09-07 08:26:51 +00:00
feat: honor Gemini retryDelay and improve rate limit handling
- Enhanced GeminiHandler to properly parse 429 errors with RetryInfo and QuotaFailure - Added GeminiError class to preserve structured error details - Updated retry logic to respect provider-suggested delays with 2-second buffer - Added distinction between temporary rate limiting and quota exhaustion - Improved user messaging for rate limit scenarios - Added comprehensive tests for rate limit handling Fixes #8012
This commit is contained in:
parent
759454b455
commit
125fa83c7c
3 changed files with 442 additions and 11 deletions
307
src/api/providers/__tests__/gemini-rate-limit.spec.ts
Normal file
307
src/api/providers/__tests__/gemini-rate-limit.spec.ts
Normal file
|
|
@ -0,0 +1,307 @@
|
|||
// npx vitest run src/api/providers/__tests__/gemini-rate-limit.spec.ts
|
||||
|
||||
import { describe, it, expect, vi, beforeEach } from "vitest"
|
||||
import { GeminiHandler, GeminiError } from "../gemini"
|
||||
import { t } from "i18next"
|
||||
|
||||
describe("GeminiHandler Rate Limit Handling", () => {
|
||||
let handler: GeminiHandler
|
||||
|
||||
beforeEach(() => {
|
||||
handler = new GeminiHandler({
|
||||
apiModelId: "gemini-1.5-flash",
|
||||
geminiApiKey: "test-key",
|
||||
})
|
||||
})
|
||||
|
||||
describe("GeminiError", () => {
|
||||
it("should properly construct error with RetryInfo", () => {
|
||||
const error = new GeminiError("Rate limit exceeded", {
|
||||
status: 429,
|
||||
error: {
|
||||
status: "RESOURCE_EXHAUSTED",
|
||||
message: "Too many requests",
|
||||
details: [
|
||||
{
|
||||
"@type": "type.googleapis.com/google.rpc.RetryInfo",
|
||||
retryDelay: "59s",
|
||||
},
|
||||
],
|
||||
},
|
||||
})
|
||||
|
||||
expect(error.status).toBe(429)
|
||||
expect(error.errorStatus).toBe("RESOURCE_EXHAUSTED")
|
||||
expect(error.errorDetails).toHaveLength(1)
|
||||
expect(error.errorDetails?.[0]).toEqual({
|
||||
"@type": "type.googleapis.com/google.rpc.RetryInfo",
|
||||
retryDelay: "59s",
|
||||
})
|
||||
})
|
||||
|
||||
it("should properly construct error with QuotaFailure", () => {
|
||||
const error = new GeminiError("Quota exceeded", {
|
||||
status: 429,
|
||||
error: {
|
||||
status: "RESOURCE_EXHAUSTED",
|
||||
message: "Quota exceeded",
|
||||
details: [
|
||||
{
|
||||
"@type": "type.googleapis.com/google.rpc.QuotaFailure",
|
||||
violations: [
|
||||
{
|
||||
subject: "tokens_per_minute",
|
||||
description: "Token limit exceeded for model",
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
},
|
||||
})
|
||||
|
||||
expect(error.status).toBe(429)
|
||||
expect(error.errorStatus).toBe("RESOURCE_EXHAUSTED")
|
||||
expect(error.errorDetails).toHaveLength(1)
|
||||
expect(error.errorDetails?.[0]).toEqual({
|
||||
"@type": "type.googleapis.com/google.rpc.QuotaFailure",
|
||||
violations: [
|
||||
{
|
||||
subject: "tokens_per_minute",
|
||||
description: "Token limit exceeded for model",
|
||||
},
|
||||
],
|
||||
})
|
||||
})
|
||||
|
||||
it("should handle both RetryInfo and QuotaFailure in same error", () => {
|
||||
const error = new GeminiError("Rate limit with retry", {
|
||||
status: 429,
|
||||
error: {
|
||||
status: "RESOURCE_EXHAUSTED",
|
||||
message: "Rate limit exceeded",
|
||||
details: [
|
||||
{
|
||||
"@type": "type.googleapis.com/google.rpc.RetryInfo",
|
||||
retryDelay: "30s",
|
||||
},
|
||||
{
|
||||
"@type": "type.googleapis.com/google.rpc.QuotaFailure",
|
||||
violations: [
|
||||
{
|
||||
subject: "requests_per_minute",
|
||||
description: "Request limit exceeded",
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
},
|
||||
})
|
||||
|
||||
expect(error.status).toBe(429)
|
||||
expect(error.errorDetails).toHaveLength(2)
|
||||
|
||||
const retryInfo = error.errorDetails?.find(
|
||||
(d: any) => d["@type"] === "type.googleapis.com/google.rpc.RetryInfo",
|
||||
)
|
||||
expect(retryInfo?.retryDelay).toBe("30s")
|
||||
|
||||
const quotaFailure = error.errorDetails?.find(
|
||||
(d: any) => d["@type"] === "type.googleapis.com/google.rpc.QuotaFailure",
|
||||
)
|
||||
expect(quotaFailure?.violations?.[0]?.subject).toBe("requests_per_minute")
|
||||
})
|
||||
|
||||
it("should handle error details in errorDetails field (alternative format)", () => {
|
||||
const error = new GeminiError("Rate limit", {
|
||||
status: 429,
|
||||
errorDetails: [
|
||||
{
|
||||
"@type": "type.googleapis.com/google.rpc.RetryInfo",
|
||||
retryDelay: "45s",
|
||||
},
|
||||
],
|
||||
})
|
||||
|
||||
expect(error.status).toBe(429)
|
||||
expect(error.errorDetails).toHaveLength(1)
|
||||
expect(error.errorDetails?.[0].retryDelay).toBe("45s")
|
||||
})
|
||||
})
|
||||
|
||||
describe("Error transformation in createMessage", () => {
|
||||
it("should transform 429 errors with proper structure", async () => {
|
||||
// Mock the GoogleGenAI client to throw an error
|
||||
const mockError = {
|
||||
status: 429,
|
||||
message: "Resource exhausted",
|
||||
error: {
|
||||
status: "RESOURCE_EXHAUSTED",
|
||||
details: [
|
||||
{
|
||||
"@type": "type.googleapis.com/google.rpc.RetryInfo",
|
||||
retryDelay: "60s",
|
||||
},
|
||||
],
|
||||
},
|
||||
}
|
||||
|
||||
// Mock the client's generateContentStream method
|
||||
const mockClient = {
|
||||
models: {
|
||||
generateContentStream: vi.fn().mockRejectedValue(mockError),
|
||||
},
|
||||
}
|
||||
;(handler as any).client = mockClient
|
||||
|
||||
// Attempt to create a message and expect it to throw GeminiError
|
||||
try {
|
||||
const stream = handler.createMessage("system", [{ role: "user", content: "test" }])
|
||||
// Consume the stream to trigger the error
|
||||
for await (const chunk of stream) {
|
||||
// This should not be reached
|
||||
}
|
||||
expect.fail("Should have thrown an error")
|
||||
} catch (error) {
|
||||
expect(error).toBeInstanceOf(GeminiError)
|
||||
const geminiError = error as GeminiError
|
||||
expect(geminiError.status).toBe(429)
|
||||
expect(geminiError.errorDetails?.[0]).toEqual({
|
||||
"@type": "type.googleapis.com/google.rpc.RetryInfo",
|
||||
retryDelay: "60s",
|
||||
})
|
||||
}
|
||||
})
|
||||
|
||||
it("should handle quota exhaustion errors", async () => {
|
||||
const mockError = {
|
||||
status: 429,
|
||||
message: "Daily quota exceeded",
|
||||
error: {
|
||||
status: "RESOURCE_EXHAUSTED",
|
||||
details: [
|
||||
{
|
||||
"@type": "type.googleapis.com/google.rpc.QuotaFailure",
|
||||
violations: [
|
||||
{
|
||||
subject: "daily_quota",
|
||||
description: "Daily quota has been exhausted",
|
||||
},
|
||||
],
|
||||
},
|
||||
],
|
||||
},
|
||||
}
|
||||
|
||||
const mockClient = {
|
||||
models: {
|
||||
generateContentStream: vi.fn().mockRejectedValue(mockError),
|
||||
},
|
||||
}
|
||||
;(handler as any).client = mockClient
|
||||
|
||||
try {
|
||||
const stream = handler.createMessage("system", [{ role: "user", content: "test" }])
|
||||
for await (const chunk of stream) {
|
||||
// Should not reach here
|
||||
}
|
||||
expect.fail("Should have thrown an error")
|
||||
} catch (error) {
|
||||
expect(error).toBeInstanceOf(GeminiError)
|
||||
const geminiError = error as GeminiError
|
||||
expect(geminiError.status).toBe(429)
|
||||
|
||||
const quotaFailure = geminiError.errorDetails?.find(
|
||||
(d: any) => d["@type"] === "type.googleapis.com/google.rpc.QuotaFailure",
|
||||
)
|
||||
expect(quotaFailure?.violations?.[0]?.description).toBe("Daily quota has been exhausted")
|
||||
}
|
||||
})
|
||||
|
||||
it("should handle generic errors without status", async () => {
|
||||
const mockError = new Error("Network error")
|
||||
|
||||
const mockClient = {
|
||||
models: {
|
||||
generateContentStream: vi.fn().mockRejectedValue(mockError),
|
||||
},
|
||||
}
|
||||
;(handler as any).client = mockClient
|
||||
|
||||
try {
|
||||
const stream = handler.createMessage("system", [{ role: "user", content: "test" }])
|
||||
for await (const chunk of stream) {
|
||||
// Should not reach here
|
||||
}
|
||||
expect.fail("Should have thrown an error")
|
||||
} catch (error) {
|
||||
expect(error).toBeInstanceOf(GeminiError)
|
||||
const geminiError = error as GeminiError
|
||||
// The message will be the translated error message
|
||||
expect(geminiError.message).toBeDefined()
|
||||
expect(geminiError.status).toBeUndefined()
|
||||
expect(geminiError.errorDetails).toBeUndefined()
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
describe("Error transformation in completePrompt", () => {
|
||||
it("should transform 429 errors in completePrompt", async () => {
|
||||
const mockError = {
|
||||
status: 429,
|
||||
message: "Rate limit",
|
||||
error: {
|
||||
status: "RESOURCE_EXHAUSTED",
|
||||
details: [
|
||||
{
|
||||
"@type": "type.googleapis.com/google.rpc.RetryInfo",
|
||||
retryDelay: "30s",
|
||||
},
|
||||
],
|
||||
},
|
||||
}
|
||||
|
||||
const mockClient = {
|
||||
models: {
|
||||
generateContent: vi.fn().mockRejectedValue(mockError),
|
||||
},
|
||||
}
|
||||
;(handler as any).client = mockClient
|
||||
|
||||
try {
|
||||
await handler.completePrompt("test prompt")
|
||||
expect.fail("Should have thrown an error")
|
||||
} catch (error) {
|
||||
expect(error).toBeInstanceOf(GeminiError)
|
||||
const geminiError = error as GeminiError
|
||||
expect(geminiError.status).toBe(429)
|
||||
expect(geminiError.errorDetails?.[0].retryDelay).toBe("30s")
|
||||
}
|
||||
})
|
||||
})
|
||||
|
||||
describe("Retry delay parsing", () => {
|
||||
it("should correctly parse various delay formats", () => {
|
||||
const testCases = [
|
||||
{ input: "59s", expected: 59 },
|
||||
{ input: "120s", expected: 120 },
|
||||
{ input: "1s", expected: 1 },
|
||||
{ input: "0s", expected: 0 },
|
||||
]
|
||||
|
||||
testCases.forEach(({ input, expected }) => {
|
||||
const match = input.match(/^(\d+)s$/)
|
||||
expect(match).toBeTruthy()
|
||||
expect(Number(match![1])).toBe(expected)
|
||||
})
|
||||
})
|
||||
|
||||
it("should not match invalid delay formats", () => {
|
||||
const invalidFormats = ["59", "s59", "59m", "59.5s", ""]
|
||||
|
||||
invalidFormats.forEach((format) => {
|
||||
const match = format.match(/^(\d+)s$/)
|
||||
expect(match).toBeFalsy()
|
||||
})
|
||||
})
|
||||
})
|
||||
})
|
||||
|
|
@ -21,6 +21,44 @@ import { getModelParams } from "../transform/model-params"
|
|||
import type { SingleCompletionHandler, ApiHandlerCreateMessageMetadata } from "../index"
|
||||
import { BaseProvider } from "./base-provider"
|
||||
|
||||
// Error detail types for Gemini API errors
|
||||
export interface GeminiRetryInfo {
|
||||
"@type": "type.googleapis.com/google.rpc.RetryInfo"
|
||||
retryDelay?: string // e.g., "59s"
|
||||
}
|
||||
|
||||
export interface GeminiQuotaFailure {
|
||||
"@type": "type.googleapis.com/google.rpc.QuotaFailure"
|
||||
violations?: Array<{
|
||||
subject?: string
|
||||
description?: string
|
||||
}>
|
||||
}
|
||||
|
||||
export interface GeminiErrorDetails {
|
||||
status?: number
|
||||
error?: {
|
||||
status?: string // e.g., "RESOURCE_EXHAUSTED"
|
||||
message?: string
|
||||
details?: Array<GeminiRetryInfo | GeminiQuotaFailure | any>
|
||||
}
|
||||
errorDetails?: Array<GeminiRetryInfo | GeminiQuotaFailure | any>
|
||||
}
|
||||
|
||||
export class GeminiError extends Error {
|
||||
status?: number
|
||||
errorStatus?: string
|
||||
errorDetails?: Array<GeminiRetryInfo | GeminiQuotaFailure | any>
|
||||
|
||||
constructor(message: string, details?: GeminiErrorDetails) {
|
||||
super(message)
|
||||
this.name = "GeminiError"
|
||||
this.status = details?.status
|
||||
this.errorStatus = details?.error?.status
|
||||
this.errorDetails = details?.error?.details || details?.errorDetails
|
||||
}
|
||||
}
|
||||
|
||||
type GeminiHandlerOptions = ApiHandlerOptions & {
|
||||
isVertex?: boolean
|
||||
}
|
||||
|
|
@ -154,8 +192,20 @@ export class GeminiHandler extends BaseProvider implements SingleCompletionHandl
|
|||
}
|
||||
}
|
||||
} catch (error) {
|
||||
if (error instanceof Error) {
|
||||
throw new Error(t("common:errors.gemini.generate_stream", { error: error.message }))
|
||||
// Parse and enhance error information for better handling upstream
|
||||
if (error && typeof error === "object" && "status" in error) {
|
||||
const errorObj = error as any
|
||||
const geminiError = new GeminiError(
|
||||
errorObj.message || t("common:errors.gemini.generate_stream", { error: "Unknown error" }),
|
||||
{
|
||||
status: errorObj.status,
|
||||
error: errorObj.error,
|
||||
errorDetails: errorObj.errorDetails || errorObj.error?.details,
|
||||
},
|
||||
)
|
||||
throw geminiError
|
||||
} else if (error instanceof Error) {
|
||||
throw new GeminiError(t("common:errors.gemini.generate_stream", { error: error.message }))
|
||||
}
|
||||
|
||||
throw error
|
||||
|
|
@ -246,8 +296,20 @@ export class GeminiHandler extends BaseProvider implements SingleCompletionHandl
|
|||
|
||||
return text
|
||||
} catch (error) {
|
||||
if (error instanceof Error) {
|
||||
throw new Error(t("common:errors.gemini.generate_complete_prompt", { error: error.message }))
|
||||
// Parse and enhance error information for better handling upstream
|
||||
if (error && typeof error === "object" && "status" in error) {
|
||||
const errorObj = error as any
|
||||
const geminiError = new GeminiError(
|
||||
errorObj.message || t("common:errors.gemini.generate_complete_prompt", { error: "Unknown error" }),
|
||||
{
|
||||
status: errorObj.status,
|
||||
error: errorObj.error,
|
||||
errorDetails: errorObj.errorDetails || errorObj.error?.details,
|
||||
},
|
||||
)
|
||||
throw geminiError
|
||||
} else if (error instanceof Error) {
|
||||
throw new GeminiError(t("common:errors.gemini.generate_complete_prompt", { error: error.message }))
|
||||
}
|
||||
|
||||
throw error
|
||||
|
|
|
|||
|
|
@ -2719,15 +2719,43 @@ export class Task extends EventEmitter<TaskEvents> implements TaskLike {
|
|||
MAX_EXPONENTIAL_BACKOFF_SECONDS,
|
||||
)
|
||||
|
||||
// If the error is a 429, and the error details contain a retry delay, use that delay instead of exponential backoff
|
||||
// If the error is a 429, check for Gemini-specific error details
|
||||
if (error.status === 429) {
|
||||
// Check for RetryInfo to get provider-suggested delay
|
||||
const geminiRetryDetails = error.errorDetails?.find(
|
||||
(detail: any) => detail["@type"] === "type.googleapis.com/google.rpc.RetryInfo",
|
||||
)
|
||||
if (geminiRetryDetails) {
|
||||
const match = geminiRetryDetails?.retryDelay?.match(/^(\d+)s$/)
|
||||
|
||||
// Check for QuotaFailure to determine if it's quota exhaustion
|
||||
const quotaFailure = error.errorDetails?.find(
|
||||
(detail: any) => detail["@type"] === "type.googleapis.com/google.rpc.QuotaFailure",
|
||||
)
|
||||
|
||||
if (geminiRetryDetails?.retryDelay) {
|
||||
// Parse the delay from format like "59s"
|
||||
const match = geminiRetryDetails.retryDelay.match(/^(\d+)s$/)
|
||||
if (match) {
|
||||
exponentialDelay = Number(match[1]) + 1
|
||||
// Add a small buffer (1-2 seconds) as recommended
|
||||
exponentialDelay = Number(match[1]) + 2
|
||||
}
|
||||
}
|
||||
|
||||
// If we have quota failure but no retry info, it might be quota exhaustion
|
||||
if (quotaFailure && !geminiRetryDetails?.retryDelay) {
|
||||
// Check if the error message indicates daily/monthly quota exhaustion
|
||||
const isQuotaExhausted =
|
||||
error.message?.toLowerCase().includes("quota") &&
|
||||
(error.message?.toLowerCase().includes("daily") ||
|
||||
error.message?.toLowerCase().includes("monthly") ||
|
||||
error.message?.toLowerCase().includes("exceeded"))
|
||||
|
||||
if (isQuotaExhausted) {
|
||||
// Don't retry for quota exhaustion - show clear message and fail
|
||||
await this.say(
|
||||
"error",
|
||||
`Gemini API quota exhausted. ${quotaFailure.violations?.[0]?.description || "Your daily or monthly quota has been exceeded."}\n\nPlease check your quota limits at: https://ai.google.dev/gemini-api/docs/rate-limits`,
|
||||
)
|
||||
throw new Error("Gemini API quota exhausted - cannot retry")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
@ -2735,11 +2763,26 @@ export class Task extends EventEmitter<TaskEvents> implements TaskLike {
|
|||
// Wait for the greater of the exponential delay or the rate limit delay
|
||||
const finalDelay = Math.max(exponentialDelay, rateLimitDelay)
|
||||
|
||||
// Show countdown timer with exponential backoff
|
||||
// Determine the reason for the delay to show appropriate message
|
||||
let delayReason = ""
|
||||
if (error.status === 429) {
|
||||
const hasRetryInfo = error.errorDetails?.find(
|
||||
(detail: any) => detail["@type"] === "type.googleapis.com/google.rpc.RetryInfo",
|
||||
)
|
||||
if (hasRetryInfo) {
|
||||
delayReason = "Rate limit reached. Waiting for the provider-recommended delay"
|
||||
} else {
|
||||
delayReason = "Rate limit reached. Using exponential backoff"
|
||||
}
|
||||
} else {
|
||||
delayReason = errorMsg
|
||||
}
|
||||
|
||||
// Show countdown timer with clear messaging
|
||||
for (let i = finalDelay; i > 0; i--) {
|
||||
await this.say(
|
||||
"api_req_retry_delayed",
|
||||
`${errorMsg}\n\nRetry attempt ${retryAttempt + 1}\nRetrying in ${i} seconds...`,
|
||||
`${delayReason}\n\nRetry attempt ${retryAttempt + 1}\nRetrying in ${i} seconds...`,
|
||||
undefined,
|
||||
true,
|
||||
)
|
||||
|
|
@ -2748,7 +2791,7 @@ export class Task extends EventEmitter<TaskEvents> implements TaskLike {
|
|||
|
||||
await this.say(
|
||||
"api_req_retry_delayed",
|
||||
`${errorMsg}\n\nRetry attempt ${retryAttempt + 1}\nRetrying now...`,
|
||||
`${delayReason}\n\nRetry attempt ${retryAttempt + 1}\nRetrying now...`,
|
||||
undefined,
|
||||
false,
|
||||
)
|
||||
|
|
@ -2759,6 +2802,25 @@ export class Task extends EventEmitter<TaskEvents> implements TaskLike {
|
|||
|
||||
return
|
||||
} else {
|
||||
// For non-auto-retry cases, check if it's a Gemini quota exhaustion
|
||||
if (error.status === 429) {
|
||||
const quotaFailure = error.errorDetails?.find(
|
||||
(detail: any) => detail["@type"] === "type.googleapis.com/google.rpc.QuotaFailure",
|
||||
)
|
||||
const retryInfo = error.errorDetails?.find(
|
||||
(detail: any) => detail["@type"] === "type.googleapis.com/google.rpc.RetryInfo",
|
||||
)
|
||||
|
||||
// If quota failure without retry info, likely quota exhaustion
|
||||
if (quotaFailure && !retryInfo) {
|
||||
await this.say(
|
||||
"error",
|
||||
`Gemini API quota exhausted. ${quotaFailure.violations?.[0]?.description || "Your daily or monthly quota has been exceeded."}\n\nPlease check your quota limits at: https://ai.google.dev/gemini-api/docs/rate-limits`,
|
||||
)
|
||||
throw new Error("Gemini API quota exhausted - cannot retry")
|
||||
}
|
||||
}
|
||||
|
||||
const { response } = await this.ask(
|
||||
"api_req_failed",
|
||||
error.message ?? JSON.stringify(serializeError(error), null, 2),
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue