From 125fa83c7cfb81b4f61f51cf0a4e43426854d879 Mon Sep 17 00:00:00 2001 From: Roo Code Date: Tue, 16 Sep 2025 03:19:30 +0000 Subject: [PATCH] feat: honor Gemini retryDelay and improve rate limit handling - Enhanced GeminiHandler to properly parse 429 errors with RetryInfo and QuotaFailure - Added GeminiError class to preserve structured error details - Updated retry logic to respect provider-suggested delays with 2-second buffer - Added distinction between temporary rate limiting and quota exhaustion - Improved user messaging for rate limit scenarios - Added comprehensive tests for rate limit handling Fixes #8012 --- .../__tests__/gemini-rate-limit.spec.ts | 307 ++++++++++++++++++ src/api/providers/gemini.ts | 70 +++- src/core/task/Task.ts | 76 ++++- 3 files changed, 442 insertions(+), 11 deletions(-) create mode 100644 src/api/providers/__tests__/gemini-rate-limit.spec.ts diff --git a/src/api/providers/__tests__/gemini-rate-limit.spec.ts b/src/api/providers/__tests__/gemini-rate-limit.spec.ts new file mode 100644 index 0000000000..dce515f6e3 --- /dev/null +++ b/src/api/providers/__tests__/gemini-rate-limit.spec.ts @@ -0,0 +1,307 @@ +// npx vitest run src/api/providers/__tests__/gemini-rate-limit.spec.ts + +import { describe, it, expect, vi, beforeEach } from "vitest" +import { GeminiHandler, GeminiError } from "../gemini" +import { t } from "i18next" + +describe("GeminiHandler Rate Limit Handling", () => { + let handler: GeminiHandler + + beforeEach(() => { + handler = new GeminiHandler({ + apiModelId: "gemini-1.5-flash", + geminiApiKey: "test-key", + }) + }) + + describe("GeminiError", () => { + it("should properly construct error with RetryInfo", () => { + const error = new GeminiError("Rate limit exceeded", { + status: 429, + error: { + status: "RESOURCE_EXHAUSTED", + message: "Too many requests", + details: [ + { + "@type": "type.googleapis.com/google.rpc.RetryInfo", + retryDelay: "59s", + }, + ], + }, + }) + + expect(error.status).toBe(429) + expect(error.errorStatus).toBe("RESOURCE_EXHAUSTED") + expect(error.errorDetails).toHaveLength(1) + expect(error.errorDetails?.[0]).toEqual({ + "@type": "type.googleapis.com/google.rpc.RetryInfo", + retryDelay: "59s", + }) + }) + + it("should properly construct error with QuotaFailure", () => { + const error = new GeminiError("Quota exceeded", { + status: 429, + error: { + status: "RESOURCE_EXHAUSTED", + message: "Quota exceeded", + details: [ + { + "@type": "type.googleapis.com/google.rpc.QuotaFailure", + violations: [ + { + subject: "tokens_per_minute", + description: "Token limit exceeded for model", + }, + ], + }, + ], + }, + }) + + expect(error.status).toBe(429) + expect(error.errorStatus).toBe("RESOURCE_EXHAUSTED") + expect(error.errorDetails).toHaveLength(1) + expect(error.errorDetails?.[0]).toEqual({ + "@type": "type.googleapis.com/google.rpc.QuotaFailure", + violations: [ + { + subject: "tokens_per_minute", + description: "Token limit exceeded for model", + }, + ], + }) + }) + + it("should handle both RetryInfo and QuotaFailure in same error", () => { + const error = new GeminiError("Rate limit with retry", { + status: 429, + error: { + status: "RESOURCE_EXHAUSTED", + message: "Rate limit exceeded", + details: [ + { + "@type": "type.googleapis.com/google.rpc.RetryInfo", + retryDelay: "30s", + }, + { + "@type": "type.googleapis.com/google.rpc.QuotaFailure", + violations: [ + { + subject: "requests_per_minute", + description: "Request limit exceeded", + }, + ], + }, + ], + }, + }) + + expect(error.status).toBe(429) + expect(error.errorDetails).toHaveLength(2) + + const retryInfo = error.errorDetails?.find( + (d: any) => d["@type"] === "type.googleapis.com/google.rpc.RetryInfo", + ) + expect(retryInfo?.retryDelay).toBe("30s") + + const quotaFailure = error.errorDetails?.find( + (d: any) => d["@type"] === "type.googleapis.com/google.rpc.QuotaFailure", + ) + expect(quotaFailure?.violations?.[0]?.subject).toBe("requests_per_minute") + }) + + it("should handle error details in errorDetails field (alternative format)", () => { + const error = new GeminiError("Rate limit", { + status: 429, + errorDetails: [ + { + "@type": "type.googleapis.com/google.rpc.RetryInfo", + retryDelay: "45s", + }, + ], + }) + + expect(error.status).toBe(429) + expect(error.errorDetails).toHaveLength(1) + expect(error.errorDetails?.[0].retryDelay).toBe("45s") + }) + }) + + describe("Error transformation in createMessage", () => { + it("should transform 429 errors with proper structure", async () => { + // Mock the GoogleGenAI client to throw an error + const mockError = { + status: 429, + message: "Resource exhausted", + error: { + status: "RESOURCE_EXHAUSTED", + details: [ + { + "@type": "type.googleapis.com/google.rpc.RetryInfo", + retryDelay: "60s", + }, + ], + }, + } + + // Mock the client's generateContentStream method + const mockClient = { + models: { + generateContentStream: vi.fn().mockRejectedValue(mockError), + }, + } + ;(handler as any).client = mockClient + + // Attempt to create a message and expect it to throw GeminiError + try { + const stream = handler.createMessage("system", [{ role: "user", content: "test" }]) + // Consume the stream to trigger the error + for await (const chunk of stream) { + // This should not be reached + } + expect.fail("Should have thrown an error") + } catch (error) { + expect(error).toBeInstanceOf(GeminiError) + const geminiError = error as GeminiError + expect(geminiError.status).toBe(429) + expect(geminiError.errorDetails?.[0]).toEqual({ + "@type": "type.googleapis.com/google.rpc.RetryInfo", + retryDelay: "60s", + }) + } + }) + + it("should handle quota exhaustion errors", async () => { + const mockError = { + status: 429, + message: "Daily quota exceeded", + error: { + status: "RESOURCE_EXHAUSTED", + details: [ + { + "@type": "type.googleapis.com/google.rpc.QuotaFailure", + violations: [ + { + subject: "daily_quota", + description: "Daily quota has been exhausted", + }, + ], + }, + ], + }, + } + + const mockClient = { + models: { + generateContentStream: vi.fn().mockRejectedValue(mockError), + }, + } + ;(handler as any).client = mockClient + + try { + const stream = handler.createMessage("system", [{ role: "user", content: "test" }]) + for await (const chunk of stream) { + // Should not reach here + } + expect.fail("Should have thrown an error") + } catch (error) { + expect(error).toBeInstanceOf(GeminiError) + const geminiError = error as GeminiError + expect(geminiError.status).toBe(429) + + const quotaFailure = geminiError.errorDetails?.find( + (d: any) => d["@type"] === "type.googleapis.com/google.rpc.QuotaFailure", + ) + expect(quotaFailure?.violations?.[0]?.description).toBe("Daily quota has been exhausted") + } + }) + + it("should handle generic errors without status", async () => { + const mockError = new Error("Network error") + + const mockClient = { + models: { + generateContentStream: vi.fn().mockRejectedValue(mockError), + }, + } + ;(handler as any).client = mockClient + + try { + const stream = handler.createMessage("system", [{ role: "user", content: "test" }]) + for await (const chunk of stream) { + // Should not reach here + } + expect.fail("Should have thrown an error") + } catch (error) { + expect(error).toBeInstanceOf(GeminiError) + const geminiError = error as GeminiError + // The message will be the translated error message + expect(geminiError.message).toBeDefined() + expect(geminiError.status).toBeUndefined() + expect(geminiError.errorDetails).toBeUndefined() + } + }) + }) + + describe("Error transformation in completePrompt", () => { + it("should transform 429 errors in completePrompt", async () => { + const mockError = { + status: 429, + message: "Rate limit", + error: { + status: "RESOURCE_EXHAUSTED", + details: [ + { + "@type": "type.googleapis.com/google.rpc.RetryInfo", + retryDelay: "30s", + }, + ], + }, + } + + const mockClient = { + models: { + generateContent: vi.fn().mockRejectedValue(mockError), + }, + } + ;(handler as any).client = mockClient + + try { + await handler.completePrompt("test prompt") + expect.fail("Should have thrown an error") + } catch (error) { + expect(error).toBeInstanceOf(GeminiError) + const geminiError = error as GeminiError + expect(geminiError.status).toBe(429) + expect(geminiError.errorDetails?.[0].retryDelay).toBe("30s") + } + }) + }) + + describe("Retry delay parsing", () => { + it("should correctly parse various delay formats", () => { + const testCases = [ + { input: "59s", expected: 59 }, + { input: "120s", expected: 120 }, + { input: "1s", expected: 1 }, + { input: "0s", expected: 0 }, + ] + + testCases.forEach(({ input, expected }) => { + const match = input.match(/^(\d+)s$/) + expect(match).toBeTruthy() + expect(Number(match![1])).toBe(expected) + }) + }) + + it("should not match invalid delay formats", () => { + const invalidFormats = ["59", "s59", "59m", "59.5s", ""] + + invalidFormats.forEach((format) => { + const match = format.match(/^(\d+)s$/) + expect(match).toBeFalsy() + }) + }) + }) +}) diff --git a/src/api/providers/gemini.ts b/src/api/providers/gemini.ts index 775d763a05..a012963fc9 100644 --- a/src/api/providers/gemini.ts +++ b/src/api/providers/gemini.ts @@ -21,6 +21,44 @@ import { getModelParams } from "../transform/model-params" import type { SingleCompletionHandler, ApiHandlerCreateMessageMetadata } from "../index" import { BaseProvider } from "./base-provider" +// Error detail types for Gemini API errors +export interface GeminiRetryInfo { + "@type": "type.googleapis.com/google.rpc.RetryInfo" + retryDelay?: string // e.g., "59s" +} + +export interface GeminiQuotaFailure { + "@type": "type.googleapis.com/google.rpc.QuotaFailure" + violations?: Array<{ + subject?: string + description?: string + }> +} + +export interface GeminiErrorDetails { + status?: number + error?: { + status?: string // e.g., "RESOURCE_EXHAUSTED" + message?: string + details?: Array + } + errorDetails?: Array +} + +export class GeminiError extends Error { + status?: number + errorStatus?: string + errorDetails?: Array + + constructor(message: string, details?: GeminiErrorDetails) { + super(message) + this.name = "GeminiError" + this.status = details?.status + this.errorStatus = details?.error?.status + this.errorDetails = details?.error?.details || details?.errorDetails + } +} + type GeminiHandlerOptions = ApiHandlerOptions & { isVertex?: boolean } @@ -154,8 +192,20 @@ export class GeminiHandler extends BaseProvider implements SingleCompletionHandl } } } catch (error) { - if (error instanceof Error) { - throw new Error(t("common:errors.gemini.generate_stream", { error: error.message })) + // Parse and enhance error information for better handling upstream + if (error && typeof error === "object" && "status" in error) { + const errorObj = error as any + const geminiError = new GeminiError( + errorObj.message || t("common:errors.gemini.generate_stream", { error: "Unknown error" }), + { + status: errorObj.status, + error: errorObj.error, + errorDetails: errorObj.errorDetails || errorObj.error?.details, + }, + ) + throw geminiError + } else if (error instanceof Error) { + throw new GeminiError(t("common:errors.gemini.generate_stream", { error: error.message })) } throw error @@ -246,8 +296,20 @@ export class GeminiHandler extends BaseProvider implements SingleCompletionHandl return text } catch (error) { - if (error instanceof Error) { - throw new Error(t("common:errors.gemini.generate_complete_prompt", { error: error.message })) + // Parse and enhance error information for better handling upstream + if (error && typeof error === "object" && "status" in error) { + const errorObj = error as any + const geminiError = new GeminiError( + errorObj.message || t("common:errors.gemini.generate_complete_prompt", { error: "Unknown error" }), + { + status: errorObj.status, + error: errorObj.error, + errorDetails: errorObj.errorDetails || errorObj.error?.details, + }, + ) + throw geminiError + } else if (error instanceof Error) { + throw new GeminiError(t("common:errors.gemini.generate_complete_prompt", { error: error.message })) } throw error diff --git a/src/core/task/Task.ts b/src/core/task/Task.ts index cf16df8dcc..99d41dbeea 100644 --- a/src/core/task/Task.ts +++ b/src/core/task/Task.ts @@ -2719,15 +2719,43 @@ export class Task extends EventEmitter implements TaskLike { MAX_EXPONENTIAL_BACKOFF_SECONDS, ) - // If the error is a 429, and the error details contain a retry delay, use that delay instead of exponential backoff + // If the error is a 429, check for Gemini-specific error details if (error.status === 429) { + // Check for RetryInfo to get provider-suggested delay const geminiRetryDetails = error.errorDetails?.find( (detail: any) => detail["@type"] === "type.googleapis.com/google.rpc.RetryInfo", ) - if (geminiRetryDetails) { - const match = geminiRetryDetails?.retryDelay?.match(/^(\d+)s$/) + + // Check for QuotaFailure to determine if it's quota exhaustion + const quotaFailure = error.errorDetails?.find( + (detail: any) => detail["@type"] === "type.googleapis.com/google.rpc.QuotaFailure", + ) + + if (geminiRetryDetails?.retryDelay) { + // Parse the delay from format like "59s" + const match = geminiRetryDetails.retryDelay.match(/^(\d+)s$/) if (match) { - exponentialDelay = Number(match[1]) + 1 + // Add a small buffer (1-2 seconds) as recommended + exponentialDelay = Number(match[1]) + 2 + } + } + + // If we have quota failure but no retry info, it might be quota exhaustion + if (quotaFailure && !geminiRetryDetails?.retryDelay) { + // Check if the error message indicates daily/monthly quota exhaustion + const isQuotaExhausted = + error.message?.toLowerCase().includes("quota") && + (error.message?.toLowerCase().includes("daily") || + error.message?.toLowerCase().includes("monthly") || + error.message?.toLowerCase().includes("exceeded")) + + if (isQuotaExhausted) { + // Don't retry for quota exhaustion - show clear message and fail + await this.say( + "error", + `Gemini API quota exhausted. ${quotaFailure.violations?.[0]?.description || "Your daily or monthly quota has been exceeded."}\n\nPlease check your quota limits at: https://ai.google.dev/gemini-api/docs/rate-limits`, + ) + throw new Error("Gemini API quota exhausted - cannot retry") } } } @@ -2735,11 +2763,26 @@ export class Task extends EventEmitter implements TaskLike { // Wait for the greater of the exponential delay or the rate limit delay const finalDelay = Math.max(exponentialDelay, rateLimitDelay) - // Show countdown timer with exponential backoff + // Determine the reason for the delay to show appropriate message + let delayReason = "" + if (error.status === 429) { + const hasRetryInfo = error.errorDetails?.find( + (detail: any) => detail["@type"] === "type.googleapis.com/google.rpc.RetryInfo", + ) + if (hasRetryInfo) { + delayReason = "Rate limit reached. Waiting for the provider-recommended delay" + } else { + delayReason = "Rate limit reached. Using exponential backoff" + } + } else { + delayReason = errorMsg + } + + // Show countdown timer with clear messaging for (let i = finalDelay; i > 0; i--) { await this.say( "api_req_retry_delayed", - `${errorMsg}\n\nRetry attempt ${retryAttempt + 1}\nRetrying in ${i} seconds...`, + `${delayReason}\n\nRetry attempt ${retryAttempt + 1}\nRetrying in ${i} seconds...`, undefined, true, ) @@ -2748,7 +2791,7 @@ export class Task extends EventEmitter implements TaskLike { await this.say( "api_req_retry_delayed", - `${errorMsg}\n\nRetry attempt ${retryAttempt + 1}\nRetrying now...`, + `${delayReason}\n\nRetry attempt ${retryAttempt + 1}\nRetrying now...`, undefined, false, ) @@ -2759,6 +2802,25 @@ export class Task extends EventEmitter implements TaskLike { return } else { + // For non-auto-retry cases, check if it's a Gemini quota exhaustion + if (error.status === 429) { + const quotaFailure = error.errorDetails?.find( + (detail: any) => detail["@type"] === "type.googleapis.com/google.rpc.QuotaFailure", + ) + const retryInfo = error.errorDetails?.find( + (detail: any) => detail["@type"] === "type.googleapis.com/google.rpc.RetryInfo", + ) + + // If quota failure without retry info, likely quota exhaustion + if (quotaFailure && !retryInfo) { + await this.say( + "error", + `Gemini API quota exhausted. ${quotaFailure.violations?.[0]?.description || "Your daily or monthly quota has been exceeded."}\n\nPlease check your quota limits at: https://ai.google.dev/gemini-api/docs/rate-limits`, + ) + throw new Error("Gemini API quota exhausted - cannot retry") + } + } + const { response } = await this.ask( "api_req_failed", error.message ?? JSON.stringify(serializeError(error), null, 2),