From e69577d246dbf276aad821eee16989364747952c Mon Sep 17 00:00:00 2001 From: eeshsaxena Date: Tue, 28 Jul 2026 21:58:50 +0530 Subject: [PATCH] fix(exception_mapping): map Gemini 429 quota errors before the 403 substring branch In _map_vertex_exception the `elif "403" in error_str` branch is an unanchored substring match over the whole serialized error body, and it ran before the 429 / RESOURCE_EXHAUSTED / quota branch. Google's AI Studio quota body carries a sub-second retry hint such as "Please retry in 18.403470473s.", whose digits contain "403", so an ordinary HTTP 429 mapped to BadRequestError(status_code=403) whenever the retry delay happened to include those digits. Because litellm._should_retry(403) is False, the Router's retry/fallback machinery then declined to retry a transient rate limit, turning it into a hard 4xx failure. Move the quota/429 branch above the "403" substring branch so a 429 with "Quota exceeded for" (or the other quota markers) is classified as RateLimitError regardless of the retry-delay digits. Genuine 403 bodies still fall through to the 403 branch unchanged. Adds regression coverage in test_exception_mapping_utils.py: a 429 quota body whose retry delay contains "403" now maps to RateLimitError(429), a control body without those digits keeps mapping correctly, and a real 403 permission-denied body still maps to BadRequestError(403). --- .../exception_mapping_utils.py | 40 +++++++-------- .../test_exception_mapping_utils.py | 49 +++++++++++++++++++ 2 files changed, 69 insertions(+), 20 deletions(-) diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index fdab3d5b9d4..029e767881a 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -1120,6 +1120,26 @@ def _map_vertex_exception( llm_provider=custom_llm_provider, litellm_debug_info=extra_information, ) + elif ( + "429 Quota exceeded" in error_str + or "Quota exceeded for" in error_str + or "Resource exhausted" in error_str + or "IndexError: list index out of range" in error_str + or "429 Unable to submit request because the service is temporarily out of capacity." in error_str + ): + raise RateLimitError( + message=f"litellm.RateLimitError: {custom_llm_provider}Exception - {error_str}", + model=model, + llm_provider=custom_llm_provider, + litellm_debug_info=extra_information, + response=httpx.Response( + status_code=429, + request=httpx.Request( + method="POST", + url=" https://cloud.google.com/vertex-ai/", + ), + ), + ) elif "403" in error_str: raise BadRequestError( message=f"{custom_llm_provider.capitalize()}Exception BadRequestError - {error_str}", @@ -1151,26 +1171,6 @@ def _map_vertex_exception( ), ), ) - elif ( - "429 Quota exceeded" in error_str - or "Quota exceeded for" in error_str - or "Resource exhausted" in error_str - or "IndexError: list index out of range" in error_str - or "429 Unable to submit request because the service is temporarily out of capacity." in error_str - ): - raise RateLimitError( - message=f"litellm.RateLimitError: {custom_llm_provider}Exception - {error_str}", - model=model, - llm_provider=custom_llm_provider, - litellm_debug_info=extra_information, - response=httpx.Response( - status_code=429, - request=httpx.Request( - method="POST", - url=" https://cloud.google.com/vertex-ai/", - ), - ), - ) elif ( isinstance(getattr(original_exception, "status_code", None), int) and 500 <= original_exception.status_code < 600 diff --git a/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py b/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py index 1fcee1b1c42..bb0457486f3 100644 --- a/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py +++ b/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py @@ -373,6 +373,55 @@ def test_vertex_ai_rate_limit_error_mapping(error_message, should_raise_rate_lim ) +# Regression tests for https://github.com/BerriAI/litellm/issues/34954 +# A Gemini/Vertex 429 whose RESOURCE_EXHAUSTED body carries a sub-second retry +# hint such as "Please retry in 18.403470473s." must still map to RateLimitError. +# The digits "403" in that delay used to hit an unanchored `"403" in error_str` +# branch that was evaluated before the quota branch, yielding BadRequestError(403) +# and disabling Router retries for a transient rate limit. +def _gemini_quota_body(retry_delay: str) -> str: + return ( + '{"error": {"code": 429, "message": "You exceeded your current quota. ' + "Quota exceeded for metric: generativelanguage.googleapis.com/generate_content_free_tier_requests, " + "limit: 15. Please retry in " + retry_delay + '.", "status": "RESOURCE_EXHAUSTED"}}' + ) + + +@pytest.mark.parametrize( + "retry_delay", + [ + "18.403470473s", # digits contain "403" - the bug case + "18.9s", # control: no "403" digits, already mapped correctly + ], +) +def test_gemini_429_quota_maps_to_rate_limit_regardless_of_retry_delay(retry_delay): + original_exception = Exception(_gemini_quota_body(retry_delay)) + + with pytest.raises(litellm.RateLimitError) as excinfo: + exception_type( + model="gemini/gemini-2.5-flash", + original_exception=original_exception, + custom_llm_provider="vertex_ai", + ) + assert excinfo.value.status_code == 429 + + +def test_vertex_genuine_403_still_maps_to_bad_request(): + body = ( + '{"error": {"code": 403, "message": "Permission denied on resource project foo.", ' + '"status": "PERMISSION_DENIED"}}' + ) + original_exception = Exception(body) + + with pytest.raises(litellm.BadRequestError) as excinfo: + exception_type( + model="gemini/gemini-2.5-flash", + original_exception=original_exception, + custom_llm_provider="vertex_ai", + ) + assert excinfo.value.status_code == 403 + + class TestGetBodyErrorCode: """Unit tests for _get_body_error_code helper."""