This commit is contained in:
eeshsaxena 2026-08-27 19:10:14 -05:00 • committed by GitHub
commit de6cb77a69
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 69 additions and 20 deletions

View file

@ -1167,6 +1167,26 @@ def _map_vertex_exception(
llm_provider=custom_llm_provider,
litellm_debug_info=extra_information,
)
elif (
"429 Quota exceeded" in error_str
or "Quota exceeded for" in error_str
or "Resource exhausted" in error_str
or "IndexError: list index out of range" in error_str
or "429 Unable to submit request because the service is temporarily out of capacity." in error_str
):
raise RateLimitError(
message=f"litellm.RateLimitError: {custom_llm_provider}Exception - {error_str}",
model=model,
llm_provider=custom_llm_provider,
litellm_debug_info=extra_information,
response=httpx.Response(
status_code=429,
request=httpx.Request(
method="POST",
url=" https://cloud.google.com/vertex-ai/",
),
),
)
elif "403" in error_str:
raise BadRequestError(
message=f"{custom_llm_provider.capitalize()}Exception BadRequestError - {error_str}",
@ -1198,26 +1218,6 @@ def _map_vertex_exception(
),
),
)
elif (
"429 Quota exceeded" in error_str
or "Quota exceeded for" in error_str
or "Resource exhausted" in error_str
or "IndexError: list index out of range" in error_str
or "429 Unable to submit request because the service is temporarily out of capacity." in error_str
):
raise RateLimitError(
message=f"litellm.RateLimitError: {custom_llm_provider}Exception - {error_str}",
model=model,
llm_provider=custom_llm_provider,
litellm_debug_info=extra_information,
response=httpx.Response(
status_code=429,
request=httpx.Request(
method="POST",
url=" https://cloud.google.com/vertex-ai/",
),
),
)
elif (
isinstance(getattr(original_exception, "status_code", None), int)
and 500 <= original_exception.status_code < 600

View file

@ -452,6 +452,55 @@ def test_vertex_ai_rate_limit_error_mapping(error_message, should_raise_rate_lim
)
# Regression tests for https://github.com/BerriAI/litellm/issues/34954
# A Gemini/Vertex 429 whose RESOURCE_EXHAUSTED body carries a sub-second retry
# hint such as "Please retry in 18.403470473s." must still map to RateLimitError.
# The digits "403" in that delay used to hit an unanchored `"403" in error_str`
# branch that was evaluated before the quota branch, yielding BadRequestError(403)
# and disabling Router retries for a transient rate limit.
def _gemini_quota_body(retry_delay: str) -> str:
return (
'{"error": {"code": 429, "message": "You exceeded your current quota. '
"Quota exceeded for metric: generativelanguage.googleapis.com/generate_content_free_tier_requests, "
"limit: 15. Please retry in " + retry_delay + '.", "status": "RESOURCE_EXHAUSTED"}}'
)
@pytest.mark.parametrize(
"retry_delay",
[
"18.403470473s", # digits contain "403" - the bug case
"18.9s", # control: no "403" digits, already mapped correctly
],
)
def test_gemini_429_quota_maps_to_rate_limit_regardless_of_retry_delay(retry_delay):
original_exception = Exception(_gemini_quota_body(retry_delay))
with pytest.raises(litellm.RateLimitError) as excinfo:
exception_type(
model="gemini/gemini-2.5-flash",
original_exception=original_exception,
custom_llm_provider="vertex_ai",
)
assert excinfo.value.status_code == 429
def test_vertex_genuine_403_still_maps_to_bad_request():
body = (
'{"error": {"code": 403, "message": "Permission denied on resource project foo.", '
'"status": "PERMISSION_DENIED"}}'
)
original_exception = Exception(body)
with pytest.raises(litellm.BadRequestError) as excinfo:
exception_type(
model="gemini/gemini-2.5-flash",
original_exception=original_exception,
custom_llm_provider="vertex_ai",
)
assert excinfo.value.status_code == 403
class TestGetBodyErrorCode:
"""Unit tests for _get_body_error_code helper."""