fix(exception-mapping): map Gemini upstream-error body code 429 to RateLimitError (#30417)

* fix(exception-mapping): map Gemini upstream-error body code 429 to RateLimitError

Some Gemini-compatible gateways (e.g. new-api) wrap a 429 rate-limit
signal from upstream inside an HTTP 500/503 envelope, with the real
code only surfaced in the JSON body:

    {"error":{"message":"...high demand...","type":"upstream_error",
              "param":"","code":429}}

Previously LiteLLM only looked at the HTTP status and mapped this to
InternalServerError, which Router treats as non-retryable for many
configs — so users got hard 500s instead of fallback/retry.

Now the Gemini/Vertex exception mapper parses error.code from the body
and routes code 429 to RateLimitError before falling through to the
HTTP-status branches. Other body codes fall through unchanged.

Tests cover:
- new-api gateway's `code:429` payload now maps to RateLimitError
- Genuine 500-body responses stay InternalServerError
- Non-JSON body strings fall through to status-code mapping unchanged

* fix(exception-mapping): scope body-code 429 promotion to 5xx envelopes

Addresses greptile P1/P2 + @Sameerlite's review on #30417. The new
elif branch was firing for any HTTP status, so a gateway response of
HTTP 400 with body {"error":{"code":429,...}} would be incorrectly
promoted to RateLimitError (retryable) instead of falling through
to BadRequestError. Same trap for 401 -> AuthenticationError.

Scoped the body-code 429 check to `500 <= status_code < 600` —
covers 500/502/503/504 (gateways wrapping upstream 429 in any 5xx
envelope) without inviting the 4xx misclassification.

Tests: parametrized table now covers 5xx (500/502/503), 4xx (400/401),
and the existing fall-through cases, asserting each maps to the
exception type that matches the HTTP status code. 50/50 pass locally.

* ci: retrigger workflows after base branch change to litellm_internal_staging
This commit is contained in:
songkuan-zheng 2026-06-17 19:33:59 +08:00 • committed by GitHub
parent 847a8ace3d
commit dbe1028804
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 150 additions and 0 deletions

View file

@ -170,6 +170,16 @@ def get_error_message(error_obj) -> Optional[str]:
####### EXCEPTION MAPPING ################
def _get_body_error_code(error_str: str) -> Optional[int]:
"""Return error.code from a JSON error body, or None if not parseable."""
try:
body = json.loads(error_str)
code = body.get("error", {}).get("code")
return int(code) if code is not None else None
except Exception:
return None
def _get_response_headers(original_exception: Exception) -> Optional[httpx.Headers]:
"""
Extract and return the response headers from an exception, if present.
@ -1415,6 +1425,29 @@ def exception_type( # type: ignore
),
),
)
elif (
isinstance(getattr(original_exception, "status_code", None), int)
and 500 <= original_exception.status_code < 600
and _get_body_error_code(error_str) == 429
):
# upstream gateway wraps a 429 inside a 5xx envelope
# e.g. HTTP 500/503 with {"error":{"code":429,...}}.
# Scoped to 5xx so HTTP 400/401 with body code:429
# still maps to BadRequestError / AuthenticationError.
exception_mapping_worked = True
raise RateLimitError(
message=f"litellm.RateLimitError: {custom_llm_provider}Exception - {error_str}",
model=model,
llm_provider=custom_llm_provider,
litellm_debug_info=extra_information,
response=httpx.Response(
status_code=429,
request=httpx.Request(
method="POST",
url=" https://cloud.google.com/vertex-ai/",
),
),
)
elif (
"500 Internal Server Error" in error_str
or "The model is overloaded." in error_str

View file

@ -11,6 +11,7 @@ sys.path.insert(
from litellm.litellm_core_utils.exception_mapping_utils import (
ExceptionCheckers,
_get_body_error_code,
exception_type,
extract_and_raise_litellm_exception,
)
@ -359,6 +360,122 @@ def test_vertex_ai_rate_limit_error_mapping(error_message, should_raise_rate_lim
)
class TestGetBodyErrorCode:
"""Unit tests for _get_body_error_code helper."""
def test_parses_int_code(self):
body = (
'{"error":{"message":"high demand","type":"upstream_error",'
'"param":"","code":429}}'
)
assert _get_body_error_code(body) == 429
def test_parses_string_code(self):
# some gateways serialize code as a string
body = '{"error":{"message":"x","code":"503"}}'
assert _get_body_error_code(body) == 503
def test_returns_none_on_non_json(self):
assert _get_body_error_code("not json") is None
def test_returns_none_when_no_error_key(self):
assert _get_body_error_code('{"ok":true}') is None
def test_returns_none_when_no_code_key(self):
assert _get_body_error_code('{"error":{"message":"x"}}') is None
# Test cases for Gemini upstream-error body-code mapping.
#
# Body code 429 wrapped in a 5xx HTTP envelope (e.g. new-api gateways)
# must map to RateLimitError so Router retries kick in. A 4xx HTTP
# envelope with body code:429 must NOT — it falls through to whatever
# the HTTP status code maps to (BadRequestError, AuthenticationError,
# etc.), matching upstream's existing semantics.
gemini_body_code_429_test_cases = [
# (status_code, error_body, expected_exception_type, description)
(
500,
'{"error":{"message":" This model is currently experiencing high demand.'
" Spikes in demand are usually temporary. Please try again later."
' (request id: x)","type":"upstream_error","param":"","code":429}}',
litellm.RateLimitError,
"HTTP 500 envelope with body code:429 -> RateLimitError",
),
(
503,
'{"error":{"message":"upstream unavailable","type":"upstream_error",'
'"param":"","code":429}}',
litellm.RateLimitError,
"HTTP 503 envelope with body code:429 -> RateLimitError",
),
(
502,
'{"error":{"message":"bad gateway","code":429}}',
litellm.RateLimitError,
"HTTP 502 envelope with body code:429 -> RateLimitError",
),
(
500,
'{"error":{"message":"server boom","code":500}}',
litellm.InternalServerError,
"HTTP 500 with body code:500 stays InternalServerError",
),
(
500,
"plain text 500 error",
litellm.InternalServerError,
"HTTP 500 with non-JSON body falls through to status_code mapping",
),
(
400,
'{"error":{"message":"malformed","code":429}}',
litellm.BadRequestError,
"HTTP 400 with body code:429 must NOT be promoted to RateLimitError",
),
(
401,
'{"error":{"message":"bad key","code":429}}',
litellm.AuthenticationError,
"HTTP 401 with body code:429 must NOT be promoted to RateLimitError",
),
]
@pytest.mark.parametrize(
"status_code, error_body, expected_exception, description",
gemini_body_code_429_test_cases,
)
def test_gemini_upstream_error_body_code_429_maps_to_rate_limit(
status_code, error_body, expected_exception, description
):
"""
Body code 429 inside a 5xx envelope -> RateLimitError so Router
retries kick in. Body code 429 inside a 4xx envelope must fall
through to the HTTP-status-code branch (P1 from greptile review).
"""
model = "gemini/gemini-2.5-flash"
custom_llm_provider = "gemini"
# Build an exception that looks like what _handle_error produces:
# a BaseLLMException-style object with .status_code and .message
class _FakeGeminiError(Exception):
def __init__(self, status_code, message):
self.status_code = status_code
self.message = message
super().__init__(message)
original_exception = _FakeGeminiError(status_code=status_code, message=error_body)
with pytest.raises(expected_exception) as excinfo:
exception_type(
model=model,
original_exception=original_exception,
custom_llm_provider=custom_llm_provider,
)
assert isinstance(excinfo.value, expected_exception), description
class TestExtractAndRaiseLitellmException:
"""Tests for extract_and_raise_litellm_exception function"""