diff --git a/litellm/litellm_core_utils/exception_mapping_utils.py b/litellm/litellm_core_utils/exception_mapping_utils.py index 82708d412c9..eb54cfc8903 100644 --- a/litellm/litellm_core_utils/exception_mapping_utils.py +++ b/litellm/litellm_core_utils/exception_mapping_utils.py @@ -1114,6 +1114,13 @@ def _map_sagemaker_exception( ) +def _is_vertex_403_error(*, original_exception: _ProviderHTTPException, error_str: str) -> bool: + status_code = getattr(original_exception, "status_code", None) + if isinstance(status_code, int): + return status_code == 403 + return re.search(r"\b403\b", error_str) is not None + + def _map_vertex_exception( *, model: str, @@ -1170,7 +1177,7 @@ def _map_vertex_exception( llm_provider=custom_llm_provider, litellm_debug_info=extra_information, ) - elif "403" in error_str: + elif _is_vertex_403_error(original_exception=original_exception, error_str=error_str): raise BadRequestError( message=f"{custom_llm_provider.capitalize()}Exception BadRequestError - {error_str}", model=model, @@ -1204,7 +1211,7 @@ def _map_vertex_exception( elif ( "429 Quota exceeded" in error_str or "Quota exceeded for" in error_str - or "Resource exhausted" in error_str + or re.search(r"resource[\s_]exhausted", error_str, re.IGNORECASE) is not None or "IndexError: list index out of range" in error_str or "429 Unable to submit request because the service is temporarily out of capacity." in error_str ): diff --git a/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py b/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py index 42d3df76902..74d4e47d3e0 100644 --- a/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py +++ b/tests/test_litellm/litellm_core_utils/test_exception_mapping_utils.py @@ -570,6 +570,85 @@ def test_gemini_upstream_error_body_code_429_maps_to_rate_limit( assert isinstance(excinfo.value, expected_exception), description +def _quota_exceeded_body(retry_delay: str) -> str: + return ( + '{"error": {"code": 429, "message": "You exceeded your current quota. ' + "Quota exceeded for metric: generativelanguage.googleapis.com/generate_content_free_tier_requests. " + f'Please retry in {retry_delay}s.", "status": "RESOURCE_EXHAUSTED"}}}}' + ) + + +# https://github.com/BerriAI/litellm/issues/34954 - "403" appearing anywhere in a 429 body +# (e.g. inside the sub-second retry hint) must not divert the mapping away from RateLimitError +vertex_403_collision_test_cases = [ + (429, _quota_exceeded_body("18.403470473"), litellm.RateLimitError, 429), + (429, _quota_exceeded_body("18.9"), litellm.RateLimitError, 429), + ( + 429, + '{"error": {"code": 429, "message": "Resource has been exhausted (e.g. check quota).",' + ' "status": "RESOURCE_EXHAUSTED"}}', + litellm.RateLimitError, + 429, + ), + ( + 403, + '{"error": {"code": 403, "message": "Permission denied on resource project foo.",' + ' "status": "PERMISSION_DENIED"}}', + litellm.BadRequestError, + 403, + ), + ( + 403, + '{"error": {"code": 403, "message": "Quota exceeded for metric: some-metric",' + ' "status": "PERMISSION_DENIED"}}', + litellm.BadRequestError, + 403, + ), +] + + +@pytest.mark.parametrize( + "status_code, error_body, expected_exception, expected_status_code", + vertex_403_collision_test_cases, +) +def test_vertex_403_substring_does_not_shadow_rate_limit( + status_code, error_body, expected_exception, expected_status_code +): + class _FakeVertexError(Exception): + def __init__(self, status_code, message): + self.status_code = status_code + self.message = message + super().__init__(message) + + with pytest.raises(expected_exception) as excinfo: + exception_type( + model="gemini/gemini-2.5-flash", + original_exception=_FakeVertexError(status_code=status_code, message=error_body), + custom_llm_provider="gemini", + ) + + assert excinfo.value.status_code == expected_status_code + assert litellm._should_retry(excinfo.value.status_code) is (expected_status_code == 429) + + +@pytest.mark.parametrize( + "error_message, expected_exception", + [ + ("403 Permission denied on resource project foo.", litellm.BadRequestError), + ("429 Quota exceeded for metric: foo. Please retry in 18.403470473s.", litellm.RateLimitError), + ("429 RESOURCE_EXHAUSTED. Please retry in 1.403s.", litellm.RateLimitError), + ], +) +def test_vertex_status_less_exception_403_mapping(error_message, expected_exception): + """google's SDK raises status-less errors whose text starts with the HTTP code.""" + with pytest.raises(expected_exception): + exception_type( + model="gemini/gemini-2.5-flash", + original_exception=Exception(error_message), + custom_llm_provider="vertex_ai", + ) + + class TestExtractAndRaiseLitellmException: """Tests for extract_and_raise_litellm_exception function"""