fix: map _handle_error exceptions to typed exceptions for Router fallback

Fixes #20507

When `model` and `custom_llm_provider` are passed to `_handle_error()`,
exceptions are now mapped through `exception_type()` before raising.
This produces typed exceptions (RateLimitError, ContextWindowExceededError,
AuthenticationError, etc.) that Router retry/fallback logic can detect
via isinstance checks.

Without this fix, anthropic_messages pass-through raises BaseLLMException,
which Router cannot match for context_window_fallbacks or retry logic.

Changes:
- Add optional `model` and `custom_llm_provider` params to `_handle_error()`
- Map via `exception_type()` when both are present; otherwise raise the raw
  exception unchanged (backward compatible)
- Thread `custom_llm_provider` through
  `_async_post_anthropic_messages_with_http_error_retry` so both the
  HTTP-error and generic-exception raise paths produce typed exceptions

Tests: 6 new tests covering 429->RateLimitError, 400->ContextWindowExceededError,
401->AuthenticationError, 500->InternalServerError, backward compat, pass-through.
This commit is contained in:
Cayman Roden 2026-05-31 19:37:32 -07:00
parent 35f6961526
commit fbf09ec127
2 changed files with 178 additions and 8 deletions

View file

@ -1864,6 +1864,7 @@ class BaseLLMHTTPHandler:
litellm_params: GenericLiteLLMParams,
api_key: Optional[str],
model: str,
custom_llm_provider: str = "",
) -> httpx.Response:
max_attempts = max(
provider_config.max_retry_on_anthropic_messages_http_error, 1
@ -1910,9 +1911,19 @@ class BaseLLMHTTPHandler:
)
logging_obj.model_call_details.update(request_body)
continue
raise self._handle_error(e=e, provider_config=provider_config)
raise self._handle_error(
e=e,
provider_config=provider_config,
model=model,
custom_llm_provider=custom_llm_provider,
)
except Exception as e:
raise self._handle_error(e=e, provider_config=provider_config)
raise self._handle_error(
e=e,
provider_config=provider_config,
model=model,
custom_llm_provider=custom_llm_provider,
)
raise RuntimeError(
"unreachable: anthropic messages HTTP retry loop exited without return"
@ -2069,6 +2080,7 @@ class BaseLLMHTTPHandler:
litellm_params=litellm_params,
api_key=api_key,
model=model,
custom_llm_provider=custom_llm_provider,
)
# used for logging + cost tracking
@ -5123,6 +5135,8 @@ class BaseLLMHTTPHandler:
"BaseContainerConfig",
BaseEvalsAPIConfig,
],
model: str = "",
custom_llm_provider: str = "",
):
status_code = getattr(e, "status_code", 500)
error_headers = getattr(e, "headers", None)
@ -5144,17 +5158,33 @@ class BaseLLMHTTPHandler:
if provider_config is None:
from litellm.llms.base_llm.chat.transformation import BaseLLMException
raise BaseLLMException(
raw_exception: Exception = BaseLLMException(
status_code=status_code,
message=error_text,
headers=error_headers,
)
else:
raw_exception = provider_config.get_error_class(
error_message=error_text,
status_code=status_code,
headers=error_headers,
)
raise provider_config.get_error_class(
error_message=error_text,
status_code=status_code,
headers=error_headers,
)
# When model and provider are known, map to typed exceptions
# (RateLimitError, ContextWindowExceededError, etc.) so Router
# retry/fallback logic works correctly.
if model and custom_llm_provider:
from litellm.litellm_core_utils.exception_mapping_utils import (
exception_type,
)
raise exception_type(
model=model,
original_exception=raw_exception,
custom_llm_provider=custom_llm_provider,
)
raise raw_exception
async def async_realtime(
self,

View file

@ -0,0 +1,140 @@
"""Test that _handle_error maps exceptions to typed exceptions when model/provider are known.
Covers fix for https://github.com/BerriAI/litellm/issues/20507:
anthropic_messages pass-through was raising BaseLLMException instead of typed
exceptions (RateLimitError, ContextWindowExceededError, etc.), breaking Router
retry/fallback logic.
"""
import httpx
import pytest
import litellm
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
class TestHandleErrorTypedExceptions:
"""Verify _handle_error produces typed exceptions when model+provider are supplied."""
def setup_method(self):
self.handler = BaseLLMHTTPHandler()
def test_rate_limit_error_typed(self):
"""429 with model+provider should raise RateLimitError, not BaseLLMException."""
mock_response = httpx.Response(
status_code=429,
request=httpx.Request("POST", "https://api.anthropic.com/v1/messages"),
text="rate_limit_error: too many requests",
)
exc = httpx.HTTPStatusError(
message="429 Too Many Requests",
request=mock_response.request,
response=mock_response,
)
with pytest.raises(litellm.RateLimitError):
self.handler._handle_error(
e=exc,
provider_config=None,
model="claude-sonnet-4-6",
custom_llm_provider="anthropic",
)
def test_context_window_error_typed(self):
"""400 with context window message should raise ContextWindowExceededError."""
mock_response = httpx.Response(
status_code=400,
request=httpx.Request("POST", "https://api.anthropic.com/v1/messages"),
text='{"error": {"type": "invalid_request_error", "message": "prompt is too long: 200000 tokens > 100000 maximum"}}',
)
exc = httpx.HTTPStatusError(
message="400 Bad Request",
request=mock_response.request,
response=mock_response,
)
with pytest.raises(litellm.ContextWindowExceededError):
self.handler._handle_error(
e=exc,
provider_config=None,
model="claude-sonnet-4-6",
custom_llm_provider="anthropic",
)
def test_auth_error_typed(self):
"""401 should raise AuthenticationError."""
mock_response = httpx.Response(
status_code=401,
request=httpx.Request("POST", "https://api.anthropic.com/v1/messages"),
text='{"error": {"type": "authentication_error", "message": "invalid x-api-key"}}',
)
exc = httpx.HTTPStatusError(
message="401 Unauthorized",
request=mock_response.request,
response=mock_response,
)
with pytest.raises(litellm.AuthenticationError):
self.handler._handle_error(
e=exc,
provider_config=None,
model="claude-sonnet-4-6",
custom_llm_provider="anthropic",
)
def test_server_error_typed(self):
"""500 should raise a litellm exception, not BaseLLMException."""
mock_response = httpx.Response(
status_code=500,
request=httpx.Request("POST", "https://api.anthropic.com/v1/messages"),
text="Internal Server Error",
)
exc = httpx.HTTPStatusError(
message="500 Internal Server Error",
request=mock_response.request,
response=mock_response,
)
with pytest.raises(litellm.InternalServerError):
self.handler._handle_error(
e=exc,
provider_config=None,
model="claude-sonnet-4-6",
custom_llm_provider="anthropic",
)
def test_without_model_raises_base_exception(self):
"""Without model/provider, should still raise BaseLLMException (backward compat)."""
mock_response = httpx.Response(
status_code=429,
request=httpx.Request("POST", "https://api.anthropic.com/v1/messages"),
text="rate limit exceeded",
)
exc = httpx.HTTPStatusError(
message="429 Too Many Requests",
request=mock_response.request,
response=mock_response,
)
with pytest.raises(BaseLLMException):
self.handler._handle_error(
e=exc,
provider_config=None,
)
def test_already_typed_exception_passes_through(self):
"""If the exception is already a litellm type, it should pass through."""
typed_exc = litellm.RateLimitError(
message="rate limited",
model="claude-sonnet-4-6",
llm_provider="anthropic",
)
with pytest.raises(litellm.RateLimitError):
self.handler._handle_error(
e=typed_exc,
provider_config=None,
model="claude-sonnet-4-6",
custom_llm_provider="anthropic",
)