mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix: map _handle_error exceptions to typed exceptions for Router fallback
Fixes #20507 When `model` and `custom_llm_provider` are passed to `_handle_error()`, exceptions are now mapped through `exception_type()` before raising. This produces typed exceptions (RateLimitError, ContextWindowExceededError, AuthenticationError, etc.) that Router retry/fallback logic can detect via isinstance checks. Without this fix, anthropic_messages pass-through raises BaseLLMException, which Router cannot match for context_window_fallbacks or retry logic. Changes: - Add optional `model` and `custom_llm_provider` params to `_handle_error()` - Map via `exception_type()` when both are present; otherwise raise the raw exception unchanged (backward compatible) - Thread `custom_llm_provider` through `_async_post_anthropic_messages_with_http_error_retry` so both the HTTP-error and generic-exception raise paths produce typed exceptions Tests: 6 new tests covering 429->RateLimitError, 400->ContextWindowExceededError, 401->AuthenticationError, 500->InternalServerError, backward compat, pass-through.
This commit is contained in:
parent
35f6961526
commit
fbf09ec127
2 changed files with 178 additions and 8 deletions
|
|
@ -1864,6 +1864,7 @@ class BaseLLMHTTPHandler:
|
|||
litellm_params: GenericLiteLLMParams,
|
||||
api_key: Optional[str],
|
||||
model: str,
|
||||
custom_llm_provider: str = "",
|
||||
) -> httpx.Response:
|
||||
max_attempts = max(
|
||||
provider_config.max_retry_on_anthropic_messages_http_error, 1
|
||||
|
|
@ -1910,9 +1911,19 @@ class BaseLLMHTTPHandler:
|
|||
)
|
||||
logging_obj.model_call_details.update(request_body)
|
||||
continue
|
||||
raise self._handle_error(e=e, provider_config=provider_config)
|
||||
raise self._handle_error(
|
||||
e=e,
|
||||
provider_config=provider_config,
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
except Exception as e:
|
||||
raise self._handle_error(e=e, provider_config=provider_config)
|
||||
raise self._handle_error(
|
||||
e=e,
|
||||
provider_config=provider_config,
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
raise RuntimeError(
|
||||
"unreachable: anthropic messages HTTP retry loop exited without return"
|
||||
|
|
@ -2069,6 +2080,7 @@ class BaseLLMHTTPHandler:
|
|||
litellm_params=litellm_params,
|
||||
api_key=api_key,
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
# used for logging + cost tracking
|
||||
|
|
@ -5123,6 +5135,8 @@ class BaseLLMHTTPHandler:
|
|||
"BaseContainerConfig",
|
||||
BaseEvalsAPIConfig,
|
||||
],
|
||||
model: str = "",
|
||||
custom_llm_provider: str = "",
|
||||
):
|
||||
status_code = getattr(e, "status_code", 500)
|
||||
error_headers = getattr(e, "headers", None)
|
||||
|
|
@ -5144,17 +5158,33 @@ class BaseLLMHTTPHandler:
|
|||
if provider_config is None:
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
|
||||
raise BaseLLMException(
|
||||
raw_exception: Exception = BaseLLMException(
|
||||
status_code=status_code,
|
||||
message=error_text,
|
||||
headers=error_headers,
|
||||
)
|
||||
else:
|
||||
raw_exception = provider_config.get_error_class(
|
||||
error_message=error_text,
|
||||
status_code=status_code,
|
||||
headers=error_headers,
|
||||
)
|
||||
|
||||
raise provider_config.get_error_class(
|
||||
error_message=error_text,
|
||||
status_code=status_code,
|
||||
headers=error_headers,
|
||||
)
|
||||
# When model and provider are known, map to typed exceptions
|
||||
# (RateLimitError, ContextWindowExceededError, etc.) so Router
|
||||
# retry/fallback logic works correctly.
|
||||
if model and custom_llm_provider:
|
||||
from litellm.litellm_core_utils.exception_mapping_utils import (
|
||||
exception_type,
|
||||
)
|
||||
|
||||
raise exception_type(
|
||||
model=model,
|
||||
original_exception=raw_exception,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
raise raw_exception
|
||||
|
||||
async def async_realtime(
|
||||
self,
|
||||
|
|
|
|||
140
tests/litellm/test_handle_error_typed_exceptions.py
Normal file
140
tests/litellm/test_handle_error_typed_exceptions.py
Normal file
|
|
@ -0,0 +1,140 @@
|
|||
"""Test that _handle_error maps exceptions to typed exceptions when model/provider are known.
|
||||
|
||||
Covers fix for https://github.com/BerriAI/litellm/issues/20507:
|
||||
anthropic_messages pass-through was raising BaseLLMException instead of typed
|
||||
exceptions (RateLimitError, ContextWindowExceededError, etc.), breaking Router
|
||||
retry/fallback logic.
|
||||
"""
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.llms.custom_httpx.llm_http_handler import BaseLLMHTTPHandler
|
||||
|
||||
|
||||
class TestHandleErrorTypedExceptions:
|
||||
"""Verify _handle_error produces typed exceptions when model+provider are supplied."""
|
||||
|
||||
def setup_method(self):
|
||||
self.handler = BaseLLMHTTPHandler()
|
||||
|
||||
def test_rate_limit_error_typed(self):
|
||||
"""429 with model+provider should raise RateLimitError, not BaseLLMException."""
|
||||
mock_response = httpx.Response(
|
||||
status_code=429,
|
||||
request=httpx.Request("POST", "https://api.anthropic.com/v1/messages"),
|
||||
text="rate_limit_error: too many requests",
|
||||
)
|
||||
exc = httpx.HTTPStatusError(
|
||||
message="429 Too Many Requests",
|
||||
request=mock_response.request,
|
||||
response=mock_response,
|
||||
)
|
||||
|
||||
with pytest.raises(litellm.RateLimitError):
|
||||
self.handler._handle_error(
|
||||
e=exc,
|
||||
provider_config=None,
|
||||
model="claude-sonnet-4-6",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
|
||||
def test_context_window_error_typed(self):
|
||||
"""400 with context window message should raise ContextWindowExceededError."""
|
||||
mock_response = httpx.Response(
|
||||
status_code=400,
|
||||
request=httpx.Request("POST", "https://api.anthropic.com/v1/messages"),
|
||||
text='{"error": {"type": "invalid_request_error", "message": "prompt is too long: 200000 tokens > 100000 maximum"}}',
|
||||
)
|
||||
exc = httpx.HTTPStatusError(
|
||||
message="400 Bad Request",
|
||||
request=mock_response.request,
|
||||
response=mock_response,
|
||||
)
|
||||
|
||||
with pytest.raises(litellm.ContextWindowExceededError):
|
||||
self.handler._handle_error(
|
||||
e=exc,
|
||||
provider_config=None,
|
||||
model="claude-sonnet-4-6",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
|
||||
def test_auth_error_typed(self):
|
||||
"""401 should raise AuthenticationError."""
|
||||
mock_response = httpx.Response(
|
||||
status_code=401,
|
||||
request=httpx.Request("POST", "https://api.anthropic.com/v1/messages"),
|
||||
text='{"error": {"type": "authentication_error", "message": "invalid x-api-key"}}',
|
||||
)
|
||||
exc = httpx.HTTPStatusError(
|
||||
message="401 Unauthorized",
|
||||
request=mock_response.request,
|
||||
response=mock_response,
|
||||
)
|
||||
|
||||
with pytest.raises(litellm.AuthenticationError):
|
||||
self.handler._handle_error(
|
||||
e=exc,
|
||||
provider_config=None,
|
||||
model="claude-sonnet-4-6",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
|
||||
def test_server_error_typed(self):
|
||||
"""500 should raise a litellm exception, not BaseLLMException."""
|
||||
mock_response = httpx.Response(
|
||||
status_code=500,
|
||||
request=httpx.Request("POST", "https://api.anthropic.com/v1/messages"),
|
||||
text="Internal Server Error",
|
||||
)
|
||||
exc = httpx.HTTPStatusError(
|
||||
message="500 Internal Server Error",
|
||||
request=mock_response.request,
|
||||
response=mock_response,
|
||||
)
|
||||
|
||||
with pytest.raises(litellm.InternalServerError):
|
||||
self.handler._handle_error(
|
||||
e=exc,
|
||||
provider_config=None,
|
||||
model="claude-sonnet-4-6",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
|
||||
def test_without_model_raises_base_exception(self):
|
||||
"""Without model/provider, should still raise BaseLLMException (backward compat)."""
|
||||
mock_response = httpx.Response(
|
||||
status_code=429,
|
||||
request=httpx.Request("POST", "https://api.anthropic.com/v1/messages"),
|
||||
text="rate limit exceeded",
|
||||
)
|
||||
exc = httpx.HTTPStatusError(
|
||||
message="429 Too Many Requests",
|
||||
request=mock_response.request,
|
||||
response=mock_response,
|
||||
)
|
||||
|
||||
with pytest.raises(BaseLLMException):
|
||||
self.handler._handle_error(
|
||||
e=exc,
|
||||
provider_config=None,
|
||||
)
|
||||
|
||||
def test_already_typed_exception_passes_through(self):
|
||||
"""If the exception is already a litellm type, it should pass through."""
|
||||
typed_exc = litellm.RateLimitError(
|
||||
message="rate limited",
|
||||
model="claude-sonnet-4-6",
|
||||
llm_provider="anthropic",
|
||||
)
|
||||
|
||||
with pytest.raises(litellm.RateLimitError):
|
||||
self.handler._handle_error(
|
||||
e=typed_exc,
|
||||
provider_config=None,
|
||||
model="claude-sonnet-4-6",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
Loading…
Add table
Reference in a new issue