mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-27 01:22:18 +00:00
RouterRateLimitError now carries the model group's deployment ids so it can tell when every deployment is cooled down, and exposes that as type=all_deployments_in_cooldown with an explicit message. A partial cooldown keeps type=rate_limit_error. Either way the proxy no longer reports type=internal_server_error next to code 429 Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
97 lines
3.4 KiB
Python
97 lines
3.4 KiB
Python
from typing import TYPE_CHECKING, Any, Final
|
|
|
|
from litellm._logging import redact_secrets, verbose_router_logger
|
|
from litellm.constants import MAX_EXCEPTION_MESSAGE_LENGTH
|
|
from litellm.router_utils.cooldown_handlers import (
|
|
_async_get_cooldown_deployments_with_debug_info,
|
|
)
|
|
from litellm.types.integrations.slack_alerting import AlertType
|
|
from litellm.types.router import RouterRateLimitError
|
|
|
|
if TYPE_CHECKING:
|
|
from opentelemetry.trace import Span as _Span
|
|
|
|
from litellm.router import Router as _Router
|
|
|
|
LitellmRouter = _Router
|
|
Span = _Span | Any
|
|
else:
|
|
LitellmRouter = Any
|
|
Span = Any
|
|
|
|
|
|
async def send_llm_exception_alert(
|
|
litellm_router_instance: LitellmRouter,
|
|
request_kwargs: dict,
|
|
error_traceback_str: str,
|
|
original_exception,
|
|
):
|
|
"""
|
|
Only runs if router.slack_alerting_logger is set
|
|
Sends a Slack / MS Teams alert for the LLM API call failure. Only if router.slack_alerting_logger is set.
|
|
|
|
Parameters:
|
|
litellm_router_instance (_Router): The LitellmRouter instance.
|
|
original_exception (Any): The original exception that occurred.
|
|
|
|
Returns:
|
|
None
|
|
"""
|
|
if litellm_router_instance is None:
|
|
return
|
|
|
|
if not hasattr(litellm_router_instance, "slack_alerting_logger"):
|
|
return
|
|
|
|
if litellm_router_instance.slack_alerting_logger is None:
|
|
return
|
|
|
|
if "proxy_server_request" in request_kwargs:
|
|
# Do not send any alert if it's a request from litellm proxy server request
|
|
# the proxy is already instrumented to send LLM API call failures
|
|
return
|
|
|
|
litellm_debug_info: Final = getattr(original_exception, "litellm_debug_info", None)
|
|
exception_str = str(original_exception)
|
|
if litellm_debug_info is not None:
|
|
exception_str += litellm_debug_info
|
|
exception_str += f"\n\n{error_traceback_str[:MAX_EXCEPTION_MESSAGE_LENGTH]}"
|
|
|
|
# Redact secrets before sending to external service (Slack / MS Teams)
|
|
exception_str = redact_secrets(exception_str)
|
|
|
|
await litellm_router_instance.slack_alerting_logger.send_alert(
|
|
message=f"LLM API call failed: `{exception_str}`",
|
|
level="High",
|
|
alert_type=AlertType.llm_exceptions,
|
|
alerting_metadata={},
|
|
)
|
|
|
|
|
|
async def async_raise_no_deployment_exception(
|
|
litellm_router_instance: LitellmRouter, model: str, parent_otel_span: Span | None
|
|
):
|
|
"""
|
|
Raises a RouterRateLimitError if no deployment is found for the given model.
|
|
"""
|
|
verbose_router_logger.info("get_available_deployment for model: %s, No deployment available", model)
|
|
model_ids: Final = litellm_router_instance.get_model_ids(model_name=model)
|
|
_cooldown_time: Final = litellm_router_instance.cooldown_cache.get_min_cooldown(
|
|
model_ids=model_ids, parent_otel_span=parent_otel_span
|
|
)
|
|
_cooldown_list: Final = await _async_get_cooldown_deployments_with_debug_info(
|
|
litellm_router_instance=litellm_router_instance,
|
|
parent_otel_span=parent_otel_span,
|
|
)
|
|
verbose_router_logger.info(
|
|
"No deployment found for model: %s, cooldown_list with debug info: %s", model, _cooldown_list
|
|
)
|
|
|
|
cooldown_list_ids: Final = [cooldown_model[0] for cooldown_model in (_cooldown_list or [])]
|
|
return RouterRateLimitError(
|
|
model=model,
|
|
cooldown_time=_cooldown_time,
|
|
enable_pre_call_checks=litellm_router_instance.enable_pre_call_checks,
|
|
cooldown_list=cooldown_list_ids,
|
|
model_ids=model_ids,
|
|
)
|