From 9955880e5adfbb413e3630ae1437107bad778540 Mon Sep 17 00:00:00 2001 From: shivam Date: Thu, 5 Feb 2026 14:45:22 -0800 Subject: [PATCH] resolved greptile comments and added docs --- docs/my-website/docs/proxy/config_settings.md | 2 + .../docs/proxy/guardrails/quick_start.md | 28 +++++++++++ litellm/proxy/guardrails/guardrail_retries.py | 48 ++++++++++++------- 3 files changed, 60 insertions(+), 18 deletions(-) diff --git a/docs/my-website/docs/proxy/config_settings.md b/docs/my-website/docs/proxy/config_settings.md index 5cdae51f448..fce1e9af6fc 100644 --- a/docs/my-website/docs/proxy/config_settings.md +++ b/docs/my-website/docs/proxy/config_settings.md @@ -351,6 +351,8 @@ router_settings: | assistants_config | AssistantsConfig | Set on proxy via `assistant_settings`. [Further docs](../assistants.md) | | set_verbose | boolean | [DEPRECATED PARAM - see debug docs](./debugging) If true, sets the logging level to verbose. | | retry_after | int | Time to wait before retrying a request in seconds. Defaults to 0. If `x-retry-after` is received from LLM API, this value is overridden. | +| guardrail_num_retries | Optional[int] | Default number of retries for failed guardrail calls when a guardrail does not set its own `num_retries`. Guardrails retry on 429, 5xx, timeouts. | +| guardrail_retry_after | Optional[float] | Default minimum seconds to wait before retrying a failed guardrail call when a guardrail does not set its own `retry_after`. | | provider_budget_config | ProviderBudgetConfig | Provider budget configuration. Use this to set llm_provider budget limits. example $100/day to OpenAI, $100/day to Azure, etc. Defaults to None. [Further Docs](./provider_budget_routing.md) | | enable_pre_call_checks | boolean | If true, checks if a call is within the model's context window before making the call. [More information here](reliability) | | model_group_retry_policy | Dict[str, RetryPolicy] | [SDK-only arg] Set retry policy for model groups. | diff --git a/docs/my-website/docs/proxy/guardrails/quick_start.md b/docs/my-website/docs/proxy/guardrails/quick_start.md index ddb215fcb66..d983300a5b3 100644 --- a/docs/my-website/docs/proxy/guardrails/quick_start.md +++ b/docs/my-website/docs/proxy/guardrails/quick_start.md @@ -88,6 +88,32 @@ Need to distribute guardrail requests across multiple accounts or regions? See [ - Weighted distribution across guardrail instances - Multi-region guardrail deployments +### Retrying failed guardrail calls + +Guardrail calls (e.g. to Bedrock, Lakera, Model Armor) are retried on transient failures (429, 5xx, timeouts, network errors), using the same retry rules as the router. You can configure retries per guardrail or set defaults for all guardrails in `router_settings`. + +**Per-guardrail** — Add `num_retries` and/or `retry_after` under `litellm_params`: + +```yaml +guardrails: + - guardrail_name: "bedrock-guard" + litellm_params: + guardrail: bedrock + mode: pre_call + num_retries: 3 # number of retries (default: 2; use 0 to disable) + retry_after: 1.0 # minimum seconds to wait before retry (used in backoff) +``` + +**Proxy-level defaults** — Set defaults in `router_settings` so guardrails that don't specify their own values use these: + +```yaml +router_settings: + guardrail_num_retries: 3 # default retries for all guardrails + guardrail_retry_after: 1.0 # default minimum wait before retry (seconds) +``` + +Retries are not attempted for policy violations (e.g. `ModifyResponseException`, `ContentPolicyViolationError`) or non-retriable 4xx (e.g. 404). + ## 2. Start LiteLLM Gateway @@ -633,6 +659,8 @@ guardrails: api_key: string # Required: API key for the guardrail service api_base: string # Optional: Base URL for the guardrail service default_on: boolean # Optional: Default False. When set to True, will run on every request, does not need client to specify guardrail in request + num_retries: int # Optional: Number of retries for failed guardrail calls (default: 2; 0 to disable). Retries on 429, 5xx, timeouts. + retry_after: float # Optional: Minimum seconds to wait before retrying (used in backoff). guardrail_info: # Optional[Dict]: Additional information about the guardrail ``` diff --git a/litellm/proxy/guardrails/guardrail_retries.py b/litellm/proxy/guardrails/guardrail_retries.py index 253dee6178c..11210f1b56c 100644 --- a/litellm/proxy/guardrails/guardrail_retries.py +++ b/litellm/proxy/guardrails/guardrail_retries.py @@ -15,6 +15,28 @@ T = TypeVar("T") DEFAULT_GUARDRAIL_NUM_RETRIES = 2 DEFAULT_GUARDRAIL_RETRY_AFTER = 0.0 +# Cached router settings to avoid importing proxy_server in the request path on every guardrail call. +_CACHED_ROUTER_SETTINGS: Optional[dict] = None + + +def _get_router_settings() -> Optional[dict]: + """Resolve router settings once per process; used for guardrail retry defaults.""" + global _CACHED_ROUTER_SETTINGS + if _CACHED_ROUTER_SETTINGS is not None: + return _CACHED_ROUTER_SETTINGS + try: + from litellm.proxy.proxy_server import llm_router + + if llm_router is not None: + s = llm_router.get_settings() + if s is not None: + _CACHED_ROUTER_SETTINGS = s + return _CACHED_ROUTER_SETTINGS + except Exception: + pass + _CACHED_ROUTER_SETTINGS = {} + return _CACHED_ROUTER_SETTINGS + def should_retry_guardrail_error(error: Exception) -> bool: """ @@ -82,7 +104,6 @@ async def run_guardrail_with_retries( coro = coro_factory() return await coro - last_error: Optional[Exception] = None attempt = 0 remaining = num_retries @@ -91,7 +112,6 @@ async def run_guardrail_with_retries( coro = coro_factory() return await coro except Exception as e: - last_error = e if not should_retry_guardrail_error(e): raise if remaining <= 0: @@ -117,9 +137,6 @@ async def run_guardrail_with_retries( remaining -= 1 attempt += 1 - if last_error is not None: - raise last_error - def get_guardrail_retry_config(guardrail_to_apply: Any) -> tuple[int, float]: """ @@ -144,18 +161,13 @@ def get_guardrail_retry_config(guardrail_to_apply: Any) -> tuple[int, float]: if not (isinstance(opts, dict) and "retry_after" in opts) and "retry_after" in guardrail_config and guardrail_config["retry_after"] is not None: retry_after = float(guardrail_config["retry_after"]) - # Proxy-level defaults from router_settings (when guardrail does not set its own) - try: - from litellm.proxy.proxy_server import llm_router - - if llm_router is not None: - s = llm_router.get_settings() - if s is not None: - if num_retries == DEFAULT_GUARDRAIL_NUM_RETRIES and s.get("guardrail_num_retries") is not None: - num_retries = int(s["guardrail_num_retries"]) - if retry_after == DEFAULT_GUARDRAIL_RETRY_AFTER and s.get("guardrail_retry_after") is not None: - retry_after = float(s["guardrail_retry_after"]) - except Exception: - pass + # Proxy-level defaults from router_settings (when guardrail does not set its own). + # Uses cached settings to avoid importing proxy_server on every guardrail invocation. + s = _get_router_settings() + if s: + if num_retries == DEFAULT_GUARDRAIL_NUM_RETRIES and s.get("guardrail_num_retries") is not None: + num_retries = int(s["guardrail_num_retries"]) + if retry_after == DEFAULT_GUARDRAIL_RETRY_AFTER and s.get("guardrail_retry_after") is not None: + retry_after = float(s["guardrail_retry_after"]) return num_retries, retry_after