resolved greptile comments and added docs

This commit is contained in:
shivam 2026-02-05 14:45:22 -08:00
parent f2370bd3ba
commit 9955880e5a
3 changed files with 60 additions and 18 deletions

View file

@ -351,6 +351,8 @@ router_settings:
| assistants_config | AssistantsConfig | Set on proxy via `assistant_settings`. [Further docs](../assistants.md) |
| set_verbose | boolean | [DEPRECATED PARAM - see debug docs](./debugging) If true, sets the logging level to verbose. |
| retry_after | int | Time to wait before retrying a request in seconds. Defaults to 0. If `x-retry-after` is received from LLM API, this value is overridden. |
| guardrail_num_retries | Optional[int] | Default number of retries for failed guardrail calls when a guardrail does not set its own `num_retries`. Guardrails retry on 429, 5xx, timeouts. |
| guardrail_retry_after | Optional[float] | Default minimum seconds to wait before retrying a failed guardrail call when a guardrail does not set its own `retry_after`. |
| provider_budget_config | ProviderBudgetConfig | Provider budget configuration. Use this to set llm_provider budget limits. example $100/day to OpenAI, $100/day to Azure, etc. Defaults to None. [Further Docs](./provider_budget_routing.md) |
| enable_pre_call_checks | boolean | If true, checks if a call is within the model's context window before making the call. [More information here](reliability) |
| model_group_retry_policy | Dict[str, RetryPolicy] | [SDK-only arg] Set retry policy for model groups. |

View file

@ -88,6 +88,32 @@ Need to distribute guardrail requests across multiple accounts or regions? See [
- Weighted distribution across guardrail instances
- Multi-region guardrail deployments
### Retrying failed guardrail calls
Guardrail calls (e.g. to Bedrock, Lakera, Model Armor) are retried on transient failures (429, 5xx, timeouts, network errors), using the same retry rules as the router. You can configure retries per guardrail or set defaults for all guardrails in `router_settings`.
**Per-guardrail** — Add `num_retries` and/or `retry_after` under `litellm_params`:
```yaml
guardrails:
- guardrail_name: "bedrock-guard"
litellm_params:
guardrail: bedrock
mode: pre_call
num_retries: 3 # number of retries (default: 2; use 0 to disable)
retry_after: 1.0 # minimum seconds to wait before retry (used in backoff)
```
**Proxy-level defaults** — Set defaults in `router_settings` so guardrails that don't specify their own values use these:
```yaml
router_settings:
guardrail_num_retries: 3 # default retries for all guardrails
guardrail_retry_after: 1.0 # default minimum wait before retry (seconds)
```
Retries are not attempted for policy violations (e.g. `ModifyResponseException`, `ContentPolicyViolationError`) or non-retriable 4xx (e.g. 404).
## 2. Start LiteLLM Gateway
@ -633,6 +659,8 @@ guardrails:
api_key: string # Required: API key for the guardrail service
api_base: string # Optional: Base URL for the guardrail service
default_on: boolean # Optional: Default False. When set to True, will run on every request, does not need client to specify guardrail in request
num_retries: int # Optional: Number of retries for failed guardrail calls (default: 2; 0 to disable). Retries on 429, 5xx, timeouts.
retry_after: float # Optional: Minimum seconds to wait before retrying (used in backoff).
guardrail_info: # Optional[Dict]: Additional information about the guardrail
```

View file

@ -15,6 +15,28 @@ T = TypeVar("T")
DEFAULT_GUARDRAIL_NUM_RETRIES = 2
DEFAULT_GUARDRAIL_RETRY_AFTER = 0.0
# Cached router settings to avoid importing proxy_server in the request path on every guardrail call.
_CACHED_ROUTER_SETTINGS: Optional[dict] = None
def _get_router_settings() -> Optional[dict]:
"""Resolve router settings once per process; used for guardrail retry defaults."""
global _CACHED_ROUTER_SETTINGS
if _CACHED_ROUTER_SETTINGS is not None:
return _CACHED_ROUTER_SETTINGS
try:
from litellm.proxy.proxy_server import llm_router
if llm_router is not None:
s = llm_router.get_settings()
if s is not None:
_CACHED_ROUTER_SETTINGS = s
return _CACHED_ROUTER_SETTINGS
except Exception:
pass
_CACHED_ROUTER_SETTINGS = {}
return _CACHED_ROUTER_SETTINGS
def should_retry_guardrail_error(error: Exception) -> bool:
"""
@ -82,7 +104,6 @@ async def run_guardrail_with_retries(
coro = coro_factory()
return await coro
last_error: Optional[Exception] = None
attempt = 0
remaining = num_retries
@ -91,7 +112,6 @@ async def run_guardrail_with_retries(
coro = coro_factory()
return await coro
except Exception as e:
last_error = e
if not should_retry_guardrail_error(e):
raise
if remaining <= 0:
@ -117,9 +137,6 @@ async def run_guardrail_with_retries(
remaining -= 1
attempt += 1
if last_error is not None:
raise last_error
def get_guardrail_retry_config(guardrail_to_apply: Any) -> tuple[int, float]:
"""
@ -144,18 +161,13 @@ def get_guardrail_retry_config(guardrail_to_apply: Any) -> tuple[int, float]:
if not (isinstance(opts, dict) and "retry_after" in opts) and "retry_after" in guardrail_config and guardrail_config["retry_after"] is not None:
retry_after = float(guardrail_config["retry_after"])
# Proxy-level defaults from router_settings (when guardrail does not set its own)
try:
from litellm.proxy.proxy_server import llm_router
if llm_router is not None:
s = llm_router.get_settings()
if s is not None:
if num_retries == DEFAULT_GUARDRAIL_NUM_RETRIES and s.get("guardrail_num_retries") is not None:
num_retries = int(s["guardrail_num_retries"])
if retry_after == DEFAULT_GUARDRAIL_RETRY_AFTER and s.get("guardrail_retry_after") is not None:
retry_after = float(s["guardrail_retry_after"])
except Exception:
pass
# Proxy-level defaults from router_settings (when guardrail does not set its own).
# Uses cached settings to avoid importing proxy_server on every guardrail invocation.
s = _get_router_settings()
if s:
if num_retries == DEFAULT_GUARDRAIL_NUM_RETRIES and s.get("guardrail_num_retries") is not None:
num_retries = int(s["guardrail_num_retries"])
if retry_after == DEFAULT_GUARDRAIL_RETRY_AFTER and s.get("guardrail_retry_after") is not None:
retry_after = float(s["guardrail_retry_after"])
return num_retries, retry_after