mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
resolved greptile comments and added docs
This commit is contained in:
parent
f2370bd3ba
commit
9955880e5a
3 changed files with 60 additions and 18 deletions
|
|
@ -351,6 +351,8 @@ router_settings:
|
|||
| assistants_config | AssistantsConfig | Set on proxy via `assistant_settings`. [Further docs](../assistants.md) |
|
||||
| set_verbose | boolean | [DEPRECATED PARAM - see debug docs](./debugging) If true, sets the logging level to verbose. |
|
||||
| retry_after | int | Time to wait before retrying a request in seconds. Defaults to 0. If `x-retry-after` is received from LLM API, this value is overridden. |
|
||||
| guardrail_num_retries | Optional[int] | Default number of retries for failed guardrail calls when a guardrail does not set its own `num_retries`. Guardrails retry on 429, 5xx, timeouts. |
|
||||
| guardrail_retry_after | Optional[float] | Default minimum seconds to wait before retrying a failed guardrail call when a guardrail does not set its own `retry_after`. |
|
||||
| provider_budget_config | ProviderBudgetConfig | Provider budget configuration. Use this to set llm_provider budget limits. example $100/day to OpenAI, $100/day to Azure, etc. Defaults to None. [Further Docs](./provider_budget_routing.md) |
|
||||
| enable_pre_call_checks | boolean | If true, checks if a call is within the model's context window before making the call. [More information here](reliability) |
|
||||
| model_group_retry_policy | Dict[str, RetryPolicy] | [SDK-only arg] Set retry policy for model groups. |
|
||||
|
|
|
|||
|
|
@ -88,6 +88,32 @@ Need to distribute guardrail requests across multiple accounts or regions? See [
|
|||
- Weighted distribution across guardrail instances
|
||||
- Multi-region guardrail deployments
|
||||
|
||||
### Retrying failed guardrail calls
|
||||
|
||||
Guardrail calls (e.g. to Bedrock, Lakera, Model Armor) are retried on transient failures (429, 5xx, timeouts, network errors), using the same retry rules as the router. You can configure retries per guardrail or set defaults for all guardrails in `router_settings`.
|
||||
|
||||
**Per-guardrail** — Add `num_retries` and/or `retry_after` under `litellm_params`:
|
||||
|
||||
```yaml
|
||||
guardrails:
|
||||
- guardrail_name: "bedrock-guard"
|
||||
litellm_params:
|
||||
guardrail: bedrock
|
||||
mode: pre_call
|
||||
num_retries: 3 # number of retries (default: 2; use 0 to disable)
|
||||
retry_after: 1.0 # minimum seconds to wait before retry (used in backoff)
|
||||
```
|
||||
|
||||
**Proxy-level defaults** — Set defaults in `router_settings` so guardrails that don't specify their own values use these:
|
||||
|
||||
```yaml
|
||||
router_settings:
|
||||
guardrail_num_retries: 3 # default retries for all guardrails
|
||||
guardrail_retry_after: 1.0 # default minimum wait before retry (seconds)
|
||||
```
|
||||
|
||||
Retries are not attempted for policy violations (e.g. `ModifyResponseException`, `ContentPolicyViolationError`) or non-retriable 4xx (e.g. 404).
|
||||
|
||||
|
||||
## 2. Start LiteLLM Gateway
|
||||
|
||||
|
|
@ -633,6 +659,8 @@ guardrails:
|
|||
api_key: string # Required: API key for the guardrail service
|
||||
api_base: string # Optional: Base URL for the guardrail service
|
||||
default_on: boolean # Optional: Default False. When set to True, will run on every request, does not need client to specify guardrail in request
|
||||
num_retries: int # Optional: Number of retries for failed guardrail calls (default: 2; 0 to disable). Retries on 429, 5xx, timeouts.
|
||||
retry_after: float # Optional: Minimum seconds to wait before retrying (used in backoff).
|
||||
guardrail_info: # Optional[Dict]: Additional information about the guardrail
|
||||
|
||||
```
|
||||
|
|
|
|||
|
|
@ -15,6 +15,28 @@ T = TypeVar("T")
|
|||
DEFAULT_GUARDRAIL_NUM_RETRIES = 2
|
||||
DEFAULT_GUARDRAIL_RETRY_AFTER = 0.0
|
||||
|
||||
# Cached router settings to avoid importing proxy_server in the request path on every guardrail call.
|
||||
_CACHED_ROUTER_SETTINGS: Optional[dict] = None
|
||||
|
||||
|
||||
def _get_router_settings() -> Optional[dict]:
|
||||
"""Resolve router settings once per process; used for guardrail retry defaults."""
|
||||
global _CACHED_ROUTER_SETTINGS
|
||||
if _CACHED_ROUTER_SETTINGS is not None:
|
||||
return _CACHED_ROUTER_SETTINGS
|
||||
try:
|
||||
from litellm.proxy.proxy_server import llm_router
|
||||
|
||||
if llm_router is not None:
|
||||
s = llm_router.get_settings()
|
||||
if s is not None:
|
||||
_CACHED_ROUTER_SETTINGS = s
|
||||
return _CACHED_ROUTER_SETTINGS
|
||||
except Exception:
|
||||
pass
|
||||
_CACHED_ROUTER_SETTINGS = {}
|
||||
return _CACHED_ROUTER_SETTINGS
|
||||
|
||||
|
||||
def should_retry_guardrail_error(error: Exception) -> bool:
|
||||
"""
|
||||
|
|
@ -82,7 +104,6 @@ async def run_guardrail_with_retries(
|
|||
coro = coro_factory()
|
||||
return await coro
|
||||
|
||||
last_error: Optional[Exception] = None
|
||||
attempt = 0
|
||||
remaining = num_retries
|
||||
|
||||
|
|
@ -91,7 +112,6 @@ async def run_guardrail_with_retries(
|
|||
coro = coro_factory()
|
||||
return await coro
|
||||
except Exception as e:
|
||||
last_error = e
|
||||
if not should_retry_guardrail_error(e):
|
||||
raise
|
||||
if remaining <= 0:
|
||||
|
|
@ -117,9 +137,6 @@ async def run_guardrail_with_retries(
|
|||
remaining -= 1
|
||||
attempt += 1
|
||||
|
||||
if last_error is not None:
|
||||
raise last_error
|
||||
|
||||
|
||||
def get_guardrail_retry_config(guardrail_to_apply: Any) -> tuple[int, float]:
|
||||
"""
|
||||
|
|
@ -144,18 +161,13 @@ def get_guardrail_retry_config(guardrail_to_apply: Any) -> tuple[int, float]:
|
|||
if not (isinstance(opts, dict) and "retry_after" in opts) and "retry_after" in guardrail_config and guardrail_config["retry_after"] is not None:
|
||||
retry_after = float(guardrail_config["retry_after"])
|
||||
|
||||
# Proxy-level defaults from router_settings (when guardrail does not set its own)
|
||||
try:
|
||||
from litellm.proxy.proxy_server import llm_router
|
||||
|
||||
if llm_router is not None:
|
||||
s = llm_router.get_settings()
|
||||
if s is not None:
|
||||
if num_retries == DEFAULT_GUARDRAIL_NUM_RETRIES and s.get("guardrail_num_retries") is not None:
|
||||
num_retries = int(s["guardrail_num_retries"])
|
||||
if retry_after == DEFAULT_GUARDRAIL_RETRY_AFTER and s.get("guardrail_retry_after") is not None:
|
||||
retry_after = float(s["guardrail_retry_after"])
|
||||
except Exception:
|
||||
pass
|
||||
# Proxy-level defaults from router_settings (when guardrail does not set its own).
|
||||
# Uses cached settings to avoid importing proxy_server on every guardrail invocation.
|
||||
s = _get_router_settings()
|
||||
if s:
|
||||
if num_retries == DEFAULT_GUARDRAIL_NUM_RETRIES and s.get("guardrail_num_retries") is not None:
|
||||
num_retries = int(s["guardrail_num_retries"])
|
||||
if retry_after == DEFAULT_GUARDRAIL_RETRY_AFTER and s.get("guardrail_retry_after") is not None:
|
||||
retry_after = float(s["guardrail_retry_after"])
|
||||
|
||||
return num_retries, retry_after
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue