mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-11 03:38:38 +00:00
feat(router): add expose_router_debug_in_errors flag (default True) to redact internal model_group/fallback names (#30418)
* feat(router)!: redact internal model_group/fallback names from exception messages
The Router was unconditionally appending internal config names onto
exception.message:
- "Received Model Group=..."
- "Available Model Group Fallbacks=..."
- "No fallback model group found... Fallbacks={...}"
- "context_window_fallbacks={...}"
- Deployment-timeout messages including model_group
- Fallback failure detail listing fallback chain
ProxyException forwards .message verbatim to clients, so gateways were
leaking their model_name / fallback wiring in every failed call.
Fix: gate all five mutation sites on a new
`litellm.expose_router_debug_in_errors` flag (default False). Set to
True to restore upstream debug behavior for local debugging.
Why: matches the redaction posture this codebase already has for
upstream model identifiers (cf. _litellm_returned_model_name) and
removes the last common error-path leak of internal model_group names.
Breaking change marker (!): if anything parses "Received Model Group="
out of client error messages, flip the flag on or migrate to the
x-litellm-* response headers instead.
Tests: 7 cases covering each of the 5 redaction sites + the flag-on
inverse path, plus a "default off" sanity check.
* test(router): cover sites 1 + 3 of expose_router_debug_in_errors gate
Addresses Greptile / codecov feedback on #30418: patch coverage was
55.6% with 4 lines uncovered in litellm/router.py. The existing tests
exercised sites 2 (ContextWindowExceededError), 4 (no-fallback-found),
and 5 (Received Model Group) — both default and flag-on. Sites 1 and 3
were declared in the PR description as covered by "site 5 also fires"
but the gate body lines for each (the `e.message +=` inside the
`if litellm.expose_router_debug_in_errors:` branch) only execute when
the flag is on AND the specific exception path is taken, which neither
existing test triggered.
Added 4 new tests (default + flag-on × 2 sites):
- test_default_does_not_leak_deployment_timeout_debug
- test_flag_on_leaks_deployment_timeout_debug
- test_default_does_not_leak_content_policy_fallback_hint
- test_flag_on_leaks_content_policy_fallback_hint
Trigger details:
- Site 1 (litellm.Timeout in _acompletion) is reached via the
Router-supported `mock_timeout=True` + `timeout=0.001` kwargs on
`acompletion(...)`. Cannot embed a Timeout instance in model_list
because Router.__init__ deep-copies it and Timeout.__reduce__ does
not preserve the required positional args.
- Site 3 (ContentPolicyViolationError without content_policy_fallbacks
set, in async_function_with_fallbacks_common_utils) is reached by
passing a `mock_response=litellm.ContentPolicyViolationError(...)`
instance via the call-site kwarg — same deepcopy-avoidance reason.
11/11 tests pass locally. Patch coverage on litellm/router.py for this
PR's diff should now be 100%.
* chore(router): flip expose_router_debug_in_errors default to True
Addresses @Sameerlite's review on #30418 — maintain backward
compat on the wire. Redact becomes opt-in via setting the flag
to False; the historical behavior (leak internal model_group /
fallback wiring through exception messages) is preserved as the
default.
- litellm/__init__.py: default flipped to True, docstring rewritten
with deprecation note pointing at a future flip to False (redact
by default) in a major release.
- tests/test_litellm/test_router_exception_redaction.py: fixture
resets to True (was False); the "off" tests now explicitly set
False; the "default_leaks_*" tests rely on the fixture default.
test_flag_defaults_off -> test_flag_defaults_on.
- No router.py change needed; the gate keys off the same flag,
only the default changes.
- PR title no longer needs the breaking-change `!` marker — no
client sees a behavior change at default settings.
11/11 pass locally.
* ci: retrigger workflows after base branch change to litellm_internal_staging
This commit is contained in:
parent
dbe1028804
commit
47771aa369
3 changed files with 334 additions and 5 deletions
|
|
@ -213,6 +213,15 @@ standard_logging_payload_excluded_fields: Optional[List[str]] = (
|
|||
log_raw_request_response: bool = False
|
||||
redact_messages_in_exceptions: Optional[bool] = False
|
||||
redact_user_api_key_info: Optional[bool] = False
|
||||
# When True (default — preserves historical behavior), the Router appends
|
||||
# internal config names (model_group, fallback model groups, deployment
|
||||
# timeouts, fallback failure details) onto exception messages and surfaces
|
||||
# them to clients via ProxyException.message. Set to False if you do NOT
|
||||
# want the proxy's internal model_name / fallback wiring visible to clients.
|
||||
# Deprecation: planned to flip to False (redact by default) in a future
|
||||
# major release; opt in early with `litellm.expose_router_debug_in_errors
|
||||
# = False`.
|
||||
expose_router_debug_in_errors: bool = True
|
||||
filter_invalid_headers: Optional[bool] = False
|
||||
add_user_information_to_llm_headers: Optional[bool] = (
|
||||
None # adds user_id, team_id, token hash (params from StandardLoggingMetadata) to request headers
|
||||
|
|
|
|||
|
|
@ -3045,7 +3045,8 @@ class Router:
|
|||
deployment_timeout_param = _timeout_debug_deployment_dict.get(
|
||||
"litellm_params", {}
|
||||
).get("timeout", None)
|
||||
e.message += f"\n\nDeployment Info: request_timeout: {deployment_request_timeout_param}\ntimeout: {deployment_timeout_param}"
|
||||
if litellm.expose_router_debug_in_errors:
|
||||
e.message += f"\n\nDeployment Info: request_timeout: {deployment_request_timeout_param}\ntimeout: {deployment_timeout_param}"
|
||||
# Set per-deployment num_retries on exception for retry logic
|
||||
if deployment is not None:
|
||||
self._set_deployment_num_retries_on_exception(e, deployment)
|
||||
|
|
@ -6644,7 +6645,8 @@ class Router:
|
|||
)
|
||||
)
|
||||
|
||||
e.message += "\n{}".format(error_message)
|
||||
if litellm.expose_router_debug_in_errors:
|
||||
e.message += "\n{}".format(error_message)
|
||||
elif isinstance(e, litellm.ContentPolicyViolationError):
|
||||
if content_policy_fallbacks is not None:
|
||||
content_policy_fallback_model_group: Optional[List[str]] = (
|
||||
|
|
@ -6679,7 +6681,8 @@ class Router:
|
|||
)
|
||||
)
|
||||
|
||||
e.message += "\n{}".format(error_message)
|
||||
if litellm.expose_router_debug_in_errors:
|
||||
e.message += "\n{}".format(error_message)
|
||||
if fallbacks is not None and model_group is not None:
|
||||
verbose_router_logger.debug(f"inside model fallbacks: {fallbacks}")
|
||||
(
|
||||
|
|
@ -6697,7 +6700,10 @@ class Router:
|
|||
verbose_router_logger.info(
|
||||
f"No fallback model group found for original model_group={model_group}. Fallbacks={fallbacks}"
|
||||
)
|
||||
if hasattr(original_exception, "message"):
|
||||
if (
|
||||
hasattr(original_exception, "message")
|
||||
and litellm.expose_router_debug_in_errors
|
||||
):
|
||||
original_exception.message += f"No fallback model group found for original model_group={model_group}. Fallbacks={fallbacks}" # type: ignore
|
||||
raise original_exception
|
||||
|
||||
|
|
@ -6728,7 +6734,10 @@ class Router:
|
|||
)
|
||||
fallback_failure_exception_str = str(new_exception)
|
||||
|
||||
if hasattr(original_exception, "message"):
|
||||
if (
|
||||
hasattr(original_exception, "message")
|
||||
and litellm.expose_router_debug_in_errors
|
||||
):
|
||||
# add the available fallbacks to the exception
|
||||
original_exception.message += ". Received Model Group={}\nAvailable Model Group Fallbacks={}".format( # type: ignore
|
||||
model_group,
|
||||
|
|
|
|||
311
tests/test_litellm/test_router_exception_redaction.py
Normal file
311
tests/test_litellm/test_router_exception_redaction.py
Normal file
|
|
@ -0,0 +1,311 @@
|
|||
"""
|
||||
Tests for `litellm.expose_router_debug_in_errors`.
|
||||
|
||||
The Router historically appended internal config names (model_group,
|
||||
fallback_model_group, fallback failure detail, deployment timeouts,
|
||||
context_window_fallbacks dict, etc.) onto the message of the exception
|
||||
it re-raises. That message is then surfaced to clients by
|
||||
ProxyException, leaking the proxy's internal wiring.
|
||||
|
||||
The flag defaults to True to preserve historical behavior (no
|
||||
breaking change for existing deployments). Set it to False to redact
|
||||
those strings from the raised exception's message.
|
||||
|
||||
These tests verify that with the flag ON (default) the historical
|
||||
leak strings appear in the raised exception's message, and with the
|
||||
flag OFF the proxy's internal wiring is redacted.
|
||||
|
||||
Five leak sites are gated in `litellm/router.py`:
|
||||
|
||||
1. Deployment timeout debug after `litellm.Timeout`
|
||||
2. ContextWindowExceededError fallback hint
|
||||
3. ContentPolicyViolationError fallback hint
|
||||
4. "No fallback model group found for..." when fallbacks dict misses
|
||||
5. "Received Model Group=...\\nAvailable Model Group Fallbacks=..."
|
||||
(always fires on terminal raise from the fallback orchestrator)
|
||||
|
||||
Site 5 is the broadest — it fires for every failing call that goes
|
||||
through the fallback orchestrator with any non-context-window /
|
||||
non-content-policy error, regardless of whether `fallbacks` is set.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm import Router
|
||||
|
||||
_RECEIVED_MODEL_GROUP_PHRASE = "Received Model Group="
|
||||
_AVAILABLE_FALLBACKS_PHRASE = "Available Model Group Fallbacks="
|
||||
_CONTEXT_WINDOW_HINT_PHRASE = "context_window_fallbacks="
|
||||
_INTERNAL_MODEL_GROUP_NAME = "all-anthropic/claude-secret-internal"
|
||||
|
||||
|
||||
def _router_with_rate_limit_failure() -> Router:
|
||||
return Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": _INTERNAL_MODEL_GROUP_NAME,
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o",
|
||||
"api_key": "key",
|
||||
"mock_response": "litellm.RateLimitError",
|
||||
},
|
||||
"model_info": {"id": "secret-deployment-id"},
|
||||
},
|
||||
],
|
||||
num_retries=0,
|
||||
)
|
||||
|
||||
|
||||
def _router_with_context_window_failure() -> Router:
|
||||
return Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": _INTERNAL_MODEL_GROUP_NAME,
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o",
|
||||
"api_key": "key",
|
||||
"mock_response": "litellm.ContextWindowExceededError",
|
||||
},
|
||||
"model_info": {"id": "secret-deployment-id"},
|
||||
},
|
||||
],
|
||||
num_retries=0,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _reset_expose_flag():
|
||||
"""Each test starts with the flag in its default (on) state."""
|
||||
original = litellm.expose_router_debug_in_errors
|
||||
litellm.expose_router_debug_in_errors = True
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
litellm.expose_router_debug_in_errors = original
|
||||
|
||||
|
||||
def test_flag_defaults_on():
|
||||
assert litellm.expose_router_debug_in_errors is True
|
||||
|
||||
|
||||
# --- Site 5: "Received Model Group=..." on terminal raise --------------------
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_flag_off_does_not_leak_received_model_group():
|
||||
litellm.expose_router_debug_in_errors = False
|
||||
router = _router_with_rate_limit_failure()
|
||||
with pytest.raises(litellm.RateLimitError) as excinfo:
|
||||
await router.acompletion(
|
||||
model=_INTERNAL_MODEL_GROUP_NAME,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
)
|
||||
msg = excinfo.value.message
|
||||
assert _RECEIVED_MODEL_GROUP_PHRASE not in msg, msg
|
||||
assert _AVAILABLE_FALLBACKS_PHRASE not in msg, msg
|
||||
assert _INTERNAL_MODEL_GROUP_NAME not in msg, msg
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_default_leaks_received_model_group():
|
||||
router = _router_with_rate_limit_failure()
|
||||
with pytest.raises(litellm.RateLimitError) as excinfo:
|
||||
await router.acompletion(
|
||||
model=_INTERNAL_MODEL_GROUP_NAME,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
)
|
||||
msg = excinfo.value.message
|
||||
assert _RECEIVED_MODEL_GROUP_PHRASE in msg, msg
|
||||
assert _AVAILABLE_FALLBACKS_PHRASE in msg, msg
|
||||
assert _INTERNAL_MODEL_GROUP_NAME in msg, msg
|
||||
|
||||
|
||||
# --- Site 2: ContextWindowExceededError fallback hint ------------------------
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_flag_off_does_not_leak_context_window_fallback_hint():
|
||||
litellm.expose_router_debug_in_errors = False
|
||||
router = _router_with_context_window_failure()
|
||||
with pytest.raises(litellm.ContextWindowExceededError) as excinfo:
|
||||
await router.acompletion(
|
||||
model=_INTERNAL_MODEL_GROUP_NAME,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
)
|
||||
msg = excinfo.value.message
|
||||
assert _CONTEXT_WINDOW_HINT_PHRASE not in msg, msg
|
||||
assert _RECEIVED_MODEL_GROUP_PHRASE not in msg, msg
|
||||
assert _INTERNAL_MODEL_GROUP_NAME not in msg, msg
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_default_leaks_context_window_fallback_hint():
|
||||
router = _router_with_context_window_failure()
|
||||
with pytest.raises(litellm.ContextWindowExceededError) as excinfo:
|
||||
await router.acompletion(
|
||||
model=_INTERNAL_MODEL_GROUP_NAME,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
)
|
||||
msg = excinfo.value.message
|
||||
assert _CONTEXT_WINDOW_HINT_PHRASE in msg, msg
|
||||
# Site 5 also fires for ContextWindow errors that exit the
|
||||
# orchestrator without fallback resolution, so the model_group
|
||||
# name leaks under the default behavior.
|
||||
assert _INTERNAL_MODEL_GROUP_NAME in msg, msg
|
||||
|
||||
|
||||
# --- Site 4: "No fallback model group found..." when fallbacks miss ---------
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_flag_off_does_not_leak_when_no_fallback_group_found():
|
||||
litellm.expose_router_debug_in_errors = False
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": _INTERNAL_MODEL_GROUP_NAME,
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o",
|
||||
"api_key": "key",
|
||||
"mock_response": "litellm.RateLimitError",
|
||||
},
|
||||
"model_info": {"id": "secret-deployment-id"},
|
||||
},
|
||||
],
|
||||
# Fallbacks defined for a different model_group, so resolution
|
||||
# ends with fallback_model_group=None and hits site 4.
|
||||
fallbacks=[{"some-other-group": ["some-other-target"]}],
|
||||
num_retries=0,
|
||||
)
|
||||
with pytest.raises(litellm.RateLimitError) as excinfo:
|
||||
await router.acompletion(
|
||||
model=_INTERNAL_MODEL_GROUP_NAME,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
)
|
||||
msg = excinfo.value.message
|
||||
assert "No fallback model group found" not in msg, msg
|
||||
assert "some-other-group" not in msg, msg
|
||||
assert _INTERNAL_MODEL_GROUP_NAME not in msg, msg
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_default_leaks_when_no_fallback_group_found():
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": _INTERNAL_MODEL_GROUP_NAME,
|
||||
"litellm_params": {
|
||||
"model": "gpt-4o",
|
||||
"api_key": "key",
|
||||
"mock_response": "litellm.RateLimitError",
|
||||
},
|
||||
"model_info": {"id": "secret-deployment-id"},
|
||||
},
|
||||
],
|
||||
fallbacks=[{"some-other-group": ["some-other-target"]}],
|
||||
num_retries=0,
|
||||
)
|
||||
with pytest.raises(litellm.RateLimitError) as excinfo:
|
||||
await router.acompletion(
|
||||
model=_INTERNAL_MODEL_GROUP_NAME,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
)
|
||||
msg = excinfo.value.message
|
||||
assert "No fallback model group found" in msg, msg
|
||||
assert _INTERNAL_MODEL_GROUP_NAME in msg, msg
|
||||
|
||||
|
||||
# --- Site 1: Deployment timeout debug on litellm.Timeout --------------------
|
||||
|
||||
|
||||
def _router_with_plain_deployment() -> Router:
|
||||
"""Plain deployment, no preconfigured mock_response — caller supplies via kwargs.
|
||||
|
||||
Exception instances cannot live in `model_list[*].litellm_params` because
|
||||
`Router.__init__` deep-copies model_list and several LiteLLM exceptions
|
||||
(Timeout, ContentPolicyViolationError) require positional args that
|
||||
`__reduce__` cannot reconstruct. Passing the trigger at call-site bypasses
|
||||
the deepcopy entirely.
|
||||
"""
|
||||
return Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": _INTERNAL_MODEL_GROUP_NAME,
|
||||
"litellm_params": {"model": "gpt-4o", "api_key": "key"},
|
||||
"model_info": {"id": "secret-deployment-id"},
|
||||
},
|
||||
],
|
||||
num_retries=0,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_flag_off_does_not_leak_deployment_timeout_debug():
|
||||
litellm.expose_router_debug_in_errors = False
|
||||
router = _router_with_plain_deployment()
|
||||
with pytest.raises(litellm.Timeout) as excinfo:
|
||||
await router.acompletion(
|
||||
model=_INTERNAL_MODEL_GROUP_NAME,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
mock_timeout=True,
|
||||
timeout=0.001,
|
||||
)
|
||||
msg = excinfo.value.message
|
||||
assert "Deployment Info: request_timeout:" not in msg, msg
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_default_leaks_deployment_timeout_debug():
|
||||
router = _router_with_plain_deployment()
|
||||
with pytest.raises(litellm.Timeout) as excinfo:
|
||||
await router.acompletion(
|
||||
model=_INTERNAL_MODEL_GROUP_NAME,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
mock_timeout=True,
|
||||
timeout=0.001,
|
||||
)
|
||||
msg = excinfo.value.message
|
||||
assert "Deployment Info: request_timeout:" in msg, msg
|
||||
|
||||
|
||||
# --- Site 3: ContentPolicyViolationError fallback hint (no fallback set) ----
|
||||
|
||||
|
||||
def _content_policy_error() -> litellm.ContentPolicyViolationError:
|
||||
return litellm.ContentPolicyViolationError(
|
||||
message="mocked policy violation",
|
||||
model="gpt-4o",
|
||||
llm_provider="openai",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_flag_off_does_not_leak_content_policy_fallback_hint():
|
||||
litellm.expose_router_debug_in_errors = False
|
||||
router = _router_with_plain_deployment()
|
||||
with pytest.raises(litellm.ContentPolicyViolationError) as excinfo:
|
||||
await router.acompletion(
|
||||
model=_INTERNAL_MODEL_GROUP_NAME,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
mock_response=_content_policy_error(),
|
||||
)
|
||||
msg = excinfo.value.message
|
||||
assert "content_policy_fallback=" not in msg, msg
|
||||
assert _INTERNAL_MODEL_GROUP_NAME not in msg, msg
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_default_leaks_content_policy_fallback_hint():
|
||||
router = _router_with_plain_deployment()
|
||||
with pytest.raises(litellm.ContentPolicyViolationError) as excinfo:
|
||||
await router.acompletion(
|
||||
model=_INTERNAL_MODEL_GROUP_NAME,
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
mock_response=_content_policy_error(),
|
||||
)
|
||||
msg = excinfo.value.message
|
||||
assert "content_policy_fallback=" in msg, msg
|
||||
assert _INTERNAL_MODEL_GROUP_NAME in msg, msg
|
||||
Loading…
Add table
Reference in a new issue