fix(pass_through): log pre-call guardrail blocks at WARNING, not ERROR with a traceback (#31500)

A pre-call guardrail block on a pass-through endpoint (e.g. OpenAI moderation
flagging disallowed content) was logged at ERROR level with a full stack trace,
even though the guardrail is working as designed and the client correctly
receives the 4xx. The generic except in pass_through_request logged every
exception via verbose_proxy_logger.exception(), so an intentional block produced
scary traceback noise for operators tailing logs.

Branch on the existing CustomGuardrail._is_guardrail_intervention classifier
(the same predicate pipeline_executor already uses) so guardrail interventions
log once at WARNING without a traceback while genuine failures keep their ERROR
and traceback. This covers every guardrail that signals a block through the
shared typed exceptions or an HTTPException 400, not just OpenAI moderation, and
leaves the client-facing response unchanged.

Resolves LIT-3538
This commit is contained in:
Yassin Kortam 2026-06-27 22:18:40 +03:00 • committed by GitHub
parent c33a7f8757
commit 63490655ad
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 145 additions and 4 deletions

View file

@ -36,6 +36,7 @@ import litellm
from litellm._logging import verbose_proxy_logger
from litellm._uuid import uuid
from litellm.constants import MAXIMUM_TRACEBACK_LINES_TO_LOG
from litellm.integrations.custom_guardrail import CustomGuardrail
from litellm.integrations.custom_logger import CustomLogger
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
from litellm.litellm_core_utils.logging_worker import GLOBAL_LOGGING_WORKER
@ -1423,11 +1424,17 @@ async def pass_through_request(
cache_key=None,
api_base=str(url._uri_reference) if url else None,
)
verbose_proxy_logger.exception(
"litellm.proxy.proxy_server.pass_through_endpoint(): Exception occured - {}".format(
str(e)
if CustomGuardrail._is_guardrail_intervention(e):
verbose_proxy_logger.warning(
"pass_through_endpoint: request blocked by guardrail - %s",
str(e),
)
else:
verbose_proxy_logger.exception(
"litellm.proxy.proxy_server.pass_through_endpoint(): Exception occured - {}".format(
str(e)
)
)
)
#########################################################
# Monitoring: Trigger post_call_failure_hook

View file

@ -3485,3 +3485,137 @@ class TestStaleRouteCleanupOnReload:
assert InitPassThroughEndpointHelpers.is_registered_pass_through_route(
"/live-passthrough/some/subpath"
)
# Regression (LIT-3538): a pre-call guardrail block on a passthrough endpoint
# must be logged at WARNING without a traceback, not as an ERROR with a full
# stack trace. The generic ``except Exception`` in ``pass_through_request`` used
# to call ``verbose_proxy_logger.exception(...)`` for every exception, so an
# intentional guardrail block (which the rest of the codebase already classifies
# via ``CustomGuardrail._is_guardrail_intervention``) produced scary error noise
# even though the client correctly receives the 4xx.
from fastapi import HTTPException as _FastAPIHTTPException
from litellm.exceptions import (
BlockedPiiEntityError,
GuardrailRaisedException,
)
_PT_MODULE = "litellm.proxy.pass_through_endpoints.pass_through_endpoints"
def _lit3538_user_api_key_dict():
d = MagicMock()
d.api_key = "sk-test"
d.user_id = "user-1"
d.team_id = "team-1"
d.org_id = None
d.metadata = {}
d.team_metadata = {}
d.parent_otel_span = None
d.request_route = "/mock/echo"
return d
def _lit3538_request():
r = MagicMock()
r.method = "POST"
r.query_params = {}
r.url = "http://testserver/mock/echo"
r.state = SimpleNamespace()
headers = MagicMock()
headers.copy.return_value = {}
r.headers = headers
return r
async def _drive_pass_through_block(raised_exception):
"""Drive the real ``pass_through_request`` so its pre_call_hook raises
``raised_exception``, returning (status_code, logger_mock)."""
proxy_logging = MagicMock()
proxy_logging.pre_call_hook = AsyncMock(side_effect=raised_exception)
proxy_logging.post_call_failure_hook = AsyncMock()
proxy_logging.get_proxy_hook = MagicMock(return_value=None)
logger = MagicMock()
patches = [
patch("litellm.proxy.proxy_server.proxy_logging_obj", proxy_logging),
patch(f"{_PT_MODULE}.verbose_proxy_logger", logger),
patch(
f"{_PT_MODULE}._read_request_body",
new_callable=AsyncMock,
return_value={},
),
patch(f"{_PT_MODULE}._safe_get_request_headers", return_value={}),
patch(
"litellm.proxy.pass_through_endpoints.passthrough_guardrails."
"PassthroughGuardrailHandler.collect_guardrails",
return_value=[],
),
]
with ExitStack() as stack:
for p in patches:
stack.enter_context(p)
status_code = None
try:
await pass_through_request(
request=_lit3538_request(),
target="https://upstream.example/echo",
custom_headers={"Content-Type": "application/json"},
user_api_key_dict=_lit3538_user_api_key_dict(),
stream=False,
)
except Exception as e: # ProxyException carrying the original status
status_code = getattr(e, "code", None) or getattr(e, "status_code", None)
return status_code, logger
@pytest.mark.asyncio
@pytest.mark.parametrize(
"guardrail_exception, expected_code",
[
(
GuardrailRaisedException(guardrail_name="g", message="blocked"),
400,
),
(
BlockedPiiEntityError(entity_type="EMAIL", guardrail_name="presidio"),
400,
),
(
_FastAPIHTTPException(
status_code=400, detail={"error": "Violated moderation policy"}
),
400,
),
],
)
async def test_pre_call_guardrail_block_logs_warning_not_exception(
guardrail_exception, expected_code
):
status_code, logger = await _drive_pass_through_block(guardrail_exception)
assert int(status_code) == expected_code
assert (
logger.exception.call_count == 0
), "guardrail block must not be logged as an ERROR with a traceback"
assert (
logger.warning.call_count == 1
), "guardrail block must be logged once at WARNING"
@pytest.mark.asyncio
async def test_non_guardrail_exception_still_logs_with_traceback():
status_code, logger = await _drive_pass_through_block(
RuntimeError("upstream connection reset")
)
assert int(status_code) == 500
assert (
logger.exception.call_count == 1
), "a genuine failure must still be logged via verbose_proxy_logger.exception"
assert (
logger.warning.call_count == 0
), "a genuine failure must not be downgraded to WARNING"