fix(guardrails): warn when only_scan_new_messages is set on a guardrail that ignores it

only_scan_new_messages is declared on BaseLitellmParams, so it validates on
any guardrail's config, but only guardrails that call
filter_new_texts_for_session actually honor it. Everywhere else the setting is
accepted in silence and the guardrail keeps scanning the full request, so the
config reads as tuned while nothing changed.

Add a supports_only_scan_new_messages() capability check, overridden by the
two guardrails that implement the feature, and report the mismatch at
initialization. This mirrors the supports_scan_only_tool_results() check in
the same function.

Warn rather than raise, unlike that sibling check: a misconfigured
scan_only_tool_results can leave nothing scanned, which is worth failing
startup over, whereas an ignored only_scan_new_messages means everything is
scanned, which fails safe. Raising would also break the boot of any existing
deployment that already carries the flag on a guardrail that ignores it.
This commit is contained in:
michelligabriele 2026-09-11 14:21:39 +02:00
parent d5578e7ae4
commit 3f7636318c
No known key found for this signature in database
5 changed files with 113 additions and 0 deletions

View file

@ -964,6 +964,15 @@ class CustomGuardrail(CustomLogger):
"""
return True
def supports_only_scan_new_messages(self) -> bool:
"""Whether this guardrail actually scans only the per-session diff.
Guardrails that never call ``filter_new_texts_for_session`` always scan the
full request, so configuring them with ``only_scan_new_messages`` is reported
at initialization instead of silently doing nothing.
"""
return False
def structured_messages_cover_full_request(self) -> bool:
"""Whether returned ``structured_messages`` span the whole request.

View file

@ -509,6 +509,9 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM):
def supports_scan_only_tool_results(self) -> bool:
return self.experimental_use_latest_role_message_only is not True
def supports_only_scan_new_messages(self) -> bool:
return True
def _prepare_guardrail_messages_for_role(
self,
messages: list[AllMessageValues] | None,

View file

@ -2144,3 +2144,6 @@ class ContentFilterGuardrail(CustomGuardrail):
GuardrailEventHooks.pre_mcp_call,
GuardrailEventHooks.post_mcp_call,
]
def supports_only_scan_new_messages(self) -> bool:
return True

View file

@ -458,6 +458,15 @@ def _configure_callback_scoping(
"skip_tool_message_in_guardrail are enabled together, which excludes every message from "
"scanning, so no request content would ever be scanned. Remove one of the two."
)
# Warn rather than raise: unlike scan_only_tool_results (which can leave nothing scanned),
# an ignored only_scan_new_messages means everything is scanned, which fails safe -- and
# raising would break the boot of deployments that already carry the flag, on upgrade.
if litellm_params.only_scan_new_messages and not custom_guardrail_callback.supports_only_scan_new_messages():
verbose_proxy_logger.warning(
"Guardrail %s: only_scan_new_messages is set but this guardrail always scans the full request; "
"the setting has no effect.",
guardrail_name,
)
_apply_configured_bool_overrides(custom_guardrail_callback, litellm_params)

View file

@ -1084,3 +1084,92 @@ def test_sync_guardrail_from_db_applies_db_dict_params_to_live_instance():
finally:
for cb_list, snapshot in zip(lists, snapshots):
cb_list[:] = snapshot
class TestOnlyScanNewMessagesInitWarning:
"""only_scan_new_messages is declared on BaseLitellmParams, so it validates on any
guardrail -- but only guardrails that call filter_new_texts_for_session honor it.
Configuring it anywhere else must say so at initialization instead of silently
scanning the full context forever while the config reads as tuned.
Warn rather than raise, unlike the scan_only_tool_results check above it: a
misconfigured scan_only_tool_results can leave nothing scanned (an open hole), while
an ignored only_scan_new_messages means everything is scanned, which fails safe --
and raising would break the boot of any deployment already carrying the flag.
"""
WARNING_FRAGMENT = "only_scan_new_messages is set but this guardrail always scans the full request"
def _initialize_capturing_warnings(self, caplog, name: str, params: dict):
import logging
lists = _all_callback_lists()
snapshots = [list(cb_list) for cb_list in lists]
try:
with caplog.at_level(logging.WARNING, logger="LiteLLM Proxy"):
result = InMemoryGuardrailHandler().initialize_guardrail(
guardrail={"guardrail_name": name, "litellm_params": params},
)
return result, [record.getMessage() for record in caplog.records]
finally:
for cb_list, snapshot in zip(lists, snapshots):
cb_list[:] = snapshot
def test_unsupported_guardrail_warns_and_still_boots(self, caplog):
result, warnings = self._initialize_capturing_warnings(
caplog,
"presidio-only-scan-new-messages",
{
"guardrail": "presidio",
"mode": "pre_call",
"presidio_analyzer_api_base": "https://fakelink.com/v1/presidio/analyze",
"presidio_anonymizer_api_base": "https://fakelink.com/v1/presidio/anonymize",
"only_scan_new_messages": True,
},
)
assert result is not None, "an ignored performance flag must not fail proxy boot"
assert any(self.WARNING_FRAGMENT in message for message in warnings)
def test_content_filter_does_not_warn(self, caplog):
_, warnings = self._initialize_capturing_warnings(
caplog,
"content-filter-only-scan-new-messages",
{
"guardrail": "litellm_content_filter",
"mode": "pre_call",
"blocked_words": [{"keyword": "hunter2", "action": "BLOCK"}],
"only_scan_new_messages": True,
},
)
assert not any(self.WARNING_FRAGMENT in message for message in warnings)
def test_bedrock_does_not_warn(self, caplog):
_, warnings = self._initialize_capturing_warnings(
caplog,
"bedrock-only-scan-new-messages",
{
"guardrail": "bedrock",
"mode": "pre_call",
"guardrailIdentifier": "gr-1",
"guardrailVersion": "1",
"only_scan_new_messages": True,
},
)
assert not any(self.WARNING_FRAGMENT in message for message in warnings)
def test_flag_absent_does_not_warn(self, caplog):
_, warnings = self._initialize_capturing_warnings(
caplog,
"presidio-no-only-scan-new-messages",
{
"guardrail": "presidio",
"mode": "pre_call",
"presidio_analyzer_api_base": "https://fakelink.com/v1/presidio/analyze",
"presidio_anonymizer_api_base": "https://fakelink.com/v1/presidio/anonymize",
},
)
assert not any(self.WARNING_FRAGMENT in message for message in warnings)