From 3f7636318cf5da64781e687e00b8ee023e42af78 Mon Sep 17 00:00:00 2001 From: michelligabriele Date: Fri, 11 Sep 2026 14:21:39 +0200 Subject: [PATCH] fix(guardrails): warn when only_scan_new_messages is set on a guardrail that ignores it only_scan_new_messages is declared on BaseLitellmParams, so it validates on any guardrail's config, but only guardrails that call filter_new_texts_for_session actually honor it. Everywhere else the setting is accepted in silence and the guardrail keeps scanning the full request, so the config reads as tuned while nothing changed. Add a supports_only_scan_new_messages() capability check, overridden by the two guardrails that implement the feature, and report the mismatch at initialization. This mirrors the supports_scan_only_tool_results() check in the same function. Warn rather than raise, unlike that sibling check: a misconfigured scan_only_tool_results can leave nothing scanned, which is worth failing startup over, whereas an ignored only_scan_new_messages means everything is scanned, which fails safe. Raising would also break the boot of any existing deployment that already carries the flag on a guardrail that ignores it. --- litellm/integrations/custom_guardrail.py | 9 ++ .../guardrail_hooks/bedrock_guardrails.py | 3 + .../litellm_content_filter/content_filter.py | 3 + .../proxy/guardrails/guardrail_registry.py | 9 ++ .../guardrails/test_guardrail_registry.py | 89 +++++++++++++++++++ 5 files changed, 113 insertions(+) diff --git a/litellm/integrations/custom_guardrail.py b/litellm/integrations/custom_guardrail.py index f4fd68b600b..3c2781bd417 100644 --- a/litellm/integrations/custom_guardrail.py +++ b/litellm/integrations/custom_guardrail.py @@ -964,6 +964,15 @@ class CustomGuardrail(CustomLogger): """ return True + def supports_only_scan_new_messages(self) -> bool: + """Whether this guardrail actually scans only the per-session diff. + + Guardrails that never call ``filter_new_texts_for_session`` always scan the + full request, so configuring them with ``only_scan_new_messages`` is reported + at initialization instead of silently doing nothing. + """ + return False + def structured_messages_cover_full_request(self) -> bool: """Whether returned ``structured_messages`` span the whole request. diff --git a/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py b/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py index b69a223f7d1..4580cf99063 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py +++ b/litellm/proxy/guardrails/guardrail_hooks/bedrock_guardrails.py @@ -509,6 +509,9 @@ class BedrockGuardrail(CustomGuardrail, BaseAWSLLM): def supports_scan_only_tool_results(self) -> bool: return self.experimental_use_latest_role_message_only is not True + def supports_only_scan_new_messages(self) -> bool: + return True + def _prepare_guardrail_messages_for_role( self, messages: list[AllMessageValues] | None, diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py index 8d3ebd93475..a4cdfb0a7a4 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py @@ -2144,3 +2144,6 @@ class ContentFilterGuardrail(CustomGuardrail): GuardrailEventHooks.pre_mcp_call, GuardrailEventHooks.post_mcp_call, ] + + def supports_only_scan_new_messages(self) -> bool: + return True diff --git a/litellm/proxy/guardrails/guardrail_registry.py b/litellm/proxy/guardrails/guardrail_registry.py index 0dc50cd6196..ecadfde3fb2 100644 --- a/litellm/proxy/guardrails/guardrail_registry.py +++ b/litellm/proxy/guardrails/guardrail_registry.py @@ -458,6 +458,15 @@ def _configure_callback_scoping( "skip_tool_message_in_guardrail are enabled together, which excludes every message from " "scanning, so no request content would ever be scanned. Remove one of the two." ) + # Warn rather than raise: unlike scan_only_tool_results (which can leave nothing scanned), + # an ignored only_scan_new_messages means everything is scanned, which fails safe -- and + # raising would break the boot of deployments that already carry the flag, on upgrade. + if litellm_params.only_scan_new_messages and not custom_guardrail_callback.supports_only_scan_new_messages(): + verbose_proxy_logger.warning( + "Guardrail %s: only_scan_new_messages is set but this guardrail always scans the full request; " + "the setting has no effect.", + guardrail_name, + ) _apply_configured_bool_overrides(custom_guardrail_callback, litellm_params) diff --git a/tests/test_litellm/proxy/guardrails/test_guardrail_registry.py b/tests/test_litellm/proxy/guardrails/test_guardrail_registry.py index 836668de0c8..56edad34755 100644 --- a/tests/test_litellm/proxy/guardrails/test_guardrail_registry.py +++ b/tests/test_litellm/proxy/guardrails/test_guardrail_registry.py @@ -1084,3 +1084,92 @@ def test_sync_guardrail_from_db_applies_db_dict_params_to_live_instance(): finally: for cb_list, snapshot in zip(lists, snapshots): cb_list[:] = snapshot + + +class TestOnlyScanNewMessagesInitWarning: + """only_scan_new_messages is declared on BaseLitellmParams, so it validates on any + guardrail -- but only guardrails that call filter_new_texts_for_session honor it. + Configuring it anywhere else must say so at initialization instead of silently + scanning the full context forever while the config reads as tuned. + + Warn rather than raise, unlike the scan_only_tool_results check above it: a + misconfigured scan_only_tool_results can leave nothing scanned (an open hole), while + an ignored only_scan_new_messages means everything is scanned, which fails safe -- + and raising would break the boot of any deployment already carrying the flag. + """ + + WARNING_FRAGMENT = "only_scan_new_messages is set but this guardrail always scans the full request" + + def _initialize_capturing_warnings(self, caplog, name: str, params: dict): + import logging + + lists = _all_callback_lists() + snapshots = [list(cb_list) for cb_list in lists] + try: + with caplog.at_level(logging.WARNING, logger="LiteLLM Proxy"): + result = InMemoryGuardrailHandler().initialize_guardrail( + guardrail={"guardrail_name": name, "litellm_params": params}, + ) + return result, [record.getMessage() for record in caplog.records] + finally: + for cb_list, snapshot in zip(lists, snapshots): + cb_list[:] = snapshot + + def test_unsupported_guardrail_warns_and_still_boots(self, caplog): + result, warnings = self._initialize_capturing_warnings( + caplog, + "presidio-only-scan-new-messages", + { + "guardrail": "presidio", + "mode": "pre_call", + "presidio_analyzer_api_base": "https://fakelink.com/v1/presidio/analyze", + "presidio_anonymizer_api_base": "https://fakelink.com/v1/presidio/anonymize", + "only_scan_new_messages": True, + }, + ) + + assert result is not None, "an ignored performance flag must not fail proxy boot" + assert any(self.WARNING_FRAGMENT in message for message in warnings) + + def test_content_filter_does_not_warn(self, caplog): + _, warnings = self._initialize_capturing_warnings( + caplog, + "content-filter-only-scan-new-messages", + { + "guardrail": "litellm_content_filter", + "mode": "pre_call", + "blocked_words": [{"keyword": "hunter2", "action": "BLOCK"}], + "only_scan_new_messages": True, + }, + ) + + assert not any(self.WARNING_FRAGMENT in message for message in warnings) + + def test_bedrock_does_not_warn(self, caplog): + _, warnings = self._initialize_capturing_warnings( + caplog, + "bedrock-only-scan-new-messages", + { + "guardrail": "bedrock", + "mode": "pre_call", + "guardrailIdentifier": "gr-1", + "guardrailVersion": "1", + "only_scan_new_messages": True, + }, + ) + + assert not any(self.WARNING_FRAGMENT in message for message in warnings) + + def test_flag_absent_does_not_warn(self, caplog): + _, warnings = self._initialize_capturing_warnings( + caplog, + "presidio-no-only-scan-new-messages", + { + "guardrail": "presidio", + "mode": "pre_call", + "presidio_analyzer_api_base": "https://fakelink.com/v1/presidio/analyze", + "presidio_anonymizer_api_base": "https://fakelink.com/v1/presidio/anonymize", + }, + ) + + assert not any(self.WARNING_FRAGMENT in message for message in warnings)