diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py index fe183a1447a..87dbe15d037 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py @@ -72,15 +72,15 @@ SENTENCE_TERMINATORS: Final = re.compile(r"[.!?]+") def _is_word_char_pattern(s: str) -> bool: """ Check if a string consists only of word characters (alphanumeric or underscore). - + Word boundaries (\\b) only work around word characters [a-zA-Z0-9_]. Punctuation-only identifiers like ">", "=" require different handling. - + This is a module-level helper for consistent keyword matching in content filter. - + Args: s: String to check - + Returns: True if all characters in s are word characters, False otherwise """ @@ -1000,10 +1000,10 @@ class ContentFilterGuardrail(CustomGuardrail): This implements logic like: if text contains both an identifier word (e.g., "minor") AND a block word (e.g., "romantic"), then block it. - NOTE on inflected forms: Word boundary matching means base forms like "alter" + NOTE on inflected forms: Word boundary matching means base forms like "alter" will NOT match inflected forms like "alters", "altered", or "altering". - This is an intentional trade-off to avoid false positives (e.g., "alter" in - "alternative"). For stronger coverage of SQL keywords, configure multiple + This is an intentional trade-off to avoid false positives (e.g., "alter" in + "alternative"). For stronger coverage of SQL keywords, configure multiple related keywords in your policy. Args: @@ -1261,6 +1261,7 @@ class ContentFilterGuardrail(CustomGuardrail): text_lower: Final = text.lower() import re + for keyword, (action, description) in self.blocked_words.items(): # Use word boundaries for word-character keywords to prevent false positives # (e.g., "alternative" matching "alter") diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/content_filter/test_content_filter_regression.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/content_filter/test_content_filter_regression.py index 82df33f6292..c1d113faca5 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/content_filter/test_content_filter_regression.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/content_filter/test_content_filter_regression.py @@ -59,7 +59,7 @@ class TestPunctuationOnlyIdentifiers: BlockedWord(keyword="=", action=ContentFilterAction.BLOCK), BlockedWord(keyword=">", action=ContentFilterAction.BLOCK), BlockedWord(keyword="<", action=ContentFilterAction.BLOCK), - ] + ], ) # These should all match (contain the punctuation) @@ -77,7 +77,7 @@ class TestPunctuationOnlyIdentifiers: class TestInflectedFormsKnownLimitation: """ Tests demonstrating the inflected forms limitation. - + This is a KNOWN LIMITATION, not a bug. Word boundary matching intentionally does not match stemmed/inflected forms. This avoids false positives. """ @@ -90,7 +90,7 @@ class TestInflectedFormsKnownLimitation: guardrail_name="test-stemming", blocked_words=[ BlockedWord(keyword="alter", action=ContentFilterAction.BLOCK), - ] + ], ) # Base form should match @@ -102,10 +102,10 @@ class TestInflectedFormsKnownLimitation: """ Inflected forms (like "alters", "altered") do NOT match the base form "alter". This is a KNOWN LIMITATION. - + Design choice: We explicitly document this as a limitation rather than implementing stemming/lemmatization because: - + 1. Stemming can cause MORE false positives (e.g., "men" matching within "recommend" after stemming) 2. Word boundaries prevent false positives with substring matches @@ -113,7 +113,7 @@ class TestInflectedFormsKnownLimitation: in their policy YAML (e.g., "alter", "alters", "altered") 4. Punctuation-only identifiers require substring matching anyway 5. Adding NLTK/stemming introduces complexity and dependencies - + The trade-off is that policy authors may need to be explicit about which forms they want to block, but gain more predictable behavior. """ @@ -121,7 +121,7 @@ class TestInflectedFormsKnownLimitation: guardrail_name="test-stemming-limited", blocked_words=[ BlockedWord(keyword="alter", action=ContentFilterAction.BLOCK), - ] + ], ) # Infected forms DO NOT match - this is documented behavior @@ -148,7 +148,7 @@ class TestFalsePositiveAvoidance: guardrail_name="test-fp-avoidance", blocked_words=[ BlockedWord(keyword="alter", action=ContentFilterAction.BLOCK), - ] + ], ) # "alternative" contains "alter" as a substring but word boundary prevents match @@ -160,7 +160,7 @@ class TestFalsePositiveAvoidance: guardrail_name="test-fp-avoidance2", blocked_words=[ BlockedWord(keyword="exec", action=ContentFilterAction.BLOCK), - ] + ], ) result = guardrail2._check_blocked_words("The executive summary") assert result is None, "executive should not match exec" @@ -170,7 +170,7 @@ class TestFalsePositiveAvoidance: guardrail_name="test-fp-avoidance3", blocked_words=[ BlockedWord(keyword="select", action=ContentFilterAction.BLOCK), - ] + ], ) result = guardrail3._check_blocked_words("The selection process") assert result is None, "selection should not match select"