diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py index e9e7851c01a..fe183a1447a 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py @@ -1260,10 +1260,21 @@ class ContentFilterGuardrail(CustomGuardrail): return None text_lower: Final = text.lower() + import re for keyword, (action, description) in self.blocked_words.items(): - if keyword in text_lower: - verbose_proxy_logger.debug("Blocked word '%s' found with action %s", keyword, action) - return (keyword, action, description) + # Use word boundaries for word-character keywords to prevent false positives + # (e.g., "alternative" matching "alter") + # Punctuation-only keywords (e.g., "=", ">") use substring matching + if _is_word_char_pattern(keyword): + pattern = r"\b" + re.escape(keyword) + r"\b" + if re.search(pattern, text_lower): + verbose_proxy_logger.debug("Blocked word '%s' found with action %s", keyword, action) + return (keyword, action, description) + else: + # Punctuation-only: use substring matching + if keyword in text_lower: + verbose_proxy_logger.debug("Blocked word '%s' found with action %s", keyword, action) + return (keyword, action, description) return None def _handle_conditional_match( diff --git a/tests/test_litellm/proxy/guardrails/guardrail_hooks/content_filter/test_content_filter_regression.py b/tests/test_litellm/proxy/guardrails/guardrail_hooks/content_filter/test_content_filter_regression.py index 0fff4b25157..82df33f6292 100644 --- a/tests/test_litellm/proxy/guardrails/guardrail_hooks/content_filter/test_content_filter_regression.py +++ b/tests/test_litellm/proxy/guardrails/guardrail_hooks/content_filter/test_content_filter_regression.py @@ -35,9 +35,9 @@ class TestWordCharPatternHelper: def test_mixed_words(self): """Words with punctuation mixed in have word characters.""" - # These should still be treated as word patterns for boundary purposes - # since they contain word characters - assert _is_word_char_pattern("drop") is False # Backslash is not word char + # 'drop' is all word characters (letters), so it should be True + assert _is_word_char_pattern("drop") is True + # '--' is punctuation-only, so it should be False assert _is_word_char_pattern("--") is False # Dashes def test_empty_string(self):