From 307b72d59e2ed7b9e668e53a10f66dd50ea34735 Mon Sep 17 00:00:00 2001 From: Karunasagar Mohansundar Date: Sun, 27 Sep 2026 13:49:58 +0000 Subject: [PATCH] fix(content_filter): add regex word boundaries for SQL injection keywords Fix false positives when SQL keywords appear as substrings in benign words. The _check_conditional_categories method was using substring matching for identifier_words, causing false positives for words like: - "alternative" containing "alter" - "executive" containing "exec" - "selection" containing "select" - "updateable" containing "update" This fix applies regex word boundaries (\b) to identifier word matching, consistent with how block_words are already handled in the same code block. Multi-word phrases (containing spaces) continue to use substring matching to preserve support for phrases like "union select". Fixes #42681 --- .../litellm_content_filter/content_filter.py | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py index 092e8eaafa1..f998c8df0d8 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py +++ b/litellm/proxy/guardrails/guardrail_hooks/litellm_content_filter/content_filter.py @@ -1026,12 +1026,21 @@ class ContentFilterGuardrail(CustomGuardrail): if not sentence_lower: continue - # Check if sentence contains ANY identifier word + # Check if sentence contains ANY identifier word (with word boundaries) identifier_found = None for identifier in identifier_words: - if identifier in sentence_lower: - identifier_found = identifier - break + # Use word boundary to avoid false positives (e.g., "alter" in "alternative") + if " " in identifier: + # Multi-word phrase - use simple substring matching + if identifier in sentence_lower: + identifier_found = identifier + break + else: + # Single word - use word boundary + pattern = r"\b" + re.escape(identifier) + r"\b" + if re.search(pattern, sentence_lower): + identifier_found = identifier + break if not identifier_found: continue