style: auto-format with ruff

This commit is contained in:
Karunasagar Mohansundar 2026-09-28 03:49:33 +00:00
parent 9ccf43403d
commit 553db8242c
2 changed files with 18 additions and 17 deletions

View file

@ -72,15 +72,15 @@ SENTENCE_TERMINATORS: Final = re.compile(r"[.!?]+")
def _is_word_char_pattern(s: str) -> bool:
"""
Check if a string consists only of word characters (alphanumeric or underscore).
Word boundaries (\\b) only work around word characters [a-zA-Z0-9_].
Punctuation-only identifiers like ">", "=" require different handling.
This is a module-level helper for consistent keyword matching in content filter.
Args:
s: String to check
Returns:
True if all characters in s are word characters, False otherwise
"""
@ -1000,10 +1000,10 @@ class ContentFilterGuardrail(CustomGuardrail):
This implements logic like: if text contains both an identifier word (e.g., "minor")
AND a block word (e.g., "romantic"), then block it.
NOTE on inflected forms: Word boundary matching means base forms like "alter"
NOTE on inflected forms: Word boundary matching means base forms like "alter"
will NOT match inflected forms like "alters", "altered", or "altering".
This is an intentional trade-off to avoid false positives (e.g., "alter" in
"alternative"). For stronger coverage of SQL keywords, configure multiple
This is an intentional trade-off to avoid false positives (e.g., "alter" in
"alternative"). For stronger coverage of SQL keywords, configure multiple
related keywords in your policy.
Args:
@ -1261,6 +1261,7 @@ class ContentFilterGuardrail(CustomGuardrail):
text_lower: Final = text.lower()
import re
for keyword, (action, description) in self.blocked_words.items():
# Use word boundaries for word-character keywords to prevent false positives
# (e.g., "alternative" matching "alter")

View file

@ -59,7 +59,7 @@ class TestPunctuationOnlyIdentifiers:
BlockedWord(keyword="=", action=ContentFilterAction.BLOCK),
BlockedWord(keyword=">", action=ContentFilterAction.BLOCK),
BlockedWord(keyword="<", action=ContentFilterAction.BLOCK),
]
],
)
# These should all match (contain the punctuation)
@ -77,7 +77,7 @@ class TestPunctuationOnlyIdentifiers:
class TestInflectedFormsKnownLimitation:
"""
Tests demonstrating the inflected forms limitation.
This is a KNOWN LIMITATION, not a bug. Word boundary matching intentionally
does not match stemmed/inflected forms. This avoids false positives.
"""
@ -90,7 +90,7 @@ class TestInflectedFormsKnownLimitation:
guardrail_name="test-stemming",
blocked_words=[
BlockedWord(keyword="alter", action=ContentFilterAction.BLOCK),
]
],
)
# Base form should match
@ -102,10 +102,10 @@ class TestInflectedFormsKnownLimitation:
"""
Inflected forms (like "alters", "altered") do NOT match
the base form "alter". This is a KNOWN LIMITATION.
Design choice: We explicitly document this as a limitation rather than
implementing stemming/lemmatization because:
1. Stemming can cause MORE false positives (e.g., "men" matching within
"recommend" after stemming)
2. Word boundaries prevent false positives with substring matches
@ -113,7 +113,7 @@ class TestInflectedFormsKnownLimitation:
in their policy YAML (e.g., "alter", "alters", "altered")
4. Punctuation-only identifiers require substring matching anyway
5. Adding NLTK/stemming introduces complexity and dependencies
The trade-off is that policy authors may need to be explicit about
which forms they want to block, but gain more predictable behavior.
"""
@ -121,7 +121,7 @@ class TestInflectedFormsKnownLimitation:
guardrail_name="test-stemming-limited",
blocked_words=[
BlockedWord(keyword="alter", action=ContentFilterAction.BLOCK),
]
],
)
# Infected forms DO NOT match - this is documented behavior
@ -148,7 +148,7 @@ class TestFalsePositiveAvoidance:
guardrail_name="test-fp-avoidance",
blocked_words=[
BlockedWord(keyword="alter", action=ContentFilterAction.BLOCK),
]
],
)
# "alternative" contains "alter" as a substring but word boundary prevents match
@ -160,7 +160,7 @@ class TestFalsePositiveAvoidance:
guardrail_name="test-fp-avoidance2",
blocked_words=[
BlockedWord(keyword="exec", action=ContentFilterAction.BLOCK),
]
],
)
result = guardrail2._check_blocked_words("The executive summary")
assert result is None, "executive should not match exec"
@ -170,7 +170,7 @@ class TestFalsePositiveAvoidance:
guardrail_name="test-fp-avoidance3",
blocked_words=[
BlockedWord(keyword="select", action=ContentFilterAction.BLOCK),
]
],
)
result = guardrail3._check_blocked_words("The selection process")
assert result is None, "selection should not match select"