fix(content_filter): use substring matching for identifier words in conditionals

- Reverted word boundary matching for identifier words in _check_conditional_categories
- Reverted word boundary matching for conditional words in _cut_breaks_wider_context
- Identifier words MUST use substring matching to match straddling-cut scenarios
- Word boundary matching is kept for BLOCK words only (not identifiers)
- Fixes test_streaming_hook_blocks_conditional_identifier_straddling_cut
This commit is contained in:
Karunasagar Mohansundar 2026-09-28 07:45:13 +00:00
parent 40bf1eded8
commit 4328131b97

View file

@ -1050,26 +1050,12 @@ class ContentFilterGuardrail(CustomGuardrail):
if not sentence_lower:
continue
# Check if sentence contains ANY identifier word (with word boundaries)
# Check if sentence contains ANY identifier word
identifier_found = None
for identifier in identifier_words:
# Use word boundary to avoid false positives (e.g., "alter" in "alternative")
if " " in identifier:
# Multi-word phrase - use simple substring matching
if identifier in sentence_lower:
identifier_found = identifier
break
else:
# Single word - use word boundary for alphanumeric words
# Punctuation-only identifiers (e.g., ">", "=", "!=") need substring matching
# since word boundaries don't work around non-word characters
if _is_word_char_pattern(identifier):
pattern = r"\b" + re.escape(identifier) + r"\b"
else:
pattern = re.escape(identifier)
if re.search(pattern, sentence_lower):
identifier_found = identifier
break
if identifier in sentence_lower:
identifier_found = identifier
break
if not identifier_found:
continue
@ -2058,22 +2044,10 @@ class ContentFilterGuardrail(CustomGuardrail):
cut_sentence: Final = (
SENTENCE_TERMINATORS.split(head.lower())[-1] + SENTENCE_TERMINATORS.split(tail_lower, maxsplit=1)[0]
)
# Use word boundary matching for single-word conditional words to match _check_conditional_categories behavior
# Check if any conditional word appears in the cut_sentence
for word in plan.conditional_words:
if " " in word:
# Multi-word phrase - use substring matching
if word in cut_sentence:
return True
else:
# Single word - use word boundary matching
if _is_word_char_pattern(word):
pattern = r"\b" + re.escape(word) + r"\b"
if re.search(pattern, cut_sentence):
return True
else:
# Punctuation-only words use substring matching
if word in cut_sentence:
return True
if word in cut_sentence:
return True
return False
def _trim_streamed_choice_buffer(