From bea65b6dccc38cbc5d6fcc0cff0cd04a4cd97c2f Mon Sep 17 00:00:00 2001 From: Abhimanyu Kapur <38531241+akapur99@users.noreply.github.com> Date: Wed, 5 Aug 2026 13:02:54 -0700 Subject: [PATCH] fix(autorouter): match CJK keyword_tier_rules that regex word boundaries miss (#35984) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(autorouter): match CJK keyword_tier_rules that regex word boundaries miss Single-word keywords were matched with a \b...\b regex. Every CJK character is a regex word character and CJK is written without spaces, so \b never fires between two of them and a rule like 发票 silently missed 我需要开发票, falling through to complexity scoring instead of the configured tier. Keywords containing CJK now match as plain substrings, the same way multi-word phrases already did. The gate reads the keyword rather than the prompt, so a keyword with no CJK in it keeps word boundary matching regardless of the script the prompt is written in. * fix(autorouter): cover Han extensions in planes 2 and 3, not just up to U+2FA1F The supplementary range stopped at U+2FA1F, so Extension G and H ideographs kept the word boundary path and stayed unmatchable. Both planes are dedicated to CJK ideographs, so covering them whole also handles later extensions without chasing each new block. --- .../complexity_router/complexity_router.py | 28 ++++--- .../router_strategy/test_complexity_router.py | 78 +++++++++++++++++++ 2 files changed, 94 insertions(+), 12 deletions(-) diff --git a/litellm/router_strategy/complexity_router/complexity_router.py b/litellm/router_strategy/complexity_router/complexity_router.py index 06b2fb53cb5..932118963c1 100644 --- a/litellm/router_strategy/complexity_router/complexity_router.py +++ b/litellm/router_strategy/complexity_router/complexity_router.py @@ -226,6 +226,8 @@ _REMINDER_CLOSE: Final = "" _TRUNCATION_MARKER: Final = "..." +_CJK_CHARACTER: Final = re.compile("[぀-ヿㇰ-ㇿ㐀-䶿一-鿿豈-﫿ヲ-ン\U00020000-\U0003ffff]") + def _message_text(content: object) -> str: """Flatten message content to plain text, joining multi-part text blocks. @@ -623,23 +625,25 @@ class ComplexityRouter(CustomLogger): return DimensionScore("tokenCount", 0, None) def _keyword_matches(self, text: str, keyword: str) -> bool: - """ - Check if a keyword matches in text using word boundary matching. + r""" + Check if a keyword matches in text. - For single-word keywords, uses regex word boundaries to avoid - false positives (e.g., "error" matching "terrorism", "class" matching "classical"). - For multi-word phrases, uses substring matching. + Single-word keywords use regex word boundaries to avoid false positives, e.g. "api" + must not match "capital" and "error" must not match "terrorism". + + Multi-word phrases and keywords containing CJK match as plain substrings. CJK is + written without spaces and every CJK character is a regex word character, so `\b` + never fires between two of them: `\b发票\b` misses "我需要开发票" entirely. The gate is + on the keyword rather than the text, so a keyword with no CJK in it keeps word + boundary matching no matter what script the prompt is written in. """ kw_lower: Final = keyword.lower() - # For single-word keywords, use word boundary matching to avoid false positives - # e.g., "api" should not match "capital", "error" should not match "terrorism" - if " " not in kw_lower: - pattern: Final = r"\b" + re.escape(kw_lower) + r"\b" - return bool(re.search(pattern, text)) + if " " in kw_lower or _CJK_CHARACTER.search(kw_lower): + return kw_lower in text - # For multi-word phrases, substring matching is fine - return kw_lower in text + pattern: Final = r"\b" + re.escape(kw_lower) + r"\b" + return bool(re.search(pattern, text)) def _score_keyword_match( self, diff --git a/tests/test_litellm/router_strategy/test_complexity_router.py b/tests/test_litellm/router_strategy/test_complexity_router.py index ff2d0aec39a..3f639c8b20c 100644 --- a/tests/test_litellm/router_strategy/test_complexity_router.py +++ b/tests/test_litellm/router_strategy/test_complexity_router.py @@ -2405,6 +2405,84 @@ class TestLexicalKeywordTierRules: assert router._lexical_tier_override("what is a k8scluster thing") is None +class TestCjkKeywordTierRules: + """CJK keyword_tier_rules must fire mid-sentence, where regex word boundaries cannot.""" + + def _router(self, mock_router_instance, basic_config, keywords: List[str]) -> ComplexityRouter: + return ComplexityRouter( + model_name="test-router", + litellm_router_instance=mock_router_instance, + complexity_router_config={ + **basic_config, + "keyword_tier_rules": [{"keywords": keywords, "tier": "REASONING"}], + }, + ) + + @pytest.mark.parametrize( + "keyword, prompt", + [ + ("发票", "我需要开发票"), + ("退款", "我要退款,谢谢"), + ("账单查询", "我的账单查询怎么做"), + ("API文档", "请问在哪里看API文档"), + ("請求", "這個請求要怎麼處理"), + ("見積", "見積をお願いします"), + ("キャンセル", "注文をキャンセルしたい"), + ("\U00030000", "这个\U00030000很少见"), + ], + ) + def test_cjk_keyword_matches_without_surrounding_whitespace( + self, mock_router_instance, basic_config, keyword, prompt + ): + """CJK is written without spaces, so `\\b\\b` never fires between two CJK characters.""" + router = self._router(mock_router_instance, basic_config, [keyword]) + assert router._lexical_tier_override(prompt) == KeywordOverride( + tier=ComplexityTier.REASONING, matched_keyword=keyword + ) + + def test_cjk_keyword_does_not_match_unrelated_prompt(self, mock_router_instance, basic_config): + """Substring matching must still be a real test, not a match-all.""" + router = self._router(mock_router_instance, basic_config, ["发票"]) + assert router._lexical_tier_override("我想查一下订单状态") is None + + @pytest.mark.asyncio + async def test_cjk_keyword_overrides_scoring_end_to_end(self, mock_router_instance, basic_config): + """The whole hook, not just the matcher: a Chinese prompt reaches the tier it was mapped to.""" + prompt = "我需要开发票" + router = self._router(mock_router_instance, basic_config, ["发票"]) + scored_tier, _, _ = router.classify(prompt) + assert scored_tier != ComplexityTier.REASONING + + result = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={}, + messages=[{"role": "user", "content": prompt}], + ) + assert result is not None + assert result.model == "o1-preview" + + def test_latin_keywords_keep_word_boundary_matching(self, mock_router_instance, basic_config): + """The CJK gate reads the keyword, so a Latin keyword is unaffected by the prompt's script.""" + router = self._router(mock_router_instance, basic_config, ["k8s"]) + assert router._lexical_tier_override("what is a k8scluster thing") is None + assert router._lexical_tier_override("running my k8s cluster") == KeywordOverride( + tier=ComplexityTier.REASONING, matched_keyword="k8s" + ) + + def test_latin_keyword_against_cjk_prompt_still_needs_a_boundary(self, mock_router_instance, basic_config): + """A Latin keyword glued to CJK characters is still a substring false positive.""" + router = self._router(mock_router_instance, basic_config, ["api"]) + assert router._lexical_tier_override("请解释一下rapid这个词") is None + assert router._lexical_tier_override("请问 api 怎么调用") == KeywordOverride( + tier=ComplexityTier.REASONING, matched_keyword="api" + ) + + def test_accented_latin_keeps_word_boundary_semantics(self, complexity_router): + """Guards the alternative fix (ASCII-only lookarounds), which would break diacritics.""" + assert complexity_router._keyword_matches("un café apiculteur", "api") is False + assert complexity_router._keyword_matches("appelle l' api maintenant", "api") is True + + def _make_embedding_response(vectors: List[List[float]]) -> "litellm.EmbeddingResponse": return litellm.EmbeddingResponse( model="fake-embed",