From afbfc3f8fa8391a232b48c66e1d715bf9477888c Mon Sep 17 00:00:00 2001 From: tin-berri Date: Wed, 19 Aug 2026 14:19:24 -0700 Subject: [PATCH] fix(complexity-router): gate the reasoning override on a non-SIMPLE score (#37500) Two or more reasoning keyword matches promoted a request straight to the REASONING tier no matter what the weighted score said, so "hi, step by step, pros and cons" scored 0.100 and still bought the most expensive tier. Require the score to clear the simple_medium boundary before the override applies. Promotion from MEDIUM or COMPLEX is unchanged; only prompts the scorer already placed in the cheapest band stay there. --- .../complexity_router/complexity_router.py | 7 ++++--- .../router_strategy/test_complexity_router.py | 18 ++++++++++++++++++ .../add_model/ClassificationMethodConfig.tsx | 2 +- .../RoutingDecisionCard.test.tsx | 12 +++++++++--- .../LogDetailsDrawer/RoutingDecisionCard.tsx | 2 +- 5 files changed, 33 insertions(+), 8 deletions(-) diff --git a/litellm/router_strategy/complexity_router/complexity_router.py b/litellm/router_strategy/complexity_router/complexity_router.py index d16063b9bd4..0573f8acf18 100644 --- a/litellm/router_strategy/complexity_router/complexity_router.py +++ b/litellm/router_strategy/complexity_router/complexity_router.py @@ -1020,13 +1020,14 @@ class ComplexityRouter(CustomLogger): weights: Final = self.config.dimension_weights weighted_score: Final = sum(d.score * weights.get(d.name, 0) for d in dimensions) - # Check for reasoning override (2+ reasoning markers) + boundaries: Final = self._effective_tier_boundaries() + scored_above_simple: Final = weighted_score >= boundaries["simple_medium"] + # Reuse match count from _score_keyword_match to avoid scanning twice - if reasoning_match_count >= 2: + if reasoning_match_count >= 2 and scored_above_simple: return ComplexityTier.REASONING, weighted_score, tuple(signals), "reasoning_override" # Map score to tier - boundaries: Final = self._effective_tier_boundaries() if weighted_score < boundaries["simple_medium"]: tier = ComplexityTier.SIMPLE elif weighted_score < boundaries["medium_complex"]: diff --git a/tests/test_litellm/router_strategy/test_complexity_router.py b/tests/test_litellm/router_strategy/test_complexity_router.py index e1e8d9553b3..a148da8b675 100644 --- a/tests/test_litellm/router_strategy/test_complexity_router.py +++ b/tests/test_litellm/router_strategy/test_complexity_router.py @@ -267,6 +267,24 @@ class TestReasoningMarkerScoring: # 2+ reasoning markers should force REASONING tier assert tier == ComplexityTier.REASONING + def test_reasoning_override_does_not_rescue_a_simple_score(self, complexity_router): + """Reasoning markers on an otherwise trivial prompt must not reach REASONING.""" + prompt = "hi, step by step, pros and cons" + tier, score, signals = complexity_router.classify(prompt) + assert score < complexity_router.config.tier_boundaries["simple_medium"] + assert any("step by step" in s and "pros and cons" in s for s in signals) + assert tier == ComplexityTier.SIMPLE + + def test_reasoning_override_applies_at_the_simple_medium_boundary(self, complexity_router): + """A score sitting exactly on simple_medium is not SIMPLE, so the override still promotes it.""" + prompt = ( + "Give me the pros and cons, step by step, of moving our checkout service " + "to an event-driven architecture." + ) + tier, score, signals = complexity_router.classify(prompt) + assert score == complexity_router.config.tier_boundaries["simple_medium"] + assert tier == ComplexityTier.REASONING + def test_system_prompt_reasoning_not_counted(self, complexity_router): """Reasoning markers in system prompt should not count for override.""" user_prompt = "What is 2+2?" diff --git a/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx b/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx index 8d42b34a14b..da0c50c7bd0 100644 --- a/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx +++ b/ui/litellm-dashboard/src/components/add_model/ClassificationMethodConfig.tsx @@ -93,7 +93,7 @@ const HowClassificationWorks: React.FC<{ value: ComplexityRouterConfigValue }> =
  • {effectiveTierLabel("REASONING", value.tier_labels)}: Score > {ranges.complexReasoning}{" "} - (or 2+ reasoning markers) + (or 2+ reasoning markers with a score of at least {ranges.simpleMedium})
  • )} diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.test.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.test.tsx index a2afe0d7a0d..ed99e414706 100644 --- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.test.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.test.tsx @@ -59,7 +59,9 @@ describe("RoutingDecisionCard", () => { }} />, ); - expect(screen.getByText("Heuristic, REASONING override (2 or more reasoning markers)")).toBeInTheDocument(); + expect( + screen.getByText("Heuristic, REASONING override (2 or more reasoning markers, score above the lowest tier)"), + ).toBeInTheDocument(); expect(screen.getByText("0.20")).toBeInTheDocument(); // The score did not decide this tier, so NO band explanation may render at all. // Asserting the absence of one specific band would pass vacuously: 0.20 sits in @@ -188,7 +190,9 @@ describe("RoutingDecisionCard", () => { // `signals` is gone under redaction; the cause alone must suppress the band. render(); expect(screen.queryByText(/SIMPLE|MEDIUM|COMPLEX|at or above/)).not.toBeInTheDocument(); - expect(screen.getByText("Heuristic, REASONING override (2 or more reasoning markers)")).toBeInTheDocument(); + expect( + screen.getByText("Heuristic, REASONING override (2 or more reasoning markers, score above the lowest tier)"), + ).toBeInTheDocument(); }); it("shows the operator's tier name on the badge instead of the canonical one", () => { @@ -212,7 +216,9 @@ describe("RoutingDecisionCard", () => { render( , ); - expect(screen.getByText("Heuristic, Deep override (2 or more reasoning markers)")).toBeInTheDocument(); + expect( + screen.getByText("Heuristic, Deep override (2 or more reasoning markers, score above the lowest tier)"), + ).toBeInTheDocument(); }); it("falls back to the raw cause for a value this build does not know", () => { diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.tsx index 0876a539653..cb680668849 100644 --- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.tsx @@ -72,7 +72,7 @@ function describeCause(decision: RoutingDecision): string { case "heuristic_scorer": return "Heuristic scorer"; case "reasoning_override": - return `Heuristic, ${tierLabel ?? "REASONING"} override (2 or more reasoning markers)`; + return `Heuristic, ${tierLabel ?? "REASONING"} override (2 or more reasoning markers, score above the lowest tier)`; case "llm_classifier": return classifierModel ? `LLM classifier (${classifierModel})` : "LLM classifier"; case "literal_keyword_match":