From 752a03ecab9854f706add2c46d9f835b50ea6060 Mon Sep 17 00:00:00 2001 From: Tin Chi Lo Date: Fri, 21 Aug 2026 18:52:43 -0700 Subject: [PATCH] fix(complexity-router): retune default tier boundaries to 0.10 / 0.25 / 0.50 Lowers the shipped simple_medium, medium_complex and complex_reasoning defaults so the heuristic scorer stops parking technical, multi-part prompts in the cheapest tiers. On the in-repo labelled eval set this moves 5 of 29 cases up a tier and takes accuracy from 28/29 to 29/29. Also points _effective_tier_boundaries() at DEFAULT_TIER_BOUNDARIES instead of restating the three numbers, so a config that overrides only one boundary no longer fills the other two from a second, now stale copy of the defaults. --- .../complexity_router/README.md | 14 +++---- .../complexity_router/complexity_router.py | 7 ++-- .../complexity_router/config.py | 6 +-- .../router_strategy/test_complexity_router.py | 41 ++++++++++++++++++- .../router_strategy/test_quality_router.py | 2 +- .../add_model/ComplexityRouterConfig.test.tsx | 8 ++-- .../add_model/HeuristicScoringConfig.test.tsx | 10 ++--- .../tests/mocks/complexityScorerDefaults.ts | 2 +- 8 files changed, 65 insertions(+), 25 deletions(-) diff --git a/litellm/router_strategy/complexity_router/README.md b/litellm/router_strategy/complexity_router/README.md index cf7bde93360..3aea7a6e482 100644 --- a/litellm/router_strategy/complexity_router/README.md +++ b/litellm/router_strategy/complexity_router/README.md @@ -29,10 +29,10 @@ The weighted sum is mapped to tiers using configurable boundaries: | Tier | Score Range | Boundary key below it | Typical Use | |------|-------------|-----------------------|-------------| -| SIMPLE | < 0.15 | - | Basic questions, greetings | -| MEDIUM | 0.15 - 0.35 | `simple_medium` | Standard queries | -| COMPLEX | 0.35 - 0.60 | `medium_complex` | Technical, multi-part requests | -| REASONING | > 0.60 | `complex_reasoning` | Chain-of-thought, analysis | +| SIMPLE | < 0.10 | - | Basic questions, greetings | +| MEDIUM | 0.10 - 0.25 | `simple_medium` | Standard queries | +| COMPLEX | 0.25 - 0.50 | `medium_complex` | Technical, multi-part requests | +| REASONING | > 0.50 | `complex_reasoning` | Chain-of-thought, analysis | Tier names are defaults you can rename with [`tier_labels`](#renaming-the-tiers). The three `tier_boundaries` keys are named after those defaults but they are scorer knobs, not tiers: each one names the gap between two rungs and is persisted by name on every routing decision, so they stay `simple_medium` / `medium_complex` / `complex_reasoning` no matter what you call the tiers. The column above tells a renamed deployment which knob it is turning. @@ -120,9 +120,9 @@ model_list: # Tier boundaries (normalized scores) tier_boundaries: - simple_medium: 0.15 - medium_complex: 0.35 - complex_reasoning: 0.60 + simple_medium: 0.10 + medium_complex: 0.25 + complex_reasoning: 0.50 # Token count thresholds token_thresholds: diff --git a/litellm/router_strategy/complexity_router/complexity_router.py b/litellm/router_strategy/complexity_router/complexity_router.py index cbaba69f696..7420c3afb7c 100644 --- a/litellm/router_strategy/complexity_router/complexity_router.py +++ b/litellm/router_strategy/complexity_router/complexity_router.py @@ -48,6 +48,7 @@ from .config import ( DEFAULT_REASONING_KEYWORDS, DEFAULT_SIMPLE_KEYWORDS, DEFAULT_TECHNICAL_KEYWORDS, + DEFAULT_TIER_BOUNDARIES, PLAN_MODE_SYSTEM_SENTINELS, PLAN_MODE_TAIL_SENTINELS, PLAN_MODE_TOOL_NAME, @@ -1097,9 +1098,9 @@ class ComplexityRouter(CustomLogger): """ boundaries: Final = self.config.tier_boundaries return StandardLoggingRoutingDecisionTierBoundaries( - simple_medium=boundaries.get("simple_medium", 0.15), - medium_complex=boundaries.get("medium_complex", 0.35), - complex_reasoning=boundaries.get("complex_reasoning", 0.60), + simple_medium=boundaries.get("simple_medium", DEFAULT_TIER_BOUNDARIES["simple_medium"]), + medium_complex=boundaries.get("medium_complex", DEFAULT_TIER_BOUNDARIES["medium_complex"]), + complex_reasoning=boundaries.get("complex_reasoning", DEFAULT_TIER_BOUNDARIES["complex_reasoning"]), ) def _build_routing_decision( diff --git a/litellm/router_strategy/complexity_router/config.py b/litellm/router_strategy/complexity_router/config.py index d3c4bd7938b..10906f14bda 100644 --- a/litellm/router_strategy/complexity_router/config.py +++ b/litellm/router_strategy/complexity_router/config.py @@ -366,9 +366,9 @@ DEFAULT_DIMENSION_WEIGHTS: Final[dict[str, float]] = { # ─── Default Tier Boundaries ─── DEFAULT_TIER_BOUNDARIES: Final[dict[str, float]] = { - "simple_medium": 0.15, # Lower threshold to catch more MEDIUM cases - "medium_complex": 0.35, # Lower threshold to catch technical COMPLEX cases - "complex_reasoning": 0.60, # Reasoning tier reserved for explicit reasoning markers + "simple_medium": 0.10, + "medium_complex": 0.25, + "complex_reasoning": 0.50, } diff --git a/tests/test_litellm/router_strategy/test_complexity_router.py b/tests/test_litellm/router_strategy/test_complexity_router.py index 64b60c75f87..3e14c462f9e 100644 --- a/tests/test_litellm/router_strategy/test_complexity_router.py +++ b/tests/test_litellm/router_strategy/test_complexity_router.py @@ -37,6 +37,7 @@ from litellm.router_strategy.complexity_router.config import ( DEFAULT_CLASSIFIER_CONTEXT_WINDOW_SIZE, DEFAULT_COMPLEXITY_CONFIG, DEFAULT_TECHNICAL_KEYWORDS, + DEFAULT_TIER_BOUNDARIES, ClassifierLLMConfig, ComplexityRouterConfig, ComplexityTier, @@ -602,6 +603,44 @@ class TestPreRoutingHook: assert result.model == "o1-preview" # REASONING tier model +class TestEffectiveTierBoundaries: + """Test how configured boundaries resolve against the shipped defaults.""" + + def test_unconfigured_boundaries_resolve_to_the_shipped_defaults(self, mock_router_instance): + """A router with no tier_boundaries runs on DEFAULT_TIER_BOUNDARIES, not on stale literals.""" + router = ComplexityRouter( + model_name="test-complexity-router", + litellm_router_instance=mock_router_instance, + complexity_router_config={"tiers": {"SIMPLE": "a", "MEDIUM": "b", "COMPLEX": "c", "REASONING": "d"}}, + ) + assert dict(router._effective_tier_boundaries()) == DEFAULT_TIER_BOUNDARIES + + def test_partial_boundaries_fill_missing_keys_from_the_defaults(self, mock_router_instance): + """Overriding one boundary must not strand the other two on a second, drifting copy of the defaults.""" + router = ComplexityRouter( + model_name="test-complexity-router", + litellm_router_instance=mock_router_instance, + complexity_router_config={ + "tiers": {"SIMPLE": "a", "MEDIUM": "b", "COMPLEX": "c", "REASONING": "d"}, + "tier_boundaries": {"simple_medium": 0.42}, + }, + ) + assert dict(router._effective_tier_boundaries()) == { + "simple_medium": 0.42, + "medium_complex": DEFAULT_TIER_BOUNDARIES["medium_complex"], + "complex_reasoning": DEFAULT_TIER_BOUNDARIES["complex_reasoning"], + } + + def test_defaults_stay_ordered_and_within_the_scoring_range(self): + """The tiers only all remain reachable while the boundaries ascend.""" + simple_medium, medium_complex, complex_reasoning = ( + DEFAULT_TIER_BOUNDARIES["simple_medium"], + DEFAULT_TIER_BOUNDARIES["medium_complex"], + DEFAULT_TIER_BOUNDARIES["complex_reasoning"], + ) + assert 0 < simple_medium < medium_complex < complex_reasoning < 1 + + class TestConfigOverrides: """Test configuration override functionality.""" @@ -5074,7 +5113,7 @@ class TestRoutingDecisionContents: assert isinstance(decision["score"], float) assert any("short" in signal for signal in decision["signals"]) # The snapshot must reflect the CONFIGURED boundaries (the fixture overrides the - # 0.15/0.35/0.60 defaults), so a logged row stays truthful after config edits. + # shipped defaults), so a logged row stays truthful after config edits. assert decision["tier_boundaries"] == { "simple_medium": 0.25, "medium_complex": 0.50, diff --git a/tests/test_litellm/router_strategy/test_quality_router.py b/tests/test_litellm/router_strategy/test_quality_router.py index a54e95ff7a1..f55f557bbb5 100644 --- a/tests/test_litellm/router_strategy/test_quality_router.py +++ b/tests/test_litellm/router_strategy/test_quality_router.py @@ -408,7 +408,7 @@ class TestPreRoutingHook: request in the session, and carries no signal about how requests differ. Before the fix this system prompt alone supplied 5 codePresence + 2 technicalTerms keyword matches, saturating both dimensions and crossing the default - simple_medium boundary (0.15) purely from harness text, independent of the ask.""" + simple_medium boundary purely from harness text, independent of the ask.""" agent_system_prompt = ( "You are Claude Code, Anthropic's official CLI for Claude.\n" "You are an interactive agent that helps users with software engineering tasks.\n\n" diff --git a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx index 0a848ed7deb..ef3c954f630 100644 --- a/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx +++ b/ui/litellm-dashboard/src/components/add_model/ComplexityRouterConfig.test.tsx @@ -69,10 +69,10 @@ describe("ComplexityRouterConfig", () => { it("should show score thresholds in the classification section", () => { renderWithProviders(); fireEvent.click(screen.getByText("Advanced: Classification Method")); - expect(screen.getByText(/Score < 0.15/)).toBeInTheDocument(); - expect(screen.getByText(/Score 0.15 - 0.35/)).toBeInTheDocument(); - expect(screen.getByText(/Score 0.35 - 0.60/)).toBeInTheDocument(); - expect(screen.getByText(/Score > 0.60/)).toBeInTheDocument(); + expect(screen.getByText(/Score < 0.10/)).toBeInTheDocument(); + expect(screen.getByText(/Score 0.10 - 0.25/)).toBeInTheDocument(); + expect(screen.getByText(/Score 0.25 - 0.50/)).toBeInTheDocument(); + expect(screen.getByText(/Score > 0.50/)).toBeInTheDocument(); }); it("should default to heuristic and hide classifier model/timeout fields", () => { diff --git a/ui/litellm-dashboard/src/components/add_model/HeuristicScoringConfig.test.tsx b/ui/litellm-dashboard/src/components/add_model/HeuristicScoringConfig.test.tsx index 9dccd767e49..bc2c155711b 100644 --- a/ui/litellm-dashboard/src/components/add_model/HeuristicScoringConfig.test.tsx +++ b/ui/litellm-dashboard/src/components/add_model/HeuristicScoringConfig.test.tsx @@ -41,7 +41,7 @@ describe("HeuristicScoringConfig", () => { it("prefills the shipped defaults", async () => { await render(BASE); - expect(screen.getByLabelText("Simple to Medium")).toHaveValue("0.15"); + expect(screen.getByLabelText("Simple to Medium")).toHaveValue("0.1"); expect(screen.getByLabelText("Long above")).toHaveValue("400"); expect(screen.getByTestId("dimension-weight-total")).toHaveTextContent("total 1.00"); }); @@ -57,8 +57,8 @@ describe("HeuristicScoringConfig", () => { expect((onChange.mock.calls.at(-1)?.[0] as ComplexityRouterConfigValue).tier_boundaries).toEqual({ simple_medium: 0.22, - medium_complex: 0.35, - complex_reasoning: 0.6, + medium_complex: 0.25, + complex_reasoning: 0.5, }); }); @@ -184,7 +184,7 @@ describe("ClassificationMethodConfig scorer gating", () => { expect(screen.getByText(/Score < 0.22/)).toBeInTheDocument(); expect(screen.getByText(/Score 0.44 - 0.66/)).toBeInTheDocument(); - expect(screen.queryByText(/0.15/)).not.toBeInTheDocument(); + expect(screen.queryByText(/0.10/)).not.toBeInTheDocument(); }); it("states the configured override floor in the reasoning-marker aside, not the boundary", () => { @@ -196,7 +196,7 @@ describe("ClassificationMethodConfig scorer gating", () => { it("falls back to the Simple to Medium boundary when no override floor is set", () => { renderWithProviders(); - expect(screen.getByText(/2\+ reasoning markers with a score of at least 0\.15/)).toBeInTheDocument(); + expect(screen.getByText(/2\+ reasoning markers with a score of at least 0\.10/)).toBeInTheDocument(); }); it("renders a row for every scored dimension", async () => { diff --git a/ui/litellm-dashboard/tests/mocks/complexityScorerDefaults.ts b/ui/litellm-dashboard/tests/mocks/complexityScorerDefaults.ts index 3eb2d781213..507d8d65f72 100644 --- a/ui/litellm-dashboard/tests/mocks/complexityScorerDefaults.ts +++ b/ui/litellm-dashboard/tests/mocks/complexityScorerDefaults.ts @@ -10,7 +10,7 @@ import type { ComplexityScorerDefaults } from "@/components/networking"; * which is how the failure path is covered. */ export const SHIPPED_SCORER_DEFAULTS: ComplexityScorerDefaults = { - tier_boundaries: { simple_medium: 0.15, medium_complex: 0.35, complex_reasoning: 0.6 }, + tier_boundaries: { simple_medium: 0.1, medium_complex: 0.25, complex_reasoning: 0.5 }, token_thresholds: { simple: 15, complex: 400 }, dimension_weights: { codePresence: 0.3,