fix(complexity-router): retune default tier boundaries to 0.10 / 0.25 / 0.50

Lowers the shipped simple_medium, medium_complex and complex_reasoning defaults so the
heuristic scorer stops parking technical, multi-part prompts in the cheapest tiers. On the
in-repo labelled eval set this moves 5 of 29 cases up a tier and takes accuracy from 28/29
to 29/29.

Also points _effective_tier_boundaries() at DEFAULT_TIER_BOUNDARIES instead of restating
the three numbers, so a config that overrides only one boundary no longer fills the other
two from a second, now stale copy of the defaults.
This commit is contained in:
Tin Chi Lo 2026-08-21 18:52:43 -07:00
parent b9bff0998c
commit 752a03ecab
8 changed files with 65 additions and 25 deletions

View file

@ -29,10 +29,10 @@ The weighted sum is mapped to tiers using configurable boundaries:
| Tier | Score Range | Boundary key below it | Typical Use |
|------|-------------|-----------------------|-------------|
| SIMPLE | < 0.15 | - | Basic questions, greetings |
| MEDIUM | 0.15 - 0.35 | `simple_medium` | Standard queries |
| COMPLEX | 0.35 - 0.60 | `medium_complex` | Technical, multi-part requests |
| REASONING | > 0.60 | `complex_reasoning` | Chain-of-thought, analysis |
| SIMPLE | < 0.10 | - | Basic questions, greetings |
| MEDIUM | 0.10 - 0.25 | `simple_medium` | Standard queries |
| COMPLEX | 0.25 - 0.50 | `medium_complex` | Technical, multi-part requests |
| REASONING | > 0.50 | `complex_reasoning` | Chain-of-thought, analysis |
Tier names are defaults you can rename with [`tier_labels`](#renaming-the-tiers). The three `tier_boundaries` keys are named after those defaults but they are scorer knobs, not tiers: each one names the gap between two rungs and is persisted by name on every routing decision, so they stay `simple_medium` / `medium_complex` / `complex_reasoning` no matter what you call the tiers. The column above tells a renamed deployment which knob it is turning.
@ -120,9 +120,9 @@ model_list:
# Tier boundaries (normalized scores)
tier_boundaries:
simple_medium: 0.15
medium_complex: 0.35
complex_reasoning: 0.60
simple_medium: 0.10
medium_complex: 0.25
complex_reasoning: 0.50
# Token count thresholds
token_thresholds:

View file

@ -48,6 +48,7 @@ from .config import (
DEFAULT_REASONING_KEYWORDS,
DEFAULT_SIMPLE_KEYWORDS,
DEFAULT_TECHNICAL_KEYWORDS,
DEFAULT_TIER_BOUNDARIES,
PLAN_MODE_SYSTEM_SENTINELS,
PLAN_MODE_TAIL_SENTINELS,
PLAN_MODE_TOOL_NAME,
@ -1097,9 +1098,9 @@ class ComplexityRouter(CustomLogger):
"""
boundaries: Final = self.config.tier_boundaries
return StandardLoggingRoutingDecisionTierBoundaries(
simple_medium=boundaries.get("simple_medium", 0.15),
medium_complex=boundaries.get("medium_complex", 0.35),
complex_reasoning=boundaries.get("complex_reasoning", 0.60),
simple_medium=boundaries.get("simple_medium", DEFAULT_TIER_BOUNDARIES["simple_medium"]),
medium_complex=boundaries.get("medium_complex", DEFAULT_TIER_BOUNDARIES["medium_complex"]),
complex_reasoning=boundaries.get("complex_reasoning", DEFAULT_TIER_BOUNDARIES["complex_reasoning"]),
)
def _build_routing_decision(

View file

@ -366,9 +366,9 @@ DEFAULT_DIMENSION_WEIGHTS: Final[dict[str, float]] = {
# ─── Default Tier Boundaries ───
DEFAULT_TIER_BOUNDARIES: Final[dict[str, float]] = {
"simple_medium": 0.15, # Lower threshold to catch more MEDIUM cases
"medium_complex": 0.35, # Lower threshold to catch technical COMPLEX cases
"complex_reasoning": 0.60, # Reasoning tier reserved for explicit reasoning markers
"simple_medium": 0.10,
"medium_complex": 0.25,
"complex_reasoning": 0.50,
}

View file

@ -37,6 +37,7 @@ from litellm.router_strategy.complexity_router.config import (
DEFAULT_CLASSIFIER_CONTEXT_WINDOW_SIZE,
DEFAULT_COMPLEXITY_CONFIG,
DEFAULT_TECHNICAL_KEYWORDS,
DEFAULT_TIER_BOUNDARIES,
ClassifierLLMConfig,
ComplexityRouterConfig,
ComplexityTier,
@ -602,6 +603,44 @@ class TestPreRoutingHook:
assert result.model == "o1-preview" # REASONING tier model
class TestEffectiveTierBoundaries:
"""Test how configured boundaries resolve against the shipped defaults."""
def test_unconfigured_boundaries_resolve_to_the_shipped_defaults(self, mock_router_instance):
"""A router with no tier_boundaries runs on DEFAULT_TIER_BOUNDARIES, not on stale literals."""
router = ComplexityRouter(
model_name="test-complexity-router",
litellm_router_instance=mock_router_instance,
complexity_router_config={"tiers": {"SIMPLE": "a", "MEDIUM": "b", "COMPLEX": "c", "REASONING": "d"}},
)
assert dict(router._effective_tier_boundaries()) == DEFAULT_TIER_BOUNDARIES
def test_partial_boundaries_fill_missing_keys_from_the_defaults(self, mock_router_instance):
"""Overriding one boundary must not strand the other two on a second, drifting copy of the defaults."""
router = ComplexityRouter(
model_name="test-complexity-router",
litellm_router_instance=mock_router_instance,
complexity_router_config={
"tiers": {"SIMPLE": "a", "MEDIUM": "b", "COMPLEX": "c", "REASONING": "d"},
"tier_boundaries": {"simple_medium": 0.42},
},
)
assert dict(router._effective_tier_boundaries()) == {
"simple_medium": 0.42,
"medium_complex": DEFAULT_TIER_BOUNDARIES["medium_complex"],
"complex_reasoning": DEFAULT_TIER_BOUNDARIES["complex_reasoning"],
}
def test_defaults_stay_ordered_and_within_the_scoring_range(self):
"""The tiers only all remain reachable while the boundaries ascend."""
simple_medium, medium_complex, complex_reasoning = (
DEFAULT_TIER_BOUNDARIES["simple_medium"],
DEFAULT_TIER_BOUNDARIES["medium_complex"],
DEFAULT_TIER_BOUNDARIES["complex_reasoning"],
)
assert 0 < simple_medium < medium_complex < complex_reasoning < 1
class TestConfigOverrides:
"""Test configuration override functionality."""
@ -5074,7 +5113,7 @@ class TestRoutingDecisionContents:
assert isinstance(decision["score"], float)
assert any("short" in signal for signal in decision["signals"])
# The snapshot must reflect the CONFIGURED boundaries (the fixture overrides the
# 0.15/0.35/0.60 defaults), so a logged row stays truthful after config edits.
# shipped defaults), so a logged row stays truthful after config edits.
assert decision["tier_boundaries"] == {
"simple_medium": 0.25,
"medium_complex": 0.50,

View file

@ -408,7 +408,7 @@ class TestPreRoutingHook:
request in the session, and carries no signal about how requests differ. Before
the fix this system prompt alone supplied 5 codePresence + 2 technicalTerms
keyword matches, saturating both dimensions and crossing the default
simple_medium boundary (0.15) purely from harness text, independent of the ask."""
simple_medium boundary purely from harness text, independent of the ask."""
agent_system_prompt = (
"You are Claude Code, Anthropic's official CLI for Claude.\n"
"You are an interactive agent that helps users with software engineering tasks.\n\n"

View file

@ -69,10 +69,10 @@ describe("ComplexityRouterConfig", () => {
it("should show score thresholds in the classification section", () => {
renderWithProviders(<ComplexityRouterConfig {...baseProps} />);
fireEvent.click(screen.getByText("Advanced: Classification Method"));
expect(screen.getByText(/Score < 0.15/)).toBeInTheDocument();
expect(screen.getByText(/Score 0.15 - 0.35/)).toBeInTheDocument();
expect(screen.getByText(/Score 0.35 - 0.60/)).toBeInTheDocument();
expect(screen.getByText(/Score > 0.60/)).toBeInTheDocument();
expect(screen.getByText(/Score < 0.10/)).toBeInTheDocument();
expect(screen.getByText(/Score 0.10 - 0.25/)).toBeInTheDocument();
expect(screen.getByText(/Score 0.25 - 0.50/)).toBeInTheDocument();
expect(screen.getByText(/Score > 0.50/)).toBeInTheDocument();
});
it("should default to heuristic and hide classifier model/timeout fields", () => {

View file

@ -41,7 +41,7 @@ describe("HeuristicScoringConfig", () => {
it("prefills the shipped defaults", async () => {
await render(BASE);
expect(screen.getByLabelText("Simple to Medium")).toHaveValue("0.15");
expect(screen.getByLabelText("Simple to Medium")).toHaveValue("0.1");
expect(screen.getByLabelText("Long above")).toHaveValue("400");
expect(screen.getByTestId("dimension-weight-total")).toHaveTextContent("total 1.00");
});
@ -57,8 +57,8 @@ describe("HeuristicScoringConfig", () => {
expect((onChange.mock.calls.at(-1)?.[0] as ComplexityRouterConfigValue).tier_boundaries).toEqual({
simple_medium: 0.22,
medium_complex: 0.35,
complex_reasoning: 0.6,
medium_complex: 0.25,
complex_reasoning: 0.5,
});
});
@ -184,7 +184,7 @@ describe("ClassificationMethodConfig scorer gating", () => {
expect(screen.getByText(/Score < 0.22/)).toBeInTheDocument();
expect(screen.getByText(/Score 0.44 - 0.66/)).toBeInTheDocument();
expect(screen.queryByText(/0.15/)).not.toBeInTheDocument();
expect(screen.queryByText(/0.10/)).not.toBeInTheDocument();
});
it("states the configured override floor in the reasoning-marker aside, not the boundary", () => {
@ -196,7 +196,7 @@ describe("ClassificationMethodConfig scorer gating", () => {
it("falls back to the Simple to Medium boundary when no override floor is set", () => {
renderWithProviders(<ClassificationMethodConfig {...props} value={BASE} />);
expect(screen.getByText(/2\+ reasoning markers with a score of at least 0\.15/)).toBeInTheDocument();
expect(screen.getByText(/2\+ reasoning markers with a score of at least 0\.10/)).toBeInTheDocument();
});
it("renders a row for every scored dimension", async () => {

View file

@ -10,7 +10,7 @@ import type { ComplexityScorerDefaults } from "@/components/networking";
* which is how the failure path is covered.
*/
export const SHIPPED_SCORER_DEFAULTS: ComplexityScorerDefaults = {
tier_boundaries: { simple_medium: 0.15, medium_complex: 0.35, complex_reasoning: 0.6 },
tier_boundaries: { simple_medium: 0.1, medium_complex: 0.25, complex_reasoning: 0.5 },
token_thresholds: { simple: 15, complex: 400 },
dimension_weights: {
codePresence: 0.3,