mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix(complexity-router): gate the reasoning override on a non-SIMPLE score (#37500)
Two or more reasoning keyword matches promoted a request straight to the REASONING tier no matter what the weighted score said, so "hi, step by step, pros and cons" scored 0.100 and still bought the most expensive tier. Require the score to clear the simple_medium boundary before the override applies. Promotion from MEDIUM or COMPLEX is unchanged; only prompts the scorer already placed in the cheapest band stay there.
This commit is contained in:
parent
eec27a9cb3
commit
afbfc3f8fa
5 changed files with 33 additions and 8 deletions
|
|
@ -1020,13 +1020,14 @@ class ComplexityRouter(CustomLogger):
|
|||
weights: Final = self.config.dimension_weights
|
||||
weighted_score: Final = sum(d.score * weights.get(d.name, 0) for d in dimensions)
|
||||
|
||||
# Check for reasoning override (2+ reasoning markers)
|
||||
boundaries: Final = self._effective_tier_boundaries()
|
||||
scored_above_simple: Final = weighted_score >= boundaries["simple_medium"]
|
||||
|
||||
# Reuse match count from _score_keyword_match to avoid scanning twice
|
||||
if reasoning_match_count >= 2:
|
||||
if reasoning_match_count >= 2 and scored_above_simple:
|
||||
return ComplexityTier.REASONING, weighted_score, tuple(signals), "reasoning_override"
|
||||
|
||||
# Map score to tier
|
||||
boundaries: Final = self._effective_tier_boundaries()
|
||||
if weighted_score < boundaries["simple_medium"]:
|
||||
tier = ComplexityTier.SIMPLE
|
||||
elif weighted_score < boundaries["medium_complex"]:
|
||||
|
|
|
|||
|
|
@ -267,6 +267,24 @@ class TestReasoningMarkerScoring:
|
|||
# 2+ reasoning markers should force REASONING tier
|
||||
assert tier == ComplexityTier.REASONING
|
||||
|
||||
def test_reasoning_override_does_not_rescue_a_simple_score(self, complexity_router):
|
||||
"""Reasoning markers on an otherwise trivial prompt must not reach REASONING."""
|
||||
prompt = "hi, step by step, pros and cons"
|
||||
tier, score, signals = complexity_router.classify(prompt)
|
||||
assert score < complexity_router.config.tier_boundaries["simple_medium"]
|
||||
assert any("step by step" in s and "pros and cons" in s for s in signals)
|
||||
assert tier == ComplexityTier.SIMPLE
|
||||
|
||||
def test_reasoning_override_applies_at_the_simple_medium_boundary(self, complexity_router):
|
||||
"""A score sitting exactly on simple_medium is not SIMPLE, so the override still promotes it."""
|
||||
prompt = (
|
||||
"Give me the pros and cons, step by step, of moving our checkout service "
|
||||
"to an event-driven architecture."
|
||||
)
|
||||
tier, score, signals = complexity_router.classify(prompt)
|
||||
assert score == complexity_router.config.tier_boundaries["simple_medium"]
|
||||
assert tier == ComplexityTier.REASONING
|
||||
|
||||
def test_system_prompt_reasoning_not_counted(self, complexity_router):
|
||||
"""Reasoning markers in system prompt should not count for override."""
|
||||
user_prompt = "What is 2+2?"
|
||||
|
|
|
|||
|
|
@ -93,7 +93,7 @@ const HowClassificationWorks: React.FC<{ value: ComplexityRouterConfigValue }> =
|
|||
</li>
|
||||
<li>
|
||||
<strong>{effectiveTierLabel("REASONING", value.tier_labels)}</strong>: Score > {ranges.complexReasoning}{" "}
|
||||
(or 2+ reasoning markers)
|
||||
(or 2+ reasoning markers with a score of at least {ranges.simpleMedium})
|
||||
</li>
|
||||
</ul>
|
||||
)}
|
||||
|
|
|
|||
|
|
@ -59,7 +59,9 @@ describe("RoutingDecisionCard", () => {
|
|||
}}
|
||||
/>,
|
||||
);
|
||||
expect(screen.getByText("Heuristic, REASONING override (2 or more reasoning markers)")).toBeInTheDocument();
|
||||
expect(
|
||||
screen.getByText("Heuristic, REASONING override (2 or more reasoning markers, score above the lowest tier)"),
|
||||
).toBeInTheDocument();
|
||||
expect(screen.getByText("0.20")).toBeInTheDocument();
|
||||
// The score did not decide this tier, so NO band explanation may render at all.
|
||||
// Asserting the absence of one specific band would pass vacuously: 0.20 sits in
|
||||
|
|
@ -188,7 +190,9 @@ describe("RoutingDecisionCard", () => {
|
|||
// `signals` is gone under redaction; the cause alone must suppress the band.
|
||||
render(<RoutingDecisionCard decision={{ ...heuristic, cause: "reasoning_override", signals: undefined }} />);
|
||||
expect(screen.queryByText(/SIMPLE|MEDIUM|COMPLEX|at or above/)).not.toBeInTheDocument();
|
||||
expect(screen.getByText("Heuristic, REASONING override (2 or more reasoning markers)")).toBeInTheDocument();
|
||||
expect(
|
||||
screen.getByText("Heuristic, REASONING override (2 or more reasoning markers, score above the lowest tier)"),
|
||||
).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("shows the operator's tier name on the badge instead of the canonical one", () => {
|
||||
|
|
@ -212,7 +216,9 @@ describe("RoutingDecisionCard", () => {
|
|||
render(
|
||||
<RoutingDecisionCard decision={{ ...heuristic, cause: "reasoning_override", score: 0.2, tier_label: "Deep" }} />,
|
||||
);
|
||||
expect(screen.getByText("Heuristic, Deep override (2 or more reasoning markers)")).toBeInTheDocument();
|
||||
expect(
|
||||
screen.getByText("Heuristic, Deep override (2 or more reasoning markers, score above the lowest tier)"),
|
||||
).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("falls back to the raw cause for a value this build does not know", () => {
|
||||
|
|
|
|||
|
|
@ -72,7 +72,7 @@ function describeCause(decision: RoutingDecision): string {
|
|||
case "heuristic_scorer":
|
||||
return "Heuristic scorer";
|
||||
case "reasoning_override":
|
||||
return `Heuristic, ${tierLabel ?? "REASONING"} override (2 or more reasoning markers)`;
|
||||
return `Heuristic, ${tierLabel ?? "REASONING"} override (2 or more reasoning markers, score above the lowest tier)`;
|
||||
case "llm_classifier":
|
||||
return classifierModel ? `LLM classifier (${classifierModel})` : "LLM classifier";
|
||||
case "literal_keyword_match":
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue