mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
feat(complexity-router): make the reasoning override floor configurable (#37537)
The reasoning override's floor was pinned to tier_boundaries.simple_medium, so an operator could not restore the unconditional promotion nor raise the bar independently of the SIMPLE/MEDIUM cut. Setting reasoning_override_min_score was accepted and echoed back by /model/info, because the config model allows extra keys, while routing ignored it. Resolve the floor through one accessor that falls back to simple_medium when the field is unset, so moving that boundary still moves the floor with it, and an explicit 0 is a real floor rather than an absent one. Record the resolved value on the routing decision so a logged row states the floor that applied, which is also what lets the Admin UI stop hardcoding the copy PR #37500 added.
This commit is contained in:
parent
aee48f5264
commit
dfeb12649b
20 changed files with 423 additions and 14 deletions
|
|
@ -165,7 +165,7 @@ response = litellm.completion(
|
|||
|
||||
### Reasoning Override
|
||||
|
||||
If 2+ reasoning markers are detected in the user message, the request is automatically routed to the REASONING tier regardless of the weighted score. This ensures complex reasoning tasks get the appropriate model.
|
||||
If 2+ reasoning markers are detected in the user message, the request is promoted to the REASONING tier even when the weighted score maps lower, so complex reasoning tasks get the appropriate model. The promotion requires the score to reach `reasoning_override_min_score`, which tracks `tier_boundaries.simple_medium` unless set, so stock phrases on an otherwise trivial prompt cannot buy the top tier. Set it to `0` to promote on the markers alone.
|
||||
|
||||
### System Prompt Handling
|
||||
|
||||
|
|
|
|||
|
|
@ -1021,10 +1021,10 @@ class ComplexityRouter(CustomLogger):
|
|||
weighted_score: Final = sum(d.score * weights.get(d.name, 0) for d in dimensions)
|
||||
|
||||
boundaries: Final = self._effective_tier_boundaries()
|
||||
scored_above_simple: Final = weighted_score >= boundaries["simple_medium"]
|
||||
clears_override_floor: Final = weighted_score >= self._effective_reasoning_override_min_score()
|
||||
|
||||
# Reuse match count from _score_keyword_match to avoid scanning twice
|
||||
if reasoning_match_count >= 2 and scored_above_simple:
|
||||
if reasoning_match_count >= 2 and clears_override_floor:
|
||||
return ComplexityTier.REASONING, weighted_score, tuple(signals), "reasoning_override"
|
||||
|
||||
# Map score to tier
|
||||
|
|
@ -1039,6 +1039,18 @@ class ComplexityRouter(CustomLogger):
|
|||
|
||||
return tier, weighted_score, tuple(signals), "heuristic_scorer"
|
||||
|
||||
def _effective_reasoning_override_min_score(self) -> float:
|
||||
"""The score a request must reach before the reasoning-marker override may promote it.
|
||||
|
||||
Unset tracks the SIMPLE/MEDIUM boundary, so moving that boundary moves this floor with it
|
||||
and the override still cannot rescue a request the mapping would call SIMPLE. An explicit
|
||||
0 is a real floor, not an absent one, so the comparison is against None.
|
||||
"""
|
||||
configured: Final = self.config.reasoning_override_min_score
|
||||
if configured is None:
|
||||
return self._effective_tier_boundaries()["simple_medium"]
|
||||
return configured
|
||||
|
||||
def _effective_tier_boundaries(self) -> StandardLoggingRoutingDecisionTierBoundaries:
|
||||
"""The tier boundaries in effect, with the documented defaults filled in.
|
||||
|
||||
|
|
@ -1095,6 +1107,7 @@ class ComplexityRouter(CustomLogger):
|
|||
if score is not None:
|
||||
decision["score"] = score
|
||||
decision["tier_boundaries"] = self._effective_tier_boundaries()
|
||||
decision["reasoning_override_min_score"] = self._effective_reasoning_override_min_score()
|
||||
if signals:
|
||||
# Stored as a list because this record is serialized to JSON for the spend
|
||||
# log and read back as an array by the dashboard; a sequence type that only
|
||||
|
|
|
|||
|
|
@ -481,6 +481,15 @@ class ComplexityRouterConfig(BaseModel):
|
|||
),
|
||||
)
|
||||
|
||||
reasoning_override_min_score: float | None = Field(
|
||||
default=None,
|
||||
description=(
|
||||
"Minimum weighted score a request must reach before 2+ reasoning markers may promote it to the "
|
||||
"reasoning tier. Unset tracks tier_boundaries.simple_medium, so the override never rescues a "
|
||||
"request the scorer placed in the cheapest tier; 0 restores the unconditional override"
|
||||
),
|
||||
)
|
||||
|
||||
# Token count thresholds
|
||||
token_thresholds: dict[str, int] = Field(
|
||||
default_factory=lambda: DEFAULT_TOKEN_THRESHOLDS.copy(),
|
||||
|
|
|
|||
|
|
@ -2836,6 +2836,7 @@ class StandardLoggingRoutingDecision(TypedDict, total=False):
|
|||
classifier_cost: float
|
||||
escalated: bool
|
||||
tier_boundaries: StandardLoggingRoutingDecisionTierBoundaries
|
||||
reasoning_override_min_score: ReadOnly[float]
|
||||
conversation_continuing: bool
|
||||
savings_baseline_model: str
|
||||
savings_baseline_deployment_id: str
|
||||
|
|
@ -2860,6 +2861,7 @@ DERIVED_ROUTING_DECISION_FIELDS: Final[frozenset[str]] = frozenset(
|
|||
"classifier_cost",
|
||||
"escalated",
|
||||
"tier_boundaries",
|
||||
"reasoning_override_min_score",
|
||||
"conversation_continuing",
|
||||
"savings_baseline_model",
|
||||
"savings_baseline_deployment_id",
|
||||
|
|
|
|||
|
|
@ -285,6 +285,73 @@ class TestReasoningMarkerScoring:
|
|||
assert score == complexity_router.config.tier_boundaries["simple_medium"]
|
||||
assert tier == ComplexityTier.REASONING
|
||||
|
||||
def test_explicit_zero_floor_restores_the_unconditional_override(self, mock_router_instance, basic_config):
|
||||
"""0 is a real floor, not an absent one, so the markers alone promote again."""
|
||||
router = ComplexityRouter(
|
||||
model_name="test-complexity-router",
|
||||
litellm_router_instance=mock_router_instance,
|
||||
complexity_router_config={**basic_config, "reasoning_override_min_score": 0.0},
|
||||
)
|
||||
tier, score, _ = router.classify("hi, step by step, pros and cons")
|
||||
assert score < router.config.tier_boundaries["simple_medium"]
|
||||
assert tier == ComplexityTier.REASONING
|
||||
|
||||
def test_floor_defaults_to_simple_medium_and_follows_it(self, mock_router_instance, basic_config):
|
||||
"""Unset tracks simple_medium, so moving that boundary moves the floor with it."""
|
||||
prompt = (
|
||||
"Give me the pros and cons, step by step, of moving our checkout service "
|
||||
"to an event-driven architecture."
|
||||
)
|
||||
low = ComplexityRouter(
|
||||
model_name="test-complexity-router",
|
||||
litellm_router_instance=mock_router_instance,
|
||||
complexity_router_config={**basic_config, "tier_boundaries": {"simple_medium": 0.20}},
|
||||
)
|
||||
high = ComplexityRouter(
|
||||
model_name="test-complexity-router",
|
||||
litellm_router_instance=mock_router_instance,
|
||||
complexity_router_config={**basic_config, "tier_boundaries": {"simple_medium": 0.30}},
|
||||
)
|
||||
assert low._effective_reasoning_override_min_score() == 0.20
|
||||
assert high._effective_reasoning_override_min_score() == 0.30
|
||||
assert low.classify(prompt)[0] == ComplexityTier.REASONING
|
||||
assert high.classify(prompt)[0] != ComplexityTier.REASONING
|
||||
|
||||
def test_explicit_floor_overrides_the_boundary(self, mock_router_instance, basic_config):
|
||||
"""A configured floor decides the override, not simple_medium."""
|
||||
prompt = (
|
||||
"Give me the pros and cons, step by step, of moving our checkout service "
|
||||
"to an event-driven architecture."
|
||||
)
|
||||
router = ComplexityRouter(
|
||||
model_name="test-complexity-router",
|
||||
litellm_router_instance=mock_router_instance,
|
||||
complexity_router_config={
|
||||
**basic_config,
|
||||
"tier_boundaries": {"simple_medium": 0.10},
|
||||
"reasoning_override_min_score": 0.90,
|
||||
},
|
||||
)
|
||||
tier, score, _ = router.classify(prompt)
|
||||
assert score > router.config.tier_boundaries["simple_medium"]
|
||||
assert router._effective_reasoning_override_min_score() == 0.90
|
||||
assert tier != ComplexityTier.REASONING
|
||||
|
||||
def test_configured_floor_is_applied_with_greater_or_equal(self, mock_router_instance, basic_config):
|
||||
"""A score landing exactly on the configured floor still promotes."""
|
||||
prompt = (
|
||||
"Give me the pros and cons, step by step, of moving our checkout service "
|
||||
"to an event-driven architecture."
|
||||
)
|
||||
router = ComplexityRouter(
|
||||
model_name="test-complexity-router",
|
||||
litellm_router_instance=mock_router_instance,
|
||||
complexity_router_config={**basic_config, "reasoning_override_min_score": 0.25},
|
||||
)
|
||||
tier, score, _ = router.classify(prompt)
|
||||
assert score == 0.25
|
||||
assert tier == ComplexityTier.REASONING
|
||||
|
||||
def test_system_prompt_reasoning_not_counted(self, complexity_router):
|
||||
"""Reasoning markers in system prompt should not count for override."""
|
||||
user_prompt = "What is 2+2?"
|
||||
|
|
|
|||
|
|
@ -58,17 +58,32 @@ const scoringExplanation = (value: ComplexityRouterConfigValue): string => {
|
|||
const boundaryRanges = (
|
||||
shipped: Record<string, number> | undefined,
|
||||
overrides: Record<string, number> | undefined,
|
||||
): { simpleMedium: string; mediumComplex: string; complexReasoning: string } | null => {
|
||||
reasoningOverrideMinScore: number | undefined,
|
||||
): {
|
||||
simpleMedium: string;
|
||||
mediumComplex: string;
|
||||
complexReasoning: string;
|
||||
reasoningOverrideFloor: string;
|
||||
} | null => {
|
||||
const effective: Record<string, number> = { ...shipped, ...overrides };
|
||||
const [low, mid, high] = [effective.simple_medium, effective.medium_complex, effective.complex_reasoning];
|
||||
if (low === undefined || mid === undefined || high === undefined) return null;
|
||||
return { simpleMedium: low.toFixed(2), mediumComplex: mid.toFixed(2), complexReasoning: high.toFixed(2) };
|
||||
return {
|
||||
simpleMedium: low.toFixed(2),
|
||||
mediumComplex: mid.toFixed(2),
|
||||
complexReasoning: high.toFixed(2),
|
||||
reasoningOverrideFloor: (reasoningOverrideMinScore ?? low).toFixed(2),
|
||||
};
|
||||
};
|
||||
|
||||
const HowClassificationWorks: React.FC<{ value: ComplexityRouterConfigValue }> = ({ value }) => {
|
||||
// The shipped boundaries come from the proxy, so this card cannot state ranges the router stopped using.
|
||||
const { data: scorerDefaults, isError } = useComplexityScorerDefaults();
|
||||
const ranges = boundaryRanges(scorerDefaults?.tier_boundaries, value.tier_boundaries);
|
||||
const ranges = boundaryRanges(
|
||||
scorerDefaults?.tier_boundaries,
|
||||
value.tier_boundaries,
|
||||
value.reasoning_override_min_score,
|
||||
);
|
||||
|
||||
return (
|
||||
<Card className="bg-gray-50 mt-4">
|
||||
|
|
@ -93,7 +108,7 @@ const HowClassificationWorks: React.FC<{ value: ComplexityRouterConfigValue }> =
|
|||
</li>
|
||||
<li>
|
||||
<strong>{effectiveTierLabel("REASONING", value.tier_labels)}</strong>: Score > {ranges.complexReasoning}{" "}
|
||||
(or 2+ reasoning markers with a score of at least {ranges.simpleMedium})
|
||||
(or 2+ reasoning markers with a score of at least {ranges.reasoningOverrideFloor})
|
||||
</li>
|
||||
</ul>
|
||||
)}
|
||||
|
|
|
|||
|
|
@ -136,6 +136,11 @@ export interface ComplexityRouterConfigValue {
|
|||
tier_boundaries?: TierBoundaries;
|
||||
token_thresholds?: TokenThresholds;
|
||||
dimension_weights?: DimensionWeights;
|
||||
/**
|
||||
* Score floor the reasoning-marker override must clear. Undefined keeps the key out of the payload, so the
|
||||
* floor tracks tier_boundaries.simple_medium; an explicit 0 is a real floor that promotes on the markers alone.
|
||||
*/
|
||||
reasoning_override_min_score?: number;
|
||||
}
|
||||
|
||||
interface ComplexityRouterConfigProps {
|
||||
|
|
|
|||
|
|
@ -95,6 +95,57 @@ describe("HeuristicScoringConfig", () => {
|
|||
expect(onChange).toHaveBeenCalledWith(expect.objectContaining({ token_thresholds: undefined }));
|
||||
});
|
||||
|
||||
it("shows the boundary an untouched override floor tracks, rather than a fixed number", async () => {
|
||||
await render(BASE);
|
||||
|
||||
const field = screen.getByLabelText("Minimum score");
|
||||
expect(field).toHaveValue("");
|
||||
expect(field).toHaveAttribute("placeholder", SHIPPED_SCORER_DEFAULTS.tier_boundaries.simple_medium.toFixed(2));
|
||||
});
|
||||
|
||||
it("tracks the operator's own Simple to Medium override, not the shipped boundary", async () => {
|
||||
await render({ ...BASE, tier_boundaries: { simple_medium: 0.42, medium_complex: 0.5, complex_reasoning: 0.7 } });
|
||||
|
||||
expect(screen.getByLabelText("Minimum score")).toHaveAttribute("placeholder", "0.42");
|
||||
});
|
||||
|
||||
// 0 restores an unconditional override, so it has to reach the config as 0 rather than as "untouched".
|
||||
it("commits an explicit 0 override floor", async () => {
|
||||
const onChange = await render(BASE);
|
||||
fireEvent.change(screen.getByLabelText("Minimum score"), { target: { value: "0" } });
|
||||
|
||||
expect(onChange.mock.calls.at(-1)?.[0]).toMatchObject({ reasoning_override_min_score: 0 });
|
||||
});
|
||||
|
||||
it("renders a stored 0 as 0 rather than as an untouched field", async () => {
|
||||
await render({ ...BASE, reasoning_override_min_score: 0 });
|
||||
|
||||
expect(screen.getByLabelText("Minimum score")).toHaveValue("0");
|
||||
});
|
||||
|
||||
it("counts a set override floor among the overrides", () => {
|
||||
renderWithProviders(
|
||||
<HeuristicScoringConfig value={{ ...BASE, reasoning_override_min_score: 0 }} onChange={vi.fn()} />,
|
||||
);
|
||||
|
||||
expect(screen.getByTestId("advanced-scoring-override-count")).toHaveTextContent("1 override");
|
||||
});
|
||||
|
||||
it("clamps the override floor to the score range", async () => {
|
||||
const onChange = await render(BASE);
|
||||
fireEvent.change(screen.getByLabelText("Minimum score"), { target: { value: "9" } });
|
||||
|
||||
expect(onChange.mock.calls.at(-1)?.[0]).toMatchObject({ reasoning_override_min_score: 1 });
|
||||
});
|
||||
|
||||
it("resets the override floor back to tracking the boundary", async () => {
|
||||
const onChange = await render({ ...BASE, reasoning_override_min_score: 0 });
|
||||
|
||||
await userEvent.click(screen.getAllByRole("button", { name: "Reset to defaults" }).at(-1)!);
|
||||
|
||||
expect(onChange).toHaveBeenCalledWith(expect.objectContaining({ reasoning_override_min_score: undefined }));
|
||||
});
|
||||
|
||||
it("flags decreasing boundaries as an error without blocking the save", async () => {
|
||||
const bad = { ...BASE, tier_boundaries: { simple_medium: 0.5, medium_complex: 0.2, complex_reasoning: 0.6 } };
|
||||
await render(bad);
|
||||
|
|
@ -136,6 +187,18 @@ describe("ClassificationMethodConfig scorer gating", () => {
|
|||
expect(screen.queryByText(/0.15/)).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("states the configured override floor in the reasoning-marker aside, not the boundary", () => {
|
||||
renderWithProviders(<ClassificationMethodConfig {...props} value={{ ...BASE, reasoning_override_min_score: 0 }} />);
|
||||
|
||||
expect(screen.getByText(/2\+ reasoning markers with a score of at least 0\.00/)).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("falls back to the Simple to Medium boundary when no override floor is set", () => {
|
||||
renderWithProviders(<ClassificationMethodConfig {...props} value={BASE} />);
|
||||
|
||||
expect(screen.getByText(/2\+ reasoning markers with a score of at least 0\.15/)).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("renders a row for every scored dimension", async () => {
|
||||
await render(BASE);
|
||||
|
||||
|
|
|
|||
|
|
@ -23,6 +23,8 @@ interface GroupSpec {
|
|||
labels: Record<string, string>;
|
||||
}
|
||||
|
||||
const OVERRIDE_FLOOR_ID = "reasoning-override-min-score";
|
||||
|
||||
const GROUPS: GroupSpec[] = [
|
||||
{
|
||||
group: "tier_boundaries",
|
||||
|
|
@ -88,7 +90,12 @@ const HeuristicScoringConfig: React.FC<HeuristicScoringConfigProps> = ({ value,
|
|||
// falls back to the default model, so there is nothing here to configure.
|
||||
const scorerRuns = heuristicScoringRole(value) !== "never";
|
||||
|
||||
const overrides = GROUPS.filter((spec) => value[spec.group] !== undefined).length;
|
||||
// What an untouched override floor follows: the boundary in effect, override included, not the shipped one.
|
||||
const trackedFloor: number | undefined = { ...defaults?.tier_boundaries, ...value.tier_boundaries }.simple_medium;
|
||||
|
||||
const overrides =
|
||||
GROUPS.filter((spec) => value[spec.group] !== undefined).length +
|
||||
(value.reasoning_override_min_score !== undefined ? 1 : 0);
|
||||
|
||||
// min/max are inert on a text input, and a plain number input renders Number("0.") as "0" so a decimal
|
||||
// cannot be typed. Hence the local draft plus an explicit clamp here.
|
||||
|
|
@ -102,6 +109,12 @@ const HeuristicScoringConfig: React.FC<HeuristicScoringConfigProps> = ({ value,
|
|||
});
|
||||
};
|
||||
|
||||
const commitOverrideFloor = (raw: string) => {
|
||||
const parsed = Number(raw);
|
||||
if (raw.trim() === "" || !Number.isFinite(parsed)) return;
|
||||
onChange({ ...value, reasoning_override_min_score: Math.min(1, Math.max(-1, parsed)) });
|
||||
};
|
||||
|
||||
if (!scorerRuns) return null;
|
||||
|
||||
return (
|
||||
|
|
@ -215,6 +228,50 @@ const HeuristicScoringConfig: React.FC<HeuristicScoringConfigProps> = ({ value,
|
|||
</section>
|
||||
);
|
||||
})}
|
||||
|
||||
<section className="space-y-2">
|
||||
<div className="flex items-center justify-between">
|
||||
<span className="text-sm font-medium">Reasoning override floor</span>
|
||||
{value.reasoning_override_min_score !== undefined && (
|
||||
<Button
|
||||
type="button"
|
||||
variant="link"
|
||||
size="xs"
|
||||
onClick={() => onChange({ ...value, reasoning_override_min_score: undefined })}
|
||||
>
|
||||
Reset to defaults
|
||||
</Button>
|
||||
)}
|
||||
</div>
|
||||
<p className="text-xs text-muted-foreground">
|
||||
Two or more reasoning markers promote a request to the reasoning tier, but only once its weighted
|
||||
score reaches this floor.{" "}
|
||||
{trackedFloor === undefined
|
||||
? "Left untouched, it tracks the Simple to Medium boundary."
|
||||
: `Left untouched, it tracks the Simple to Medium boundary, currently ${trackedFloor.toFixed(2)}.`}{" "}
|
||||
Set it to 0 to promote on the markers alone.
|
||||
</p>
|
||||
<div className="flex items-center gap-3">
|
||||
<Label htmlFor={OVERRIDE_FLOOR_ID} className="w-44 text-xs font-normal">
|
||||
Minimum score
|
||||
</Label>
|
||||
<Input
|
||||
id={OVERRIDE_FLOOR_ID}
|
||||
type="text"
|
||||
inputMode="decimal"
|
||||
className="w-28"
|
||||
placeholder={trackedFloor === undefined ? undefined : trackedFloor.toFixed(2)}
|
||||
value={
|
||||
draft?.id === OVERRIDE_FLOOR_ID ? draft.raw : value.reasoning_override_min_score?.toString() ?? ""
|
||||
}
|
||||
onChange={(event) => {
|
||||
setDraft({ id: OVERRIDE_FLOOR_ID, raw: event.target.value });
|
||||
commitOverrideFloor(event.target.value);
|
||||
}}
|
||||
onBlur={() => setDraft(null)}
|
||||
/>
|
||||
</div>
|
||||
</section>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
|
|
|
|||
|
|
@ -285,6 +285,28 @@ describe("AddAutoRouterTab", () => {
|
|||
});
|
||||
});
|
||||
|
||||
// The scalar floor is the one scorer knob with no group dict behind it, so its wiring into the create
|
||||
// payload is only proven end to end. 0 is the case a truthy check would silently drop.
|
||||
it("carries a reasoning override floor of 0 through to the create payload", async () => {
|
||||
const user = userEvent.setup();
|
||||
vi.mocked(getMissingTiersError).mockReturnValue(null);
|
||||
|
||||
renderWithProviders(<Harness />);
|
||||
|
||||
await user.type(screen.getByPlaceholderText(/smart_router/i), "override-floor-router");
|
||||
expandDetailedConfiguration();
|
||||
await user.click(screen.getByText("Advanced: Classification Method"));
|
||||
await user.click(await screen.findByText("Advanced scoring"));
|
||||
fireEvent.change(await screen.findByLabelText("Minimum score"), { target: { value: "0" } });
|
||||
|
||||
await user.click(screen.getByRole("button", { name: /add auto router/i }));
|
||||
|
||||
await waitFor(() => expect(handleAddAutoRouterSubmit).toHaveBeenCalled());
|
||||
expect(vi.mocked(handleAddAutoRouterSubmit).mock.calls.at(-1)?.[0].complexity_router_config).toMatchObject({
|
||||
reasoning_override_min_score: 0,
|
||||
});
|
||||
});
|
||||
|
||||
it("carries session affinity turned on through to the create payload", async () => {
|
||||
const user = userEvent.setup();
|
||||
vi.mocked(getMissingTiersError).mockReturnValue(null);
|
||||
|
|
|
|||
|
|
@ -361,6 +361,7 @@ const AddAutoRouterTab: React.FC<AddAutoRouterTabProps> = ({
|
|||
tierBoundaries: complexityRouterConfig.tier_boundaries,
|
||||
tokenThresholds: complexityRouterConfig.token_thresholds,
|
||||
dimensionWeights: complexityRouterConfig.dimension_weights,
|
||||
reasoningOverrideMinScore: complexityRouterConfig.reasoning_override_min_score,
|
||||
};
|
||||
|
||||
const submitRecommendedRouter = async (name: string) => {
|
||||
|
|
|
|||
|
|
@ -601,6 +601,27 @@ describe("buildComplexityRouterConfig scorer knobs", () => {
|
|||
it("drops them when the classifier falls back to the default model and nothing is scored", () => {
|
||||
expect(buildComplexityRouterConfig(llmWithDefaultFallback)).not.toHaveProperty("tier_boundaries");
|
||||
});
|
||||
|
||||
it("omits the reasoning override floor while untouched, so it keeps tracking simple_medium", () => {
|
||||
expect(buildComplexityRouterConfig(baseParams)).not.toHaveProperty("reasoning_override_min_score");
|
||||
});
|
||||
|
||||
it("emits the reasoning override floor that was set", () => {
|
||||
const config = buildComplexityRouterConfig({ ...baseParams, reasoningOverrideMinScore: 0.4 });
|
||||
expect(config.reasoning_override_min_score).toBe(0.4);
|
||||
});
|
||||
|
||||
// 0 is an unconditional override, not an absent knob, so a falsy check here would silently discard it.
|
||||
it("emits an explicit 0 reasoning override floor", () => {
|
||||
const config = buildComplexityRouterConfig({ ...baseParams, reasoningOverrideMinScore: 0 });
|
||||
expect(config.reasoning_override_min_score).toBe(0);
|
||||
});
|
||||
|
||||
it("drops the reasoning override floor when nothing is scored", () => {
|
||||
expect(buildComplexityRouterConfig({ ...llmWithDefaultFallback, reasoningOverrideMinScore: 0 })).not.toHaveProperty(
|
||||
"reasoning_override_min_score",
|
||||
);
|
||||
});
|
||||
});
|
||||
|
||||
describe("plan-mode minimum tier", () => {
|
||||
|
|
|
|||
|
|
@ -46,6 +46,7 @@ interface ScorerKnobInputs {
|
|||
tierBoundaries: TierBoundaries | undefined;
|
||||
tokenThresholds: TokenThresholds | undefined;
|
||||
dimensionWeights: DimensionWeights | undefined;
|
||||
reasoningOverrideMinScore: number | undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
|
|
@ -59,6 +60,7 @@ const scorerKnobPayload = ({
|
|||
tierBoundaries,
|
||||
tokenThresholds,
|
||||
dimensionWeights,
|
||||
reasoningOverrideMinScore,
|
||||
}: ScorerKnobInputs) =>
|
||||
heuristicScoringRoleFor(classifierType, classifierFallback) === "never"
|
||||
? {}
|
||||
|
|
@ -66,6 +68,7 @@ const scorerKnobPayload = ({
|
|||
...(tierBoundaries && { tier_boundaries: tierBoundaries }),
|
||||
...(tokenThresholds && { token_thresholds: tokenThresholds }),
|
||||
...(dimensionWeights && { dimension_weights: dimensionWeights }),
|
||||
...(reasoningOverrideMinScore !== undefined && { reasoning_override_min_score: reasoningOverrideMinScore }),
|
||||
};
|
||||
|
||||
export interface BuildComplexityRouterConfigParams {
|
||||
|
|
@ -95,6 +98,7 @@ export interface BuildComplexityRouterConfigParams {
|
|||
tierBoundaries?: TierBoundaries;
|
||||
tokenThresholds?: TokenThresholds;
|
||||
dimensionWeights?: DimensionWeights;
|
||||
reasoningOverrideMinScore?: number;
|
||||
}
|
||||
|
||||
export interface ComplexityRouterConfigPayload {
|
||||
|
|
@ -124,6 +128,7 @@ export interface ComplexityRouterConfigPayload {
|
|||
tier_boundaries?: TierBoundaries;
|
||||
token_thresholds?: TokenThresholds;
|
||||
dimension_weights?: DimensionWeights;
|
||||
reasoning_override_min_score?: number;
|
||||
}
|
||||
|
||||
const TIER_KEYS: Array<keyof ComplexityTiers> = ["SIMPLE", "MEDIUM", "COMPLEX", "REASONING"];
|
||||
|
|
@ -230,11 +235,19 @@ export const buildComplexityRouterConfig = ({
|
|||
tierBoundaries,
|
||||
tokenThresholds,
|
||||
dimensionWeights,
|
||||
reasoningOverrideMinScore,
|
||||
}: BuildComplexityRouterConfigParams): ComplexityRouterConfigPayload => {
|
||||
const cleanedEscalationKeywords = escalationKeywords.map((keyword) => keyword.trim()).filter(Boolean);
|
||||
const cleanedKeywordTierRules = serializeKeywordTierRules(keywordTierRules);
|
||||
const cleanedTierLabels = serializeTierLabels(tierLabels);
|
||||
const scorerInputs = { classifierType, classifierFallback, tierBoundaries, tokenThresholds, dimensionWeights };
|
||||
const scorerInputs = {
|
||||
classifierType,
|
||||
classifierFallback,
|
||||
tierBoundaries,
|
||||
tokenThresholds,
|
||||
dimensionWeights,
|
||||
reasoningOverrideMinScore,
|
||||
};
|
||||
const scorerKnobs = scorerKnobPayload(scorerInputs);
|
||||
|
||||
return {
|
||||
|
|
|
|||
|
|
@ -4,6 +4,7 @@ import { heuristicScoringRoleFor } from "./ComplexityRouterConfig";
|
|||
import {
|
||||
dimensionLabel,
|
||||
hydrateDimensionWeights,
|
||||
hydrateReasoningOverrideMinScore,
|
||||
hydrateTierBoundaries,
|
||||
hydrateTokenThresholds,
|
||||
weightTotal,
|
||||
|
|
@ -46,6 +47,21 @@ describe("hydrating the scorer knobs", () => {
|
|||
expect(weightTotal({ a: 0.1, b: 0.2 })).toBe(0.3);
|
||||
});
|
||||
|
||||
it.each([[undefined], [null], ["0.15"], [Number.NaN], [Number.POSITIVE_INFINITY], [{ value: 0.15 }]])(
|
||||
"hydrates the reasoning override floor %s to undefined",
|
||||
(raw) => {
|
||||
expect(hydrateReasoningOverrideMinScore(raw)).toBeUndefined();
|
||||
},
|
||||
);
|
||||
|
||||
// A stored 0 is an unconditional override, so hydrating it to undefined would silently retune the router
|
||||
// back to tracking simple_medium on the next save.
|
||||
it("hydrates a stored reasoning override floor, zero and negatives included", () => {
|
||||
expect(hydrateReasoningOverrideMinScore(0)).toBe(0);
|
||||
expect(hydrateReasoningOverrideMinScore(-0.3)).toBe(-0.3);
|
||||
expect(hydrateReasoningOverrideMinScore(0.42)).toBe(0.42);
|
||||
});
|
||||
|
||||
it("falls back to the raw key when a dimension has no label yet", () => {
|
||||
expect(dimensionLabel("codePresence")).toBe("Code presence");
|
||||
expect(dimensionLabel("somethingNew")).toBe("somethingNew");
|
||||
|
|
|
|||
|
|
@ -44,5 +44,12 @@ export const hydrateTokenThresholds = (raw: unknown): TokenThresholds | undefine
|
|||
|
||||
export const hydrateDimensionWeights = (raw: unknown): DimensionWeights | undefined => hydrateNumericMap(raw);
|
||||
|
||||
/**
|
||||
* The scalar counterpart of hydrateNumericMap: absent hydrates to undefined so an untouched save keeps the
|
||||
* floor tracking tier_boundaries.simple_medium, while a stored 0 hydrates to 0, which is a real floor.
|
||||
*/
|
||||
export const hydrateReasoningOverrideMinScore = (raw: unknown): number | undefined =>
|
||||
typeof raw === "number" && Number.isFinite(raw) ? raw : undefined;
|
||||
|
||||
export const weightTotal = (weights: DimensionWeights): number =>
|
||||
Math.round(Object.values(weights).reduce((total, weight) => total + weight, 0) * 100) / 100;
|
||||
|
|
|
|||
|
|
@ -313,6 +313,42 @@ describe("buildUpdatedComplexityRouterConfig scorer knobs", () => {
|
|||
it("never invents knobs for a router that never had them", () => {
|
||||
expect(buildUpdatedComplexityRouterConfig(STORED, FORM_VALUE)).not.toHaveProperty("tier_boundaries");
|
||||
});
|
||||
|
||||
// 0 is an unconditional reasoning override. Treating it as unset here would quietly retune the router
|
||||
// back to tracking simple_medium on the next save.
|
||||
it("round-trips a stored reasoning override floor of 0", () => {
|
||||
const result = buildUpdatedComplexityRouterConfig(
|
||||
{ ...STORED, reasoning_override_min_score: 0 },
|
||||
{ ...FORM_VALUE, reasoning_override_min_score: 0 },
|
||||
);
|
||||
expect(result.reasoning_override_min_score).toBe(0);
|
||||
});
|
||||
|
||||
it("writes a newly set reasoning override floor over the stored one", () => {
|
||||
const result = buildUpdatedComplexityRouterConfig(
|
||||
{ ...STORED, reasoning_override_min_score: 0 },
|
||||
{ ...FORM_VALUE, reasoning_override_min_score: 0.5 },
|
||||
);
|
||||
expect(result.reasoning_override_min_score).toBe(0.5);
|
||||
});
|
||||
|
||||
it("drops a stored reasoning override floor when the operator resets it", () => {
|
||||
const result = buildUpdatedComplexityRouterConfig({ ...STORED, reasoning_override_min_score: 0.5 }, FORM_VALUE);
|
||||
|
||||
expect(result).not.toHaveProperty("reasoning_override_min_score");
|
||||
expect(result.some_future_backend_key).toEqual({ nested: true });
|
||||
});
|
||||
|
||||
it("drops the reasoning override floor on a router whose scorer never runs", () => {
|
||||
const neverScores = {
|
||||
...FORM_VALUE,
|
||||
reasoning_override_min_score: 0,
|
||||
classifier_type: "llm" as const,
|
||||
classifier_fallback: "default_model" as const,
|
||||
};
|
||||
const result = buildUpdatedComplexityRouterConfig({ ...STORED, reasoning_override_min_score: 0 }, neverScores);
|
||||
expect(result).not.toHaveProperty("reasoning_override_min_score");
|
||||
});
|
||||
});
|
||||
|
||||
describe("buildUpdatedComplexityRouterConfig plan-mode minimum tier", () => {
|
||||
|
|
|
|||
|
|
@ -30,6 +30,7 @@ import { DEFAULT_MATCH_THRESHOLD } from "../add_model/SemanticKeywordMatching";
|
|||
import { hydrateKeywordTierRules, serializeKeywordTierRules } from "../add_model/complexity_router_keywords";
|
||||
import {
|
||||
hydrateDimensionWeights,
|
||||
hydrateReasoningOverrideMinScore,
|
||||
hydrateTierBoundaries,
|
||||
hydrateTokenThresholds,
|
||||
} from "../add_model/heuristic_scoring_knobs";
|
||||
|
|
@ -84,6 +85,7 @@ const MANAGED_COMPLEXITY_ROUTER_KEYS = new Set([
|
|||
"tier_boundaries",
|
||||
"token_thresholds",
|
||||
"dimension_weights",
|
||||
"reasoning_override_min_score",
|
||||
]);
|
||||
|
||||
// Managed only when the caller passes the corresponding state. A caller that does not render
|
||||
|
|
@ -200,6 +202,10 @@ export const buildUpdatedComplexityRouterConfig = (
|
|||
...(scorerRuns && value.tier_boundaries !== undefined && { tier_boundaries: value.tier_boundaries }),
|
||||
...(scorerRuns && value.token_thresholds !== undefined && { token_thresholds: value.token_thresholds }),
|
||||
...(scorerRuns && value.dimension_weights !== undefined && { dimension_weights: value.dimension_weights }),
|
||||
...(scorerRuns &&
|
||||
value.reasoning_override_min_score !== undefined && {
|
||||
reasoning_override_min_score: value.reasoning_override_min_score,
|
||||
}),
|
||||
};
|
||||
};
|
||||
|
||||
|
|
@ -367,6 +373,7 @@ const EditAutoRouterModal: React.FC<EditAutoRouterModalProps> = ({
|
|||
tier_boundaries: hydrateTierBoundaries(parsedConfig.tier_boundaries),
|
||||
token_thresholds: hydrateTokenThresholds(parsedConfig.token_thresholds),
|
||||
dimension_weights: hydrateDimensionWeights(parsedConfig.dimension_weights),
|
||||
reasoning_override_min_score: hydrateReasoningOverrideMinScore(parsedConfig.reasoning_override_min_score),
|
||||
session_affinity:
|
||||
typeof parsedConfig.session_affinity === "boolean"
|
||||
? parsedConfig.session_affinity
|
||||
|
|
|
|||
|
|
@ -60,7 +60,9 @@ describe("RoutingDecisionCard", () => {
|
|||
/>,
|
||||
);
|
||||
expect(
|
||||
screen.getByText("Heuristic, REASONING override (2 or more reasoning markers, score above the lowest tier)"),
|
||||
screen.getByText(
|
||||
"Heuristic, REASONING override (2 or more reasoning markers, score of at least the Simple to Medium boundary)",
|
||||
),
|
||||
).toBeInTheDocument();
|
||||
expect(screen.getByText("0.20")).toBeInTheDocument();
|
||||
// The score did not decide this tier, so NO band explanation may render at all.
|
||||
|
|
@ -191,7 +193,9 @@ describe("RoutingDecisionCard", () => {
|
|||
render(<RoutingDecisionCard decision={{ ...heuristic, cause: "reasoning_override", signals: undefined }} />);
|
||||
expect(screen.queryByText(/SIMPLE|MEDIUM|COMPLEX|at or above/)).not.toBeInTheDocument();
|
||||
expect(
|
||||
screen.getByText("Heuristic, REASONING override (2 or more reasoning markers, score above the lowest tier)"),
|
||||
screen.getByText(
|
||||
"Heuristic, REASONING override (2 or more reasoning markers, score of at least the Simple to Medium boundary)",
|
||||
),
|
||||
).toBeInTheDocument();
|
||||
});
|
||||
|
||||
|
|
@ -217,10 +221,41 @@ describe("RoutingDecisionCard", () => {
|
|||
<RoutingDecisionCard decision={{ ...heuristic, cause: "reasoning_override", score: 0.2, tier_label: "Deep" }} />,
|
||||
);
|
||||
expect(
|
||||
screen.getByText("Heuristic, Deep override (2 or more reasoning markers, score above the lowest tier)"),
|
||||
screen.getByText(
|
||||
"Heuristic, Deep override (2 or more reasoning markers, score of at least the Simple to Medium boundary)",
|
||||
),
|
||||
).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("states the floor the override actually cleared", () => {
|
||||
render(
|
||||
<RoutingDecisionCard
|
||||
decision={{ ...heuristic, cause: "reasoning_override", score: 0.2, reasoning_override_min_score: 0.05 }}
|
||||
/>,
|
||||
);
|
||||
expect(
|
||||
screen.getByText("Heuristic, REASONING override (2 or more reasoning markers, score of at least 0.05)"),
|
||||
).toBeInTheDocument();
|
||||
});
|
||||
|
||||
// A floor of 0 is an unconditional override, so a falsy check here would print the "before this change"
|
||||
// wording on a row that recorded a real floor.
|
||||
it("states a recorded floor of 0 rather than treating it as unrecorded", () => {
|
||||
render(
|
||||
<RoutingDecisionCard
|
||||
decision={{ ...heuristic, cause: "reasoning_override", score: 0.2, reasoning_override_min_score: 0 }}
|
||||
/>,
|
||||
);
|
||||
expect(
|
||||
screen.getByText("Heuristic, REASONING override (2 or more reasoning markers, score of at least 0)"),
|
||||
).toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("never prints undefined on a row logged before the floor was recorded", () => {
|
||||
render(<RoutingDecisionCard decision={{ ...heuristic, cause: "reasoning_override", score: 0.2 }} />);
|
||||
expect(screen.queryByText(/undefined/)).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("falls back to the raw cause for a value this build does not know", () => {
|
||||
render(<RoutingDecisionCard decision={{ cause: "some_future_cause", routed_model: "m" }} />);
|
||||
expect(screen.getByText("some_future_cause")).toBeInTheDocument();
|
||||
|
|
|
|||
|
|
@ -26,6 +26,7 @@ export interface RoutingDecision {
|
|||
classifier_model?: string;
|
||||
escalated?: boolean;
|
||||
tier_boundaries?: RoutingDecisionTierBoundaries;
|
||||
reasoning_override_min_score?: number;
|
||||
}
|
||||
|
||||
const ROUTER_TYPE_LABELS: Record<string, string> = {
|
||||
|
|
@ -65,14 +66,26 @@ function describePlanModeFloor(matchedKeyword: string | undefined): string {
|
|||
return "Plan-mode floor";
|
||||
}
|
||||
|
||||
/** Rows logged before the floor was recorded name what it tracked back then instead of a number. */
|
||||
function describeReasoningOverride(tierLabel: string | undefined, floor: number | undefined): string {
|
||||
const stated = floor === undefined ? "the Simple to Medium boundary" : String(floor);
|
||||
return `Heuristic, ${tierLabel ?? "REASONING"} override (2 or more reasoning markers, score of at least ${stated})`;
|
||||
}
|
||||
|
||||
function describeCause(decision: RoutingDecision): string {
|
||||
const { cause, classifier_model: classifierModel, matched_keyword: matchedKeyword, tier_label: tierLabel } = decision;
|
||||
const {
|
||||
cause,
|
||||
classifier_model: classifierModel,
|
||||
matched_keyword: matchedKeyword,
|
||||
tier_label: tierLabel,
|
||||
reasoning_override_min_score: overrideFloor,
|
||||
} = decision;
|
||||
|
||||
switch (cause) {
|
||||
case "heuristic_scorer":
|
||||
return "Heuristic scorer";
|
||||
case "reasoning_override":
|
||||
return `Heuristic, ${tierLabel ?? "REASONING"} override (2 or more reasoning markers, score above the lowest tier)`;
|
||||
return describeReasoningOverride(tierLabel, overrideFloor);
|
||||
case "llm_classifier":
|
||||
return classifierModel ? `LLM classifier (${classifierModel})` : "LLM classifier";
|
||||
case "literal_keyword_match":
|
||||
|
|
|
|||
7
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
7
ui/litellm-dashboard/src/lib/http/schema.d.ts
generated
vendored
|
|
@ -32480,6 +32480,11 @@ export interface components {
|
|||
* @description Keywords indicating reasoning-required content
|
||||
*/
|
||||
reasoning_keywords?: string[] | null;
|
||||
/**
|
||||
* Reasoning Override Min Score
|
||||
* @description Minimum weighted score a request must reach before 2+ reasoning markers may promote it to the reasoning tier. Unset tracks tier_boundaries.simple_medium, so the override never rescues a request the scorer placed in the cheapest tier; 0 restores the unconditional override
|
||||
*/
|
||||
reasoning_override_min_score?: number | null;
|
||||
/**
|
||||
* Reminder Markers
|
||||
* @description Override the delimiter pairs used to recognize and strip harness-injected reminder blocks before classification. A harness that wraps injected context differently per agent type (main, subagent, cron) lists every pair it emits. Replaces, rather than adds to, the built-in default of ('<system-reminder>', '</system-reminder>'), so a harness that also emits that pair lists it too. Matching is case-insensitive.
|
||||
|
|
@ -33502,6 +33507,8 @@ export interface components {
|
|||
escalation_keyword?: string;
|
||||
/** Matched Keyword */
|
||||
matched_keyword?: string;
|
||||
/** Reasoning Override Min Score */
|
||||
reasoning_override_min_score?: number;
|
||||
/** Request Type */
|
||||
request_type?: string;
|
||||
/** Routed Model */
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue