mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
fix: escalate a stalled keyword-forced tier, and let a blocked toggle clear
Two issues Bugbot found on #39809. A keyword_tier_rule forces its tier and returns before any classification runs, so stall escalation never reached that path even though keyword escalation did. That left the one path that can pin a weak model to a whole conversation as the one path a stall could not lift. Stall detection now resolves before the override branch and both paths bump. The dashboard switch disabled itself whenever session pinning or user-turn classification was on, including for a router that already had stall escalation enabled. The conflicting keys stayed set, the backend rejected the save, and the disabled switch was the only way to clear them. It now disables only the off-to-on direction.
This commit is contained in:
parent
939039f492
commit
544822b1a8
4 changed files with 50 additions and 11 deletions
|
|
@ -3053,6 +3053,14 @@ class ComplexityRouter(CustomLogger):
|
|||
|
||||
newest_ask: Final = _newest_turn_ask(resolved_messages, self._reminder_markers)
|
||||
escalation_keyword: Final = self._matched_escalation_keyword(newest_ask) if newest_ask is not None else None
|
||||
# Resolved here rather than beside the classifier because the keyword-override path below
|
||||
# returns before any classification runs, and a forced tier gets stuck for the same reason
|
||||
# a classified one does.
|
||||
stalled: Final = self.config.stall_escalation_enabled and detect_stalled_task(
|
||||
resolved_messages,
|
||||
window=self.config.stall_escalation_window,
|
||||
repeat_threshold=self.config.stall_escalation_repeat_threshold,
|
||||
)
|
||||
|
||||
plan_mode_sentinel: Final = self._matched_plan_mode_signal(request_kwargs, resolved_messages)
|
||||
plan_floor: Final = self._resolve_plan_mode_floor() if plan_mode_sentinel is not None else None
|
||||
|
|
@ -3082,10 +3090,11 @@ class ComplexityRouter(CustomLogger):
|
|||
|
||||
override: Final = await self._resolve_keyword_tier_override(user_message, request_kwargs)
|
||||
if override is not None:
|
||||
escalated_tier: Final = (
|
||||
keyword_bumped_tier: Final = (
|
||||
self._escalate_tier(override.tier) if escalation_keyword is not None else override.tier
|
||||
)
|
||||
keyword_escalated: Final = escalated_tier != override.tier
|
||||
escalated_tier: Final = self._escalate_tier(keyword_bumped_tier) if stalled else keyword_bumped_tier
|
||||
keyword_escalated: Final = keyword_bumped_tier != override.tier
|
||||
routed_tier: Final = (
|
||||
self._apply_plan_mode_floor(escalated_tier) if plan_floor is not None else escalated_tier
|
||||
)
|
||||
|
|
@ -3113,6 +3122,7 @@ class ComplexityRouter(CustomLogger):
|
|||
conversation_continuing=conversation_continuing,
|
||||
cause=keyword_cause,
|
||||
tier=routed_tier,
|
||||
signals=("stall_escalation",) if stalled else None,
|
||||
matched_keyword=plan_mode_sentinel if keyword_plan_floored else override.matched_keyword,
|
||||
escalation_keyword=escalation_keyword,
|
||||
escalated=keyword_escalated,
|
||||
|
|
@ -3136,14 +3146,6 @@ class ComplexityRouter(CustomLogger):
|
|||
escalated: Final = tier != classified_tier
|
||||
if escalated:
|
||||
signals = (*signals, "escalation")
|
||||
# Recomputed from this request's own tool calls, not remembered from a prior turn: the
|
||||
# bump lasts only as long as the recent tool calls still look stuck, and lifts itself
|
||||
# the moment they don't, with nothing to expire or leak past the task that earned it.
|
||||
stalled: Final = self.config.stall_escalation_enabled and detect_stalled_task(
|
||||
resolved_messages,
|
||||
window=self.config.stall_escalation_window,
|
||||
repeat_threshold=self.config.stall_escalation_repeat_threshold,
|
||||
)
|
||||
if stalled:
|
||||
tier = self._escalate_tier(tier)
|
||||
signals = (*signals, "stall_escalation")
|
||||
|
|
|
|||
|
|
@ -5990,6 +5990,33 @@ class TestStallEscalation:
|
|||
result = await router.async_pre_routing_hook(model="test-model", request_kwargs={}, messages=messages)
|
||||
assert result.model == "claude-sonnet-4-20250514" # SIMPLE -> MEDIUM (keyword) -> COMPLEX (stall)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_keyword_forced_tier_still_escalates_when_stalled(self, mock_router_instance, basic_config):
|
||||
"""A keyword rule forces its tier and returns before any classification runs, so
|
||||
without its own bump the one path that can pin a weak model to a whole conversation
|
||||
would be the one path a stall could never lift."""
|
||||
router = ComplexityRouter(
|
||||
model_name="test-router",
|
||||
litellm_router_instance=mock_router_instance,
|
||||
complexity_router_config={
|
||||
**basic_config,
|
||||
"stall_escalation_enabled": True,
|
||||
"keyword_tier_rules": [{"keywords": ["billing"], "tier": "SIMPLE"}],
|
||||
},
|
||||
)
|
||||
healthy = await router.async_pre_routing_hook(
|
||||
model="test-model", request_kwargs={}, messages=[{"role": "user", "content": "a billing question"}]
|
||||
)
|
||||
assert healthy.model == "gpt-4o-mini" # forced SIMPLE, nothing stuck
|
||||
|
||||
stalled = await router.async_pre_routing_hook(
|
||||
model="test-model",
|
||||
request_kwargs={},
|
||||
messages=[*_stalled_tool_history(), {"role": "user", "content": "a billing question"}],
|
||||
)
|
||||
assert stalled.model == "gpt-4o" # forced SIMPLE bumped to MEDIUM
|
||||
assert "stall_escalation" in stalled.routing_decision["signals"]
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_evidence_survives_a_new_human_ask(self, mock_router_instance, basic_config):
|
||||
"""A plain follow-up like 'try again' must not erase the stall evidence that came
|
||||
|
|
|
|||
|
|
@ -105,4 +105,11 @@ describe("StallEscalationConfig", () => {
|
|||
renderConfig({ stall_escalation_enabled: true, session_affinity: true });
|
||||
expect(screen.queryByLabelText("Repeats before escalating")).not.toBeInTheDocument();
|
||||
});
|
||||
|
||||
it("still lets an already-on router turn it off once a blocker appears, which the save needs", () => {
|
||||
const onChange = renderConfig({ stall_escalation_enabled: true, session_affinity: true });
|
||||
expect(toggle()).not.toHaveAttribute("aria-disabled", "true");
|
||||
fireEvent.click(toggle());
|
||||
expect(onChange).toHaveBeenCalledWith(expect.objectContaining({ stall_escalation_enabled: undefined }));
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -65,7 +65,10 @@ const StallEscalationConfig: React.FC<{
|
|||
<div className="flex items-center gap-2 mb-2">
|
||||
<Switch
|
||||
checked={enabled}
|
||||
disabled={blockedReason !== null}
|
||||
// Blocked only prevents turning it on: an already-on router that just became
|
||||
// blocked (e.g. session pinning turned on afterward) still needs a way to turn
|
||||
// this back off, since the backend rejects saving both together.
|
||||
disabled={blockedReason !== null && !enabled}
|
||||
onCheckedChange={toggle}
|
||||
aria-label="Escalate a stalled task to a stronger model"
|
||||
/>
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue