diff --git a/litellm/router_strategy/complexity_router/complexity_router.py b/litellm/router_strategy/complexity_router/complexity_router.py index 202dcff5157..c829e6d9442 100644 --- a/litellm/router_strategy/complexity_router/complexity_router.py +++ b/litellm/router_strategy/complexity_router/complexity_router.py @@ -46,6 +46,9 @@ from .config import ( DEFAULT_REASONING_KEYWORDS, DEFAULT_SIMPLE_KEYWORDS, DEFAULT_TECHNICAL_KEYWORDS, + PLAN_MODE_SYSTEM_SENTINELS, + PLAN_MODE_TAIL_SENTINELS, + PLAN_MODE_TOOL_NAME, TIER_SEVERITY_ORDER, ClassificationRubric, ComplexityRouterConfig, @@ -421,6 +424,123 @@ def _extract_current_ask_and_system_prompt( return current_ask, system_prompt +def _last_human_ask_index( + messages: Sequence[Mapping[str, object]], + marker_pairs: tuple[tuple[str, str], ...] = _DEFAULT_REMINDER_MARKERS, +) -> int | None: + """Index of the newest user turn carrying a real human ask, or None when every turn is plumbing. + + Tool-result carriers and reminder-only turns flatten to empty human text, so an agentic loop's + tail of tool traffic never counts as the ask. Plan-mode staleness detection anchors here: the + sentinel a client re-injects each turn lands at or after this index, while a sentinel that only + survives in history from an exited plan session sits before it. + """ + return next( + ( + index + for index in range(len(messages) - 1, -1, -1) + if messages[index].get("role") == "user" and _human_text(messages[index].get("content"), marker_pairs) + ), + None, + ) + + +def _iter_system_scope_texts( + body_system: object, + messages: Sequence[Mapping[str, object]], +) -> Iterator[str]: + """Text of the request's leading system prompt content: the top-level system param (Anthropic + dialect carries one alongside the messages array) plus system-role messages before the first + non-system turn. + + Leading only, because that is the content clients rebuild on every request, so a sentinel + matched here is current by construction. A system message sitting later in the conversation is + transcript history (Claude Code's injected reminders survive there after plan mode exits) and + must go through the staleness-aware tail scan instead -- scanning it here would floor every + turn of a session that once planned, for any pattern whose client injects mid-conversation. + """ + if isinstance(body_system, str): + yield body_system + elif isinstance(body_system, list): + yield _message_text(body_system) + for msg in messages: + if msg.get("role") != "system": + return + if text := _message_text(msg.get("content")): + yield text + + +def _matched_plan_mode_sentinel( + body: Mapping[str, object] | None, + resolved_messages: Sequence[Mapping[str, object]] | None, + extra_patterns: tuple[str, ...], + marker_pairs: tuple[tuple[str, str], ...] = _DEFAULT_REMINDER_MARKERS, +) -> str | None: + """The plan-mode sentinel this request carries, or None when it carries none. + + Reads the raw wire body when the proxy captured one, because the sentinels ride in + client-injected plumbing that the ask-extraction path deliberately strips: Claude Code injects + a system-role message mid-conversation (older versions a reminder block inside the user turn), + and both are invisible to `_extract_current_ask_and_system_prompt`. Resolved messages are only + the fallback for direct SDK callers with no proxy capture. + + Three signals with different staleness behavior, so they scan different scopes: + - Copilot CLI advertises plan mode in the tools array (`exit_plan_mode`), rebuilt per request. + - Copilot's ``modeInstructions`` preamble rides the leading system prompt, rebuilt per + request, so an occurrence there is current by construction. + - Claude Code's injected reminders persist in transcript history after the user exits plan + mode, so only an occurrence at or after the newest human ask counts: while plan mode is + active the client re-injects the reminder with every turn, and after exit the newest ask has + no reminder at or after it. Matching is raw text on purpose -- the current injection style is + a system-role message, the older one a reminder block, and stripping would delete the latter. + + Every pattern, built-in and operator-supplied, is matched in both scopes; each scope is + staleness-safe on its own terms, so the union cannot resurrect an exited plan session. + + Matches are case-sensitive substrings, same rationale as escalation keywords: these exact + client-owned strings, not incidental prose. A caller can still paste one deliberately; that + only raises the tier within pools the operator configured, so it spends up, never sideways. + """ + from litellm.litellm_core_utils.prompt_templates.factory import has_tool_with_name + + tools: Final = body.get("tools") if body is not None else None + if has_tool_with_name(tools, PLAN_MODE_TOOL_NAME): + return PLAN_MODE_TOOL_NAME + + body_messages: Final = body.get("messages") if body is not None else None + messages: Final[Sequence[Mapping[str, object]]] = ( + tuple(msg for msg in body_messages if isinstance(msg, Mapping)) + if isinstance(body_messages, list) + else (resolved_messages or ()) + ) + + patterns: Final = (*PLAN_MODE_SYSTEM_SENTINELS, *PLAN_MODE_TAIL_SENTINELS, *extra_patterns) + system_match: Final = next( + ( + pattern + for text in _iter_system_scope_texts(body.get("system") if body is not None else None, messages) + for pattern in patterns + if pattern in text + ), + None, + ) + if system_match is not None: + return system_match + + newest_ask_index: Final = _last_human_ask_index(messages, marker_pairs) + tail_start: Final = 0 if newest_ask_index is None else newest_ask_index + return next( + ( + pattern + for msg in islice(messages, tail_start, None) + if (text := _message_text(msg.get("content"))) + for pattern in patterns + if pattern in text + ), + None, + ) + + def _truncate(text: str, limit: int) -> str: """Cap text at limit characters, marking it so the classifier can tell the turn was cut short.""" return text if len(text) <= limit else f"{text[:limit]}{_TRUNCATION_MARKER}" @@ -489,8 +609,14 @@ def _decision_is_pinnable(decision: StandardLoggingRoutingDecision | None) -> bo A classifier that timed out did not decide anything, so pinning where its fallback landed would let one transient failure hold the session on default_model for the whole TTL. Those turns stay unpinned and the next one classifies again. + + A plan-mode floor is transient the other way around: it describes the state the client is + in right now, not what the session's traffic looks like. Pinning it would hold the session + on the floor's premium model after the user exits plan mode; leaving it unpinned means the + floor re-detects while plan mode lasts and the first ordinary turn classifies and pins as + if plan mode had never happened. """ - return decision is None or decision.get("cause") != "default_model_fallback" + return decision is None or decision.get("cause") not in ("default_model_fallback", "plan_mode") class DimensionScore: @@ -1369,7 +1495,13 @@ class ComplexityRouter(CustomLogger): classified_tier: ComplexityTier | str, user_message: str, request_kwargs: dict[str, Any] | None = None, + hard_floor: ComplexityTier | str | None = None, ) -> str: + """hard_floor excludes every candidate whose tiers all sit below it, turning this pick's + soft floors (a distance penalty a high-scoring cheap model can outweigh) into a hard + minimum for requests that carry one, e.g. the plan-mode floor. classified_tier arrives + already clamped to the floor, so the cold-start pool and the classified_tier eligibility + mode satisfy it by construction; only the "all" eligibility mode can reach below.""" from litellm.router_strategy.adaptive_router.bandit import ( normalized_cost, thompson_sample, @@ -1424,10 +1556,16 @@ class ComplexityRouter(CustomLogger): cost_weight: Final = self.config.adaptive_weights.cost penalty_weight: Final = self.config.tier_distance_penalty + floor_severity: Final = self._active_tier_severity(hard_floor) if hard_floor is not None else None best_model: str | None = None best_score = float("-inf") candidate_scores: Final[list[dict[str, Any]]] = [] for model in candidates: + if floor_severity is not None and all( + self._active_tier_severity(model_tier) < floor_severity + for model_tier in self._model_tiers.get(model, (classified_tier,)) + ): + continue cell = adaptive._cells[(request_type, model)] quality_sample = thompson_sample(cell) cost_score = normalized_cost(adaptive.model_to_cost.get(model, 0.0), all_costs) @@ -1469,6 +1607,55 @@ class ComplexityRouter(CustomLogger): } return best_model + def _resolve_plan_mode_floor(self) -> ComplexityTier | str | None: + """The configured floor as an active tier: the built-in enum member, or the defined + name itself for a custom tier set; None when the feature is off.""" + name: Final = self.config.plan_mode_min_tier + if name is None: + return None + return name if self.config.has_custom_tiers else ComplexityTier(name) + + def _active_tier_severity(self, tier: ComplexityTier | str) -> int: + """Position of a tier in the active severity order: TIER_SEVERITY_ORDER for the built-in + set, tier_definitions list order (ascending) for a custom set -- the same order + keyword_tier_rules resolve severity against.""" + return self.config.tier_names().index(_tier_name(tier)) + + def _matched_plan_mode_signal( + self, + request_kwargs: Mapping[str, object], + resolved_messages: Sequence[Mapping[str, object]] | None, + ) -> str | None: + """The plan-mode sentinel on this request, or None; always None when the floor is unset, + so routers that never opted in pay nothing for detection.""" + if self.config.plan_mode_min_tier is None: + return None + proxy_request: Final = request_kwargs.get("proxy_server_request") + body: Final = proxy_request.get("body") if isinstance(proxy_request, dict) else None + return _matched_plan_mode_sentinel( + body if isinstance(body, Mapping) else None, + resolved_messages, + tuple(self.config.plan_mode_patterns or ()), + self._reminder_markers, + ) + + def _apply_plan_mode_floor(self, tier: ComplexityTier | str) -> ComplexityTier | str: + """The higher of the decided tier and the plan-mode floor; identity when the floor is unset.""" + floor: Final = self._resolve_plan_mode_floor() + if floor is None: + return tier + return tier if self._active_tier_severity(tier) >= self._active_tier_severity(floor) else floor + + def _plan_mode_floor_is_top_tier(self) -> bool: + """Whether no configured tier outranks the plan-mode floor, i.e. the classifier's answer + could never rise above it and classification would be pure spend.""" + floor: Final = self._resolve_plan_mode_floor() + if floor is None: + return False + configured: Final = frozenset(self.config.tiers) + names: Final = self.config.tier_names() + return all(name not in configured for name in names[self._active_tier_severity(floor) + 1 :]) + def _matched_escalation_keyword(self, user_message: str) -> str | None: """The escalation keyword the prompt contains, or None when escalation is off. @@ -1815,11 +2002,25 @@ class ComplexityRouter(CustomLogger): if pin_escalation_keyword is not None: routed_model = self._escalated_pin(pinned_model) if routed_model is not None: + escalated: Final = routed_model != pinned_model + # The floor outranks the pin because plan mode is a transient state of the + # session, not a request to move it: the turns carrying the sentinel route at + # the floor, and the stored pin deliberately keeps the session's own model so + # the first turn after plan mode exits auto-routes exactly as it would have. + # Escalation is the opposite on purpose -- an explicit ask to re-pin higher. + pin_plan_sentinel: Final = self._matched_plan_mode_signal(request_kwargs, resolved_messages) + pinned_tier: Final = self._tier_for_model(routed_model) if pin_plan_sentinel is not None else None + plan_floored: Final = ( + pinned_tier is not None and self._apply_plan_mode_floor(pinned_tier) != pinned_tier + ) + session_model: Final = routed_model + if plan_floored and pinned_tier is not None: + routed_model = self.get_model_for_tier(self._apply_plan_mode_floor(pinned_tier)) # Refresh the TTL on every hit so an active session doesn't lose its # pin mid-conversation just because it outlives the original write. await self.litellm_router_instance.cache.async_set_cache( key=cache_key, - value=routed_model, + value=session_model, ttl=self.config.session_affinity_ttl_seconds, ) if self.config.adaptive: @@ -1830,8 +2031,11 @@ class ComplexityRouter(CustomLogger): kwargs_metadata: Final = request_kwargs.setdefault("metadata", {}) if isinstance(kwargs_metadata, dict): kwargs_metadata[ADAPTIVE_ROUTER_CHOSEN_MODEL_KEY] = routed_model - escalated: Final = routed_model != pinned_model - cause: RoutingDecisionCause = "session_affinity_escalation" if escalated else "session_affinity_pin" + cause: RoutingDecisionCause = ( + "plan_mode" + if plan_floored + else ("session_affinity_escalation" if escalated else "session_affinity_pin") + ) verbose_router_logger.info( "ComplexityRouter: routing decision cause=%s, routed_model=%s", cause, routed_model ) @@ -1844,6 +2048,7 @@ class ComplexityRouter(CustomLogger): routed_model=routed_model, cause=cause, tier=self._tier_for_model(routed_model), + matched_keyword=pin_plan_sentinel if plan_floored else None, escalation_keyword=pin_escalation_keyword, escalated=escalated, conversation_continuing=conversation_continuing, @@ -1860,7 +2065,17 @@ class ComplexityRouter(CustomLogger): conversation_continuing=conversation_continuing, resolved_messages=resolved_messages, ) - if cache_key is not None and response is not None and _decision_is_pinnable(response.routing_decision): + # Sentinel presence, not the plan_mode cause, gates the pin write: a plan-mode turn + # classified at or above the floor keeps its ordinary cause, yet on an adaptive router + # the hard floor constrained its pick, so pinning it would carry a plan-mode-shaped + # choice past plan mode's exit. No sentinel turn writes the pin, whatever its cause. + pinnable: Final = ( + cache_key is not None + and response is not None + and _decision_is_pinnable(response.routing_decision) + and self._matched_plan_mode_signal(request_kwargs, resolved_messages) is None + ) + if pinnable and cache_key is not None and response is not None: await self.litellm_router_instance.cache.async_set_cache( key=cache_key, value=response.model, @@ -1940,13 +2155,47 @@ class ComplexityRouter(CustomLogger): newest_ask: Final = _newest_turn_ask(resolved_messages, self._reminder_markers) escalation_keyword: Final = self._matched_escalation_keyword(newest_ask) if newest_ask is not None else None + plan_mode_sentinel: Final = self._matched_plan_mode_signal(request_kwargs, resolved_messages) + plan_floor: Final = self._resolve_plan_mode_floor() if plan_mode_sentinel is not None else None + if plan_floor is not None and plan_mode_sentinel is not None and self._plan_mode_floor_is_top_tier(): + # No configured tier outranks the floor, so neither the keyword rules nor the + # classifier could change the answer -- routing directly saves the classifier call + # on every plan-mode turn. + routed_model = await self._pick_model_for_tier(plan_floor, messages, resolved_messages, request_kwargs) + verbose_router_logger.info( + "ComplexityRouter: routing decision cause=plan_mode, tier=%s, routed_model=%s", + _tier_name(plan_floor), + routed_model, + ) + return PreRoutingHookResponse( + model=routed_model, + messages=messages if has_original_messages else None, + routing_decision=self._build_routing_decision( + routed_model=routed_model, + conversation_continuing=conversation_continuing, + cause="plan_mode", + tier=plan_floor, + matched_keyword=plan_mode_sentinel, + escalation_keyword=escalation_keyword, + escalated=False, + ), + ) + override: Final = await self._resolve_keyword_tier_override(user_message, request_kwargs) if override is not None: - routed_tier: Final = self._escalate_tier(override.tier) if escalation_keyword is not None else override.tier - keyword_escalated: Final = routed_tier != override.tier + escalated_tier: Final = ( + self._escalate_tier(override.tier) if escalation_keyword is not None else override.tier + ) + keyword_escalated: Final = escalated_tier != override.tier + routed_tier: Final = ( + self._apply_plan_mode_floor(escalated_tier) if plan_floor is not None else escalated_tier + ) + keyword_plan_floored: Final = routed_tier != escalated_tier routed_model = await self._pick_model_for_tier(routed_tier, messages, resolved_messages, request_kwargs) keyword_cause: Final[RoutingDecisionCause] = ( - "semantic_keyword_match" if self.config.semantic_keyword_matching else "literal_keyword_match" + "plan_mode" + if keyword_plan_floored + else ("semantic_keyword_match" if self.config.semantic_keyword_matching else "literal_keyword_match") ) verbose_router_logger.info( "ComplexityRouter: routing decision cause=%s, escalated=%s, tier=%s, routed_model=%s", @@ -1963,7 +2212,7 @@ class ComplexityRouter(CustomLogger): conversation_continuing=conversation_continuing, cause=keyword_cause, tier=routed_tier, - matched_keyword=override.matched_keyword, + matched_keyword=plan_mode_sentinel if keyword_plan_floored else override.matched_keyword, escalation_keyword=escalation_keyword, escalated=keyword_escalated, ), @@ -1977,9 +2226,20 @@ class ComplexityRouter(CustomLogger): escalated: Final = tier != classified_tier if escalated: signals = (*signals, "escalation") + pre_floor_tier: Final = tier + if plan_floor is not None: + tier = self._apply_plan_mode_floor(tier) + plan_floored: Final = tier != pre_floor_tier + if plan_floored: + signals = (*signals, "plan_mode_floor") score_repr: Final = f"{score:.3f}" if score is not None else "n/a" fallback_model: Final = self.config.default_model if not self.config.plugins else None - if outcome.cause == "default_model_fallback" and fallback_model is not None: + # A sentinel-carrying request skips the failure exit below, whether or not the floor + # moved the tier: default_model carries no tier guarantee (its placeholder tier is the + # pool that holds it, or MEDIUM when none does), so a placeholder at or above the floor + # would otherwise route a plan-mode request to a model the floor cannot vouch for. The + # clamped tier's pool is the destination the floor can guarantee. + if outcome.cause == "default_model_fallback" and fallback_model is not None and plan_mode_sentinel is None: # Classification failed and the operator asked for default_model, so route there # directly. Neither the tier pool nor the adaptive bandit gets a say: both answer # "which model suits this tier", and no tier was decided. Escalation is skipped for @@ -2008,7 +2268,12 @@ class ComplexityRouter(CustomLogger): ), ) if self.config.adaptive: - routed_model = self._soft_floor_pick(tier, user_message, request_kwargs) + # hard_floor rather than a hard pick, and passed whenever the sentinel is present + # rather than only when the floor moved the tier: a request classified AT the floor + # has plan_floored False, yet adaptive_eligible="all" scores every model and only + # penalizes tier distance, so without the floor the bandit could still route below + # it -- and a floor a bandit can slide under is not a floor. + routed_model = self._soft_floor_pick(tier, user_message, request_kwargs, hard_floor=plan_floor) adaptive: Final = self._ensure_adaptive_router() if adaptive is not None: kwargs_metadata: Final = request_kwargs.setdefault("metadata", {}) @@ -2044,22 +2309,29 @@ class ComplexityRouter(CustomLogger): # short-circuited above), and there `tier` exists solely to name a pool for the plugins to # filter. Reporting it as the request's tier would attribute a classification to a request # that never got one, so the record names the pool in its signals instead. - classified_pool_tier: Final = None if outcome.cause == "default_model_fallback" else tier + # A floored failure still reports its tier: the floor decided it, unlike the plain + # failure path where no tier was decided and reporting one would fabricate a + # classification. + classified_pool_tier: Final = ( + None if outcome.cause == "default_model_fallback" and plan_mode_sentinel is None else tier + ) decision_signals: Final = ( (*signals, f"plugin-filtered-pool:{_tier_name(tier)}") - if outcome.cause == "default_model_fallback" + if outcome.cause == "default_model_fallback" and self.config.plugins else signals ) + decision_cause: Final[RoutingDecisionCause] = "plan_mode" if plan_floored else outcome.cause return PreRoutingHookResponse( model=routed_model, messages=messages if has_original_messages else None, routing_decision=self._build_routing_decision( routed_model=routed_model, conversation_continuing=conversation_continuing, - cause=outcome.cause, + cause=decision_cause, tier=classified_pool_tier, score=score, signals=decision_signals, + matched_keyword=plan_mode_sentinel if plan_floored else None, escalation_keyword=escalation_keyword, escalated=escalated, classifier_model=classifier_model, diff --git a/litellm/router_strategy/complexity_router/config.py b/litellm/router_strategy/complexity_router/config.py index 01dd8cb1548..20329de34e2 100644 --- a/litellm/router_strategy/complexity_router/config.py +++ b/litellm/router_strategy/complexity_router/config.py @@ -267,6 +267,16 @@ DEFAULT_TECHNICAL_KEYWORDS: Final[list[str]] = [ DEFAULT_ESCALATION_KEYWORDS: Final[list[str]] = ["LITELLM ESCALATE"] +# Verified against Claude Code 2.1.233 wire captures and vscode-copilot-chat source +# (agentPrompt.tsx / planAgentProvider.ts). These are client-owned strings that drift with +# client releases; operators extend coverage via plan_mode_patterns rather than editing these. +PLAN_MODE_TAIL_SENTINELS: Final[tuple[str, ...]] = ( + "Plan mode is active", + "Plan mode still active", +) +PLAN_MODE_SYSTEM_SENTINELS: Final[tuple[str, ...]] = ('You are currently running in "Plan" mode.',) +PLAN_MODE_TOOL_NAME: Final[str] = "exit_plan_mode" + DEFAULT_SIMPLE_KEYWORDS: Final[list[str]] = [ "what is", @@ -623,6 +633,31 @@ class ComplexityRouterConfig(BaseModel): description="Rules that force a specific tier when their keywords match the prompt", ) + plan_mode_min_tier: str | None = Field( + default=None, + description=( + "When set, requests carrying a coding-agent plan-mode sentinel (Claude Code plan " + "mode, VS Code Copilot Plan mode, Copilot CLI's exit_plan_mode tool) are routed to " + "at least this tier: the classified tier still wins when it is higher, and the " + "floor also overrides a session-affinity pin to a lower tier for exactly the turns " + "carrying the sentinel, without rewriting the pin -- the first turn after plan mode " + "exits routes as if plan mode had never happened. Names a built-in tier, or with " + "tier_definitions set, one of the defined tier names (list order is ascending " + "severity, same as keyword_tier_rules). Unset disables detection entirely. The " + "sentinels ride in client-injected prompt text, so a caller who pastes one can " + "spend up to this tier's models -- never down, and never outside the configured " + "pools." + ), + ) + plan_mode_patterns: tuple[str, ...] | None = Field( + default=None, + description=( + "Additional case-sensitive literal sentinels that mark a request as plan mode, on " + "top of the built-in Claude Code and Copilot ones. For clients whose plan-mode " + "wording the built-ins don't cover, or after a client release changes its strings." + ), + ) + # Semantic (embedding) matching for keyword_tier_rules instead of literal text matching semantic_keyword_matching: bool = Field( default=False, @@ -723,6 +758,42 @@ class ComplexityRouterConfig(BaseModel): return None return [stripped for keyword in value if (stripped := keyword.strip())] + @field_validator("plan_mode_min_tier", mode="before") + @classmethod + def _coerce_plan_mode_min_tier(cls, value: object) -> object: + if isinstance(value, ComplexityTier): + return value.value + if isinstance(value, str): + return value.strip() + return value + + @field_validator("plan_mode_patterns") + @classmethod + def _normalize_plan_mode_patterns(cls, value: tuple[str, ...] | None) -> tuple[str, ...] | None: + """Blank patterns are dropped rather than kept: an empty string substring-matches every + request, which would silently floor all traffic (same failure mode keyword_tier_rules + rejects).""" + if value is None: + return None + return tuple(stripped for pattern in value if (stripped := pattern.strip())) + + @model_validator(mode="after") + def _validate_plan_mode_min_tier(self) -> "ComplexityRouterConfig": + if self.plan_mode_min_tier is None: + return self + if self.plan_mode_min_tier not in self.tier_names(): + raise ValueError( + f"plan_mode_min_tier {self.plan_mode_min_tier!r} is not an active tier: it must name " + f"one of {', '.join(self.tier_names())}" + ) + if self.plan_mode_min_tier not in self.tiers: + raise ValueError( + f"plan_mode_min_tier {self.plan_mode_min_tier} has no model configured in tiers; " + "a floor pointing at an unconfigured tier would route every plan-mode request to the " + "default fallback instead of the premium pool the operator intended" + ) + return self + @model_validator(mode="after") def _validate_llm_classifier_config(self) -> "ComplexityRouterConfig": if self.classifier_type == "llm" and self.classifier_llm_config is None: diff --git a/litellm/types/utils.py b/litellm/types/utils.py index c818d6a151f..13831799c7f 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -2776,6 +2776,11 @@ RoutingDecisionCause = Literal[ "default_model_fallback", "literal_keyword_match", "semantic_keyword_match", + # A plan-mode sentinel (Claude Code / Copilot plan mode) was detected on the request and + # plan_mode_min_tier decided the tier: either it raised what the pipeline chose (classifier, + # keyword rule, or session pin), or the floor was already the top configured tier and the + # classifier was skipped. The matched sentinel rides in matched_keyword. + "plan_mode", "session_affinity_pin", "session_affinity_escalation", "default_fallback", diff --git a/tests/test_litellm/router_strategy/test_complexity_router.py b/tests/test_litellm/router_strategy/test_complexity_router.py index 44a8ed94e7c..7034ebc9bd7 100644 --- a/tests/test_litellm/router_strategy/test_complexity_router.py +++ b/tests/test_litellm/router_strategy/test_complexity_router.py @@ -29,6 +29,7 @@ from litellm.router_strategy.complexity_router.complexity_router import ( DimensionScore, KeywordOverride, _built_in_prompt, + _matched_plan_mode_sentinel, classification_system_prompt, ) from litellm.router_strategy.complexity_router.config import ( @@ -7179,3 +7180,552 @@ class TestTierDefinitions: "classifier_llm_config": {"model": "haiku-classifier", "system_prompt": "grade it"}, } ) + + +class TestPlanModeDetection: + """Wire-shape detection for coding-agent plan mode. + + Fixture bodies are sanitized minimal replicas of real captures: Claude Code 2.1.233 via an + ANTHROPIC_BASE_URL logging stub (mid-conversation system-role message on the Anthropic + dialect), and vscode-copilot-chat source for the Copilot shapes. + """ + + CLAUDE_CODE_SENTINEL = ( + "Plan mode is active. The user indicated that they do not want you to execute yet -- " + "you MUST NOT make any edits, run any non-readonly tools" + ) + COPILOT_PREAMBLE = ( + '\nYou are currently running in "Plan" mode. Below are your ' + "instructions for this mode, they must take precedence over any instructions above.\n" + "You are a PLANNING AGENT.\n" + ) + + def test_claude_code_mid_conversation_system_message_matches(self): + body = { + "system": [{"type": "text", "text": "You are a coding agent."}], + "messages": [ + {"role": "user", "content": [{"type": "text", "text": "add a hello endpoint"}]}, + {"role": "system", "content": [{"type": "text", "text": self.CLAUDE_CODE_SENTINEL}]}, + ], + } + assert _matched_plan_mode_sentinel(body, None, ()) == "Plan mode is active" + + def test_claude_code_sparse_reminder_on_later_turn_matches(self): + body = { + "messages": [ + {"role": "user", "content": "plan the refactor"}, + {"role": "system", "content": "Plan mode still active (see full instructions earlier)."}, + {"role": "assistant", "content": [{"type": "tool_use", "id": "t1", "name": "Read", "input": {}}]}, + {"role": "user", "content": [{"type": "tool_result", "tool_use_id": "t1", "content": "file body"}]}, + ] + } + assert _matched_plan_mode_sentinel(body, None, ()) == "Plan mode still active" + + def test_claude_code_legacy_reminder_block_inside_user_turn_matches(self): + body = { + "messages": [ + { + "role": "user", + "content": [ + { + "type": "text", + "text": f"{self.CLAUDE_CODE_SENTINEL}\nplan my feature", + } + ], + } + ] + } + assert _matched_plan_mode_sentinel(body, None, ()) == "Plan mode is active" + + def test_exited_plan_mode_history_does_not_match(self): + """After the user exits plan mode, the old reminder survives in history but sits before + the newest human ask, so it must not keep flooring the session.""" + body = { + "messages": [ + {"role": "user", "content": "plan the migration"}, + {"role": "system", "content": self.CLAUDE_CODE_SENTINEL}, + {"role": "assistant", "content": "Here is the plan."}, + {"role": "user", "content": "looks good, implement it"}, + ] + } + assert _matched_plan_mode_sentinel(body, None, ()) is None + + def test_copilot_system_message_preamble_matches_regardless_of_position(self): + """Copilot rebuilds its system message per request, so a match anywhere in system scope is + current -- including the usual position before the user turns, which the tail rule alone + would miss.""" + body = { + "messages": [ + {"role": "system", "content": f"You are an expert.\n{self.COPILOT_PREAMBLE}"}, + {"role": "user", "content": "refactor the auth flow"}, + {"role": "assistant", "content": "Looking."}, + {"role": "user", "content": "continue"}, + ] + } + assert _matched_plan_mode_sentinel(body, None, ()) == 'You are currently running in "Plan" mode.' + + def test_copilot_cli_exit_plan_mode_tool_matches_openai_and_anthropic_tool_shapes(self): + openai_shape = {"tools": [{"type": "function", "function": {"name": "exit_plan_mode"}}], "messages": []} + anthropic_shape = {"tools": [{"name": "exit_plan_mode", "input_schema": {}}], "messages": []} + assert _matched_plan_mode_sentinel(openai_shape, None, ()) == "exit_plan_mode" + assert _matched_plan_mode_sentinel(anthropic_shape, None, ()) == "exit_plan_mode" + + def test_operator_extra_patterns_match_in_system_scope_and_tail(self): + in_system = { + "messages": [{"role": "system", "content": "CUSTOM AGENT PLANNING"}, {"role": "user", "content": "hi"}] + } + in_tail = { + "messages": [{"role": "user", "content": "hi"}, {"role": "system", "content": "CUSTOM AGENT PLANNING"}] + } + assert _matched_plan_mode_sentinel(in_system, None, ("CUSTOM AGENT PLANNING",)) == "CUSTOM AGENT PLANNING" + assert _matched_plan_mode_sentinel(in_tail, None, ("CUSTOM AGENT PLANNING",)) == "CUSTOM AGENT PLANNING" + + def test_stale_custom_pattern_in_mid_conversation_system_message_does_not_match(self): + """Only the leading system prompt is staleness-exempt: a custom pattern surviving in a + mid-conversation system message from an exited plan session must not keep flooring.""" + stale = { + "messages": [ + {"role": "user", "content": "plan it"}, + {"role": "system", "content": "CUSTOM AGENT PLANNING"}, + {"role": "assistant", "content": "planned"}, + {"role": "user", "content": "implement it"}, + ] + } + assert _matched_plan_mode_sentinel(stale, None, ("CUSTOM AGENT PLANNING",)) is None + + def test_plain_request_does_not_match(self): + body = { + "system": "You are helpful.", + "messages": [{"role": "user", "content": "what is the plan for dinner?"}], + } + assert _matched_plan_mode_sentinel(body, None, ()) is None + + def test_sentinel_quoted_in_newest_ask_matches_by_design(self): + """A caller pasting the sentinel can floor their own request. Deliberate: the floor only + raises the tier within operator-configured pools, so this spends up, never sideways.""" + body = {"messages": [{"role": "user", "content": "why do I see 'Plan mode is active' in my logs?"}]} + assert _matched_plan_mode_sentinel(body, None, ()) == "Plan mode is active" + + def test_resolved_messages_fallback_when_no_proxy_body(self): + resolved = ( + {"role": "user", "content": "plan it"}, + {"role": "system", "content": self.CLAUDE_CODE_SENTINEL}, + ) + assert _matched_plan_mode_sentinel(None, resolved, ()) == "Plan mode is active" + + +class TestPlanModeTierFloor: + """End-to-end plan_mode_min_tier behavior through async_pre_routing_hook.""" + + PLAN_BODY = { + "messages": [ + {"role": "user", "content": [{"type": "text", "text": "add a hello endpoint"}]}, + {"role": "system", "content": [{"type": "text", "text": "Plan mode is active. Do not execute."}]}, + ] + } + + @pytest.fixture + def floor_config(self, basic_config) -> dict: + return {**basic_config, "plan_mode_min_tier": "COMPLEX"} + + def _router(self, mock_router_instance, config: dict) -> ComplexityRouter: + return ComplexityRouter( + model_name="test-router", + litellm_router_instance=mock_router_instance, + complexity_router_config=config, + ) + + @pytest.mark.asyncio + async def test_floor_raises_simple_prompt_and_records_plan_mode_cause(self, mock_router_instance, floor_config): + router = self._router(mock_router_instance, floor_config) + result = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={"proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "add a hello endpoint"}], + ) + assert result is not None + assert result.model == "claude-sonnet-4-20250514" + assert result.routing_decision is not None + assert result.routing_decision["cause"] == "plan_mode" + assert result.routing_decision["matched_keyword"] == "Plan mode is active" + assert "plan_mode_floor" in result.routing_decision["signals"] + + @pytest.mark.asyncio + async def test_classifier_result_above_floor_wins(self, mock_router_instance, basic_config): + """The floor is a floor, not a pin: a keyword rule routing above it is untouched.""" + config = { + **basic_config, + "plan_mode_min_tier": "MEDIUM", + "keyword_tier_rules": [{"keywords": ["kubernetes"], "tier": "REASONING"}], + } + router = self._router(mock_router_instance, config) + result = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={"proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "plan the kubernetes migration"}], + ) + assert result is not None + assert result.model == "o1-preview" + assert result.routing_decision is not None + assert result.routing_decision["cause"] == "literal_keyword_match" + + @pytest.mark.asyncio + async def test_keyword_rule_below_floor_gets_floored(self, mock_router_instance, basic_config): + config = { + **basic_config, + "plan_mode_min_tier": "COMPLEX", + "keyword_tier_rules": [{"keywords": ["hello endpoint"], "tier": "SIMPLE"}], + } + router = self._router(mock_router_instance, config) + result = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={"proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "add a hello endpoint"}], + ) + assert result is not None + assert result.model == "claude-sonnet-4-20250514" + assert result.routing_decision is not None + assert result.routing_decision["cause"] == "plan_mode" + + @pytest.mark.asyncio + async def test_top_tier_floor_skips_classification(self, mock_router_instance, basic_config): + config = {**basic_config, "plan_mode_min_tier": "REASONING"} + router = self._router(mock_router_instance, config) + with patch.object(router, "aclassify") as classify_spy: + result = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={"proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "add a hello endpoint"}], + ) + classify_spy.assert_not_called() + assert result is not None + assert result.model == "o1-preview" + assert result.routing_decision is not None + assert result.routing_decision["cause"] == "plan_mode" + + @pytest.mark.asyncio + async def test_no_sentinel_routes_normally(self, mock_router_instance, floor_config): + router = self._router(mock_router_instance, floor_config) + result = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={}, + messages=[{"role": "user", "content": "Hello!"}], + ) + assert result is not None + assert result.model == "gpt-4o-mini" + + @pytest.mark.asyncio + async def test_unset_floor_ignores_sentinel(self, mock_router_instance, basic_config): + router = self._router(mock_router_instance, basic_config) + result = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={"proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "Hello!"}], + ) + assert result is not None + assert result.model == "gpt-4o-mini" + + @pytest.mark.asyncio + async def test_floor_overrides_session_pin_only_while_plan_mode_lasts(self, mock_router_instance, basic_config): + """Mid-session shift+tab into plan mode: the plan turns route at the floor, but the + stored pin keeps the session's own model, so the first turn after plan mode exits + auto-routes back to it instead of staying premium.""" + from litellm.caching.dual_cache import DualCache + + mock_router_instance.cache = DualCache() + config = {**basic_config, "plan_mode_min_tier": "COMPLEX", "session_affinity": True} + router = self._router(mock_router_instance, config) + session_kwargs = {"metadata": {"session_id": "plan-session"}} + first = await router.async_pre_routing_hook( + model="test-model", + request_kwargs=dict(session_kwargs), + messages=[{"role": "user", "content": "Hello!"}], + ) + assert first is not None and first.model == "gpt-4o-mini" + second = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={**session_kwargs, "proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "add a hello endpoint"}], + ) + assert second is not None + assert second.model == "claude-sonnet-4-20250514" + assert second.routing_decision is not None + assert second.routing_decision["cause"] == "plan_mode" + third = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={**session_kwargs, "proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "add auth to the endpoint"}], + ) + assert third is not None and third.model == "claude-sonnet-4-20250514" + fourth = await router.async_pre_routing_hook( + model="test-model", + request_kwargs=dict(session_kwargs), + messages=[{"role": "user", "content": "Hello!"}], + ) + assert fourth is not None + assert fourth.model == "gpt-4o-mini" + assert fourth.routing_decision is not None + assert fourth.routing_decision["cause"] == "session_affinity_pin" + + @pytest.mark.asyncio + async def test_plan_mode_first_turn_does_not_seed_the_session_pin(self, mock_router_instance, basic_config): + """A session whose first turn is already in plan mode must not pin the floored model: + the first ordinary turn classifies and pins as if plan mode had never happened.""" + from litellm.caching.dual_cache import DualCache + + mock_router_instance.cache = DualCache() + config = {**basic_config, "plan_mode_min_tier": "COMPLEX", "session_affinity": True} + router = self._router(mock_router_instance, config) + session_kwargs = {"metadata": {"session_id": "plan-first-session"}} + first = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={**session_kwargs, "proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "add a hello endpoint"}], + ) + assert first is not None and first.model == "claude-sonnet-4-20250514" + second = await router.async_pre_routing_hook( + model="test-model", + request_kwargs=dict(session_kwargs), + messages=[{"role": "user", "content": "Hello!"}], + ) + assert second is not None + assert second.model == "gpt-4o-mini" + assert second.routing_decision is not None + assert second.routing_decision["cause"] in ("heuristic_scorer", "reasoning_override") + + @pytest.mark.asyncio + async def test_pinned_session_at_or_above_floor_keeps_pin_cause(self, mock_router_instance, basic_config): + from litellm.caching.dual_cache import DualCache + + mock_router_instance.cache = DualCache() + config = {**basic_config, "plan_mode_min_tier": "MEDIUM", "session_affinity": True} + router = self._router(mock_router_instance, config) + session_kwargs = {"metadata": {"session_id": "premium-session"}} + first = await router.async_pre_routing_hook( + model="test-model", + request_kwargs=dict(session_kwargs), + messages=[ + {"role": "user", "content": "Let's think step by step and reason through this problem carefully."} + ], + ) + assert first is not None and first.model == "o1-preview" + second = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={**session_kwargs, "proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "plan the next step"}], + ) + assert second is not None + assert second.model == "o1-preview" + assert second.routing_decision is not None + assert second.routing_decision["cause"] == "session_affinity_pin" + + @pytest.mark.asyncio + async def test_floor_supports_custom_tier_sets_via_list_order_severity(self, mock_router_instance): + """With tier_definitions, the floor names a defined tier and severity is the list order + (ascending), the same resolution keyword_tier_rules use.""" + config = { + "tier_definitions": [ + {"name": "LIGHT", "description": "trivial lookups"}, + {"name": "HEAVY", "description": "multi-step engineering work"}, + ], + "tiers": {"LIGHT": "gpt-4o-mini", "HEAVY": "claude-sonnet-4-20250514"}, + "classifier_type": "llm", + "classifier_llm_config": {"model": "gpt-4o-mini"}, + "fallback_tier": "LIGHT", + "plan_mode_min_tier": "HEAVY", + } + router = self._router(mock_router_instance, config) + with patch.object(router, "aclassify") as classify_spy: + result = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={"proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "add a hello endpoint"}], + ) + classify_spy.assert_not_called() + assert result is not None + assert result.model == "claude-sonnet-4-20250514" + assert result.routing_decision is not None + assert result.routing_decision["cause"] == "plan_mode" + assert result.routing_decision["tier"] == "HEAVY" + + def test_floor_must_name_an_active_tier_on_a_custom_set(self): + with pytest.raises(ValueError, match="plan_mode_min_tier"): + ComplexityRouterConfig( + tier_definitions=[ + {"name": "LIGHT", "description": "trivial lookups"}, + {"name": "HEAVY", "description": "multi-step engineering work"}, + ], + tiers={"LIGHT": "gpt-4o-mini", "HEAVY": "claude-sonnet-4-20250514"}, + classifier_type="llm", + classifier_llm_config={"model": "gpt-4o-mini"}, + fallback_tier="LIGHT", + plan_mode_min_tier="COMPLEX", + ) + + def test_floor_must_point_at_a_configured_tier(self, basic_config): + config = {**basic_config, "plan_mode_min_tier": "REASONING"} + config["tiers"] = {"SIMPLE": "gpt-4o-mini"} + with pytest.raises(ValueError, match="plan_mode_min_tier"): + ComplexityRouterConfig(**config) + + def test_blank_extra_patterns_are_dropped(self): + config = ComplexityRouterConfig( + tiers={"SIMPLE": "gpt-4o-mini", "COMPLEX": "claude-sonnet-4-20250514"}, + plan_mode_min_tier="COMPLEX", + plan_mode_patterns=[" ", "REAL PATTERN", ""], + ) + assert config.plan_mode_patterns == ("REAL PATTERN",) + + @pytest.mark.asyncio + async def test_floored_classifier_failure_routes_floor_not_default_model(self, mock_router_instance, basic_config): + """A failed classification doesn't retract the floor: the request routes to the floor's + pool, not default_model, and no plugin-filtered-pool signal is fabricated.""" + from litellm.router_strategy.complexity_router.complexity_router import ClassificationOutcome + + config = {**basic_config, "plan_mode_min_tier": "COMPLEX", "default_model": "gpt-4o-mini"} + router = self._router(mock_router_instance, config) + failure = ClassificationOutcome( + tier=ComplexityTier.MEDIUM, score=None, signals=(), cause="default_model_fallback", classifier_cost=None + ) + with patch.object(router, "aclassify", return_value=failure): + result = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={"proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "add a hello endpoint"}], + ) + assert result is not None + assert result.model == "claude-sonnet-4-20250514" + assert result.routing_decision is not None + assert result.routing_decision["cause"] == "plan_mode" + assert result.routing_decision["tier"] == "COMPLEX" + assert not any(s.startswith("plugin-filtered-pool") for s in result.routing_decision.get("signals", ())) + + @pytest.mark.asyncio + async def test_hard_floor_reaches_the_bandit_even_when_classified_at_the_floor( + self, mock_router_instance, basic_config + ): + """A request classified exactly AT the floor has plan_floored False, yet the bandit must + still receive the floor: adaptive_eligible="all" scores every model and could otherwise + route below it.""" + from litellm.router_strategy.complexity_router.complexity_router import ClassificationOutcome + + config = {**basic_config, "plan_mode_min_tier": "COMPLEX", "adaptive": True} + router = self._router(mock_router_instance, config) + at_floor = ClassificationOutcome( + tier=ComplexityTier.COMPLEX, score=None, signals=(), cause="llm_classifier", classifier_cost=None + ) + with ( + patch.object(router, "aclassify", return_value=at_floor), + patch.object(router, "_soft_floor_pick", return_value="claude-sonnet-4-20250514") as bandit_spy, + patch.object(router, "_ensure_adaptive_router", return_value=None), + ): + result = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={"proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "add a hello endpoint"}], + ) + bandit_spy.assert_called_once() + assert bandit_spy.call_args.kwargs["hard_floor"] == ComplexityTier.COMPLEX + assert result is not None + assert result.model == "claude-sonnet-4-20250514" + + def test_hard_floor_excludes_below_floor_candidates_from_the_bandit(self, mock_router_instance): + """With a dominant posterior on a cheap model and adaptive_eligible="all", the pick must + still refuse every candidate whose tiers all sit below the hard floor.""" + from litellm.router_strategy.adaptive_router.bandit import BanditCell + from litellm.types.router import RequestType + + adaptive_instance = MagicMock() + adaptive_instance.model_list = [ + { + "model_name": "cheap", + "litellm_params": {"model": "openai/gpt-4o-mini", "input_cost_per_token": 0.00000015}, + "model_info": {"adaptive_router_preferences": {"quality_tier": 1, "strengths": []}}, + }, + { + "model_name": "premium", + "litellm_params": {"model": "openai/gpt-4o", "input_cost_per_token": 0.000005}, + "model_info": {"adaptive_router_preferences": {"quality_tier": 3, "strengths": []}}, + }, + ] + adaptive_instance.model_name_to_deployment_indices = {"cheap": [0], "premium": [1]} + router = ComplexityRouter( + model_name="hybrid", + litellm_router_instance=adaptive_instance, + complexity_router_config={ + "adaptive": True, + "tiers": {"SIMPLE": ["cheap"], "MEDIUM": ["cheap"], "COMPLEX": ["premium"]}, + "plan_mode_min_tier": "COMPLEX", + }, + ) + adaptive = router._ensure_adaptive_router() + assert adaptive is not None + adaptive._cells[(RequestType.GENERAL, "cheap")] = BanditCell(alpha=20.0, beta=1.0) + adaptive._cells[(RequestType.GENERAL, "premium")] = BanditCell(alpha=1.0, beta=20.0) + with patch( + "litellm.router_strategy.adaptive_router.bandit.thompson_sample", + side_effect=lambda cell, rng=None: cell.alpha / (cell.alpha + cell.beta), + ): + unfloored = router._soft_floor_pick(ComplexityTier.COMPLEX, "hi") + floored = router._soft_floor_pick(ComplexityTier.COMPLEX, "hi", hard_floor=ComplexityTier.COMPLEX) + assert unfloored == "cheap" + assert floored == "premium" + + @pytest.mark.asyncio + async def test_at_floor_plan_mode_turn_does_not_write_the_session_pin(self, mock_router_instance, basic_config): + """A plan-mode turn routed at or above the floor keeps its ordinary cause, but it still + must not pin: on an adaptive router the hard floor shaped that pick, and any sentinel + turn's pin would carry plan mode past its exit.""" + from litellm.caching.dual_cache import DualCache + + mock_router_instance.cache = DualCache() + config = { + **basic_config, + "plan_mode_min_tier": "MEDIUM", + "session_affinity": True, + "keyword_tier_rules": [{"keywords": ["kubernetes"], "tier": "REASONING"}], + } + router = self._router(mock_router_instance, config) + session_kwargs = {"metadata": {"session_id": "at-floor-session"}} + first = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={**session_kwargs, "proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "plan the kubernetes migration"}], + ) + assert first is not None and first.model == "o1-preview" + assert first.routing_decision is not None + assert first.routing_decision["cause"] == "literal_keyword_match" + second = await router.async_pre_routing_hook( + model="test-model", + request_kwargs=dict(session_kwargs), + messages=[{"role": "user", "content": "Hello!"}], + ) + assert second is not None + assert second.model == "gpt-4o-mini" + assert second.routing_decision is not None + assert second.routing_decision["cause"] in ("heuristic_scorer", "reasoning_override") + + @pytest.mark.asyncio + async def test_failure_exit_skipped_when_placeholder_tier_equals_the_floor( + self, mock_router_instance, basic_config + ): + """default_model outside every pool reports the MEDIUM placeholder; a MEDIUM floor then + leaves plan_floored False, and the exit must still not route a sentinel-carrying request + to a model the floor cannot vouch for.""" + from litellm.router_strategy.complexity_router.complexity_router import ClassificationOutcome + + config = {**basic_config, "plan_mode_min_tier": "MEDIUM", "default_model": "untiered-fallback"} + router = self._router(mock_router_instance, config) + failure = ClassificationOutcome( + tier=ComplexityTier.MEDIUM, score=None, signals=(), cause="default_model_fallback", classifier_cost=None + ) + with patch.object(router, "aclassify", return_value=failure): + result = await router.async_pre_routing_hook( + model="test-model", + request_kwargs={"proxy_server_request": {"body": self.PLAN_BODY}}, + messages=[{"role": "user", "content": "add a hello endpoint"}], + ) + assert result is not None + assert result.model == "gpt-4o" + assert result.routing_decision is not None + assert result.routing_decision["tier"] == "MEDIUM" diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.test.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.test.tsx index 3cd7faa5583..a2afe0d7a0d 100644 --- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.test.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.test.tsx @@ -129,6 +129,33 @@ describe("RoutingDecisionCard", () => { expect(screen.getByText('Keyword match: "deploy to k8s"')).toBeInTheDocument(); }); + it("shows the plan-mode sentinel that floored the tier", () => { + render( + , + ); + expect(screen.getByText('Plan-mode floor: "Plan mode is active"')).toBeInTheDocument(); + }); + + it("names the exit_plan_mode tool instead of quoting it as a sentinel", () => { + render( + , + ); + expect(screen.getByText("Plan-mode floor (exit_plan_mode tool)")).toBeInTheDocument(); + }); + + it("does not claim the score chose the tier on a plan-mode floored row", () => { + // The score's band can name a lower tier than the floored badge; the cause suppresses it. + render( + , + ); + expect(screen.queryByText(/below|to 0|at or above/)).not.toBeInTheDocument(); + expect(screen.getByText('Plan-mode floor: "Plan mode is active"')).toBeInTheDocument(); + }); + it("shows the escalation keyword", () => { render( , diff --git a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.tsx b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.tsx index 219900cad0a..0876a539653 100644 --- a/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.tsx +++ b/ui/litellm-dashboard/src/components/view_logs/LogDetailsDrawer/RoutingDecisionCard.tsx @@ -59,6 +59,12 @@ function describeScoreAgainstBoundaries( return named(`at or above ${complexReasoning}`, "REASONING"); } +function describePlanModeFloor(matchedKeyword: string | undefined): string { + if (matchedKeyword === "exit_plan_mode") return "Plan-mode floor (exit_plan_mode tool)"; + if (matchedKeyword) return `Plan-mode floor: "${matchedKeyword}"`; + return "Plan-mode floor"; +} + function describeCause(decision: RoutingDecision): string { const { cause, classifier_model: classifierModel, matched_keyword: matchedKeyword, tier_label: tierLabel } = decision; @@ -73,6 +79,8 @@ function describeCause(decision: RoutingDecision): string { return matchedKeyword ? `Keyword match: "${matchedKeyword}"` : "Keyword match"; case "semantic_keyword_match": return "Semantic keyword match"; + case "plan_mode": + return describePlanModeFloor(matchedKeyword); case "session_affinity_pin": return "Pinned to session"; case "session_affinity_escalation": @@ -141,7 +149,7 @@ export function RoutingDecisionCard({ // boundary would claim something untrue. Keyed off the cause rather than a marker // inside `signals`, which redaction can remove. const scoreExplanation = - score !== undefined && decision.cause !== "reasoning_override" + score !== undefined && decision.cause !== "reasoning_override" && decision.cause !== "plan_mode" ? describeScoreAgainstBoundaries(score, tierBoundaries, tierLabel !== undefined) : null; diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index a57adca3c0a..ca3974f24de 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -32333,6 +32333,16 @@ export interface components { * @default 0.5 */ match_threshold: number; + /** + * Plan Mode Min Tier + * @description When set, requests carrying a coding-agent plan-mode sentinel (Claude Code plan mode, VS Code Copilot Plan mode, Copilot CLI's exit_plan_mode tool) are routed to at least this tier: the classified tier still wins when it is higher, and the floor also overrides a session-affinity pin to a lower tier for exactly the turns carrying the sentinel, without rewriting the pin -- the first turn after plan mode exits routes as if plan mode had never happened. Names a built-in tier, or with tier_definitions set, one of the defined tier names (list order is ascending severity, same as keyword_tier_rules). Unset disables detection entirely. The sentinels ride in client-injected prompt text, so a caller who pastes one can spend up to this tier's models -- never down, and never outside the configured pools. + */ + plan_mode_min_tier?: string | null; + /** + * Plan Mode Patterns + * @description Additional case-sensitive literal sentinels that mark a request as plan mode, on top of the built-in Claude Code and Copilot ones. For clients whose plan-mode wording the built-ins don't cover, or after a client release changes its strings. + */ + plan_mode_patterns?: string[] | null; /** * Plugins * @description Not settable over HTTP; routing plugins are runtime objects @@ -33317,7 +33327,7 @@ export interface components { * Cause * @enum {string} */ - cause?: "heuristic_scorer" | "reasoning_override" | "llm_classifier" | "classifier_fallback" | "default_model_fallback" | "literal_keyword_match" | "semantic_keyword_match" | "session_affinity_pin" | "session_affinity_escalation" | "default_fallback" | "keyword" | "quality_tier" | "bandit"; + cause?: "heuristic_scorer" | "reasoning_override" | "llm_classifier" | "classifier_fallback" | "default_model_fallback" | "literal_keyword_match" | "semantic_keyword_match" | "plan_mode" | "session_affinity_pin" | "session_affinity_escalation" | "default_fallback" | "keyword" | "quality_tier" | "bandit"; /** Classifier Cost */ classifier_cost?: number; /** Classifier Model */