feat(router): per-group supported reasoning efforts via model-map intersection

This commit is contained in:
Tin Chi Lo 2026-08-20 15:16:42 -07:00
parent 9821b451e3
commit a5f0e228b5
11 changed files with 295 additions and 28 deletions

View file

@ -165,6 +165,10 @@ from litellm.router_utils.pre_call_checks.model_rate_limit_check import (
from litellm.router_utils.pre_call_checks.prompt_caching_deployment_check import (
PromptCachingDeploymentCheck,
)
from litellm.router_utils.reasoning_effort_capability import (
intersect_supported_reasoning_efforts,
resolve_supported_reasoning_efforts,
)
from litellm.router_utils.router_callbacks.track_deployment_metrics import (
increment_deployment_failures_for_current_minute,
increment_deployment_successes_for_current_minute,
@ -9541,6 +9545,11 @@ class Router:
if model_info.get("rpm", None) is not None and _deployment_rpm is None:
_deployment_rpm = model_info.get("rpm")
model_group_info.supported_reasoning_efforts = intersect_supported_reasoning_efforts(
model_group_info.supported_reasoning_efforts,
resolve_supported_reasoning_efforts(model_info),
)
if _deployment_tpm is not None:
if total_tpm is None:
total_tpm = 0

View file

@ -0,0 +1,52 @@
"""Resolve which reasoning_effort values a deployment, and by intersection a model group, accepts.
The model-map flags carry different polarity per level, mirroring the provider gates
(gpt_5_transformation.py restricts xhigh to explicit opt-in and treats minimal/low as opt-out;
anthropic/chat/transformation.py rejects only xhigh/max without an explicit flag): medium and high
are unconditional for any reasoning model, none/minimal/low are supported unless the map explicitly
says false, and xhigh/max require an explicit true. Shipping the resolved list keeps that polarity
in one place instead of re-encoding it in every consumer.
"""
from collections.abc import Mapping, Sequence
from typing import Final
REASONING_EFFORT_CAPABILITY_ORDER: Final = ("none", "minimal", "low", "medium", "high", "xhigh", "max")
_OPT_OUT_FLAGS: Final = (
("none", "supports_none_reasoning_effort"),
("minimal", "supports_minimal_reasoning_effort"),
("low", "supports_low_reasoning_effort"),
)
_OPT_IN_FLAGS: Final = (
("xhigh", "supports_xhigh_reasoning_effort"),
("max", "supports_max_reasoning_effort"),
)
_UNCONDITIONAL_EFFORTS: Final = frozenset(("medium", "high"))
def resolve_supported_reasoning_efforts(model_info: Mapping[str, object]) -> tuple[str, ...] | None:
"""None = no capability metadata for this deployment (e.g. a model absent from the model map,
whose stub info carries no supports_reasoning key at all); () = reasoning unsupported."""
if "supports_reasoning" not in model_info:
return None
if model_info.get("supports_reasoning") is not True:
return ()
opt_out: Final = frozenset(effort for effort, flag in _OPT_OUT_FLAGS if model_info.get(flag) is not False)
opt_in: Final = frozenset(effort for effort, flag in _OPT_IN_FLAGS if model_info.get(flag) is True)
allowed: Final = opt_out | _UNCONDITIONAL_EFFORTS | opt_in
return tuple(effort for effort in REASONING_EFFORT_CAPABILITY_ORDER if effort in allowed)
def intersect_supported_reasoning_efforts(
current: Sequence[str] | None,
resolved: Sequence[str] | None,
) -> tuple[str, ...] | None:
"""Deployments without metadata (None) never narrow the group; an effort survives only when
every deployment with metadata accepts it, so the group offers nothing routing could reject."""
if resolved is None:
return tuple(current) if current is not None else None
if current is None:
return tuple(resolved)
keep: Final = frozenset(current) & frozenset(resolved)
return tuple(effort for effort in REASONING_EFFORT_CAPABILITY_ORDER if effort in keep)

View file

@ -635,6 +635,7 @@ class ModelGroupInfo(BaseModel):
supports_url_context: bool = Field(default=False)
supports_reasoning: bool = Field(default=False)
supports_function_calling: bool = Field(default=False)
supported_reasoning_efforts: tuple[str, ...] | None = Field(default=None)
supported_openai_params: list[str] | None = Field(default=[])
configurable_clientside_auth_params: CONFIGURABLE_CLIENTSIDE_AUTH_PARAMS = None

View file

@ -0,0 +1,71 @@
from litellm.router_utils.reasoning_effort_capability import (
intersect_supported_reasoning_efforts,
resolve_supported_reasoning_efforts,
)
class TestResolveSupportedReasoningEfforts:
def test_no_metadata_resolves_to_unknown(self):
assert resolve_supported_reasoning_efforts({}) is None
def test_non_reasoning_model_supports_no_efforts(self):
assert resolve_supported_reasoning_efforts({"supports_reasoning": None}) == ()
assert resolve_supported_reasoning_efforts({"supports_reasoning": False}) == ()
def test_reasoning_model_with_no_flags_gets_the_opt_out_levels_only(self):
# The kimi shape: supports_reasoning true, zero effort flags. medium/high are unconditional,
# none/minimal/low are opt-out so absence means supported, xhigh/max are opt-in so absence
# means unsupported.
assert resolve_supported_reasoning_efforts({"supports_reasoning": True}) == (
"none",
"minimal",
"low",
"medium",
"high",
)
def test_explicit_false_removes_an_opt_out_level(self):
# The gpt-5.5-pro shape from the model map: only medium/high/xhigh are accepted upstream.
resolved = resolve_supported_reasoning_efforts(
{
"supports_reasoning": True,
"supports_none_reasoning_effort": False,
"supports_minimal_reasoning_effort": False,
"supports_low_reasoning_effort": False,
"supports_xhigh_reasoning_effort": True,
}
)
assert resolved == ("medium", "high", "xhigh")
def test_explicit_true_adds_the_opt_in_levels(self):
# The claude-opus shape: xhigh and max explicitly true, everything else absent.
resolved = resolve_supported_reasoning_efforts(
{
"supports_reasoning": True,
"supports_xhigh_reasoning_effort": True,
"supports_max_reasoning_effort": True,
}
)
assert resolved == ("none", "minimal", "low", "medium", "high", "xhigh", "max")
def test_opt_in_flag_set_false_stays_excluded(self):
resolved = resolve_supported_reasoning_efforts(
{"supports_reasoning": True, "supports_xhigh_reasoning_effort": False}
)
assert resolved is not None
assert "xhigh" not in resolved
class TestIntersectSupportedReasoningEfforts:
def test_unknown_never_narrows(self):
assert intersect_supported_reasoning_efforts(["medium", "high"], None) == ("medium", "high")
assert intersect_supported_reasoning_efforts(None, ["medium", "high"]) == ("medium", "high")
assert intersect_supported_reasoning_efforts(None, None) is None
def test_intersection_keeps_canonical_order(self):
assert intersect_supported_reasoning_efforts(
["max", "high", "medium", "xhigh"], ["xhigh", "medium", "minimal"]
) == ("medium", "xhigh")
def test_disjoint_sets_intersect_to_empty(self):
assert intersect_supported_reasoning_efforts(["max"], ["minimal"]) == ()

View file

@ -8721,3 +8721,87 @@ class TestAzureBaseModelFallbackLogging:
deployment=None, received_model_name="my-group", id="azure-base-model-test-id"
)
assert model_info["max_input_tokens"] == litellm.model_cost["azure/gpt-4o-mini"]["max_input_tokens"]
def test_model_group_info_intersects_supported_reasoning_efforts():
router = litellm.Router(
model_list=[
{
"model_name": "smart-group",
"litellm_params": {"model": "anthropic/opus-like"},
"model_info": {"id": "opus-like-deployment"},
},
{
"model_name": "smart-group",
"litellm_params": {"model": "openai/mini-like"},
"model_info": {"id": "mini-like-deployment"},
},
]
)
def _model_info(model_id: str, model_name: str):
if model_id == "opus-like-deployment":
return {
"key": model_name,
"litellm_provider": "anthropic",
"mode": "chat",
"supports_reasoning": True,
"supports_xhigh_reasoning_effort": True,
"supports_max_reasoning_effort": True,
}
return {
"key": model_name,
"litellm_provider": "openai",
"mode": "chat",
"supports_reasoning": True,
"supports_none_reasoning_effort": False,
"supports_minimal_reasoning_effort": True,
"supports_xhigh_reasoning_effort": False,
}
with patch.object(router, "get_deployment_model_info", side_effect=_model_info):
result = router._set_model_group_info(
model_group="smart-group",
user_facing_model_group_name="smart-group",
)
assert result is not None
# opus-like offers all seven levels, mini-like lacks none/xhigh/max; only the common set survives,
# so the group never advertises an effort routing could hand to a deployment that rejects it.
assert result.supported_reasoning_efforts == ("minimal", "low", "medium", "high")
def test_model_group_info_reasoning_efforts_ignore_deployments_without_metadata():
router = litellm.Router(
model_list=[
{
"model_name": "smart-group",
"litellm_params": {"model": "anthropic/opus-like"},
"model_info": {"id": "opus-like-deployment"},
},
{
"model_name": "smart-group",
"litellm_params": {"model": "openai/unmapped-model"},
"model_info": {"id": "unmapped-deployment"},
},
]
)
def _model_info(model_id: str, model_name: str):
if model_id == "opus-like-deployment":
return {
"key": model_name,
"litellm_provider": "anthropic",
"mode": "chat",
"supports_reasoning": True,
"supports_max_reasoning_effort": True,
}
return {"key": model_name, "litellm_provider": "openai", "mode": "chat"}
with patch.object(router, "get_deployment_model_info", side_effect=_model_info):
result = router._set_model_group_info(
model_group="smart-group",
user_facing_model_group_name="smart-group",
)
assert result is not None
assert result.supported_reasoning_efforts == ("none", "minimal", "low", "medium", "high", "max")

View file

@ -8,7 +8,12 @@ vi.mock(
);
const mockModelInfo = [
{ model_group: "gpt-4", mode: "chat", supports_reasoning: true },
{
model_group: "gpt-4",
mode: "chat",
supports_reasoning: true,
supported_reasoning_efforts: ["medium", "high", "xhigh"],
},
{ model_group: "gpt-3.5-turbo", mode: "chat" },
{ model_group: "claude-3-opus", mode: "chat", supports_reasoning: true },
{ model_group: "text-embedding-3-small", mode: "embedding" },
@ -943,3 +948,37 @@ describe("ComplexityRouterConfig reasoning effort gating", () => {
).toHaveTextContent("low");
});
});
describe("ComplexityRouterConfig per-model effort filtering", () => {
it("offers only the efforts the model group supports", async () => {
renderWithProviders(<ComplexityRouterConfig {...baseProps} />);
const user = userEvent.setup();
await user.click(screen.getByRole("combobox", { name: "Reasoning effort for gpt-4 in the Complex tier" }));
const options = (await screen.findAllByRole("option")).map((option) => option.textContent);
expect(options).toEqual(["Default", "medium", "high", "xhigh"]);
});
it("falls back to every effort when the group only reports supports_reasoning", async () => {
renderWithProviders(<ComplexityRouterConfig {...baseProps} />);
const user = userEvent.setup();
await user.click(
screen.getByRole("combobox", { name: "Reasoning effort for claude-3-opus in the Reasoning tier" }),
);
const options = (await screen.findAllByRole("option")).map((option) => option.textContent);
expect(options).toEqual(["Default", "none", "minimal", "low", "medium", "high", "xhigh"]);
});
// Hand-authored configs can carry a level outside the supported set (e.g. max); it must render
// and stay clearable rather than being masked as Default.
it("keeps showing a stored effort outside the supported set", () => {
renderWithProviders(
<ComplexityRouterConfig
{...baseProps}
value={{ ...defaultValue, tier_model_params: { COMPLEX: { "gpt-4": { reasoning_effort: "max" } } } }}
/>,
);
expect(screen.getByRole("combobox", { name: "Reasoning effort for gpt-4 in the Complex tier" })).toHaveTextContent(
"max",
);
});
});

View file

@ -13,7 +13,7 @@ import { ModelGroup } from "@/components/llm_calls/fetch_models";
import AdaptiveRoutingConfig from "./AdaptiveRoutingConfig";
import ClassificationMethodConfig from "./ClassificationMethodConfig";
import {
ReasoningEffort,
REASONING_EFFORT_OPTIONS,
TierModelParamsByTier,
pruneTierModelParams,
resolveComplexityDefaultModel,
@ -251,8 +251,13 @@ const ComplexityRouterConfig: React.FC<ComplexityRouterConfigProps> = ({
const defaultModel = resolveComplexityDefaultModel(value.tiers, value.default_model);
// Embedding models can't serve a chat-completion role, so they're excluded here.
const reasoningModels = new Set(
modelInfo.filter((model) => model.supports_reasoning).map((model) => model.model_group),
// The backend list is the per-group intersection of accepted effort levels; when a proxy does not
// send it yet, fall back to the coarse supports_reasoning gate with every level offered.
const effortOptionsByModel: Record<string, string[]> = Object.fromEntries(
modelInfo.map((model) => [
model.model_group,
model.supported_reasoning_efforts ?? (model.supports_reasoning ? [...REASONING_EFFORT_OPTIONS] : []),
]),
);
const modelOptions = modelInfo
@ -270,11 +275,7 @@ const ComplexityRouterConfig: React.FC<ComplexityRouterConfigProps> = ({
});
};
const handleTierModelEffortChange = (
tier: keyof ComplexityTiers,
model: string,
effort: ReasoningEffort | undefined,
) => {
const handleTierModelEffortChange = (tier: keyof ComplexityTiers, model: string, effort: string | undefined) => {
onChange({
...value,
tier_model_params: setTierModelReasoningEffort(value.tier_model_params, tier, model, effort),
@ -365,7 +366,7 @@ const ComplexityRouterConfig: React.FC<ComplexityRouterConfigProps> = ({
<TierModelEffortRows
tierLabel={label}
models={value.tiers[tier]}
reasoningModels={reasoningModels}
effortOptionsByModel={effortOptionsByModel}
paramsByModel={value.tier_model_params?.[tier]}
onEffortChange={(model, effort) => handleTierModelEffortChange(tier, model, effort)}
/>

View file

@ -2,35 +2,41 @@ import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@
import { SimpleTooltip } from "@/components/ui/tooltip";
import { Info } from "lucide-react";
import React from "react";
import { REASONING_EFFORT_OPTIONS, ReasoningEffort, TierModelParams } from "./complexity_router_tiers";
import { TierModelParams } from "./complexity_router_tiers";
const PROVIDER_DEFAULT = "__provider_default__";
const asEffort = (params: TierModelParams | undefined): ReasoningEffort | undefined => {
const storedEffort = (params: TierModelParams | undefined): string | undefined => {
const stored = params?.reasoning_effort;
if (typeof stored !== "string") return undefined;
return REASONING_EFFORT_OPTIONS.find((option) => option === stored);
return typeof stored === "string" && stored ? stored : undefined;
};
interface TierModelEffortRowsProps {
tierLabel: string;
models: string[];
reasoningModels: ReadonlySet<string>;
effortOptionsByModel: Record<string, string[]>;
paramsByModel: Record<string, TierModelParams> | undefined;
onEffortChange: (model: string, effort: ReasoningEffort | undefined) => void;
onEffortChange: (model: string, effort: string | undefined) => void;
}
const TierModelEffortRows: React.FC<TierModelEffortRowsProps> = ({
tierLabel,
models,
reasoningModels,
effortOptionsByModel,
paramsByModel,
onEffortChange,
}) => {
const shown = models.filter(
(model) => reasoningModels.has(model) || Object.keys(paramsByModel?.[model] ?? {}).length > 0,
);
if (shown.length === 0) return null;
const rows = models
.map((model) => {
const effort = storedEffort(paramsByModel?.[model]);
const supported = effortOptionsByModel[model] ?? [];
// A stored effort outside the supported set (hand-authored, or capabilities changed since it
// was saved) stays listed so it renders and can be cleared.
const options = effort !== undefined && !supported.includes(effort) ? [...supported, effort] : supported;
return { model, effort, options };
})
.filter(({ model, options }) => options.length > 0 || Object.keys(paramsByModel?.[model] ?? {}).length > 0);
if (rows.length === 0) return null;
return (
<div className="mt-2 space-y-1">
<div className="flex items-center gap-1">
@ -41,18 +47,17 @@ const TierModelEffortRows: React.FC<TierModelEffortRowsProps> = ({
<Info className="size-3 text-muted-foreground/70" />
</SimpleTooltip>
</div>
{shown.map((model) => (
{rows.map(({ model, effort, options }) => (
<div key={model} className="flex items-center justify-between gap-2">
<span className="truncate text-xs">{model}</span>
<Select
items={[
{ value: PROVIDER_DEFAULT, label: "Default" },
...REASONING_EFFORT_OPTIONS.map((option) => ({ value: option, label: option })),
...options.map((option) => ({ value: option, label: option })),
]}
value={asEffort(paramsByModel?.[model]) ?? PROVIDER_DEFAULT}
value={effort ?? PROVIDER_DEFAULT}
onValueChange={(selected: string | null) =>
selected !== null &&
onEffortChange(model, selected === PROVIDER_DEFAULT ? undefined : (selected as ReasoningEffort))
selected !== null && onEffortChange(model, selected === PROVIDER_DEFAULT ? undefined : selected)
}
>
<SelectTrigger
@ -64,7 +69,7 @@ const TierModelEffortRows: React.FC<TierModelEffortRowsProps> = ({
</SelectTrigger>
<SelectContent>
<SelectItem value={PROVIDER_DEFAULT}>Default</SelectItem>
{REASONING_EFFORT_OPTIONS.map((option) => (
{options.map((option) => (
<SelectItem key={option} value={option}>
{option}
</SelectItem>

View file

@ -90,7 +90,7 @@ export const setTierModelReasoningEffort = (
current: TierModelParamsByTier | undefined,
tier: string,
model: string,
effort: ReasoningEffort | undefined,
effort: string | undefined,
): TierModelParamsByTier | undefined => {
const { reasoning_effort: _dropped, ...rest } = current?.[tier]?.[model] ?? {};
const params = effort === undefined ? rest : { ...rest, reasoning_effort: effort };

View file

@ -7,6 +7,7 @@ export interface ModelGroup {
model_group: string;
mode?: string;
supports_reasoning?: boolean;
supported_reasoning_efforts?: string[];
}
interface AvailableModel {
@ -15,6 +16,7 @@ interface AvailableModel {
id?: string | null;
mode?: string | null;
supports_reasoning?: boolean | null;
supported_reasoning_efforts?: string[] | null;
}
export const fetchAvailableModelsForTeam = async (accessToken: string, teamId: string): Promise<ModelGroup[]> => {
@ -39,6 +41,7 @@ export const fetchAvailableModels = async (accessToken: string): Promise<ModelGr
model_group: item.model_group || item.id || item.model_name || "",
mode: item.mode || undefined,
supports_reasoning: item.supports_reasoning === true || undefined,
supported_reasoning_efforts: item.supported_reasoning_efforts ?? undefined,
}))
.filter((model: ModelGroup) => model.model_group !== "");

View file

@ -29522,6 +29522,8 @@ export interface components {
* @default []
*/
supported_openai_params: string[] | null;
/** Supported Reasoning Efforts */
supported_reasoning_efforts?: string[] | null;
/**
* Supports Function Calling
* @default false