fix(reasoning): keep the chat gate on xhigh and stop empty groups zeroing the picker

The chat-completions gate only ever owned xhigh. Widening it to max and ultra
made gpt-5.6 answer 400 on requests litellm itself converts to /v1/responses,
where max is valid, because the gate runs before the bridge decision. No map
entry asserts either flag, so the widened gate could only ever reject.

An empty per-group intersection now falls back to the capability-blind level
list in the dashboard, matching what the picker showed before the field
existed, and ModelGroupInfo tolerates whatever shape an operator writes under
supported_reasoning_efforts instead of failing the whole /model_group/info
response.
This commit is contained in:
mateo-berri 2026-08-21 19:16:28 -07:00 • committed by Tin Chi Lo
parent 6ec49ea7dd
commit 18528b1a63
7 changed files with 137 additions and 41 deletions

View file

@ -222,8 +222,8 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
if "reasoning_effort" in optional_params:
optional_params["reasoning_effort"] = normalized
if effective_effort in ("xhigh", "max", "ultra"):
# xhigh/max/ultra are opt-in capabilities: only allow if the model explicitly supports them.
if effective_effort == "xhigh":
# xhigh is an opt-in capability: only allow if model explicitly supports it.
if not self._supports_reasoning_effort_level(model, effective_effort):
if litellm.drop_params or drop_params:
non_default_params.pop("reasoning_effort", None)

View file

@ -54,8 +54,12 @@ def _supports_none_reasoning_effort(model_info: Mapping[str, object]) -> bool:
def resolve_supported_reasoning_efforts(model_info: Mapping[str, object]) -> tuple[str, ...] | None:
"""None = no capability metadata for this deployment (e.g. a model absent from the model map,
whose stub info carries no supports_reasoning key at all); () = reasoning unsupported."""
"""None = the caller passed no supports_reasoning key at all; () = this deployment adds no
effort levels to its group. The router always supplies the key, so a deployment absent from the
model map arrives with supports_reasoning None and lands on (), the same answer a mapped
non-reasoning model gets: the group cannot promise a level on behalf of a model nothing is known
about. () therefore reads as "no usable answer" downstream, which is why the dashboard falls
back to the capability-blind level list on an empty group rather than hiding the control."""
if "supports_reasoning" not in model_info:
return None
if model_info.get("supports_reasoning") is not True:

View file

@ -641,6 +641,19 @@ class ModelGroupInfo(BaseModel):
supported_openai_params: list[str] | None = Field(default=[])
configurable_clientside_auth_params: CONFIGURABLE_CLIENTSIDE_AUTH_PARAMS = None
@field_validator("supported_reasoning_efforts", mode="before")
@classmethod
def _accept_only_well_formed_reasoning_efforts(cls, value: object) -> tuple[str, ...] | None:
"""Deployment model_info reaches this model through a **kwargs splat, so an operator can put
any shape under this key. Anything that is not a level or a list of levels resolves to None
and lets the computed intersection stand, because raising here fails the whole
/model_group/info response rather than the one group that carries the bad value."""
if isinstance(value, str):
return (value,)
if isinstance(value, (list, tuple)) and all(isinstance(level, str) for level in value):
return tuple(value)
return None
def __init__(self, **data) -> None:
for field_name, field_type in get_type_hints(self.__class__).items():
if field_type is bool and data.get(field_name) is None:

View file

@ -1312,18 +1312,32 @@ def test_responses_gpt54_allow_temperature_effort_none(
@pytest.mark.parametrize("model", ["gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"])
def test_gpt5_6_rejects_reasoning_effort_max_on_chat_completions(config: OpenAIConfig, model: str):
def test_gpt5_6_forwards_reasoning_effort_max_for_the_responses_bridge(config: OpenAIConfig, model: str):
"""A chat request carrying tools or a reasoning summary is converted to /v1/responses further
down main.py, and that surface accepts max. This runs before litellm has decided to bridge, so
refusing max here would break the cursor thinking-max shape that works today. Plain chat
completions still answer max with a provider 400, and the capability list below is what keeps
the level out of the picker."""
params = config.map_openai_params(
non_default_params={"reasoning_effort": "max"},
optional_params={},
model=model,
drop_params=False,
)
assert params["reasoning_effort"] == "max"
@pytest.mark.parametrize("model", ["gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"])
def test_gpt5_6_never_advertises_reasoning_effort_max(model: str):
"""/v1/chat/completions answers max with "Unsupported value: 'reasoning_effort' does not support
'max' with this model. Supported values are: 'none', 'low', 'medium', 'high', and 'xhigh'" on
every gpt-5.6 snapshot, so no gpt-5.6 map entry asserts supports_max_reasoning_effort and the
level is refused here instead of being forwarded into a provider 400."""
with pytest.raises(litellm.utils.UnsupportedParamsError):
config.map_openai_params(
non_default_params={"reasoning_effort": "max"},
optional_params={},
model=model,
drop_params=False,
)
'max' with this model. Supported values are: 'none', 'low', 'medium', 'high', and 'xhigh'", so no
gpt-5.6 entry asserts supports_max_reasoning_effort and the advertised set stops at xhigh."""
from litellm.router_utils.reasoning_effort_capability import resolve_supported_reasoning_efforts
resolved = resolve_supported_reasoning_efforts(litellm.get_model_info(model))
assert resolved is not None
assert "max" not in resolved
assert "xhigh" in resolved
def test_gpt5_6_keeps_reasoning_effort_max_on_the_responses_api(
@ -1341,34 +1355,43 @@ def test_gpt5_6_keeps_reasoning_effort_max_on_the_responses_api(
assert params["reasoning"] == {"effort": "max"}
def test_gpt5_6_rejects_reasoning_effort_ultra_until_the_map_opts_in(config: OpenAIConfig):
# ultra is plumbed as an opt-in level but no map entry asserts it: OpenAI's model guidance
# documents effort values only up to max, and the builder guide frames ultra as multi-agent
# orchestration. A verified supports_ultra_reasoning_effort flag lights this up with no code.
with pytest.raises(litellm.utils.UnsupportedParamsError):
config.map_openai_params(
non_default_params={"reasoning_effort": "ultra"},
optional_params={},
model="gpt-5.6",
drop_params=False,
)
def test_gpt5_6_never_advertises_reasoning_effort_ultra():
"""ultra is plumbed as an opt-in level but no map entry asserts it: OpenAI's model guidance
documents effort values only up to max, and /v1/responses answers ultra with a 400. A verified
supports_ultra_reasoning_effort flag lights it up with no code change."""
from litellm.router_utils.reasoning_effort_capability import resolve_supported_reasoning_efforts
resolved = resolve_supported_reasoning_efforts(litellm.get_model_info("gpt-5.6"))
assert resolved is not None
assert "ultra" not in resolved
@pytest.mark.parametrize("effort", ["max", "ultra"])
def test_gpt5_rejects_opt_in_reasoning_efforts_for_other_models(config: OpenAIConfig, effort: str):
def test_gpt5_forwards_levels_the_chat_gate_does_not_own(config: OpenAIConfig, effort: str):
"""Only xhigh is gated on this surface. max and ultra reach the provider (or the responses
bridge) and are answered there, which is what happened before per-group capabilities existed."""
params = config.map_openai_params(
non_default_params={"reasoning_effort": effort},
optional_params={},
model="gpt-5.1",
drop_params=False,
)
assert params["reasoning_effort"] == effort
def test_gpt5_rejects_xhigh_for_models_without_the_flag(config: OpenAIConfig):
with pytest.raises(litellm.utils.UnsupportedParamsError):
config.map_openai_params(
non_default_params={"reasoning_effort": effort},
non_default_params={"reasoning_effort": "xhigh"},
optional_params={},
model="gpt-5.1",
drop_params=False,
)
@pytest.mark.parametrize("effort", ["max", "ultra"])
def test_gpt5_drops_opt_in_reasoning_efforts_when_requested(config: OpenAIConfig, effort: str):
def test_gpt5_drops_xhigh_when_requested(config: OpenAIConfig):
params = config.map_openai_params(
non_default_params={"reasoning_effort": effort},
non_default_params={"reasoning_effort": "xhigh"},
optional_params={},
model="gpt-5.1",
drop_params=True,

View file

@ -8927,7 +8927,11 @@ def test_model_group_info_intersects_supported_reasoning_efforts():
assert result.supported_reasoning_efforts == ("minimal", "low", "medium", "high")
def test_model_group_info_reasoning_efforts_ignore_deployments_without_metadata():
def test_model_group_info_reasoning_efforts_empty_when_a_deployment_is_off_the_map():
"""The router fills every ModelInfo key, so a deployment absent from the model map arrives with
supports_reasoning None rather than with the key missing, and the group can no longer promise a
level on its behalf. The dashboard reads the empty result as "nothing known" and falls back to
the capability-blind picker, which is what it showed before this field existed."""
router = litellm.Router(
model_list=[
{
@ -8952,7 +8956,7 @@ def test_model_group_info_reasoning_efforts_ignore_deployments_without_metadata(
"supports_reasoning": True,
"supports_max_reasoning_effort": True,
}
return {"key": model_name, "litellm_provider": "openai", "mode": "chat"}
return {"key": model_name, "litellm_provider": "openai", "mode": None, "supports_reasoning": None}
with patch.object(router, "get_deployment_model_info", side_effect=_model_info):
result = router._set_model_group_info(
@ -8961,4 +8965,29 @@ def test_model_group_info_reasoning_efforts_ignore_deployments_without_metadata(
)
assert result is not None
assert result.supported_reasoning_efforts == ("none", "minimal", "low", "medium", "high", "max")
assert result.supported_reasoning_efforts == ()
@pytest.mark.parametrize(
"configured, expected",
[
("high", ("high",)),
(["low", "high"], ("low", "high")),
(17, None),
({"effort": "high"}, None),
([1, 2], None),
],
)
def test_model_group_info_tolerates_any_configured_reasoning_efforts_shape(configured, expected):
"""Deployment model_info is splatted into ModelGroupInfo, so this key arrives with whatever an
operator wrote in the config. A shape pydantic cannot validate used to fail the whole
/model_group/info response, not just the group carrying it."""
from litellm.types.router import ModelGroupInfo
info = ModelGroupInfo(
model_group="g",
providers=["openai"],
supported_reasoning_efforts=configured,
)
assert info.supported_reasoning_efforts == expected

View file

@ -968,6 +968,24 @@ describe("ComplexityRouterConfig per-model effort filtering", () => {
expect(options).toEqual(["Default", "none", "minimal", "low", "medium", "high", "xhigh"]);
});
it("falls back to every effort when the group intersects to nothing", async () => {
renderWithProviders(
<ComplexityRouterConfig
{...baseProps}
modelInfo={[
...mockModelInfo.filter((model) => model.model_group !== "claude-3-opus"),
{ model_group: "claude-3-opus", mode: "chat", supports_reasoning: true, supported_reasoning_efforts: [] },
]}
/>,
);
const user = userEvent.setup();
await user.click(
screen.getByRole("combobox", { name: "Reasoning effort for claude-3-opus in the Reasoning tier" }),
);
const options = (await screen.findAllByRole("option")).map((option) => option.textContent);
expect(options).toEqual(["Default", "none", "minimal", "low", "medium", "high", "xhigh"]);
});
// Hand-authored configs can carry a level outside the supported set (e.g. max); it must render
// and stay clearable rather than being masked as Default.
it("keeps showing a stored effort outside the supported set", () => {

View file

@ -226,6 +226,20 @@ export const effectiveTierLabel = (tier: keyof ComplexityTiers, tierLabels: Comp
export const planModeEligibleTiers = (tiers: ComplexityTiers): Array<keyof ComplexityTiers> =>
TIER_KEYS.filter((tier) => (tiers[tier] ?? []).length > 0);
/**
* The backend list is the per-group intersection of accepted effort levels. An empty intersection
* carries no more information than a proxy that does not send the field yet (one deployment absent
* from the model map empties it), so both fall back to the coarse supports_reasoning gate with every
* level offered, which is what this dropdown showed before the field existed.
*/
const effortOptionsForModel = (model: ModelGroup): string[] => {
const advertised = model.supported_reasoning_efforts ?? [];
if (advertised.length > 0) {
return [...advertised];
}
return model.supports_reasoning ? [...REASONING_EFFORT_OPTIONS] : [];
};
const ComplexityRouterConfig: React.FC<ComplexityRouterConfigProps> = ({
modelInfo,
value,
@ -251,16 +265,11 @@ const ComplexityRouterConfig: React.FC<ComplexityRouterConfigProps> = ({
const derivedDefaultModel = resolveComplexityDefaultModel(value.tiers);
const defaultModel = resolveComplexityDefaultModel(value.tiers, value.default_model);
// Embedding models can't serve a chat-completion role, so they're excluded here.
// The backend list is the per-group intersection of accepted effort levels; when a proxy does not
// send it yet, fall back to the coarse supports_reasoning gate with every level offered.
const effortOptionsByModel: Record<string, string[]> = Object.fromEntries(
modelInfo.map((model) => [
model.model_group,
model.supported_reasoning_efforts ?? (model.supports_reasoning ? [...REASONING_EFFORT_OPTIONS] : []),
]),
modelInfo.map((model) => [model.model_group, effortOptionsForModel(model)]),
);
// Embedding models can't serve a chat-completion role, so they're excluded here.
const modelOptions = modelInfo
.filter((model) => model.mode !== "embedding")
.map((model) => ({