mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(reasoning): keep the chat gate on xhigh and stop empty groups zeroing the picker
The chat-completions gate only ever owned xhigh. Widening it to max and ultra made gpt-5.6 answer 400 on requests litellm itself converts to /v1/responses, where max is valid, because the gate runs before the bridge decision. No map entry asserts either flag, so the widened gate could only ever reject. An empty per-group intersection now falls back to the capability-blind level list in the dashboard, matching what the picker showed before the field existed, and ModelGroupInfo tolerates whatever shape an operator writes under supported_reasoning_efforts instead of failing the whole /model_group/info response.
This commit is contained in:
parent
6ec49ea7dd
commit
18528b1a63
7 changed files with 137 additions and 41 deletions
|
|
@ -222,8 +222,8 @@ class OpenAIGPT5Config(OpenAIGPTConfig):
|
|||
if "reasoning_effort" in optional_params:
|
||||
optional_params["reasoning_effort"] = normalized
|
||||
|
||||
if effective_effort in ("xhigh", "max", "ultra"):
|
||||
# xhigh/max/ultra are opt-in capabilities: only allow if the model explicitly supports them.
|
||||
if effective_effort == "xhigh":
|
||||
# xhigh is an opt-in capability: only allow if model explicitly supports it.
|
||||
if not self._supports_reasoning_effort_level(model, effective_effort):
|
||||
if litellm.drop_params or drop_params:
|
||||
non_default_params.pop("reasoning_effort", None)
|
||||
|
|
|
|||
|
|
@ -54,8 +54,12 @@ def _supports_none_reasoning_effort(model_info: Mapping[str, object]) -> bool:
|
|||
|
||||
|
||||
def resolve_supported_reasoning_efforts(model_info: Mapping[str, object]) -> tuple[str, ...] | None:
|
||||
"""None = no capability metadata for this deployment (e.g. a model absent from the model map,
|
||||
whose stub info carries no supports_reasoning key at all); () = reasoning unsupported."""
|
||||
"""None = the caller passed no supports_reasoning key at all; () = this deployment adds no
|
||||
effort levels to its group. The router always supplies the key, so a deployment absent from the
|
||||
model map arrives with supports_reasoning None and lands on (), the same answer a mapped
|
||||
non-reasoning model gets: the group cannot promise a level on behalf of a model nothing is known
|
||||
about. () therefore reads as "no usable answer" downstream, which is why the dashboard falls
|
||||
back to the capability-blind level list on an empty group rather than hiding the control."""
|
||||
if "supports_reasoning" not in model_info:
|
||||
return None
|
||||
if model_info.get("supports_reasoning") is not True:
|
||||
|
|
|
|||
|
|
@ -641,6 +641,19 @@ class ModelGroupInfo(BaseModel):
|
|||
supported_openai_params: list[str] | None = Field(default=[])
|
||||
configurable_clientside_auth_params: CONFIGURABLE_CLIENTSIDE_AUTH_PARAMS = None
|
||||
|
||||
@field_validator("supported_reasoning_efforts", mode="before")
|
||||
@classmethod
|
||||
def _accept_only_well_formed_reasoning_efforts(cls, value: object) -> tuple[str, ...] | None:
|
||||
"""Deployment model_info reaches this model through a **kwargs splat, so an operator can put
|
||||
any shape under this key. Anything that is not a level or a list of levels resolves to None
|
||||
and lets the computed intersection stand, because raising here fails the whole
|
||||
/model_group/info response rather than the one group that carries the bad value."""
|
||||
if isinstance(value, str):
|
||||
return (value,)
|
||||
if isinstance(value, (list, tuple)) and all(isinstance(level, str) for level in value):
|
||||
return tuple(value)
|
||||
return None
|
||||
|
||||
def __init__(self, **data) -> None:
|
||||
for field_name, field_type in get_type_hints(self.__class__).items():
|
||||
if field_type is bool and data.get(field_name) is None:
|
||||
|
|
|
|||
|
|
@ -1312,18 +1312,32 @@ def test_responses_gpt54_allow_temperature_effort_none(
|
|||
|
||||
|
||||
@pytest.mark.parametrize("model", ["gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"])
|
||||
def test_gpt5_6_rejects_reasoning_effort_max_on_chat_completions(config: OpenAIConfig, model: str):
|
||||
def test_gpt5_6_forwards_reasoning_effort_max_for_the_responses_bridge(config: OpenAIConfig, model: str):
|
||||
"""A chat request carrying tools or a reasoning summary is converted to /v1/responses further
|
||||
down main.py, and that surface accepts max. This runs before litellm has decided to bridge, so
|
||||
refusing max here would break the cursor thinking-max shape that works today. Plain chat
|
||||
completions still answer max with a provider 400, and the capability list below is what keeps
|
||||
the level out of the picker."""
|
||||
params = config.map_openai_params(
|
||||
non_default_params={"reasoning_effort": "max"},
|
||||
optional_params={},
|
||||
model=model,
|
||||
drop_params=False,
|
||||
)
|
||||
assert params["reasoning_effort"] == "max"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"])
|
||||
def test_gpt5_6_never_advertises_reasoning_effort_max(model: str):
|
||||
"""/v1/chat/completions answers max with "Unsupported value: 'reasoning_effort' does not support
|
||||
'max' with this model. Supported values are: 'none', 'low', 'medium', 'high', and 'xhigh'" on
|
||||
every gpt-5.6 snapshot, so no gpt-5.6 map entry asserts supports_max_reasoning_effort and the
|
||||
level is refused here instead of being forwarded into a provider 400."""
|
||||
with pytest.raises(litellm.utils.UnsupportedParamsError):
|
||||
config.map_openai_params(
|
||||
non_default_params={"reasoning_effort": "max"},
|
||||
optional_params={},
|
||||
model=model,
|
||||
drop_params=False,
|
||||
)
|
||||
'max' with this model. Supported values are: 'none', 'low', 'medium', 'high', and 'xhigh'", so no
|
||||
gpt-5.6 entry asserts supports_max_reasoning_effort and the advertised set stops at xhigh."""
|
||||
from litellm.router_utils.reasoning_effort_capability import resolve_supported_reasoning_efforts
|
||||
|
||||
resolved = resolve_supported_reasoning_efforts(litellm.get_model_info(model))
|
||||
assert resolved is not None
|
||||
assert "max" not in resolved
|
||||
assert "xhigh" in resolved
|
||||
|
||||
|
||||
def test_gpt5_6_keeps_reasoning_effort_max_on_the_responses_api(
|
||||
|
|
@ -1341,34 +1355,43 @@ def test_gpt5_6_keeps_reasoning_effort_max_on_the_responses_api(
|
|||
assert params["reasoning"] == {"effort": "max"}
|
||||
|
||||
|
||||
def test_gpt5_6_rejects_reasoning_effort_ultra_until_the_map_opts_in(config: OpenAIConfig):
|
||||
# ultra is plumbed as an opt-in level but no map entry asserts it: OpenAI's model guidance
|
||||
# documents effort values only up to max, and the builder guide frames ultra as multi-agent
|
||||
# orchestration. A verified supports_ultra_reasoning_effort flag lights this up with no code.
|
||||
with pytest.raises(litellm.utils.UnsupportedParamsError):
|
||||
config.map_openai_params(
|
||||
non_default_params={"reasoning_effort": "ultra"},
|
||||
optional_params={},
|
||||
model="gpt-5.6",
|
||||
drop_params=False,
|
||||
)
|
||||
def test_gpt5_6_never_advertises_reasoning_effort_ultra():
|
||||
"""ultra is plumbed as an opt-in level but no map entry asserts it: OpenAI's model guidance
|
||||
documents effort values only up to max, and /v1/responses answers ultra with a 400. A verified
|
||||
supports_ultra_reasoning_effort flag lights it up with no code change."""
|
||||
from litellm.router_utils.reasoning_effort_capability import resolve_supported_reasoning_efforts
|
||||
|
||||
resolved = resolve_supported_reasoning_efforts(litellm.get_model_info("gpt-5.6"))
|
||||
assert resolved is not None
|
||||
assert "ultra" not in resolved
|
||||
|
||||
|
||||
@pytest.mark.parametrize("effort", ["max", "ultra"])
|
||||
def test_gpt5_rejects_opt_in_reasoning_efforts_for_other_models(config: OpenAIConfig, effort: str):
|
||||
def test_gpt5_forwards_levels_the_chat_gate_does_not_own(config: OpenAIConfig, effort: str):
|
||||
"""Only xhigh is gated on this surface. max and ultra reach the provider (or the responses
|
||||
bridge) and are answered there, which is what happened before per-group capabilities existed."""
|
||||
params = config.map_openai_params(
|
||||
non_default_params={"reasoning_effort": effort},
|
||||
optional_params={},
|
||||
model="gpt-5.1",
|
||||
drop_params=False,
|
||||
)
|
||||
assert params["reasoning_effort"] == effort
|
||||
|
||||
|
||||
def test_gpt5_rejects_xhigh_for_models_without_the_flag(config: OpenAIConfig):
|
||||
with pytest.raises(litellm.utils.UnsupportedParamsError):
|
||||
config.map_openai_params(
|
||||
non_default_params={"reasoning_effort": effort},
|
||||
non_default_params={"reasoning_effort": "xhigh"},
|
||||
optional_params={},
|
||||
model="gpt-5.1",
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("effort", ["max", "ultra"])
|
||||
def test_gpt5_drops_opt_in_reasoning_efforts_when_requested(config: OpenAIConfig, effort: str):
|
||||
def test_gpt5_drops_xhigh_when_requested(config: OpenAIConfig):
|
||||
params = config.map_openai_params(
|
||||
non_default_params={"reasoning_effort": effort},
|
||||
non_default_params={"reasoning_effort": "xhigh"},
|
||||
optional_params={},
|
||||
model="gpt-5.1",
|
||||
drop_params=True,
|
||||
|
|
|
|||
|
|
@ -8927,7 +8927,11 @@ def test_model_group_info_intersects_supported_reasoning_efforts():
|
|||
assert result.supported_reasoning_efforts == ("minimal", "low", "medium", "high")
|
||||
|
||||
|
||||
def test_model_group_info_reasoning_efforts_ignore_deployments_without_metadata():
|
||||
def test_model_group_info_reasoning_efforts_empty_when_a_deployment_is_off_the_map():
|
||||
"""The router fills every ModelInfo key, so a deployment absent from the model map arrives with
|
||||
supports_reasoning None rather than with the key missing, and the group can no longer promise a
|
||||
level on its behalf. The dashboard reads the empty result as "nothing known" and falls back to
|
||||
the capability-blind picker, which is what it showed before this field existed."""
|
||||
router = litellm.Router(
|
||||
model_list=[
|
||||
{
|
||||
|
|
@ -8952,7 +8956,7 @@ def test_model_group_info_reasoning_efforts_ignore_deployments_without_metadata(
|
|||
"supports_reasoning": True,
|
||||
"supports_max_reasoning_effort": True,
|
||||
}
|
||||
return {"key": model_name, "litellm_provider": "openai", "mode": "chat"}
|
||||
return {"key": model_name, "litellm_provider": "openai", "mode": None, "supports_reasoning": None}
|
||||
|
||||
with patch.object(router, "get_deployment_model_info", side_effect=_model_info):
|
||||
result = router._set_model_group_info(
|
||||
|
|
@ -8961,4 +8965,29 @@ def test_model_group_info_reasoning_efforts_ignore_deployments_without_metadata(
|
|||
)
|
||||
|
||||
assert result is not None
|
||||
assert result.supported_reasoning_efforts == ("none", "minimal", "low", "medium", "high", "max")
|
||||
assert result.supported_reasoning_efforts == ()
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"configured, expected",
|
||||
[
|
||||
("high", ("high",)),
|
||||
(["low", "high"], ("low", "high")),
|
||||
(17, None),
|
||||
({"effort": "high"}, None),
|
||||
([1, 2], None),
|
||||
],
|
||||
)
|
||||
def test_model_group_info_tolerates_any_configured_reasoning_efforts_shape(configured, expected):
|
||||
"""Deployment model_info is splatted into ModelGroupInfo, so this key arrives with whatever an
|
||||
operator wrote in the config. A shape pydantic cannot validate used to fail the whole
|
||||
/model_group/info response, not just the group carrying it."""
|
||||
from litellm.types.router import ModelGroupInfo
|
||||
|
||||
info = ModelGroupInfo(
|
||||
model_group="g",
|
||||
providers=["openai"],
|
||||
supported_reasoning_efforts=configured,
|
||||
)
|
||||
|
||||
assert info.supported_reasoning_efforts == expected
|
||||
|
|
|
|||
|
|
@ -968,6 +968,24 @@ describe("ComplexityRouterConfig per-model effort filtering", () => {
|
|||
expect(options).toEqual(["Default", "none", "minimal", "low", "medium", "high", "xhigh"]);
|
||||
});
|
||||
|
||||
it("falls back to every effort when the group intersects to nothing", async () => {
|
||||
renderWithProviders(
|
||||
<ComplexityRouterConfig
|
||||
{...baseProps}
|
||||
modelInfo={[
|
||||
...mockModelInfo.filter((model) => model.model_group !== "claude-3-opus"),
|
||||
{ model_group: "claude-3-opus", mode: "chat", supports_reasoning: true, supported_reasoning_efforts: [] },
|
||||
]}
|
||||
/>,
|
||||
);
|
||||
const user = userEvent.setup();
|
||||
await user.click(
|
||||
screen.getByRole("combobox", { name: "Reasoning effort for claude-3-opus in the Reasoning tier" }),
|
||||
);
|
||||
const options = (await screen.findAllByRole("option")).map((option) => option.textContent);
|
||||
expect(options).toEqual(["Default", "none", "minimal", "low", "medium", "high", "xhigh"]);
|
||||
});
|
||||
|
||||
// Hand-authored configs can carry a level outside the supported set (e.g. max); it must render
|
||||
// and stay clearable rather than being masked as Default.
|
||||
it("keeps showing a stored effort outside the supported set", () => {
|
||||
|
|
|
|||
|
|
@ -226,6 +226,20 @@ export const effectiveTierLabel = (tier: keyof ComplexityTiers, tierLabels: Comp
|
|||
export const planModeEligibleTiers = (tiers: ComplexityTiers): Array<keyof ComplexityTiers> =>
|
||||
TIER_KEYS.filter((tier) => (tiers[tier] ?? []).length > 0);
|
||||
|
||||
/**
|
||||
* The backend list is the per-group intersection of accepted effort levels. An empty intersection
|
||||
* carries no more information than a proxy that does not send the field yet (one deployment absent
|
||||
* from the model map empties it), so both fall back to the coarse supports_reasoning gate with every
|
||||
* level offered, which is what this dropdown showed before the field existed.
|
||||
*/
|
||||
const effortOptionsForModel = (model: ModelGroup): string[] => {
|
||||
const advertised = model.supported_reasoning_efforts ?? [];
|
||||
if (advertised.length > 0) {
|
||||
return [...advertised];
|
||||
}
|
||||
return model.supports_reasoning ? [...REASONING_EFFORT_OPTIONS] : [];
|
||||
};
|
||||
|
||||
const ComplexityRouterConfig: React.FC<ComplexityRouterConfigProps> = ({
|
||||
modelInfo,
|
||||
value,
|
||||
|
|
@ -251,16 +265,11 @@ const ComplexityRouterConfig: React.FC<ComplexityRouterConfigProps> = ({
|
|||
const derivedDefaultModel = resolveComplexityDefaultModel(value.tiers);
|
||||
const defaultModel = resolveComplexityDefaultModel(value.tiers, value.default_model);
|
||||
|
||||
// Embedding models can't serve a chat-completion role, so they're excluded here.
|
||||
// The backend list is the per-group intersection of accepted effort levels; when a proxy does not
|
||||
// send it yet, fall back to the coarse supports_reasoning gate with every level offered.
|
||||
const effortOptionsByModel: Record<string, string[]> = Object.fromEntries(
|
||||
modelInfo.map((model) => [
|
||||
model.model_group,
|
||||
model.supported_reasoning_efforts ?? (model.supports_reasoning ? [...REASONING_EFFORT_OPTIONS] : []),
|
||||
]),
|
||||
modelInfo.map((model) => [model.model_group, effortOptionsForModel(model)]),
|
||||
);
|
||||
|
||||
// Embedding models can't serve a chat-completion role, so they're excluded here.
|
||||
const modelOptions = modelInfo
|
||||
.filter((model) => model.mode !== "embedding")
|
||||
.map((model) => ({
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue