mirror of
https://github.com/usestrix/strix.git
synced 2026-10-05 02:41:38 +00:00
Fix Bedrock reasoning for adaptive-thinking models (Claude Opus 4.8)
Models such as Claude Opus 4.8 on Bedrock reject the legacy
"thinking.type=enabled" payload that LiteLLM emits when it converts the
"reasoning_effort" argument, returning:
"thinking.type.enabled" is not supported for this model. Use
"thinking.type.adaptive" and "output_config.effort" to control
thinking behavior.
LiteLLM's model metadata already flags these models with
"supports_adaptive_thinking", but its request conversion still sends the
old shape. Detect that flag and pass the adaptive-thinking API via
extra_args (thinking.type=adaptive + output_config.effort) instead of the
reasoning_effort scalar. All other models keep their existing behavior.
This commit is contained in:
parent
cc23eeb65d
commit
25674b52f4
2 changed files with 38 additions and 5 deletions
|
|
@ -138,7 +138,7 @@ def uses_chat_completions_tool_schema(model_name: str, settings: Settings) -> bo
|
|||
return not model_supports_reasoning(model_name)
|
||||
|
||||
|
||||
def model_supports_reasoning(model_name: str) -> bool:
|
||||
def _model_cost_entry(model_name: str) -> dict[str, object] | None:
|
||||
import litellm
|
||||
|
||||
name = model_name.strip().lower()
|
||||
|
|
@ -149,9 +149,25 @@ def model_supports_reasoning(model_name: str) -> bool:
|
|||
entry = litellm.model_cost.get(name)
|
||||
if entry is None and "/" in name:
|
||||
entry = litellm.model_cost.get(name.rsplit("/", 1)[1])
|
||||
return entry
|
||||
|
||||
|
||||
def model_supports_reasoning(model_name: str) -> bool:
|
||||
entry = _model_cost_entry(model_name)
|
||||
return bool(entry and entry.get("supports_reasoning"))
|
||||
|
||||
|
||||
def model_supports_adaptive_thinking(model_name: str) -> bool:
|
||||
"""Whether the model uses Anthropic's newer adaptive-thinking API.
|
||||
|
||||
Models like Claude Opus 4.8 reject the legacy ``thinking.type=enabled`` shape
|
||||
that LiteLLM emits when it converts ``reasoning_effort``. They require
|
||||
``thinking.type=adaptive`` plus ``output_config.effort`` instead.
|
||||
"""
|
||||
entry = _model_cost_entry(model_name)
|
||||
return bool(entry and entry.get("supports_adaptive_thinking"))
|
||||
|
||||
|
||||
def is_known_openai_bare_model(model_name: str) -> bool:
|
||||
import litellm
|
||||
|
||||
|
|
|
|||
|
|
@ -8,7 +8,11 @@ from typing import TYPE_CHECKING, Any
|
|||
from agents.model_settings import ModelSettings
|
||||
from openai.types.shared import Reasoning
|
||||
|
||||
from strix.config.models import DEFAULT_MODEL_RETRY, model_supports_reasoning
|
||||
from strix.config.models import (
|
||||
DEFAULT_MODEL_RETRY,
|
||||
model_supports_adaptive_thinking,
|
||||
model_supports_reasoning,
|
||||
)
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
|
|
@ -121,9 +125,22 @@ def make_model_settings(
|
|||
and reasoning_effort != "none"
|
||||
and model_supports_reasoning(model_name)
|
||||
):
|
||||
model_settings = model_settings.resolve(
|
||||
ModelSettings(reasoning=Reasoning(effort=reasoning_effort)),
|
||||
)
|
||||
if model_supports_adaptive_thinking(model_name):
|
||||
# Newer Anthropic models (e.g. Claude Opus 4.8) reject the legacy
|
||||
# ``thinking.type=enabled`` shape that LiteLLM emits from
|
||||
# ``reasoning_effort``. Send the adaptive-thinking API instead.
|
||||
model_settings = model_settings.resolve(
|
||||
ModelSettings(
|
||||
extra_args={
|
||||
"thinking": {"type": "adaptive"},
|
||||
"output_config": {"effort": reasoning_effort},
|
||||
},
|
||||
),
|
||||
)
|
||||
else:
|
||||
model_settings = model_settings.resolve(
|
||||
ModelSettings(reasoning=Reasoning(effort=reasoning_effort)),
|
||||
)
|
||||
return model_settings
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue