fix(inputs): send reasoning_effort as configured; no route-specific handling

This commit is contained in:
Ahmed Allam 2026-10-02 15:14:18 +00:00 • committed by Ahmed Allam
parent e1ec259ac2
commit 8ab5e39c06
5 changed files with 17 additions and 81 deletions

View file

@ -125,20 +125,6 @@ _TRANSIENT_MODEL_RETRY_BASE_DELAY_S = 2.0
_TRANSIENT_MODEL_RETRY_MAX_DELAY_S = 90.0
_TOOLS_WITH_REASONING_EFFORT = "Function tools with reasoning_effort are not supported"
_TOOLS_WITH_REASONING_EFFORT_HINT = (
"This model takes reasoning_effort together with tools on the Responses API only: "
"set STRIX_API_TYPE=responses, or STRIX_REASONING_EFFORT=none to stay on chat completions."
)
def _failure_text(exc: BaseException) -> str:
text = request_log.failure_text(exc)
if _model_error_status_code(exc) == 400 and _TOOLS_WITH_REASONING_EFFORT in text:
text = f"{text} {_TOOLS_WITH_REASONING_EFFORT_HINT}"
return text
def _model_error_status_code(exc: BaseException) -> int | None:
code = getattr(exc, "status_code", None)
return code if isinstance(code, int) else None
@ -707,7 +693,7 @@ async def _run_cycle_parked(
raise
except Exception as exc:
logger.exception("error escaped the run cycle for %s; parking as failed", agent_id)
await coordinator.set_status(agent_id, "failed", error=_failure_text(exc))
await coordinator.set_status(agent_id, "failed", error=request_log.failure_text(exc))
await notify_parent_on_terminal(coordinator, agent_id, "failed")
return None
@ -868,7 +854,9 @@ async def _run_cycle( # noqa: PLR0912, PLR0915
await _salvage_stream_to_session(session, pre_run_items, stream, agent_id)
if isinstance(exc, ProviderRefusalError):
logger.warning("agent %s refused by the model provider: %s", agent_id, exc)
await coordinator.set_status(agent_id, "failed", error=_failure_text(exc))
await coordinator.set_status(
agent_id, "failed", error=request_log.failure_text(exc)
)
await notify_parent_on_terminal(coordinator, agent_id, "failed")
return None
if isinstance(exc, MaxTurnsExceeded):
@ -882,7 +870,7 @@ async def _run_cycle( # noqa: PLR0912, PLR0915
# non-interactive agent's task: a child that dies still owes its parent a
# report, and the parent would otherwise wait out its timeout on a message
# the dead child can no longer send.
await coordinator.set_status(agent_id, status, error=_failure_text(exc))
await coordinator.set_status(agent_id, status, error=request_log.failure_text(exc))
await notify_parent_on_terminal(coordinator, agent_id, status)
if not interactive:
raise

View file

@ -28,7 +28,7 @@ logger = logging.getLogger(__name__)
if TYPE_CHECKING:
from strix.config.settings import ApiType, ReasoningEffort
from strix.config.settings import ReasoningEffort
def _accepts_required_tool_choice(model_name: str | None) -> bool:
@ -257,9 +257,7 @@ def make_model_settings(
prompt_cache: bool = True,
extra_headers: dict[str, str] | None = None,
has_tools: bool = True,
api_type: ApiType | None = None,
) -> ModelSettings:
"""``api_type`` is the resolved SDK-native OpenAI route, when known."""
headers = _request_headers(model_name, extra_headers)
model_settings = ModelSettings(
parallel_tool_calls=False if has_tools else None,
@ -268,13 +266,12 @@ def make_model_settings(
extra_args=request_timeout_extra_args(request_timeout),
extra_headers=headers,
)
if reasoning_effort is not None and model_supports_reasoning(model_name):
if reasoning_effort != "none":
model_settings = model_settings.resolve(_reasoning_settings(reasoning_effort))
elif _explicit_none_required(model_name, api_type):
model_settings = model_settings.resolve(
ModelSettings(reasoning=Reasoning(effort="none"))
)
if (
reasoning_effort is not None
and reasoning_effort != "none"
and model_supports_reasoning(model_name)
):
model_settings = model_settings.resolve(_reasoning_settings(reasoning_effort))
if force_required_tool_choice and _accepts_required_tool_choice(model_name):
model_settings = model_settings.resolve(ModelSettings(tool_choice="required"))
@ -288,14 +285,6 @@ def make_model_settings(
return model_settings
def _explicit_none_required(model_name: str, api_type: ApiType | None) -> bool:
"""OpenAI's chat completions reason at a default effort when the field is
absent, and the newer reasoning models reject function tools at any effort
but ``none``, so ``none`` has to be sent, not left out. LiteLLM's own route
maps the field per provider and is left alone."""
return api_type == "chat_completions" and not routes_through_litellm(model_name)
def _request_headers(
model_name: str, extra_headers: dict[str, str] | None
) -> dict[str, str] | None:

View file

@ -377,7 +377,6 @@ async def run_strix_scan(
request_timeout=settings.llm.timeout,
prompt_cache=settings.llm.prompt_cache,
extra_headers=settings.llm.extra_headers,
api_type="chat_completions" if chat_completions_tools else "responses",
)
run_config = RunConfig(
model=resolved_model,

View file

@ -88,23 +88,6 @@ def test_client_errors_are_not_transient() -> None:
assert execution._is_transient_model_error(ValueError("nope")) is False
def test_tools_with_reasoning_effort_rejection_carries_a_route_hint() -> None:
rejected = BadRequestError(
"Error code: 400 - {'error': {'message': \"Function tools with reasoning_effort are "
"not supported for gpt-5.6-sol in /v1/chat/completions. To use function tools, use "
"/v1/responses or set reasoning_effort to 'none'.\", 'param': 'reasoning_effort'}}",
response=httpx.Response(400, request=_request()),
body=None,
)
text = execution._failure_text(rejected)
assert text.startswith("Error code: 400")
assert "STRIX_API_TYPE=responses" in text
assert "STRIX_REASONING_EFFORT=none" in text
other = BadRequestError("bad", response=httpx.Response(400, request=_request()), body=None)
assert execution._failure_text(other) == "bad"
class _FakeStream:
def __init__(self, exc: BaseException | None = None) -> None:
self._exc = exc

View file

@ -432,31 +432,8 @@ def test_user_headers_override_openrouter_attribution() -> None:
assert headers["HTTP-Referer"] == "https://strix.ai"
def test_reasoning_effort_none_sent_explicitly_on_chat_completions() -> None:
# Absent the field, OpenAI's chat completions reason at a default effort,
# which the newer models reject together with function tools.
for model in ("gpt-daybreak-blue-latest", "gpt-5.6-sol", "openai/gpt-5.4"):
chat = make_model_settings("none", model_name=model, api_type="chat_completions")
assert chat.reasoning is not None, model
assert chat.reasoning.effort == "none"
def test_reasoning_effort_none_omitted_off_chat_completions() -> None:
assert (
make_model_settings("none", model_name="gpt-5.6-sol", api_type="responses").reasoning
is None
)
assert make_model_settings("none", model_name="gpt-5.6-sol", api_type=None).reasoning is None
litellm_route = make_model_settings(
"none", model_name="litellm/gpt-5.6-sol", api_type="chat_completions"
)
assert litellm_route.reasoning is None
def test_reasoning_effort_sent_as_configured_on_every_route() -> None:
for api_type in ("chat_completions", "responses", None):
settings = make_model_settings(
"high", model_name="gpt-daybreak-blue-latest", api_type=api_type
)
assert settings.reasoning is not None, api_type
assert settings.reasoning.effort == "high"
def test_reasoning_effort_sent_as_configured_and_none_omitted() -> None:
assert make_model_settings("none", model_name="gpt-5.6-sol").reasoning is None
settings = make_model_settings("high", model_name="gpt-daybreak-blue-latest")
assert settings.reasoning is not None
assert settings.reasoning.effort == "high"