fix(bedrock): refuse or drop a non-string reasoning_effort before the native chat completions call
Some checks are pending
LiteLLM Rust / rust-lint (push) Waiting to run
LiteLLM Rust / rust-test (push) Waiting to run
LiteLLM Rust / rust-wheel (push) Waiting to run

A reasoning_effort sent as an int, a list, or an object on a GPT 5.6+ deployment the native
route serves now answers 400 from litellm before any wire request, naming the type and the
drop_params way out, and is dropped under drop_params so AWS applies its default effort, the
way Converse dropped it on main. The tip since a0cef91f0b forwarded it for AWS to refuse
This commit is contained in:
mateo-berri 2026-10-01 19:00:09 -07:00
parent 2f27eb3f36
commit 6daf0b7136
3 changed files with 102 additions and 16 deletions

View file

@ -114,6 +114,18 @@ def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) -
return _without_params(params, frozenset(("reasoning_effort",)))
def non_string_reasoning_effort(params: Mapping[str, object]) -> frozenset[str]:
"""``reasoning_effort`` when the request sends it as anything but a string (an int, a list, an object).
AWS's Chat Completions endpoint answers such a value with a 400 where Converse silently dropped it, so the
native config refuses it before the call, or drops it under ``drop_params`` so AWS applies its default effort.
"""
effort: Final = params.get("reasoning_effort")
if effort is None or isinstance(effort, str):
return frozenset()
return frozenset(("reasoning_effort",))
def _held_close_tag_prefix(text: str) -> int:
return next(
(
@ -355,7 +367,17 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
drop_params=drop_params,
replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens,
)
malformed_effort: Final = non_string_reasoning_effort(non_default_params)
refused_while_reasoning: Final = chat_completions_params_refused_while_reasoning(model, non_default_params)
if malformed_effort and not (litellm.drop_params or drop_params):
raise litellm.utils.UnsupportedParamsError(
message=(
f"{model} takes reasoning_effort as a string on Bedrock's Chat Completions endpoint, not "
f"{type(non_default_params['reasoning_effort']).__name__}. Send one of its named efforts, or "
"set `litellm.drop_params = True` to drop it"
),
status_code=400,
)
if refused_while_reasoning and not (litellm.drop_params or drop_params):
raise litellm.utils.UnsupportedParamsError(
message=(
@ -367,7 +389,8 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
)
return dict( # mutable-ok: get_optional_params keeps filling this dict
without_refused_reasoning_effort(
model, with_max_completion_tokens(_without_params(mapped, refused_while_reasoning))
model,
with_max_completion_tokens(_without_params(mapped, refused_while_reasoning | malformed_effort)),
)
)

View file

@ -304,17 +304,9 @@ def test_bad_key_on_the_long_version_model_is_refused_before_any_route(gateway:
assert [marker_of(request) for request in received] == [control_marker], _routes(received)
@pytest.mark.parametrize(
"effort",
[
pytest.param(7, id="int"),
pytest.param(["high"], id="list"),
pytest.param("", id="empty"),
pytest.param("x" * 5120, id="five_kb"),
],
)
@pytest.mark.parametrize("effort", [pytest.param("", id="empty"), pytest.param("x" * 5120, id="five_kb")])
def test_invalid_reasoning_effort_reaches_the_peer_and_its_400_reaches_the_caller(
gateway: Gateway, effort: JsonValue
gateway: Gateway, effort: str
) -> None:
marker: Final = uuid.uuid4().hex
with wire_server(respond) as wire, gateway.scenario() as scenario:
@ -328,6 +320,36 @@ def test_invalid_reasoning_effort_reaches_the_peer_and_its_400_reaches_the_calle
_assert_row(_call_id(response), model, "failure")
NON_STRING_EFFORTS: Final = (pytest.param(7, id="int"), pytest.param(["high"], id="list"))
@pytest.mark.parametrize("effort", NON_STRING_EFFORTS)
def test_non_string_reasoning_effort_is_refused_before_any_wire_request(gateway: Gateway, effort: JsonValue) -> None:
marker: Final = uuid.uuid4().hex
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = _deployment(scenario, wire)
response: Final = _chat(gateway, model, marker, reasoning_effort=effort)
assert response.status_code == 400, response.text
message: Final = _error_message(response)
assert message.startswith("litellm.UnsupportedParamsError"), response.text
assert "reasoning_effort as a string" in message and "drop_params" in message, response.text
_assert_row(_call_id(response), model, "failure")
assert _routes(wire.drain()) == []
@pytest.mark.parametrize("effort", NON_STRING_EFFORTS)
def test_drop_params_deployment_drops_a_non_string_reasoning_effort(gateway: Gateway, effort: JsonValue) -> None:
marker: Final = uuid.uuid4().hex
with wire_server(respond) as wire, gateway.scenario() as scenario:
model: Final = _deployment(scenario, wire, drop_params=True)
response: Final = _chat(gateway, model, marker, reasoning_effort=effort)
assert _content(response) == answer(marker), response.text
request: Final = _only_request(wire, marker)
assert target_of(request) == NATIVE_TARGET, request.body
assert "reasoning_effort" not in _body(request), request.body
_assert_row(string_value(_payload(response)["id"]), model, "success")
def test_duplicated_reasoning_effort_key_lets_the_last_value_win(gateway: Gateway) -> None:
marker: Final = uuid.uuid4().hex
with wire_server(respond) as wire, gateway.scenario() as scenario:

View file

@ -668,16 +668,35 @@ def test_map_openai_params_keeps_reasoning_effort_low_for_grok():
@pytest.mark.parametrize("model", ["us.xai.grok-4.6", "global.openai.gpt-5.6-sol"])
@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5])
def test_map_openai_params_forwards_a_malformed_reasoning_effort_for_aws_to_refuse(model, reasoning_effort):
@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5], ids=["list", "object", "int"])
def test_map_openai_params_refuses_a_non_string_reasoning_effort_without_drop_params(model, reasoning_effort):
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
mapped = cfg.map_openai_params(
with pytest.raises(litellm.UnsupportedParamsError, match="drop_params") as refused:
cfg.map_openai_params(
non_default_params={"reasoning_effort": reasoning_effort, "max_tokens": 64},
optional_params={},
model=model,
drop_params=False,
)
assert refused.value.status_code == 400
assert type(reasoning_effort).__name__ in str(refused.value)
@pytest.mark.parametrize("model", ["us.xai.grok-4.6", "global.openai.gpt-5.6-sol"])
@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5], ids=["list", "object", "int"])
@pytest.mark.parametrize("drop_params_via", ["request", "litellm.drop_params"])
def test_map_openai_params_drops_a_non_string_reasoning_effort_under_drop_params(
monkeypatch, model, reasoning_effort, drop_params_via
):
monkeypatch.setattr(litellm, "drop_params", drop_params_via == "litellm.drop_params")
mapped = AmazonBedrockRuntimeChatCompletionsConfig().map_openai_params(
non_default_params={"reasoning_effort": reasoning_effort, "max_tokens": 64},
optional_params={},
model=model,
drop_params=False,
drop_params=drop_params_via == "request",
)
assert mapped["reasoning_effort"] == reasoning_effort
assert "reasoning_effort" not in mapped
assert mapped["max_completion_tokens"] == 64
def test_map_openai_params_keeps_reasoning_effort_none_for_gpt56():
@ -752,6 +771,28 @@ def test_refused_params_are_dropped_or_refused_before_reaching_aws(local_cost_ma
assert param.keys().isdisjoint(json.loads(requests[0].content))
@pytest.mark.parametrize("reasoning_effort", [3, ["high"]], ids=["int", "list"])
def test_non_string_reasoning_effort_is_refused_or_dropped_before_reaching_aws(
local_cost_map, fake_aws_env, reasoning_effort
):
requests, client = _recording_client(json=_chat_completion_json("ok", "global.openai.gpt-5.6-sol"))
request = {
"model": "bedrock/global.openai.gpt-5.6-sol",
"messages": [{"role": "user", "content": "hello"}],
"reasoning_effort": reasoning_effort,
"client": client,
}
with pytest.raises(litellm.UnsupportedParamsError, match="reasoning_effort") as refused:
litellm.completion(**request)
assert refused.value.status_code == 400
assert requests == []
litellm.completion(**request, drop_params=True)
assert str(requests[0].url).endswith("/openai/v1/chat/completions")
assert "reasoning_effort" not in json.loads(requests[0].content)
GPT_PARAMS_TIED_TO_REASONING_OFF = {
"temperature": 0.2,
"top_p": 0.9,