diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index c47fa3944c1..4dea3e80cfe 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -114,6 +114,18 @@ def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) - return _without_params(params, frozenset(("reasoning_effort",))) +def non_string_reasoning_effort(params: Mapping[str, object]) -> frozenset[str]: + """``reasoning_effort`` when the request sends it as anything but a string (an int, a list, an object). + + AWS's Chat Completions endpoint answers such a value with a 400 where Converse silently dropped it, so the + native config refuses it before the call, or drops it under ``drop_params`` so AWS applies its default effort. + """ + effort: Final = params.get("reasoning_effort") + if effort is None or isinstance(effort, str): + return frozenset() + return frozenset(("reasoning_effort",)) + + def _held_close_tag_prefix(text: str) -> int: return next( ( @@ -355,7 +367,17 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): drop_params=drop_params, replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens, ) + malformed_effort: Final = non_string_reasoning_effort(non_default_params) refused_while_reasoning: Final = chat_completions_params_refused_while_reasoning(model, non_default_params) + if malformed_effort and not (litellm.drop_params or drop_params): + raise litellm.utils.UnsupportedParamsError( + message=( + f"{model} takes reasoning_effort as a string on Bedrock's Chat Completions endpoint, not " + f"{type(non_default_params['reasoning_effort']).__name__}. Send one of its named efforts, or " + "set `litellm.drop_params = True` to drop it" + ), + status_code=400, + ) if refused_while_reasoning and not (litellm.drop_params or drop_params): raise litellm.utils.UnsupportedParamsError( message=( @@ -367,7 +389,8 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): ) return dict( # mutable-ok: get_optional_params keeps filling this dict without_refused_reasoning_effort( - model, with_max_completion_tokens(_without_params(mapped, refused_while_reasoning)) + model, + with_max_completion_tokens(_without_params(mapped, refused_while_reasoning | malformed_effort)), ) ) diff --git a/tests/integration/providers/test_bedrock_runtime_chat_completions_sad_wire.py b/tests/integration/providers/test_bedrock_runtime_chat_completions_sad_wire.py index a96fe884c59..357d0ea4d0c 100644 --- a/tests/integration/providers/test_bedrock_runtime_chat_completions_sad_wire.py +++ b/tests/integration/providers/test_bedrock_runtime_chat_completions_sad_wire.py @@ -304,17 +304,9 @@ def test_bad_key_on_the_long_version_model_is_refused_before_any_route(gateway: assert [marker_of(request) for request in received] == [control_marker], _routes(received) -@pytest.mark.parametrize( - "effort", - [ - pytest.param(7, id="int"), - pytest.param(["high"], id="list"), - pytest.param("", id="empty"), - pytest.param("x" * 5120, id="five_kb"), - ], -) +@pytest.mark.parametrize("effort", [pytest.param("", id="empty"), pytest.param("x" * 5120, id="five_kb")]) def test_invalid_reasoning_effort_reaches_the_peer_and_its_400_reaches_the_caller( - gateway: Gateway, effort: JsonValue + gateway: Gateway, effort: str ) -> None: marker: Final = uuid.uuid4().hex with wire_server(respond) as wire, gateway.scenario() as scenario: @@ -328,6 +320,36 @@ def test_invalid_reasoning_effort_reaches_the_peer_and_its_400_reaches_the_calle _assert_row(_call_id(response), model, "failure") +NON_STRING_EFFORTS: Final = (pytest.param(7, id="int"), pytest.param(["high"], id="list")) + + +@pytest.mark.parametrize("effort", NON_STRING_EFFORTS) +def test_non_string_reasoning_effort_is_refused_before_any_wire_request(gateway: Gateway, effort: JsonValue) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, reasoning_effort=effort) + assert response.status_code == 400, response.text + message: Final = _error_message(response) + assert message.startswith("litellm.UnsupportedParamsError"), response.text + assert "reasoning_effort as a string" in message and "drop_params" in message, response.text + _assert_row(_call_id(response), model, "failure") + assert _routes(wire.drain()) == [] + + +@pytest.mark.parametrize("effort", NON_STRING_EFFORTS) +def test_drop_params_deployment_drops_a_non_string_reasoning_effort(gateway: Gateway, effort: JsonValue) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire, drop_params=True) + response: Final = _chat(gateway, model, marker, reasoning_effort=effort) + assert _content(response) == answer(marker), response.text + request: Final = _only_request(wire, marker) + assert target_of(request) == NATIVE_TARGET, request.body + assert "reasoning_effort" not in _body(request), request.body + _assert_row(string_value(_payload(response)["id"]), model, "success") + + def test_duplicated_reasoning_effort_key_lets_the_last_value_win(gateway: Gateway) -> None: marker: Final = uuid.uuid4().hex with wire_server(respond) as wire, gateway.scenario() as scenario: diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 06fff49d500..16ae1114402 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -668,16 +668,35 @@ def test_map_openai_params_keeps_reasoning_effort_low_for_grok(): @pytest.mark.parametrize("model", ["us.xai.grok-4.6", "global.openai.gpt-5.6-sol"]) -@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5]) -def test_map_openai_params_forwards_a_malformed_reasoning_effort_for_aws_to_refuse(model, reasoning_effort): +@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5], ids=["list", "object", "int"]) +def test_map_openai_params_refuses_a_non_string_reasoning_effort_without_drop_params(model, reasoning_effort): cfg = AmazonBedrockRuntimeChatCompletionsConfig() - mapped = cfg.map_openai_params( + with pytest.raises(litellm.UnsupportedParamsError, match="drop_params") as refused: + cfg.map_openai_params( + non_default_params={"reasoning_effort": reasoning_effort, "max_tokens": 64}, + optional_params={}, + model=model, + drop_params=False, + ) + assert refused.value.status_code == 400 + assert type(reasoning_effort).__name__ in str(refused.value) + + +@pytest.mark.parametrize("model", ["us.xai.grok-4.6", "global.openai.gpt-5.6-sol"]) +@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5], ids=["list", "object", "int"]) +@pytest.mark.parametrize("drop_params_via", ["request", "litellm.drop_params"]) +def test_map_openai_params_drops_a_non_string_reasoning_effort_under_drop_params( + monkeypatch, model, reasoning_effort, drop_params_via +): + monkeypatch.setattr(litellm, "drop_params", drop_params_via == "litellm.drop_params") + mapped = AmazonBedrockRuntimeChatCompletionsConfig().map_openai_params( non_default_params={"reasoning_effort": reasoning_effort, "max_tokens": 64}, optional_params={}, model=model, - drop_params=False, + drop_params=drop_params_via == "request", ) - assert mapped["reasoning_effort"] == reasoning_effort + assert "reasoning_effort" not in mapped + assert mapped["max_completion_tokens"] == 64 def test_map_openai_params_keeps_reasoning_effort_none_for_gpt56(): @@ -752,6 +771,28 @@ def test_refused_params_are_dropped_or_refused_before_reaching_aws(local_cost_ma assert param.keys().isdisjoint(json.loads(requests[0].content)) +@pytest.mark.parametrize("reasoning_effort", [3, ["high"]], ids=["int", "list"]) +def test_non_string_reasoning_effort_is_refused_or_dropped_before_reaching_aws( + local_cost_map, fake_aws_env, reasoning_effort +): + requests, client = _recording_client(json=_chat_completion_json("ok", "global.openai.gpt-5.6-sol")) + request = { + "model": "bedrock/global.openai.gpt-5.6-sol", + "messages": [{"role": "user", "content": "hello"}], + "reasoning_effort": reasoning_effort, + "client": client, + } + with pytest.raises(litellm.UnsupportedParamsError, match="reasoning_effort") as refused: + litellm.completion(**request) + assert refused.value.status_code == 400 + assert requests == [] + + litellm.completion(**request, drop_params=True) + + assert str(requests[0].url).endswith("/openai/v1/chat/completions") + assert "reasoning_effort" not in json.loads(requests[0].content) + + GPT_PARAMS_TIED_TO_REASONING_OFF = { "temperature": 0.2, "top_p": 0.9,