mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(bedrock): refuse or drop a non-string reasoning_effort before the native chat completions call
A reasoning_effort sent as an int, a list, or an object on a GPT 5.6+ deployment the native
route serves now answers 400 from litellm before any wire request, naming the type and the
drop_params way out, and is dropped under drop_params so AWS applies its default effort, the
way Converse dropped it on main. The tip since a0cef91f0b forwarded it for AWS to refuse
This commit is contained in:
parent
2f27eb3f36
commit
6daf0b7136
3 changed files with 102 additions and 16 deletions
|
|
@ -114,6 +114,18 @@ def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) -
|
|||
return _without_params(params, frozenset(("reasoning_effort",)))
|
||||
|
||||
|
||||
def non_string_reasoning_effort(params: Mapping[str, object]) -> frozenset[str]:
|
||||
"""``reasoning_effort`` when the request sends it as anything but a string (an int, a list, an object).
|
||||
|
||||
AWS's Chat Completions endpoint answers such a value with a 400 where Converse silently dropped it, so the
|
||||
native config refuses it before the call, or drops it under ``drop_params`` so AWS applies its default effort.
|
||||
"""
|
||||
effort: Final = params.get("reasoning_effort")
|
||||
if effort is None or isinstance(effort, str):
|
||||
return frozenset()
|
||||
return frozenset(("reasoning_effort",))
|
||||
|
||||
|
||||
def _held_close_tag_prefix(text: str) -> int:
|
||||
return next(
|
||||
(
|
||||
|
|
@ -355,7 +367,17 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
|
|||
drop_params=drop_params,
|
||||
replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens,
|
||||
)
|
||||
malformed_effort: Final = non_string_reasoning_effort(non_default_params)
|
||||
refused_while_reasoning: Final = chat_completions_params_refused_while_reasoning(model, non_default_params)
|
||||
if malformed_effort and not (litellm.drop_params or drop_params):
|
||||
raise litellm.utils.UnsupportedParamsError(
|
||||
message=(
|
||||
f"{model} takes reasoning_effort as a string on Bedrock's Chat Completions endpoint, not "
|
||||
f"{type(non_default_params['reasoning_effort']).__name__}. Send one of its named efforts, or "
|
||||
"set `litellm.drop_params = True` to drop it"
|
||||
),
|
||||
status_code=400,
|
||||
)
|
||||
if refused_while_reasoning and not (litellm.drop_params or drop_params):
|
||||
raise litellm.utils.UnsupportedParamsError(
|
||||
message=(
|
||||
|
|
@ -367,7 +389,8 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
|
|||
)
|
||||
return dict( # mutable-ok: get_optional_params keeps filling this dict
|
||||
without_refused_reasoning_effort(
|
||||
model, with_max_completion_tokens(_without_params(mapped, refused_while_reasoning))
|
||||
model,
|
||||
with_max_completion_tokens(_without_params(mapped, refused_while_reasoning | malformed_effort)),
|
||||
)
|
||||
)
|
||||
|
||||
|
|
|
|||
|
|
@ -304,17 +304,9 @@ def test_bad_key_on_the_long_version_model_is_refused_before_any_route(gateway:
|
|||
assert [marker_of(request) for request in received] == [control_marker], _routes(received)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"effort",
|
||||
[
|
||||
pytest.param(7, id="int"),
|
||||
pytest.param(["high"], id="list"),
|
||||
pytest.param("", id="empty"),
|
||||
pytest.param("x" * 5120, id="five_kb"),
|
||||
],
|
||||
)
|
||||
@pytest.mark.parametrize("effort", [pytest.param("", id="empty"), pytest.param("x" * 5120, id="five_kb")])
|
||||
def test_invalid_reasoning_effort_reaches_the_peer_and_its_400_reaches_the_caller(
|
||||
gateway: Gateway, effort: JsonValue
|
||||
gateway: Gateway, effort: str
|
||||
) -> None:
|
||||
marker: Final = uuid.uuid4().hex
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
|
|
@ -328,6 +320,36 @@ def test_invalid_reasoning_effort_reaches_the_peer_and_its_400_reaches_the_calle
|
|||
_assert_row(_call_id(response), model, "failure")
|
||||
|
||||
|
||||
NON_STRING_EFFORTS: Final = (pytest.param(7, id="int"), pytest.param(["high"], id="list"))
|
||||
|
||||
|
||||
@pytest.mark.parametrize("effort", NON_STRING_EFFORTS)
|
||||
def test_non_string_reasoning_effort_is_refused_before_any_wire_request(gateway: Gateway, effort: JsonValue) -> None:
|
||||
marker: Final = uuid.uuid4().hex
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = _deployment(scenario, wire)
|
||||
response: Final = _chat(gateway, model, marker, reasoning_effort=effort)
|
||||
assert response.status_code == 400, response.text
|
||||
message: Final = _error_message(response)
|
||||
assert message.startswith("litellm.UnsupportedParamsError"), response.text
|
||||
assert "reasoning_effort as a string" in message and "drop_params" in message, response.text
|
||||
_assert_row(_call_id(response), model, "failure")
|
||||
assert _routes(wire.drain()) == []
|
||||
|
||||
|
||||
@pytest.mark.parametrize("effort", NON_STRING_EFFORTS)
|
||||
def test_drop_params_deployment_drops_a_non_string_reasoning_effort(gateway: Gateway, effort: JsonValue) -> None:
|
||||
marker: Final = uuid.uuid4().hex
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
model: Final = _deployment(scenario, wire, drop_params=True)
|
||||
response: Final = _chat(gateway, model, marker, reasoning_effort=effort)
|
||||
assert _content(response) == answer(marker), response.text
|
||||
request: Final = _only_request(wire, marker)
|
||||
assert target_of(request) == NATIVE_TARGET, request.body
|
||||
assert "reasoning_effort" not in _body(request), request.body
|
||||
_assert_row(string_value(_payload(response)["id"]), model, "success")
|
||||
|
||||
|
||||
def test_duplicated_reasoning_effort_key_lets_the_last_value_win(gateway: Gateway) -> None:
|
||||
marker: Final = uuid.uuid4().hex
|
||||
with wire_server(respond) as wire, gateway.scenario() as scenario:
|
||||
|
|
|
|||
|
|
@ -668,16 +668,35 @@ def test_map_openai_params_keeps_reasoning_effort_low_for_grok():
|
|||
|
||||
|
||||
@pytest.mark.parametrize("model", ["us.xai.grok-4.6", "global.openai.gpt-5.6-sol"])
|
||||
@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5])
|
||||
def test_map_openai_params_forwards_a_malformed_reasoning_effort_for_aws_to_refuse(model, reasoning_effort):
|
||||
@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5], ids=["list", "object", "int"])
|
||||
def test_map_openai_params_refuses_a_non_string_reasoning_effort_without_drop_params(model, reasoning_effort):
|
||||
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
|
||||
mapped = cfg.map_openai_params(
|
||||
with pytest.raises(litellm.UnsupportedParamsError, match="drop_params") as refused:
|
||||
cfg.map_openai_params(
|
||||
non_default_params={"reasoning_effort": reasoning_effort, "max_tokens": 64},
|
||||
optional_params={},
|
||||
model=model,
|
||||
drop_params=False,
|
||||
)
|
||||
assert refused.value.status_code == 400
|
||||
assert type(reasoning_effort).__name__ in str(refused.value)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["us.xai.grok-4.6", "global.openai.gpt-5.6-sol"])
|
||||
@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5], ids=["list", "object", "int"])
|
||||
@pytest.mark.parametrize("drop_params_via", ["request", "litellm.drop_params"])
|
||||
def test_map_openai_params_drops_a_non_string_reasoning_effort_under_drop_params(
|
||||
monkeypatch, model, reasoning_effort, drop_params_via
|
||||
):
|
||||
monkeypatch.setattr(litellm, "drop_params", drop_params_via == "litellm.drop_params")
|
||||
mapped = AmazonBedrockRuntimeChatCompletionsConfig().map_openai_params(
|
||||
non_default_params={"reasoning_effort": reasoning_effort, "max_tokens": 64},
|
||||
optional_params={},
|
||||
model=model,
|
||||
drop_params=False,
|
||||
drop_params=drop_params_via == "request",
|
||||
)
|
||||
assert mapped["reasoning_effort"] == reasoning_effort
|
||||
assert "reasoning_effort" not in mapped
|
||||
assert mapped["max_completion_tokens"] == 64
|
||||
|
||||
|
||||
def test_map_openai_params_keeps_reasoning_effort_none_for_gpt56():
|
||||
|
|
@ -752,6 +771,28 @@ def test_refused_params_are_dropped_or_refused_before_reaching_aws(local_cost_ma
|
|||
assert param.keys().isdisjoint(json.loads(requests[0].content))
|
||||
|
||||
|
||||
@pytest.mark.parametrize("reasoning_effort", [3, ["high"]], ids=["int", "list"])
|
||||
def test_non_string_reasoning_effort_is_refused_or_dropped_before_reaching_aws(
|
||||
local_cost_map, fake_aws_env, reasoning_effort
|
||||
):
|
||||
requests, client = _recording_client(json=_chat_completion_json("ok", "global.openai.gpt-5.6-sol"))
|
||||
request = {
|
||||
"model": "bedrock/global.openai.gpt-5.6-sol",
|
||||
"messages": [{"role": "user", "content": "hello"}],
|
||||
"reasoning_effort": reasoning_effort,
|
||||
"client": client,
|
||||
}
|
||||
with pytest.raises(litellm.UnsupportedParamsError, match="reasoning_effort") as refused:
|
||||
litellm.completion(**request)
|
||||
assert refused.value.status_code == 400
|
||||
assert requests == []
|
||||
|
||||
litellm.completion(**request, drop_params=True)
|
||||
|
||||
assert str(requests[0].url).endswith("/openai/v1/chat/completions")
|
||||
assert "reasoning_effort" not in json.loads(requests[0].content)
|
||||
|
||||
|
||||
GPT_PARAMS_TIED_TO_REASONING_OFF = {
|
||||
"temperature": 0.2,
|
||||
"top_p": 0.9,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue