mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-02 02:11:58 +00:00
fix(bedrock): drop reasoning_effort none for grok on the native chat completions route
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
c60249fa88
commit
f4d84f8f5d
2 changed files with 70 additions and 1 deletions
|
|
@ -60,6 +60,31 @@ def chat_completions_params_refused_for(model: str) -> frozenset[str]:
|
|||
)
|
||||
|
||||
|
||||
CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY: Final = MappingProxyType({"xai.": frozenset(("none",))})
|
||||
|
||||
|
||||
def chat_completions_reasoning_efforts_refused_for(model: str) -> frozenset[str]:
|
||||
"""The ``reasoning_effort`` values AWS's Chat Completions endpoint rejects for this model.
|
||||
|
||||
Grok answers ``"none"`` with a 400 (it takes low, medium, high, and xhigh) where Converse dropped every
|
||||
``reasoning_effort`` for it, so the native config drops the value and AWS applies its default effort as before.
|
||||
"""
|
||||
model_id: Final = split_bedrock_region_path(model)[1]
|
||||
return frozenset().union(
|
||||
*(
|
||||
refused
|
||||
for family, refused in CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY.items()
|
||||
if family in model_id
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) -> Mapping[str, object]:
|
||||
if params.get("reasoning_effort") not in chat_completions_reasoning_efforts_refused_for(model):
|
||||
return params
|
||||
return MappingProxyType({key: value for key, value in params.items() if key != "reasoning_effort"})
|
||||
|
||||
|
||||
def _held_close_tag_prefix(text: str) -> int:
|
||||
return next(
|
||||
(
|
||||
|
|
@ -277,7 +302,9 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
|
|||
drop_params=drop_params,
|
||||
replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens,
|
||||
)
|
||||
return dict(with_max_completion_tokens(mapped)) # mutable-ok: get_optional_params keeps filling this dict
|
||||
return dict( # mutable-ok: get_optional_params keeps filling this dict
|
||||
without_refused_reasoning_effort(model, with_max_completion_tokens(mapped))
|
||||
)
|
||||
|
||||
def _inference_params(
|
||||
self, optional_params: Mapping[str, object]
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ from litellm.llms.bedrock.chat.chat_completions.transformation import (
|
|||
AmazonBedrockRuntimeChatCompletionsConfig,
|
||||
BedrockRuntimeChatCompletionsStreamingHandler,
|
||||
ReasoningTagSplitter,
|
||||
chat_completions_reasoning_efforts_refused_for,
|
||||
split_reasoning_tag,
|
||||
with_max_completion_tokens,
|
||||
)
|
||||
|
|
@ -333,6 +334,47 @@ def test_with_max_completion_tokens_leaves_other_params_alone():
|
|||
assert with_max_completion_tokens({"temperature": 0.5}) == {"temperature": 0.5}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
["us.xai.grok-4.6", "bedrock/us-gov-west-1/us.xai.grok-4.6"],
|
||||
)
|
||||
def test_map_openai_params_drops_reasoning_effort_none_for_grok(model):
|
||||
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
|
||||
mapped = cfg.map_openai_params(
|
||||
non_default_params={"reasoning_effort": "none", "max_tokens": 64},
|
||||
optional_params={},
|
||||
model=model,
|
||||
drop_params=False,
|
||||
)
|
||||
assert "reasoning_effort" not in mapped
|
||||
|
||||
|
||||
def test_map_openai_params_keeps_reasoning_effort_low_for_grok():
|
||||
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
|
||||
mapped = cfg.map_openai_params(
|
||||
non_default_params={"reasoning_effort": "low", "max_tokens": 64},
|
||||
optional_params={},
|
||||
model="us.xai.grok-4.6",
|
||||
drop_params=False,
|
||||
)
|
||||
assert mapped["reasoning_effort"] == "low"
|
||||
|
||||
|
||||
def test_map_openai_params_keeps_reasoning_effort_none_for_gpt56():
|
||||
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
|
||||
mapped = cfg.map_openai_params(
|
||||
non_default_params={"reasoning_effort": "none", "max_tokens": 64},
|
||||
optional_params={},
|
||||
model="global.openai.gpt-5.6-sol",
|
||||
drop_params=False,
|
||||
)
|
||||
assert mapped["reasoning_effort"] == "none"
|
||||
|
||||
|
||||
def test_reasoning_efforts_refused_for_is_empty_outside_xai():
|
||||
assert chat_completions_reasoning_efforts_refused_for("openai.gpt-oss-20b-1:0") == frozenset()
|
||||
|
||||
|
||||
def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map):
|
||||
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
|
||||
assert "reasoning_effort" in cfg.get_supported_openai_params("global.openai.gpt-5.6-sol")
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue