Merge pull request #42000 from BerriAI/litellm_cherrypick_1_100_x

fix(bedrock): backport #41870 and the GPT-6 reasoning gate fix to stable/1.100.x for v1.100.2
This commit is contained in:
Mateo Wang 2026-09-19 15:34:17 -07:00 • committed by GitHub
commit 44d3c4f290
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 57 additions and 8 deletions

View file

@ -4,6 +4,7 @@ Translating between OpenAI's `/chat/completion` format and Amazon's `/converse`
import copy
import json
import re
import time
import types
from collections.abc import Mapping
@ -104,6 +105,7 @@ BEDROCK_COMPUTER_USE_TOOLS: Final = [
"bash_",
"text_editor_",
]
BEDROCK_OPENAI_COMPAT_MIN_MAX_TOKENS: Final = 16
# Beta header patterns that are not supported by Bedrock Converse API
# These will be filtered out to prevent errors
@ -292,6 +294,14 @@ class AmazonConverseConfig(BaseConfig):
llm_provider="bedrock",
)
@staticmethod
def _requires_min_max_tokens(model: str) -> bool:
return re.search(r"openai\.gpt-\d|xai\.grok-", model) is not None
@staticmethod
def _is_openai_gpt_reasoning_model(model: str) -> bool:
return re.search(r"openai\.gpt-\d", model) is not None
def _is_nova_2_model(self, model: str) -> bool:
"""
Check if the model is a Nova 2 model that supports reasoningConfig.
@ -422,14 +432,14 @@ class AmazonConverseConfig(BaseConfig):
Handle the reasoning_effort parameter based on the model type.
- GPT-OSS models: passed through unchanged via additionalModelRequestFields.
- OpenAI GPT-5.x models: mapped to ``reasoning.effort`` via additionalModelRequestFields.
- OpenAI GPT-5.x and GPT-6 models: mapped to ``reasoning.effort`` via additionalModelRequestFields.
- Nova 2 models: transformed to reasoningConfig.
- Anthropic models: mapped to ``thinking`` (and ``output_config.effort`` on
adaptive Claude 4.6 / 4.7).
"""
if "gpt-oss" in model:
optional_params["reasoning_effort"] = reasoning_effort
elif "openai.gpt-5" in model:
elif self._is_openai_gpt_reasoning_model(model):
reasoning: Final[BedrockConverseGptReasoningEffortBlock] = {"effort": reasoning_effort}
optional_params["reasoning"] = reasoning
elif self._is_nova_2_model(model):
@ -563,7 +573,11 @@ class AmazonConverseConfig(BaseConfig):
# only anthropic and mistral support tool choice config. otherwise (E.g. cohere) will fail the call - https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_ToolChoice.html
supported_params.append("tool_choice")
if "gpt-oss" in model or "openai.gpt-5" in model or "openai.gpt-5" in base_model:
if (
"gpt-oss" in model
or self._is_openai_gpt_reasoning_model(model)
or self._is_openai_gpt_reasoning_model(base_model)
):
supported_params.append("reasoning_effort")
elif self._is_nova_2_model(model):
# Nova 2 models support reasoning_effort (transformed to reasoningConfig)
@ -874,7 +888,11 @@ class AmazonConverseConfig(BaseConfig):
is_thinking_enabled=is_thinking_enabled,
)
if param == "max_tokens" or param == "max_completion_tokens":
optional_params["maxTokens"] = value
optional_params["maxTokens"] = (
max(value, BEDROCK_OPENAI_COMPAT_MIN_MAX_TOKENS)
if isinstance(value, int) and self._requires_min_max_tokens(model)
else value
)
if param == "stream":
optional_params["stream"] = value
if param == "stop":
@ -911,7 +929,7 @@ class AmazonConverseConfig(BaseConfig):
optional_params["_parallel_tool_use_config"] = {
"tool_choice": {"type": "auto", "disable_parallel_tool_use": not value}
}
if param == "thinking" and "openai.gpt-5" not in model:
if param == "thinking" and not self._is_openai_gpt_reasoning_model(model):
if (
isinstance(value, dict)
and value.get("type") == "adaptive"

View file

@ -1,6 +1,6 @@
[project]
name = "litellm"
version = "1.100.1"
version = "1.100.2"
description = "Library to easily interface with LLM API providers"
readme = "README.md"
requires-python = ">=3.10, <3.15"
@ -311,7 +311,7 @@ members = ["enterprise", "litellm-proxy-extras"]
profile = "black"
[tool.commitizen]
version = "1.100.1"
version = "1.100.2"
version_files = [
"pyproject.toml:^version",
]

View file

@ -375,12 +375,42 @@ def test_reasoning_with_forced_tool_choice_switches_to_auto():
assert optional_params["tool_choice"] == {"auto": {}}
@pytest.mark.parametrize(
"model, param, value, expected_max_tokens",
[
("us.openai.gpt-6-astra", "max_tokens", 1, 16),
("us.openai.gpt-6-astra", "max_completion_tokens", 1, 16),
("us.openai.gpt-6-astra", "max_tokens", 64, 64),
("us.xai.grok-4.6", "max_tokens", 1, 16),
("global.xai.grok-4.6", "max_completion_tokens", 1, 16),
("us.xai.grok-4.6", "max_tokens", 32, 32),
("anthropic.claude-sonnet-4-5-20250929-v1:0", "max_tokens", 1, 1),
("arn:aws:bedrock:us-east-1:123456789012:inference-profile/us.openai.gpt-6-astra", "max_tokens", 1, 16),
("arn:aws:bedrock:us-east-1:123456789012:inference-profile/global.xai.grok-4.6", "max_tokens", 1, 16),
("arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/abc123xyz", "max_tokens", 1, 1),
],
)
def test_map_openai_params_enforces_minimum_max_tokens_for_openai_compat_models(
model: str, param: str, value: int, expected_max_tokens: int
):
optional_params = AmazonConverseConfig().map_openai_params(
non_default_params={param: value},
optional_params={},
model=model,
drop_params=False,
)
assert optional_params["maxTokens"] == expected_max_tokens
@pytest.mark.parametrize(
"model",
[
"us.openai.gpt-5.6-sol",
"global.openai.gpt-5.6-terra",
"bedrock/converse/us.openai.gpt-5.6-luna",
"us.openai.gpt-6-astra",
"bedrock/converse/global.openai.gpt-6-astra",
],
)
def test_reasoning_effort_maps_to_reasoning_effort_for_openai_gpt5_converse(model, local_model_cost_map):
@ -411,6 +441,7 @@ def test_reasoning_effort_maps_to_reasoning_effort_for_openai_gpt5_converse(mode
[
"us.openai.gpt-5.6-sol",
"bedrock/converse/global.openai.gpt-5.6-luna",
"us.openai.gpt-6-astra",
],
)
def test_openai_gpt5_converse_never_forwards_thinking(model, local_model_cost_map):

2
uv.lock generated
View file

@ -4266,7 +4266,7 @@ wheels = [
[[package]]
name = "litellm"
version = "1.100.1"
version = "1.100.2"
source = { editable = "." }
dependencies = [
{ name = "aiohttp" },