mirror of
https://github.com/BerriAI/litellm.git
synced 2026-09-24 00:52:24 +00:00
Merge pull request #42000 from BerriAI/litellm_cherrypick_1_100_x
fix(bedrock): backport #41870 and the GPT-6 reasoning gate fix to stable/1.100.x for v1.100.2
This commit is contained in:
commit
44d3c4f290
4 changed files with 57 additions and 8 deletions
|
|
@ -4,6 +4,7 @@ Translating between OpenAI's `/chat/completion` format and Amazon's `/converse`
|
|||
|
||||
import copy
|
||||
import json
|
||||
import re
|
||||
import time
|
||||
import types
|
||||
from collections.abc import Mapping
|
||||
|
|
@ -104,6 +105,7 @@ BEDROCK_COMPUTER_USE_TOOLS: Final = [
|
|||
"bash_",
|
||||
"text_editor_",
|
||||
]
|
||||
BEDROCK_OPENAI_COMPAT_MIN_MAX_TOKENS: Final = 16
|
||||
|
||||
# Beta header patterns that are not supported by Bedrock Converse API
|
||||
# These will be filtered out to prevent errors
|
||||
|
|
@ -292,6 +294,14 @@ class AmazonConverseConfig(BaseConfig):
|
|||
llm_provider="bedrock",
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _requires_min_max_tokens(model: str) -> bool:
|
||||
return re.search(r"openai\.gpt-\d|xai\.grok-", model) is not None
|
||||
|
||||
@staticmethod
|
||||
def _is_openai_gpt_reasoning_model(model: str) -> bool:
|
||||
return re.search(r"openai\.gpt-\d", model) is not None
|
||||
|
||||
def _is_nova_2_model(self, model: str) -> bool:
|
||||
"""
|
||||
Check if the model is a Nova 2 model that supports reasoningConfig.
|
||||
|
|
@ -422,14 +432,14 @@ class AmazonConverseConfig(BaseConfig):
|
|||
Handle the reasoning_effort parameter based on the model type.
|
||||
|
||||
- GPT-OSS models: passed through unchanged via additionalModelRequestFields.
|
||||
- OpenAI GPT-5.x models: mapped to ``reasoning.effort`` via additionalModelRequestFields.
|
||||
- OpenAI GPT-5.x and GPT-6 models: mapped to ``reasoning.effort`` via additionalModelRequestFields.
|
||||
- Nova 2 models: transformed to reasoningConfig.
|
||||
- Anthropic models: mapped to ``thinking`` (and ``output_config.effort`` on
|
||||
adaptive Claude 4.6 / 4.7).
|
||||
"""
|
||||
if "gpt-oss" in model:
|
||||
optional_params["reasoning_effort"] = reasoning_effort
|
||||
elif "openai.gpt-5" in model:
|
||||
elif self._is_openai_gpt_reasoning_model(model):
|
||||
reasoning: Final[BedrockConverseGptReasoningEffortBlock] = {"effort": reasoning_effort}
|
||||
optional_params["reasoning"] = reasoning
|
||||
elif self._is_nova_2_model(model):
|
||||
|
|
@ -563,7 +573,11 @@ class AmazonConverseConfig(BaseConfig):
|
|||
# only anthropic and mistral support tool choice config. otherwise (E.g. cohere) will fail the call - https://docs.aws.amazon.com/bedrock/latest/APIReference/API_runtime_ToolChoice.html
|
||||
supported_params.append("tool_choice")
|
||||
|
||||
if "gpt-oss" in model or "openai.gpt-5" in model or "openai.gpt-5" in base_model:
|
||||
if (
|
||||
"gpt-oss" in model
|
||||
or self._is_openai_gpt_reasoning_model(model)
|
||||
or self._is_openai_gpt_reasoning_model(base_model)
|
||||
):
|
||||
supported_params.append("reasoning_effort")
|
||||
elif self._is_nova_2_model(model):
|
||||
# Nova 2 models support reasoning_effort (transformed to reasoningConfig)
|
||||
|
|
@ -874,7 +888,11 @@ class AmazonConverseConfig(BaseConfig):
|
|||
is_thinking_enabled=is_thinking_enabled,
|
||||
)
|
||||
if param == "max_tokens" or param == "max_completion_tokens":
|
||||
optional_params["maxTokens"] = value
|
||||
optional_params["maxTokens"] = (
|
||||
max(value, BEDROCK_OPENAI_COMPAT_MIN_MAX_TOKENS)
|
||||
if isinstance(value, int) and self._requires_min_max_tokens(model)
|
||||
else value
|
||||
)
|
||||
if param == "stream":
|
||||
optional_params["stream"] = value
|
||||
if param == "stop":
|
||||
|
|
@ -911,7 +929,7 @@ class AmazonConverseConfig(BaseConfig):
|
|||
optional_params["_parallel_tool_use_config"] = {
|
||||
"tool_choice": {"type": "auto", "disable_parallel_tool_use": not value}
|
||||
}
|
||||
if param == "thinking" and "openai.gpt-5" not in model:
|
||||
if param == "thinking" and not self._is_openai_gpt_reasoning_model(model):
|
||||
if (
|
||||
isinstance(value, dict)
|
||||
and value.get("type") == "adaptive"
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
[project]
|
||||
name = "litellm"
|
||||
version = "1.100.1"
|
||||
version = "1.100.2"
|
||||
description = "Library to easily interface with LLM API providers"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.10, <3.15"
|
||||
|
|
@ -311,7 +311,7 @@ members = ["enterprise", "litellm-proxy-extras"]
|
|||
profile = "black"
|
||||
|
||||
[tool.commitizen]
|
||||
version = "1.100.1"
|
||||
version = "1.100.2"
|
||||
version_files = [
|
||||
"pyproject.toml:^version",
|
||||
]
|
||||
|
|
|
|||
|
|
@ -375,12 +375,42 @@ def test_reasoning_with_forced_tool_choice_switches_to_auto():
|
|||
assert optional_params["tool_choice"] == {"auto": {}}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model, param, value, expected_max_tokens",
|
||||
[
|
||||
("us.openai.gpt-6-astra", "max_tokens", 1, 16),
|
||||
("us.openai.gpt-6-astra", "max_completion_tokens", 1, 16),
|
||||
("us.openai.gpt-6-astra", "max_tokens", 64, 64),
|
||||
("us.xai.grok-4.6", "max_tokens", 1, 16),
|
||||
("global.xai.grok-4.6", "max_completion_tokens", 1, 16),
|
||||
("us.xai.grok-4.6", "max_tokens", 32, 32),
|
||||
("anthropic.claude-sonnet-4-5-20250929-v1:0", "max_tokens", 1, 1),
|
||||
("arn:aws:bedrock:us-east-1:123456789012:inference-profile/us.openai.gpt-6-astra", "max_tokens", 1, 16),
|
||||
("arn:aws:bedrock:us-east-1:123456789012:inference-profile/global.xai.grok-4.6", "max_tokens", 1, 16),
|
||||
("arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/abc123xyz", "max_tokens", 1, 1),
|
||||
],
|
||||
)
|
||||
def test_map_openai_params_enforces_minimum_max_tokens_for_openai_compat_models(
|
||||
model: str, param: str, value: int, expected_max_tokens: int
|
||||
):
|
||||
optional_params = AmazonConverseConfig().map_openai_params(
|
||||
non_default_params={param: value},
|
||||
optional_params={},
|
||||
model=model,
|
||||
drop_params=False,
|
||||
)
|
||||
|
||||
assert optional_params["maxTokens"] == expected_max_tokens
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
"us.openai.gpt-5.6-sol",
|
||||
"global.openai.gpt-5.6-terra",
|
||||
"bedrock/converse/us.openai.gpt-5.6-luna",
|
||||
"us.openai.gpt-6-astra",
|
||||
"bedrock/converse/global.openai.gpt-6-astra",
|
||||
],
|
||||
)
|
||||
def test_reasoning_effort_maps_to_reasoning_effort_for_openai_gpt5_converse(model, local_model_cost_map):
|
||||
|
|
@ -411,6 +441,7 @@ def test_reasoning_effort_maps_to_reasoning_effort_for_openai_gpt5_converse(mode
|
|||
[
|
||||
"us.openai.gpt-5.6-sol",
|
||||
"bedrock/converse/global.openai.gpt-5.6-luna",
|
||||
"us.openai.gpt-6-astra",
|
||||
],
|
||||
)
|
||||
def test_openai_gpt5_converse_never_forwards_thinking(model, local_model_cost_map):
|
||||
|
|
|
|||
2
uv.lock
generated
2
uv.lock
generated
|
|
@ -4266,7 +4266,7 @@ wheels = [
|
|||
|
||||
[[package]]
|
||||
name = "litellm"
|
||||
version = "1.100.1"
|
||||
version = "1.100.2"
|
||||
source = { editable = "." }
|
||||
dependencies = [
|
||||
{ name = "aiohttp" },
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue