From 84084d9a82548e364bea390ba021ff926e87c468 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Fri, 11 Sep 2026 12:22:16 -0700
Subject: [PATCH 01/18] feat(bedrock): send grok chat completions through
runtime openai path
Unspecified bedrock grok was rewritten to Converse. Chat completions now hit bedrock-runtime /openai/v1/chat/completions, and converse/ still uses Converse
---
ci_cd/generate_model_prices_schema.py | 1 +
litellm/__init__.py | 3 +
litellm/_lazy_imports_registry.py | 5 +
.../chat/chat_completions/transformation.py | 174 ++++++++++++++++++
litellm/llms/bedrock/common_utils.py | 21 +++
...odel_prices_and_context_window_backup.json | 3 +
model_prices_and_context_window.json | 3 +
model_prices_and_context_window.schema.json | 3 +
...bedrock_chat_completions_transformation.py | 150 +++++++++++++++
9 files changed, 363 insertions(+)
create mode 100644 litellm/llms/bedrock/chat/chat_completions/transformation.py
create mode 100644 tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py
index ab29b70bdd4..b7f7972d907 100644
--- a/ci_cd/generate_model_prices_schema.py
+++ b/ci_cd/generate_model_prices_schema.py
@@ -26,6 +26,7 @@ EXTRA_BOOLEAN_KEYS = frozenset(
"gemini_audio_only_live",
"uses_embed_content",
"use_openai_responses_path",
+ "use_bedrock_runtime_chat_completions",
"bedrock_converse_supports_strict_tools",
"thinking_always_on",
}
diff --git a/litellm/__init__.py b/litellm/__init__.py
index 1dfd146a00e..fea9150a6c0 100644
--- a/litellm/__init__.py
+++ b/litellm/__init__.py
@@ -1736,6 +1736,9 @@ if TYPE_CHECKING:
from .llms.bedrock.chat.invoke_transformations.amazon_openai_transformation import (
AmazonBedrockOpenAIConfig as AmazonBedrockOpenAIConfig,
)
+ from .llms.bedrock.chat.chat_completions.transformation import (
+ AmazonBedrockRuntimeChatCompletionsConfig as AmazonBedrockRuntimeChatCompletionsConfig,
+ )
from .llms.bedrock.image_generation.amazon_stability1_transformation import (
AmazonStabilityConfig as AmazonStabilityConfig,
)
diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py
index dc323c8cc15..0880734fd9f 100644
--- a/litellm/_lazy_imports_registry.py
+++ b/litellm/_lazy_imports_registry.py
@@ -204,6 +204,7 @@ LLM_CONFIG_NAMES: Final = (
"AmazonTwelveLabsPegasusConfig",
"AmazonInvokeConfig",
"AmazonBedrockOpenAIConfig",
+ "AmazonBedrockRuntimeChatCompletionsConfig",
"AmazonStabilityConfig",
"AmazonStability3Config",
"AmazonNovaCanvasConfig",
@@ -847,6 +848,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
".llms.bedrock.chat.invoke_transformations.amazon_openai_transformation",
"AmazonBedrockOpenAIConfig",
),
+ "AmazonBedrockRuntimeChatCompletionsConfig": (
+ ".llms.bedrock.chat.chat_completions.transformation",
+ "AmazonBedrockRuntimeChatCompletionsConfig",
+ ),
"AmazonStabilityConfig": (
".llms.bedrock.image_generation.amazon_stability1_transformation",
"AmazonStabilityConfig",
diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py
new file mode 100644
index 00000000000..3bf6b2a2ffd
--- /dev/null
+++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py
@@ -0,0 +1,174 @@
+"""
+Native OpenAI Chat Completions on Amazon Bedrock Runtime.
+
+AWS serves this surface at
+``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions``.
+Grok 4.6 on runtime is one of the models that uses it: chat completions stay
+chat completions instead of being rewritten to Converse.
+
+Usage: model="us.xai.grok-4.6" or model="bedrock/us.xai.grok-4.6"
+Explicit ``bedrock/converse/...`` still uses Converse.
+"""
+
+from collections.abc import AsyncIterator, Iterator
+from typing import Any, Final
+
+import httpx
+
+import litellm
+from litellm._logging import verbose_logger
+from litellm.llms.base_llm.chat.transformation import BaseLLMException
+from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
+from litellm.llms.bedrock.common_utils import BedrockError, strip_bedrock_routing_prefix
+from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig
+from litellm.types.llms.openai import AllMessageValues
+
+
+class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
+ def __init__(self, aws_signer: BaseAWSLLM | None = None):
+ super().__init__()
+ self._aws_signer: Final = aws_signer or BaseAWSLLM()
+
+ @property
+ def custom_llm_provider(self) -> str | None:
+ return "bedrock"
+
+ def get_error_class(
+ self, error_message: str, status_code: int, headers: dict[str, object] | httpx.Headers
+ ) -> BaseLLMException:
+ return BedrockError(status_code=status_code, message=error_message, headers=headers)
+
+ def get_complete_url(
+ self,
+ api_base: str | None,
+ api_key: str | None,
+ model: str,
+ optional_params: dict,
+ litellm_params: dict,
+ stream: bool | None = None,
+ ) -> str:
+ if api_base is not None and "chat/completions" in api_base:
+ return api_base.rstrip("/")
+ aws_region_name: Final = self._aws_signer._get_aws_region_name(optional_params=optional_params, model=model)
+ endpoint_url, _ = self._aws_signer.get_runtime_endpoint(
+ api_base=api_base,
+ aws_bedrock_runtime_endpoint=optional_params.get("aws_bedrock_runtime_endpoint"),
+ aws_region_name=aws_region_name,
+ )
+ base: Final = endpoint_url.rstrip("/")
+ if base.endswith("/openai/v1/chat/completions"):
+ return base
+ if base.endswith("/openai/v1"):
+ return f"{base}/chat/completions"
+ return f"{base}/openai/v1/chat/completions"
+
+ def sign_request(
+ self,
+ headers: dict,
+ optional_params: dict,
+ request_data: dict,
+ api_base: str,
+ api_key: str | None = None,
+ model: str | None = None,
+ stream: bool | None = None,
+ fake_stream: bool | None = None,
+ ) -> tuple[dict, bytes | None]:
+ return self._aws_signer._sign_request(
+ service_name="bedrock",
+ headers=headers,
+ optional_params=optional_params,
+ request_data=request_data,
+ api_base=api_base,
+ api_key=api_key,
+ model=model,
+ stream=stream,
+ fake_stream=fake_stream,
+ )
+
+ def transform_request(
+ self,
+ model: str,
+ messages: list[AllMessageValues],
+ optional_params: dict,
+ litellm_params: dict,
+ headers: dict,
+ ) -> dict:
+ inference_params: Final = {
+ k: v for k, v in optional_params.items() if k not in self._aws_signer.aws_authentication_params
+ }
+ return super().transform_request(
+ model=strip_bedrock_routing_prefix(model),
+ messages=messages,
+ optional_params=inference_params,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ async def async_transform_request(
+ self,
+ model: str,
+ messages: list[AllMessageValues],
+ optional_params: dict,
+ litellm_params: dict,
+ headers: dict,
+ ) -> dict:
+ inference_params: Final = {
+ k: v for k, v in optional_params.items() if k not in self._aws_signer.aws_authentication_params
+ }
+ return await super().async_transform_request(
+ model=strip_bedrock_routing_prefix(model),
+ messages=messages,
+ optional_params=inference_params,
+ litellm_params=litellm_params,
+ headers=headers,
+ )
+
+ def validate_environment(
+ self,
+ headers: dict,
+ model: str,
+ messages: list[AllMessageValues],
+ optional_params: dict,
+ litellm_params: dict,
+ api_key: str | None = None,
+ api_base: str | None = None,
+ ) -> dict:
+ headers = super().validate_environment(
+ headers=headers,
+ model=model,
+ messages=messages,
+ optional_params=optional_params,
+ litellm_params=litellm_params,
+ api_key=api_key,
+ api_base=api_base,
+ )
+ project_id: Final = litellm_params.get("aws_bedrock_project_id")
+ if project_id:
+ headers["OpenAI-Project"] = project_id
+ return headers
+
+ def get_supported_openai_params(self, model: str) -> list:
+ base_params: Final = super().get_supported_openai_params(model)
+ try:
+ if litellm.supports_reasoning(model=model, custom_llm_provider=self.custom_llm_provider):
+ if "reasoning_effort" not in base_params:
+ base_params.append("reasoning_effort")
+ except Exception as e:
+ verbose_logger.debug("AmazonBedrockRuntimeChatCompletionsConfig: error checking reasoning support: %s", e)
+ return base_params
+
+ def get_model_response_iterator(
+ self,
+ streaming_response: Iterator[str] | AsyncIterator[str] | Any,
+ sync_stream: bool,
+ json_mode: bool | None = False,
+ ) -> Any:
+ from litellm.llms.openai.chat.gpt_transformation import (
+ OpenAIChatCompletionStreamingHandler,
+ )
+
+ return OpenAIChatCompletionStreamingHandler(
+ streaming_response=streaming_response,
+ sync_stream=sync_stream,
+ json_mode=json_mode,
+ )
diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py
index cb2c70e74c8..366ccadfbb7 100644
--- a/litellm/llms/bedrock/common_utils.py
+++ b/litellm/llms/bedrock/common_utils.py
@@ -780,6 +780,20 @@ def strip_bedrock_routing_prefix(model: str) -> str:
return model
+def uses_bedrock_runtime_chat_completions(model: str) -> bool:
+ """Whether this Bedrock model should use runtime native Chat Completions.
+
+ Data-driven from the price-map ``use_bedrock_runtime_chat_completions`` flag
+ so onboarding a model is a JSON change. Explicit ``converse/`` still wins in
+ ``get_bedrock_route`` because prefix routes are checked first.
+ """
+ stripped: Final = strip_bedrock_routing_prefix(model)
+ return any(
+ (litellm.model_cost.get(key) or {}).get("use_bedrock_runtime_chat_completions") is True
+ for key in (model, stripped)
+ )
+
+
def strip_bedrock_throughput_suffix(model: str) -> str:
"""Strip throughput tier suffixes and context window suffixes from Bedrock model names."""
import re
@@ -1107,6 +1121,7 @@ class BedrockModelInfo(BaseLLMModelInfo):
"async_invoke",
"openai",
"mantle",
+ "chat_completions",
]:
"""
Get the bedrock route for the given model.
@@ -1123,6 +1138,7 @@ class BedrockModelInfo(BaseLLMModelInfo):
"async_invoke",
"openai",
"mantle",
+ "chat_completions",
],
] = {
"invoke/": "invoke",
@@ -1152,6 +1168,9 @@ class BedrockModelInfo(BaseLLMModelInfo):
if is_bedrock_application_inference_profile_arn(model):
return "converse"
+ if uses_bedrock_runtime_chat_completions(model):
+ return "chat_completions"
+
base_model: Final = BedrockModelInfo.get_base_model(model)
alt_model: Final = BedrockModelInfo.get_non_litellm_routing_model_name(model=model)
if base_model in litellm.bedrock_converse_models or alt_model in litellm.bedrock_converse_models:
@@ -1328,6 +1347,8 @@ def get_bedrock_chat_config(model: str):
return litellm.AmazonConverseConfig()
elif bedrock_route == "openai":
return litellm.AmazonBedrockOpenAIConfig()
+ elif bedrock_route == "chat_completions":
+ return litellm.AmazonBedrockRuntimeChatCompletionsConfig()
elif bedrock_route == "agent":
from litellm.llms.bedrock.chat.invoke_agent.transformation import (
AmazonInvokeAgentConfig,
diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json
index ad992ff92eb..3c6900cb136 100644
--- a/litellm/model_prices_and_context_window_backup.json
+++ b/litellm/model_prices_and_context_window_backup.json
@@ -44462,6 +44462,7 @@
"input_cost_per_token": 2.64e-06,
"output_cost_per_token": 7.92e-06,
"cache_read_input_token_cost": 6.6e-07,
+ "use_bedrock_runtime_chat_completions": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@@ -55840,6 +55841,7 @@
"input_cost_per_token": 2.2e-06,
"output_cost_per_token": 6.6e-06,
"cache_read_input_token_cost": 5.5e-07,
+ "use_bedrock_runtime_chat_completions": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@@ -55855,6 +55857,7 @@
"input_cost_per_token": 2e-06,
"output_cost_per_token": 6e-06,
"cache_read_input_token_cost": 5e-07,
+ "use_bedrock_runtime_chat_completions": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json
index ad992ff92eb..3c6900cb136 100644
--- a/model_prices_and_context_window.json
+++ b/model_prices_and_context_window.json
@@ -44462,6 +44462,7 @@
"input_cost_per_token": 2.64e-06,
"output_cost_per_token": 7.92e-06,
"cache_read_input_token_cost": 6.6e-07,
+ "use_bedrock_runtime_chat_completions": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@@ -55840,6 +55841,7 @@
"input_cost_per_token": 2.2e-06,
"output_cost_per_token": 6.6e-06,
"cache_read_input_token_cost": 5.5e-07,
+ "use_bedrock_runtime_chat_completions": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@@ -55855,6 +55857,7 @@
"input_cost_per_token": 2e-06,
"output_cost_per_token": 6e-06,
"cache_read_input_token_cost": 5e-07,
+ "use_bedrock_runtime_chat_completions": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json
index 7ed1e7e568b..59d6b75a5f6 100644
--- a/model_prices_and_context_window.schema.json
+++ b/model_prices_and_context_window.schema.json
@@ -855,6 +855,9 @@
"minimum": 0,
"description": "Provider default tokens-per-minute limit."
},
+ "use_bedrock_runtime_chat_completions": {
+ "type": "boolean"
+ },
"use_openai_responses_path": {
"type": "boolean"
},
diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
new file mode 100644
index 00000000000..7724b3f0a52
--- /dev/null
+++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -0,0 +1,150 @@
+"""Native Bedrock Runtime Chat Completions: Grok stays on /openai/v1/chat/completions."""
+
+import json
+from unittest.mock import patch
+
+import httpx
+import pytest
+
+import litellm
+from litellm.llms.bedrock.chat.chat_completions.transformation import (
+ AmazonBedrockRuntimeChatCompletionsConfig,
+)
+from litellm.llms.bedrock.common_utils import (
+ BedrockModelInfo,
+ get_bedrock_chat_config,
+ uses_bedrock_runtime_chat_completions,
+)
+
+
+@pytest.fixture
+def local_cost_map(monkeypatch):
+ original_model_cost = litellm.model_cost
+ try:
+ monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true")
+ litellm.model_cost = litellm.get_model_cost_map(url="")
+ litellm.get_model_info.cache_clear()
+ yield
+ finally:
+ litellm.model_cost = original_model_cost
+ litellm.get_model_info.cache_clear()
+
+
+@pytest.mark.parametrize(
+ "model",
+ [
+ "us.xai.grok-4.6",
+ "global.xai.grok-4.6",
+ "us-gov.xai.grok-4.6",
+ "bedrock/us.xai.grok-4.6",
+ ],
+)
+def test_grok_runtime_models_use_chat_completions_route(local_cost_map, model):
+ assert uses_bedrock_runtime_chat_completions(model) is True
+ assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions"
+ assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig)
+
+
+def test_explicit_converse_prefix_still_uses_converse(local_cost_map):
+ assert BedrockModelInfo.get_bedrock_route("bedrock/converse/us.xai.grok-4.6") == "converse"
+ assert BedrockModelInfo.get_bedrock_route("converse/us.xai.grok-4.6") == "converse"
+
+
+def test_claude_stays_on_converse(local_cost_map):
+ assert uses_bedrock_runtime_chat_completions("us.anthropic.claude-3-sonnet-20240229-v1:0") is False
+ assert BedrockModelInfo.get_bedrock_route("us.anthropic.claude-3-sonnet-20240229-v1:0") == "converse"
+
+
+def test_flag_absent_means_no_chat_completions_route(monkeypatch):
+ monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": {"litellm_provider": "bedrock_converse"}})
+ assert uses_bedrock_runtime_chat_completions("us.xai.grok-4.6") is False
+
+
+def test_complete_url_is_runtime_openai_chat_completions(monkeypatch):
+ monkeypatch.setenv("AWS_REGION_NAME", "us-east-1")
+ monkeypatch.delenv("AWS_BEDROCK_RUNTIME_ENDPOINT", raising=False)
+ cfg = AmazonBedrockRuntimeChatCompletionsConfig()
+ url = cfg.get_complete_url(
+ api_base=None,
+ api_key=None,
+ model="us.xai.grok-4.6",
+ optional_params={},
+ litellm_params={},
+ )
+ assert url == "https://bedrock-runtime.us-east-1.amazonaws.com/openai/v1/chat/completions"
+
+
+def test_complete_url_appends_to_openai_v1_base():
+ cfg = AmazonBedrockRuntimeChatCompletionsConfig()
+ url = cfg.get_complete_url(
+ api_base="https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1",
+ api_key=None,
+ model="us.xai.grok-4.6",
+ optional_params={},
+ litellm_params={},
+ )
+ assert url == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
+
+
+def test_transform_request_is_openai_chat_body_not_converse():
+ cfg = AmazonBedrockRuntimeChatCompletionsConfig()
+ body = cfg.transform_request(
+ model="bedrock/us.xai.grok-4.6",
+ messages=[{"role": "user", "content": "hello"}],
+ optional_params={"temperature": 0.2, "aws_region_name": "us-east-1"},
+ litellm_params={},
+ headers={},
+ )
+ assert body["model"] == "us.xai.grok-4.6"
+ assert body["messages"] == [{"role": "user", "content": "hello"}]
+ assert body["temperature"] == 0.2
+ assert "aws_region_name" not in body
+ assert "inferenceConfig" not in body
+ assert "messages" in body
+
+
+def test_completion_posts_runtime_chat_completions(local_cost_map, monkeypatch):
+ monkeypatch.setenv("AWS_REGION_NAME", "us-west-2")
+ monkeypatch.delenv("AWS_BEDROCK_RUNTIME_ENDPOINT", raising=False)
+ monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False)
+ monkeypatch.setenv("AWS_ACCESS_KEY_ID", "testing")
+ monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing")
+ monkeypatch.setenv("AWS_SESSION_TOKEN", "testing")
+
+ requests: list[dict] = []
+
+ def mock_post(self, url, data=None, json=None, headers=None, **kwargs):
+ requests.append({"url": url, "data": data, "json": json, "headers": headers or {}})
+ return httpx.Response(
+ status_code=200,
+ json={
+ "id": "chatcmpl-test",
+ "object": "chat.completion",
+ "created": 1733529600,
+ "model": "us.xai.grok-4.6",
+ "choices": [
+ {
+ "index": 0,
+ "message": {"role": "assistant", "content": "ok"},
+ "finish_reason": "stop",
+ }
+ ],
+ "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
+ },
+ request=httpx.Request("POST", url),
+ )
+
+ with patch("litellm.llms.custom_httpx.http_handler.HTTPHandler.post", mock_post):
+ response = litellm.completion(
+ model="us.xai.grok-4.6",
+ messages=[{"role": "user", "content": "hello"}],
+ )
+
+ assert response.choices[0].message.content == "ok"
+ assert len(requests) == 1
+ assert requests[0]["url"] == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
+ raw = requests[0]["data"]
+ body = json.loads(raw) if isinstance(raw, (str, bytes, bytearray)) else (requests[0]["json"] or {})
+ assert body["model"] == "us.xai.grok-4.6"
+ assert body["messages"] == [{"role": "user", "content": "hello"}]
+ assert "inferenceConfig" not in body
From 6b9f067c80eea6ce302eec5205aaf7892f1131e9 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Sat, 19 Sep 2026 18:05:46 -0700
Subject: [PATCH 02/18] feat(bedrock): serve gpt-oss and gpt-5.6 chat
completions on runtime's native openai path
---
ci_cd/generate_model_prices_schema.py | 1 +
.../chat/chat_completions/transformation.py | 319 ++++++++--
litellm/llms/bedrock/common_utils.py | 81 ++-
litellm/main.py | 2 +-
...odel_prices_and_context_window_backup.json | 14 +
litellm/utils.py | 26 +-
model_prices_and_context_window.json | 14 +
model_prices_and_context_window.schema.json | 3 +
...bedrock_chat_completions_transformation.py | 554 ++++++++++++++++--
..._cross_region_inference_profile_mapping.py | 9 +-
tests/test_litellm/test_utils.py | 2 +
11 files changed, 897 insertions(+), 128 deletions(-)
diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py
index 5a2a56a5b21..648cceb79b0 100644
--- a/ci_cd/generate_model_prices_schema.py
+++ b/ci_cd/generate_model_prices_schema.py
@@ -31,6 +31,7 @@ EXTRA_BOOLEAN_KEYS = frozenset(
"uses_embed_content",
"use_openai_responses_path",
"use_bedrock_runtime_chat_completions",
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none",
"bedrock_converse_supports_strict_tools",
"thinking_always_on",
}
diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py
index 3bf6b2a2ffd..49218270064 100644
--- a/litellm/llms/bedrock/chat/chat_completions/transformation.py
+++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py
@@ -2,30 +2,170 @@
Native OpenAI Chat Completions on Amazon Bedrock Runtime.
AWS serves this surface at
-``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions``.
-Grok 4.6 on runtime is one of the models that uses it: chat completions stay
-chat completions instead of being rewritten to Converse.
+``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions``
+for the models whose price-map entry sets ``use_bedrock_runtime_chat_completions``
+(Grok 4.6, gpt-oss, the GPT-5.6 family): chat completions stay chat completions
+instead of being rewritten to Converse.
-Usage: model="us.xai.grok-4.6" or model="bedrock/us.xai.grok-4.6"
-Explicit ``bedrock/converse/...`` still uses Converse.
+Usage: model="us.xai.grok-4.6", model="bedrock/openai.gpt-oss-20b-1:0" or
+model="bedrock/global.openai.gpt-5.6-sol". Explicit ``bedrock/converse/...``
+still uses Converse, and so does a request that needs a Converse-only feature
+(``bedrock_request_needs_converse`` in ``common_utils``).
"""
-from collections.abc import AsyncIterator, Iterator
-from typing import Any, Final
+from collections.abc import AsyncIterator, Iterator, Mapping
+from dataclasses import dataclass, replace
+from types import MappingProxyType
+from typing import TYPE_CHECKING, Final, Literal
import httpx
import litellm
-from litellm._logging import verbose_logger
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
from litellm.llms.bedrock.common_utils import BedrockError, strip_bedrock_routing_prefix
+from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler
from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig
from litellm.types.llms.openai import AllMessageValues
+from litellm.types.utils import Choices, ModelResponse, ModelResponseStream
+
+if TYPE_CHECKING:
+ import tiktoken
+
+ from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
+
+REASONING_OPEN_TAG: Final = ""
+REASONING_CLOSE_TAG: Final = ""
+
+
+def _held_close_tag_prefix(text: str) -> int:
+ return next(
+ (
+ size
+ for size in range(min(len(text), len(REASONING_CLOSE_TAG) - 1), 0, -1)
+ if REASONING_CLOSE_TAG.startswith(text[-size:])
+ ),
+ 0,
+ )
+
+
+@dataclass(frozen=True, slots=True)
+class ReasoningTagSplitter:
+ """
+ The same split for a stream of content deltas, where a tag can arrive across chunks.
+
+ ``feed`` returns the next state plus the reasoning and content text the delta contributes;
+ ``flush`` releases what the stream ended on before a tag resolved.
+ """
+
+ phase: Literal["start", "reasoning", "after_close", "content"] = "start"
+ pending: str = ""
+
+ def feed(self, text: str) -> tuple["ReasoningTagSplitter", str, str]:
+ match self.phase:
+ case "content":
+ return self, "", text
+ case "after_close":
+ content: Final = text.lstrip()
+ return (replace(self, phase="content") if content else self), "", content
+ case "start":
+ return self._feed_start(self.pending + text)
+ case "reasoning":
+ return self._feed_reasoning(self.pending + text)
+
+ def _feed_start(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]:
+ if buffered.startswith(REASONING_OPEN_TAG):
+ return replace(self, phase="reasoning", pending="")._feed_reasoning(buffered[len(REASONING_OPEN_TAG) :])
+ if REASONING_OPEN_TAG.startswith(buffered):
+ return replace(self, pending=buffered), "", ""
+ return replace(self, phase="content", pending=""), "", buffered
+
+ def _feed_reasoning(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]:
+ close_at: Final = buffered.find(REASONING_CLOSE_TAG)
+ if close_at >= 0:
+ after_close: Final = replace(self, phase="after_close", pending="")
+ next_state, _, content = after_close.feed(buffered[close_at + len(REASONING_CLOSE_TAG) :])
+ return next_state, buffered[:close_at], content
+ held: Final = _held_close_tag_prefix(buffered)
+ return replace(self, pending=buffered[len(buffered) - held :]), buffered[: len(buffered) - held], ""
+
+ def flush(self) -> tuple["ReasoningTagSplitter", str, str]:
+ drained: Final = replace(self, phase="content", pending="")
+ if self.phase == "reasoning":
+ return drained, self.pending, ""
+ return drained, "", self.pending
+
+
+def _split_streamed_content(
+ splitter: ReasoningTagSplitter, content: str | None, finished: bool
+) -> tuple[ReasoningTagSplitter, str, str]:
+ fed_state, fed_reasoning, fed_content = splitter.feed(content or "")
+ if not finished:
+ return fed_state, fed_reasoning, fed_content
+ drained, flushed_reasoning, flushed_content = fed_state.flush()
+ return drained, fed_reasoning + flushed_reasoning, fed_content + flushed_content
+
+
+def split_reasoning_tag(content: str) -> tuple[str | None, str]:
+ """
+ Split gpt-oss's inline ``...`` prefix out of a complete message.
+
+ Runs the streaming splitter over the whole message, so a streamed and a non-streamed
+ response to the same completion split identically. Returns ``(None, content)`` when the
+ message does not start with the tag.
+ """
+ _, reasoning, body = _split_streamed_content(ReasoningTagSplitter(), content, finished=True)
+ return reasoning or None, body
+
+
+class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler):
+ """OpenAI chunk parsing plus the ```` split, tracked per choice index."""
+
+ def __init__(
+ self,
+ streaming_response: Iterator[str] | AsyncIterator[str] | ModelResponse,
+ sync_stream: bool,
+ json_mode: bool | None = False,
+ ) -> None:
+ super().__init__(streaming_response=streaming_response, sync_stream=sync_stream, json_mode=json_mode)
+ self._splitters: Mapping[int, ReasoningTagSplitter] = MappingProxyType({})
+
+ def chunk_parser(self, chunk: dict) -> ModelResponseStream: # mutable-ok: BaseModelResponseIterator signature
+ parsed: Final = super().chunk_parser(chunk)
+ for choice in parsed.choices:
+ next_state, reasoning, content = _split_streamed_content(
+ self._splitters.get(choice.index, ReasoningTagSplitter()),
+ choice.delta.content,
+ choice.finish_reason is not None,
+ )
+ self._splitters = MappingProxyType({**self._splitters, choice.index: next_state})
+ if reasoning:
+ choice.delta.reasoning_content = f"{getattr(choice.delta, 'reasoning_content', None) or ''}{reasoning}"
+ if content or choice.delta.content is not None:
+ choice.delta.content = content
+ return parsed
+
+
+def with_max_completion_tokens(params: Mapping[str, object]) -> Mapping[str, object]:
+ """
+ Send the caller's ``max_tokens`` as ``max_completion_tokens``.
+
+ Every model on this surface accepts ``max_completion_tokens`` and the GPT-5.6 family
+ rejects ``max_tokens``; an explicit ``max_completion_tokens`` wins when both are set.
+ """
+ if "max_tokens" not in params:
+ return params
+ return MappingProxyType(
+ {
+ key: value
+ for key, value in (("max_completion_tokens", params["max_tokens"]), *params.items())
+ if key != "max_tokens"
+ }
+ )
class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
- def __init__(self, aws_signer: BaseAWSLLM | None = None):
+ def __init__(self, aws_signer: BaseAWSLLM | None = None) -> None:
super().__init__()
self._aws_signer: Final = aws_signer or BaseAWSLLM()
@@ -34,7 +174,10 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
return "bedrock"
def get_error_class(
- self, error_message: str, status_code: int, headers: dict[str, object] | httpx.Headers
+ self,
+ error_message: str,
+ status_code: int,
+ headers: dict[str, object] | httpx.Headers, # mutable-ok: BaseConfig signature
) -> BaseLLMException:
return BedrockError(status_code=status_code, message=error_message, headers=headers)
@@ -43,13 +186,15 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
api_base: str | None,
api_key: str | None,
model: str,
- optional_params: dict,
- litellm_params: dict,
+ optional_params: dict, # mutable-ok: BaseConfig signature
+ litellm_params: dict, # mutable-ok: BaseConfig signature
stream: bool | None = None,
) -> str:
if api_base is not None and "chat/completions" in api_base:
return api_base.rstrip("/")
- aws_region_name: Final = self._aws_signer._get_aws_region_name(optional_params=optional_params, model=model)
+ aws_region_name: Final = self._aws_signer._get_aws_region_name( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public region resolver
+ optional_params=optional_params, model=model
+ )
endpoint_url, _ = self._aws_signer.get_runtime_endpoint(
api_base=api_base,
aws_bedrock_runtime_endpoint=optional_params.get("aws_bedrock_runtime_endpoint"),
@@ -64,16 +209,16 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
def sign_request(
self,
- headers: dict,
- optional_params: dict,
- request_data: dict,
+ headers: dict, # mutable-ok: BaseConfig signature
+ optional_params: dict, # mutable-ok: BaseConfig signature
+ request_data: dict, # mutable-ok: BaseConfig signature
api_base: str,
api_key: str | None = None,
model: str | None = None,
stream: bool | None = None,
fake_stream: bool | None = None,
- ) -> tuple[dict, bytes | None]:
- return self._aws_signer._sign_request(
+ ) -> tuple[dict, bytes | None]: # mutable-ok: BaseConfig signature
+ return self._aws_signer._sign_request( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public signer
service_name="bedrock",
headers=headers,
optional_params=optional_params,
@@ -85,21 +230,44 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
fake_stream=fake_stream,
)
+ def map_openai_params(
+ self,
+ non_default_params: dict, # mutable-ok: BaseConfig signature
+ optional_params: dict, # mutable-ok: BaseConfig signature
+ model: str,
+ drop_params: bool,
+ replace_max_completion_tokens_with_max_tokens: bool = False,
+ ) -> dict: # mutable-ok: BaseConfig signature
+ mapped: Final = super().map_openai_params(
+ non_default_params=non_default_params,
+ optional_params=optional_params,
+ model=model,
+ drop_params=drop_params,
+ replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens,
+ )
+ return dict(with_max_completion_tokens(mapped)) # mutable-ok: get_optional_params keeps filling this dict
+
+ def _inference_params(
+ self, optional_params: Mapping[str, object]
+ ) -> dict[str, object]: # mutable-ok: BaseConfig signature of transform_request
+ return { # mutable-ok: OpenAILikeChatConfig.transform_request takes a plain dict
+ key: value
+ for key, value in optional_params.items()
+ if key not in self._aws_signer.aws_authentication_params
+ }
+
def transform_request(
self,
model: str,
- messages: list[AllMessageValues],
- optional_params: dict,
- litellm_params: dict,
- headers: dict,
- ) -> dict:
- inference_params: Final = {
- k: v for k, v in optional_params.items() if k not in self._aws_signer.aws_authentication_params
- }
+ messages: list[AllMessageValues], # mutable-ok: BaseConfig signature
+ optional_params: dict, # mutable-ok: BaseConfig signature
+ litellm_params: dict, # mutable-ok: BaseConfig signature
+ headers: dict, # mutable-ok: BaseConfig signature
+ ) -> dict: # mutable-ok: BaseConfig signature
return super().transform_request(
model=strip_bedrock_routing_prefix(model),
messages=messages,
- optional_params=inference_params,
+ optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
headers=headers,
)
@@ -107,33 +275,68 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
async def async_transform_request(
self,
model: str,
- messages: list[AllMessageValues],
- optional_params: dict,
- litellm_params: dict,
- headers: dict,
- ) -> dict:
- inference_params: Final = {
- k: v for k, v in optional_params.items() if k not in self._aws_signer.aws_authentication_params
- }
+ messages: list[AllMessageValues], # mutable-ok: BaseConfig signature
+ optional_params: dict, # mutable-ok: BaseConfig signature
+ litellm_params: dict, # mutable-ok: BaseConfig signature
+ headers: dict, # mutable-ok: BaseConfig signature
+ ) -> dict: # mutable-ok: BaseConfig signature
return await super().async_transform_request(
model=strip_bedrock_routing_prefix(model),
messages=messages,
- optional_params=inference_params,
+ optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
headers=headers,
)
+ def transform_response(
+ self,
+ model: str,
+ raw_response: httpx.Response,
+ model_response: ModelResponse,
+ logging_obj: "LiteLLMLoggingObj",
+ request_data: dict, # mutable-ok: BaseConfig signature
+ messages: list[AllMessageValues], # mutable-ok: BaseConfig signature
+ optional_params: dict, # mutable-ok: BaseConfig signature
+ litellm_params: dict, # mutable-ok: BaseConfig signature
+ encoding: "tiktoken.Encoding | None",
+ api_key: str | None = None,
+ json_mode: bool | None = None,
+ ) -> ModelResponse:
+ response: Final = super().transform_response(
+ model=model,
+ raw_response=raw_response,
+ model_response=model_response,
+ logging_obj=logging_obj,
+ request_data=request_data,
+ messages=messages,
+ optional_params=optional_params,
+ litellm_params=litellm_params,
+ encoding=encoding,
+ api_key=api_key,
+ json_mode=json_mode,
+ )
+ for choice in response.choices:
+ if not isinstance(choice, Choices) or not isinstance(choice.message.content, str):
+ continue
+ reasoning, content = split_reasoning_tag(choice.message.content)
+ if reasoning is not None:
+ choice.message.reasoning_content = (
+ f"{getattr(choice.message, 'reasoning_content', None) or ''}{reasoning}"
+ )
+ choice.message.content = content
+ return response
+
def validate_environment(
self,
- headers: dict,
+ headers: dict, # mutable-ok: BaseConfig signature
model: str,
- messages: list[AllMessageValues],
- optional_params: dict,
- litellm_params: dict,
+ messages: list[AllMessageValues], # mutable-ok: BaseConfig signature
+ optional_params: dict, # mutable-ok: BaseConfig signature
+ litellm_params: dict, # mutable-ok: BaseConfig signature
api_key: str | None = None,
api_base: str | None = None,
- ) -> dict:
- headers = super().validate_environment(
+ ) -> dict: # mutable-ok: BaseConfig signature
+ validated: Final = super().validate_environment(
headers=headers,
model=model,
messages=messages,
@@ -143,31 +346,25 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
api_base=api_base,
)
project_id: Final = litellm_params.get("aws_bedrock_project_id")
- if project_id:
- headers["OpenAI-Project"] = project_id
- return headers
+ if not project_id:
+ return validated
+ return {**validated, "OpenAI-Project": project_id} # mutable-ok: BaseConfig signature returns a dict
- def get_supported_openai_params(self, model: str) -> list:
- base_params: Final = super().get_supported_openai_params(model)
- try:
- if litellm.supports_reasoning(model=model, custom_llm_provider=self.custom_llm_provider):
- if "reasoning_effort" not in base_params:
- base_params.append("reasoning_effort")
- except Exception as e:
- verbose_logger.debug("AmazonBedrockRuntimeChatCompletionsConfig: error checking reasoning support: %s", e)
- return base_params
+ def get_supported_openai_params(self, model: str) -> list: # mutable-ok: BaseConfig signature
+ base_params: Final = [param for param in super().get_supported_openai_params(model) if param != "n"]
+ if "reasoning_effort" in base_params or not litellm.supports_reasoning(
+ model=model, custom_llm_provider=self.custom_llm_provider
+ ):
+ return base_params
+ return [*base_params, "reasoning_effort"] # mutable-ok: BaseConfig signature returns a list
def get_model_response_iterator(
self,
- streaming_response: Iterator[str] | AsyncIterator[str] | Any,
+ streaming_response: Iterator[str] | AsyncIterator[str] | ModelResponse,
sync_stream: bool,
json_mode: bool | None = False,
- ) -> Any:
- from litellm.llms.openai.chat.gpt_transformation import (
- OpenAIChatCompletionStreamingHandler,
- )
-
- return OpenAIChatCompletionStreamingHandler(
+ ) -> BedrockRuntimeChatCompletionsStreamingHandler:
+ return BedrockRuntimeChatCompletionsStreamingHandler(
streaming_response=streaming_response,
sync_stream=sync_stream,
json_mode=json_mode,
diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py
index f6384aa97f1..124ef7c646c 100644
--- a/litellm/llms/bedrock/common_utils.py
+++ b/litellm/llms/bedrock/common_utils.py
@@ -28,6 +28,7 @@ from litellm.llms.base_llm.anthropic_messages.transformation import (
)
from litellm.llms.base_llm.base_utils import BaseLLMModelInfo, BaseTokenCounter
from litellm.llms.base_llm.chat.transformation import BaseLLMException
+from litellm.llms.bedrock.request_metadata import bedrock_request_metadata_is_owned
from litellm.secret_managers.main import get_secret, get_secret_str
from litellm.types.llms.bedrock import AWS_AUTH_PARAM_KEYS, AwsAuthParams
@@ -37,6 +38,18 @@ if TYPE_CHECKING:
_ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs"
_OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.")
+BedrockRoute = Literal[
+ "converse",
+ "invoke",
+ "claude_platform",
+ "converse_like",
+ "agent",
+ "agentcore",
+ "async_invoke",
+ "openai",
+ "mantle",
+ "chat_completions",
+]
def error_response_text(response: httpx.Response) -> str:
@@ -782,17 +795,54 @@ def strip_bedrock_routing_prefix(model: str) -> str:
return model
+def _bedrock_price_map_flag(model: str, flag: str) -> bool:
+ entries: Final = (litellm.model_cost.get(key) for key in (model, strip_bedrock_routing_prefix(model)))
+ return any(entry is not None and entry.get(flag) is True for entry in entries)
+
+
def uses_bedrock_runtime_chat_completions(model: str) -> bool:
"""Whether this Bedrock model should use runtime native Chat Completions.
Data-driven from the price-map ``use_bedrock_runtime_chat_completions`` flag
so onboarding a model is a JSON change. Explicit ``converse/`` still wins in
- ``get_bedrock_route`` because prefix routes are checked first.
+ ``get_bedrock_route`` because prefix routes are checked first, and a request
+ that needs a Converse-only feature (``bedrock_request_needs_converse``) is
+ served by Converse even on a flagged model.
"""
- stripped: Final = strip_bedrock_routing_prefix(model)
- return any(
- (litellm.model_cost.get(key) or {}).get("use_bedrock_runtime_chat_completions") is True
- for key in (model, stripped)
+ return _bedrock_price_map_flag(model, "use_bedrock_runtime_chat_completions")
+
+
+def bedrock_runtime_chat_completions_tools_require_reasoning_none(model: str) -> bool:
+ """Whether AWS's native Chat Completions only serves this model's function tools with ``reasoning_effort="none"``.
+
+ Data-driven from the price-map ``bedrock_runtime_chat_completions_tools_require_reasoning_none``
+ flag (the GPT-5.6 family). Converse serves tools with any effort, so those requests fall back to it.
+ """
+ return _bedrock_price_map_flag(model, "bedrock_runtime_chat_completions_tools_require_reasoning_none")
+
+
+BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
+ ("guardrailConfig", "performanceConfig", "serviceTier", "requestMetadata", "outputConfig")
+)
+
+
+def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool:
+ """Whether a request on a runtime-Chat-Completions model must still be served by Converse.
+
+ Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``) are rejected as malformed input by
+ AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body,
+ and function tools on a ``bedrock_runtime_chat_completions_tools_require_reasoning_none`` model are
+ rejected there unless ``reasoning_effort`` is exactly ``"none"``.
+ """
+ if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS):
+ return True
+ if bedrock_request_metadata_is_owned():
+ return True
+ if not request_params.get("tools"):
+ return False
+ return (
+ bedrock_runtime_chat_completions_tools_require_reasoning_none(model)
+ and request_params.get("reasoning_effort") != "none"
)
@@ -1130,20 +1180,13 @@ class BedrockModelInfo(BaseLLMModelInfo):
@staticmethod
def get_bedrock_route(
model: str,
- ) -> Literal[
- "converse",
- "invoke",
- "claude_platform",
- "converse_like",
- "agent",
- "agentcore",
- "async_invoke",
- "openai",
- "mantle",
- "chat_completions",
- ]:
+ request_params: Mapping[str, object] | None = None,
+ ) -> BedrockRoute:
"""
Get the bedrock route for the given model.
+
+ ``request_params`` (the caller's chat params) lets a runtime Chat Completions
+ model fall back to Converse for the requests only Converse can serve.
"""
route_mappings: dict[
str,
@@ -1187,7 +1230,9 @@ class BedrockModelInfo(BaseLLMModelInfo):
if is_bedrock_application_inference_profile_arn(model):
return "converse"
- if uses_bedrock_runtime_chat_completions(model):
+ if uses_bedrock_runtime_chat_completions(model) and not (
+ request_params is not None and bedrock_request_needs_converse(model, request_params)
+ ):
return "chat_completions"
base_model: Final = BedrockModelInfo.get_base_model(model)
diff --git a/litellm/main.py b/litellm/main.py
index 34410f9497c..ed78f209ca2 100644
--- a/litellm/main.py
+++ b/litellm/main.py
@@ -4078,7 +4078,7 @@ def _complete_bedrock(ctx: _CompletionDispatchContext) -> _CompletionDispatchRes
if "aws_region_name" not in optional_params or optional_params["aws_region_name"] is None:
optional_params["aws_region_name"] = aws_bedrock_client.meta.region_name
- bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model)
+ bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model, optional_params)
if bedrock_route == "claude_platform":
provider_config = ProviderConfigManager.get_provider_chat_config(
model=model,
diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json
index 70314a2e823..05658ce2d15 100644
--- a/litellm/model_prices_and_context_window_backup.json
+++ b/litellm/model_prices_and_context_window_backup.json
@@ -40870,6 +40870,7 @@
"output_cost_per_token": 0.0
},
"openai.gpt-oss-120b-1:0": {
+ "use_bedrock_runtime_chat_completions": true,
"input_cost_per_token": 1.5e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@@ -40883,6 +40884,7 @@
"supports_tool_choice": true
},
"openai.gpt-oss-20b-1:0": {
+ "use_bedrock_runtime_chat_completions": true,
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@@ -58441,6 +58443,8 @@
"supports_web_search": true
},
"us.openai.gpt-5.6-sol": {
+ "use_bedrock_runtime_chat_completions": true,
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
"input_cost_per_token": 4.4e-06,
"input_cost_per_token_above_272k_tokens": 8.8e-06,
"cache_creation_input_token_cost": 5.5e-06,
@@ -58470,6 +58474,8 @@
"supports_vision": true
},
"global.openai.gpt-5.6-sol": {
+ "use_bedrock_runtime_chat_completions": true,
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
"input_cost_per_token": 4e-06,
"input_cost_per_token_above_272k_tokens": 8e-06,
"cache_creation_input_token_cost": 5e-06,
@@ -58499,6 +58505,8 @@
"supports_vision": true
},
"us.openai.gpt-5.6-terra": {
+ "use_bedrock_runtime_chat_completions": true,
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_above_272k_tokens": 4.4e-06,
"cache_creation_input_token_cost": 2.75e-06,
@@ -58528,6 +58536,8 @@
"supports_vision": true
},
"global.openai.gpt-5.6-terra": {
+ "use_bedrock_runtime_chat_completions": true,
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
"input_cost_per_token": 2e-06,
"input_cost_per_token_above_272k_tokens": 4e-06,
"cache_creation_input_token_cost": 2.5e-06,
@@ -58557,6 +58567,8 @@
"supports_vision": true
},
"us.openai.gpt-5.6-luna": {
+ "use_bedrock_runtime_chat_completions": true,
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
"input_cost_per_token": 2.2e-07,
"input_cost_per_token_above_272k_tokens": 4.4e-07,
"cache_creation_input_token_cost": 2.75e-07,
@@ -58586,6 +58598,8 @@
"supports_vision": true
},
"global.openai.gpt-5.6-luna": {
+ "use_bedrock_runtime_chat_completions": true,
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_272k_tokens": 4e-07,
"cache_creation_input_token_cost": 2.5e-07,
diff --git a/litellm/utils.py b/litellm/utils.py
index b724313641f..82c8069770d 100644
--- a/litellm/utils.py
+++ b/litellm/utils.py
@@ -401,7 +401,7 @@ if TYPE_CHECKING:
BaseVectorStoreFilesConfig,
)
from litellm.llms.base_llm.videos.transformation import BaseVideoConfig
- from litellm.llms.bedrock.common_utils import BedrockModelInfo
+ from litellm.llms.bedrock.common_utils import BedrockModelInfo, BedrockRoute
from litellm.llms.bedrock.embed.amazon_nova_transformation import (
AmazonNovaEmbeddingConfig,
)
@@ -3350,6 +3350,17 @@ def _should_drop_param(k, additional_drop_params) -> bool:
return False
+def _bedrock_route_for_request(
+ model: str, passed_params: Mapping[str, object], additional_drop_params: list | None
+) -> BedrockRoute:
+ from litellm.llms.bedrock.common_utils import BedrockModelInfo
+
+ return BedrockModelInfo.get_bedrock_route(
+ model,
+ {k: v for k, v in passed_params.items() if not _should_drop_param(k, additional_drop_params)},
+ )
+
+
def _get_non_default_params(passed_params: dict, default_params: dict, additional_drop_params: list | None) -> dict:
non_default_params: Final = {}
for k, v in passed_params.items():
@@ -4401,9 +4412,17 @@ def get_optional_params(
message=f"{custom_llm_provider} does not support parameters: {list(unsupported_params.keys())}, for model={model}. To drop these, set `litellm.drop_params=True` or for proxy:\n\n`litellm_settings:\n drop_params: true`\n. \n If you want to use these params dynamically send allowed_openai_params={list(unsupported_params.keys())} in your request.",
)
+ bedrock_route: Final = (
+ _bedrock_route_for_request(model, passed_params, additional_drop_params)
+ if custom_llm_provider == "bedrock"
+ else None
+ )
get_supported_openai_params: Final = getattr(sys.modules[__name__], "get_supported_openai_params")
- supported_params = get_supported_openai_params(
- model=model, custom_llm_provider=custom_llm_provider, base_model=base_model
+ supported_params = (
+ litellm.AmazonConverseConfig().get_supported_openai_params(model=model)
+ if bedrock_route == "converse"
+ and isinstance(provider_config, litellm.AmazonBedrockRuntimeChatCompletionsConfig)
+ else get_supported_openai_params(model=model, custom_llm_provider=custom_llm_provider, base_model=base_model)
)
if supported_params is None:
supported_params = get_supported_openai_params(model=model, custom_llm_provider="openai")
@@ -4573,7 +4592,6 @@ def get_optional_params(
)
elif custom_llm_provider == "bedrock":
BedrockModelInfo: Final = getattr(sys.modules[__name__], "BedrockModelInfo")
- bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model)
bedrock_base_model: Final = BedrockModelInfo.get_base_model(model)
if bedrock_route == "converse" or bedrock_route == "converse_like":
optional_params = litellm.AmazonConverseConfig().map_openai_params(
diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json
index 70314a2e823..05658ce2d15 100644
--- a/model_prices_and_context_window.json
+++ b/model_prices_and_context_window.json
@@ -40870,6 +40870,7 @@
"output_cost_per_token": 0.0
},
"openai.gpt-oss-120b-1:0": {
+ "use_bedrock_runtime_chat_completions": true,
"input_cost_per_token": 1.5e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@@ -40883,6 +40884,7 @@
"supports_tool_choice": true
},
"openai.gpt-oss-20b-1:0": {
+ "use_bedrock_runtime_chat_completions": true,
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@@ -58441,6 +58443,8 @@
"supports_web_search": true
},
"us.openai.gpt-5.6-sol": {
+ "use_bedrock_runtime_chat_completions": true,
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
"input_cost_per_token": 4.4e-06,
"input_cost_per_token_above_272k_tokens": 8.8e-06,
"cache_creation_input_token_cost": 5.5e-06,
@@ -58470,6 +58474,8 @@
"supports_vision": true
},
"global.openai.gpt-5.6-sol": {
+ "use_bedrock_runtime_chat_completions": true,
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
"input_cost_per_token": 4e-06,
"input_cost_per_token_above_272k_tokens": 8e-06,
"cache_creation_input_token_cost": 5e-06,
@@ -58499,6 +58505,8 @@
"supports_vision": true
},
"us.openai.gpt-5.6-terra": {
+ "use_bedrock_runtime_chat_completions": true,
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_above_272k_tokens": 4.4e-06,
"cache_creation_input_token_cost": 2.75e-06,
@@ -58528,6 +58536,8 @@
"supports_vision": true
},
"global.openai.gpt-5.6-terra": {
+ "use_bedrock_runtime_chat_completions": true,
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
"input_cost_per_token": 2e-06,
"input_cost_per_token_above_272k_tokens": 4e-06,
"cache_creation_input_token_cost": 2.5e-06,
@@ -58557,6 +58567,8 @@
"supports_vision": true
},
"us.openai.gpt-5.6-luna": {
+ "use_bedrock_runtime_chat_completions": true,
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
"input_cost_per_token": 2.2e-07,
"input_cost_per_token_above_272k_tokens": 4.4e-07,
"cache_creation_input_token_cost": 2.75e-07,
@@ -58586,6 +58598,8 @@
"supports_vision": true
},
"global.openai.gpt-5.6-luna": {
+ "use_bedrock_runtime_chat_completions": true,
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_272k_tokens": 4e-07,
"cache_creation_input_token_cost": 2.5e-07,
diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json
index eca89e23887..9db1c8eedc4 100644
--- a/model_prices_and_context_window.schema.json
+++ b/model_prices_and_context_window.schema.json
@@ -74,6 +74,9 @@
"xhigh"
]
},
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": {
+ "type": "boolean"
+ },
"cache_creation_input_audio_token_cost": {
"type": "number",
"minimum": 0
diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
index 7724b3f0a52..230e323ace3 100644
--- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
+++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -1,7 +1,6 @@
-"""Native Bedrock Runtime Chat Completions: Grok stays on /openai/v1/chat/completions."""
+"""Native Bedrock Runtime Chat Completions: Grok, gpt-oss and GPT-5.6 stay on /openai/v1/chat/completions."""
import json
-from unittest.mock import patch
import httpx
import pytest
@@ -9,25 +8,28 @@ import pytest
import litellm
from litellm.llms.bedrock.chat.chat_completions.transformation import (
AmazonBedrockRuntimeChatCompletionsConfig,
+ BedrockRuntimeChatCompletionsStreamingHandler,
+ ReasoningTagSplitter,
+ split_reasoning_tag,
+ with_max_completion_tokens,
)
from litellm.llms.bedrock.common_utils import (
+ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS,
BedrockModelInfo,
+ bedrock_request_needs_converse,
get_bedrock_chat_config,
uses_bedrock_runtime_chat_completions,
)
+from litellm.llms.custom_httpx.http_handler import HTTPHandler
@pytest.fixture
def local_cost_map(monkeypatch):
- original_model_cost = litellm.model_cost
- try:
- monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true")
- litellm.model_cost = litellm.get_model_cost_map(url="")
- litellm.get_model_info.cache_clear()
- yield
- finally:
- litellm.model_cost = original_model_cost
- litellm.get_model_info.cache_clear()
+ monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true")
+ monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
+ litellm.get_model_info.cache_clear()
+ yield
+ litellm.get_model_info.cache_clear()
@pytest.mark.parametrize(
@@ -103,7 +105,27 @@ def test_transform_request_is_openai_chat_body_not_converse():
assert "messages" in body
-def test_completion_posts_runtime_chat_completions(local_cost_map, monkeypatch):
+def _chat_completion_json(content, model, tool_calls=None):
+ message = {"role": "assistant", "content": content, **({"tool_calls": tool_calls} if tool_calls else {})}
+ return {
+ "id": "chatcmpl-test",
+ "object": "chat.completion",
+ "created": 1733529600,
+ "model": model,
+ "choices": [{"index": 0, "message": message, "finish_reason": "tool_calls" if tool_calls else "stop"}],
+ "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
+ }
+
+
+CONVERSE_JSON = {
+ "output": {"message": {"role": "assistant", "content": [{"text": "ok"}]}},
+ "stopReason": "end_turn",
+ "usage": {"inputTokens": 1, "outputTokens": 1, "totalTokens": 2},
+}
+
+
+@pytest.fixture
+def fake_aws_env(monkeypatch):
monkeypatch.setenv("AWS_REGION_NAME", "us-west-2")
monkeypatch.delenv("AWS_BEDROCK_RUNTIME_ENDPOINT", raising=False)
monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False)
@@ -111,40 +133,490 @@ def test_completion_posts_runtime_chat_completions(local_cost_map, monkeypatch):
monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing")
monkeypatch.setenv("AWS_SESSION_TOKEN", "testing")
- requests: list[dict] = []
- def mock_post(self, url, data=None, json=None, headers=None, **kwargs):
- requests.append({"url": url, "data": data, "json": json, "headers": headers or {}})
- return httpx.Response(
- status_code=200,
- json={
- "id": "chatcmpl-test",
- "object": "chat.completion",
- "created": 1733529600,
- "model": "us.xai.grok-4.6",
- "choices": [
- {
- "index": 0,
- "message": {"role": "assistant", "content": "ok"},
- "finish_reason": "stop",
- }
- ],
- "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
- },
- request=httpx.Request("POST", url),
- )
+def _recording_client(**response_kwargs):
+ requests: list[httpx.Request] = []
- with patch("litellm.llms.custom_httpx.http_handler.HTTPHandler.post", mock_post):
- response = litellm.completion(
- model="us.xai.grok-4.6",
- messages=[{"role": "user", "content": "hello"}],
- )
+ def handle(request):
+ requests.append(request)
+ return httpx.Response(200, **response_kwargs)
+
+ return requests, HTTPHandler(client=httpx.Client(transport=httpx.MockTransport(handle)))
+
+
+def test_completion_posts_runtime_chat_completions(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=_chat_completion_json("ok", "us.xai.grok-4.6"))
+ response = litellm.completion(
+ model="us.xai.grok-4.6",
+ messages=[{"role": "user", "content": "hello"}],
+ client=client,
+ )
assert response.choices[0].message.content == "ok"
assert len(requests) == 1
- assert requests[0]["url"] == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
- raw = requests[0]["data"]
- body = json.loads(raw) if isinstance(raw, (str, bytes, bytearray)) else (requests[0]["json"] or {})
+ assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
+ body = json.loads(requests[0].content)
assert body["model"] == "us.xai.grok-4.6"
assert body["messages"] == [{"role": "user", "content": "hello"}]
assert "inferenceConfig" not in body
+
+
+OPENAI_RUNTIME_MODELS = (
+ "openai.gpt-oss-20b-1:0",
+ "openai.gpt-oss-120b-1:0",
+ "us.openai.gpt-5.6-sol",
+ "global.openai.gpt-5.6-sol",
+ "us.openai.gpt-5.6-terra",
+ "global.openai.gpt-5.6-terra",
+ "us.openai.gpt-5.6-luna",
+ "global.openai.gpt-5.6-luna",
+)
+GET_WEATHER_TOOL = {
+ "type": "function",
+ "function": {
+ "name": "get_weather",
+ "parameters": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"]},
+ },
+}
+
+
+@pytest.mark.parametrize("model", [*OPENAI_RUNTIME_MODELS, "bedrock/openai.gpt-oss-20b-1:0"])
+def test_openai_runtime_models_use_chat_completions_route(local_cost_map, model):
+ assert uses_bedrock_runtime_chat_completions(model) is True
+ assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions"
+ assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig)
+
+
+@pytest.mark.parametrize("model", ["us.amazon.nova-micro-v1:0", "us.anthropic.claude-haiku-4-5-20251001-v1:0"])
+def test_nova_and_claude_stay_on_converse(local_cost_map, model):
+ assert uses_bedrock_runtime_chat_completions(model) is False
+ assert BedrockModelInfo.get_bedrock_route(model, {"tools": [GET_WEATHER_TOOL]}) == "converse"
+
+
+@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "global.openai.gpt-5.6-sol"])
+def test_guardrail_config_falls_back_to_converse(local_cost_map, model):
+ guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"}
+ assert bedrock_request_needs_converse(model, {"guardrailConfig": guardrail}) is True
+ assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": guardrail}) == "converse"
+ assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": None}) == "chat_completions"
+
+
+@pytest.mark.parametrize(
+ "request_params, expected_route",
+ [
+ ({"tools": [GET_WEATHER_TOOL]}, "converse"),
+ ({"tools": [GET_WEATHER_TOOL], "reasoning_effort": "low"}, "converse"),
+ ({"tools": [GET_WEATHER_TOOL], "reasoning_effort": None}, "converse"),
+ ({"tools": [GET_WEATHER_TOOL], "reasoning_effort": "none"}, "chat_completions"),
+ ({"reasoning_effort": "low"}, "chat_completions"),
+ ({"tools": None, "reasoning_effort": "low"}, "chat_completions"),
+ ({"tools": [], "reasoning_effort": "low"}, "chat_completions"),
+ ({}, "chat_completions"),
+ ],
+)
+def test_gpt56_tools_need_reasoning_none_on_chat_completions(local_cost_map, request_params, expected_route):
+ assert BedrockModelInfo.get_bedrock_route("global.openai.gpt-5.6-sol", request_params) == expected_route
+ assert BedrockModelInfo.get_bedrock_route("bedrock/us.openai.gpt-5.6-terra", request_params) == expected_route
+
+
+@pytest.mark.parametrize("reasoning_effort", ["low", "high", None])
+def test_gpt_oss_tools_with_any_reasoning_effort_stay_on_chat_completions(local_cost_map, reasoning_effort):
+ params = {"tools": [GET_WEATHER_TOOL], "reasoning_effort": reasoning_effort}
+ assert bedrock_request_needs_converse("openai.gpt-oss-120b-1:0", params) is False
+ assert BedrockModelInfo.get_bedrock_route("openai.gpt-oss-120b-1:0", params) == "chat_completions"
+
+
+def test_explicit_converse_prefix_wins_for_openai_models(local_cost_map):
+ assert BedrockModelInfo.get_bedrock_route("bedrock/converse/openai.gpt-oss-20b-1:0") == "converse"
+ assert BedrockModelInfo.get_bedrock_route("converse/global.openai.gpt-5.6-sol", {}) == "converse"
+
+
+def test_map_openai_params_sends_max_tokens_as_max_completion_tokens():
+ cfg = AmazonBedrockRuntimeChatCompletionsConfig()
+ mapped = cfg.map_openai_params(
+ non_default_params={"max_tokens": 64, "temperature": 0.1},
+ optional_params={},
+ model="global.openai.gpt-5.6-sol",
+ drop_params=False,
+ )
+ assert mapped == {"max_completion_tokens": 64, "temperature": 0.1}
+
+
+def test_map_openai_params_keeps_explicit_max_completion_tokens():
+ cfg = AmazonBedrockRuntimeChatCompletionsConfig()
+ mapped = cfg.map_openai_params(
+ non_default_params={"max_tokens": 64, "max_completion_tokens": 32},
+ optional_params={},
+ model="openai.gpt-oss-20b-1:0",
+ drop_params=False,
+ )
+ assert mapped == {"max_completion_tokens": 32}
+
+
+def test_with_max_completion_tokens_leaves_other_params_alone():
+ assert with_max_completion_tokens({"temperature": 0.5}) == {"temperature": 0.5}
+
+
+def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map):
+ cfg = AmazonBedrockRuntimeChatCompletionsConfig()
+ assert "reasoning_effort" in cfg.get_supported_openai_params("global.openai.gpt-5.6-sol")
+ assert "reasoning_effort" in cfg.get_supported_openai_params("openai.gpt-oss-20b-1:0")
+
+
+def test_split_reasoning_tag_splits_leading_tag():
+ assert split_reasoning_tag("plan it\n\n\nHello") == ("plan it\n", "Hello")
+
+
+def test_split_reasoning_tag_drops_an_empty_tag():
+ assert split_reasoning_tag("Hello") == (None, "Hello")
+
+
+@pytest.mark.parametrize(
+ "content",
+ [
+ "plan it\n\n\nHello",
+ "never closed",
+ "later",
+ "",
+ ],
+)
+@pytest.mark.parametrize("chunk_size", [1, 3, 7])
+def test_split_reasoning_tag_matches_the_streamed_split(content, chunk_size):
+ chunks = [content[start : start + chunk_size] for start in range(0, len(content), chunk_size)]
+ streamed_reasoning, streamed_content = _run_splitter(chunks)
+
+ assert split_reasoning_tag(content) == (streamed_reasoning or None, streamed_content)
+
+
+def test_split_reasoning_tag_passes_plain_content_through():
+ assert split_reasoning_tag("Hello") == (None, "Hello")
+
+
+def test_split_reasoning_tag_ignores_tag_after_content_starts():
+ content = "Hello not mine"
+ assert split_reasoning_tag(content) == (None, content)
+
+
+def _run_splitter(chunks):
+ state = ReasoningTagSplitter()
+ reasoning = ""
+ content = ""
+ for chunk in chunks:
+ state, fed_reasoning, fed_content = state.feed(chunk)
+ reasoning += fed_reasoning
+ content += fed_content
+ state, flushed_reasoning, flushed_content = state.flush()
+ return reasoning + flushed_reasoning, content + flushed_content
+
+
+def test_reasoning_tag_splitter_handles_tags_split_across_chunks():
+ assert _run_splitter(["I think", " so\n\nHel", "lo"]) == ("I think so", "Hello")
+
+
+def test_reasoning_tag_splitter_passes_plain_content_through():
+ assert _run_splitter(["Hel", "lo later"]) == ("", "Hello later")
+
+
+def test_reasoning_tag_splitter_flushes_unclosed_reasoning():
+ assert _run_splitter(["never clo", "sed"]) == ("never closed", "")
+
+
+def test_reasoning_tag_splitter_releases_a_false_tag_prefix():
+ assert _run_splitter(["<", "b>x"]) == ("", "x")
+
+
+def _stream_chunk(delta, finish_reason=None, index=0):
+ return {
+ "id": "chatcmpl-test",
+ "object": "chat.completion.chunk",
+ "created": 1733529600,
+ "model": "openai.gpt-oss-20b-1:0",
+ "choices": [{"index": index, "delta": delta, "finish_reason": finish_reason}],
+ }
+
+
+def test_streaming_handler_splits_reasoning_deltas_per_choice():
+ handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True)
+
+ first = handler.chunk_parser(_stream_chunk({"role": "assistant", "content": "I think"}))
+ assert first.choices[0].delta.reasoning_content == "I think"
+ assert not first.choices[0].delta.content
+
+ second = handler.chunk_parser(_stream_chunk({"content": " so\n\nHello"}))
+ assert second.choices[0].delta.reasoning_content == " so"
+ assert second.choices[0].delta.content == "Hello"
+
+ tool_call = {"index": 0, "id": "call_0", "type": "function", "function": {"name": "get_weather", "arguments": "{}"}}
+ third = handler.chunk_parser(_stream_chunk({"content": None, "tool_calls": [tool_call]}))
+ assert third.choices[0].delta.tool_calls[0].function.name == "get_weather"
+
+ last = handler.chunk_parser(_stream_chunk({}, finish_reason="stop"))
+ assert last.choices[0].finish_reason == "stop"
+
+
+def _reasoning_of(parsed):
+ return getattr(parsed.choices[0].delta, "reasoning_content", None)
+
+
+def test_streaming_handler_keeps_split_state_per_choice_index():
+ handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True)
+
+ opened = handler.chunk_parser(_stream_chunk({"content": "first"}, index=0))
+ assert _reasoning_of(opened) == "first"
+
+ plain = handler.chunk_parser(_stream_chunk({"content": "plain answer"}, index=1))
+ assert _reasoning_of(plain) is None
+ assert plain.choices[0].delta.content == "plain answer"
+
+ still_reasoning = handler.chunk_parser(_stream_chunk({"content": " more"}, index=0))
+ assert _reasoning_of(still_reasoning) == " more"
+ assert not still_reasoning.choices[0].delta.content
+
+
+def test_streaming_handler_flushes_held_text_on_an_empty_final_delta():
+ handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True)
+
+ held = handler.chunk_parser(_stream_chunk({"content": "almost doneplan\n\nHi", "openai.gpt-oss-20b-1:0")
+ )
+ response = litellm.completion(
+ model="bedrock/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ max_tokens=64,
+ reasoning_effort="low",
+ tools=[GET_WEATHER_TOOL],
+ client=client,
+ )
+
+ assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
+ body = json.loads(requests[0].content)
+ assert body["model"] == "openai.gpt-oss-20b-1:0"
+ assert body["max_completion_tokens"] == 64
+ assert "max_tokens" not in body
+ assert body["reasoning_effort"] == "low"
+ assert body["tools"] == [GET_WEATHER_TOOL]
+ assert response.choices[0].message.reasoning_content == "plan"
+ assert response.choices[0].message.content == "Hi"
+
+
+def test_gpt56_tools_with_reasoning_effort_go_to_converse(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=CONVERSE_JSON)
+ response = litellm.completion(
+ model="bedrock/global.openai.gpt-5.6-sol",
+ messages=[{"role": "user", "content": "hello"}],
+ tools=[GET_WEATHER_TOOL],
+ reasoning_effort="low",
+ client=client,
+ )
+
+ assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse")
+ assert json.loads(requests[0].content)["toolConfig"]["tools"][0]["toolSpec"]["name"] == "get_weather"
+ assert response.choices[0].message.content == "ok"
+
+
+def test_gpt56_tools_with_reasoning_none_stay_on_chat_completions(local_cost_map, fake_aws_env):
+ tool_calls = [
+ {"id": "call_0", "type": "function", "function": {"name": "get_weather", "arguments": '{"city": "Paris"}'}}
+ ]
+ requests, client = _recording_client(json=_chat_completion_json(None, "global.openai.gpt-5.6-sol", tool_calls))
+ response = litellm.completion(
+ model="bedrock/global.openai.gpt-5.6-sol",
+ messages=[{"role": "user", "content": "weather in Paris"}],
+ tools=[GET_WEATHER_TOOL],
+ reasoning_effort="none",
+ max_tokens=64,
+ client=client,
+ )
+
+ assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
+ body = json.loads(requests[0].content)
+ assert body["tools"] == [GET_WEATHER_TOOL]
+ assert body["reasoning_effort"] == "none"
+ assert body["max_completion_tokens"] == 64
+ assert response.choices[0].message.tool_calls[0].function.name == "get_weather"
+
+
+@pytest.mark.parametrize(
+ "converse_only_param",
+ [
+ {"guardrailConfig": {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"}},
+ {"performanceConfig": {"latency": "optimized"}},
+ {"requestMetadata": {"team": "search"}},
+ {"serviceTier": {"type": "priority"}},
+ ],
+ ids=lambda param: next(iter(param)),
+)
+def test_converse_only_request_keys_go_to_converse(local_cost_map, fake_aws_env, converse_only_param):
+ requests, client = _recording_client(json=CONVERSE_JSON)
+ litellm.completion(
+ model="bedrock/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ client=client,
+ **converse_only_param,
+ )
+
+ assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse")
+ (key, value), = converse_only_param.items()
+ assert json.loads(requests[0].content)[key] == value
+
+
+def test_converse_only_keys_cover_every_converse_config_block():
+ assert set(litellm.AmazonConverseConfig.get_config_blocks()) <= BEDROCK_CONVERSE_ONLY_REQUEST_KEYS
+
+
+def test_operator_owned_request_metadata_goes_to_converse(local_cost_map, fake_aws_env, monkeypatch):
+ monkeypatch.setattr(litellm, "bedrock_request_metadata_fields", ["user_api_key_team_alias"])
+ requests, client = _recording_client(json=CONVERSE_JSON)
+ litellm.completion(
+ model="bedrock/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ metadata={"user_api_key_team_alias": "search"},
+ client=client,
+ )
+
+ assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse")
+ assert json.loads(requests[0].content)["requestMetadata"] == {"user_api_key_team_alias": "search"}
+
+
+def test_dropped_converse_only_key_keeps_the_request_on_chat_completions(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0"))
+ litellm.completion(
+ model="bedrock/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ guardrailConfig={"guardrailIdentifier": "gr-1", "guardrailVersion": "1"},
+ additional_drop_params=["guardrailConfig"],
+ max_tokens=8,
+ client=client,
+ )
+
+ assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
+ body = json.loads(requests[0].content)
+ assert "guardrailConfig" not in body
+ assert body["max_completion_tokens"] == 8
+ assert "inferenceConfig" not in body
+
+
+def test_dropped_tools_keep_gpt56_reasoning_request_on_chat_completions(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=_chat_completion_json("ok", "global.openai.gpt-5.6-sol"))
+ litellm.completion(
+ model="bedrock/global.openai.gpt-5.6-sol",
+ messages=[{"role": "user", "content": "hello"}],
+ tools=[GET_WEATHER_TOOL],
+ reasoning_effort="low",
+ additional_drop_params=["tools"],
+ client=client,
+ )
+
+ assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
+ body = json.loads(requests[0].content)
+ assert "tools" not in body
+ assert body["reasoning_effort"] == "low"
+
+
+def test_legacy_functions_stay_on_chat_completions(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0"))
+ litellm.completion(
+ model="bedrock/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ functions=[GET_WEATHER_TOOL["function"]],
+ client=client,
+ )
+
+ assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
+ assert json.loads(requests[0].content)["functions"] == [GET_WEATHER_TOOL["function"]]
+
+
+def test_converse_fallback_validates_against_converse_params(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=CONVERSE_JSON)
+ guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"}
+ with pytest.raises(litellm.UnsupportedParamsError, match="seed"):
+ litellm.completion(
+ model="bedrock/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ guardrailConfig=guardrail,
+ seed=7,
+ client=client,
+ )
+ litellm.completion(
+ model="bedrock/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ guardrailConfig=guardrail,
+ seed=7,
+ drop_params=True,
+ client=client,
+ )
+
+ assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse")
+ assert "seed" not in json.loads(requests[0].content)
+
+
+def test_n_is_rejected_before_reaching_chat_completions(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0"))
+ with pytest.raises(litellm.UnsupportedParamsError, match="'n'"):
+ litellm.completion(
+ model="bedrock/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ n=2,
+ client=client,
+ )
+ litellm.completion(
+ model="bedrock/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ n=2,
+ drop_params=True,
+ client=client,
+ )
+
+ assert "n" not in json.loads(requests[0].content)
+
+
+def _sse(chunks):
+ return ("".join(f"data: {json.dumps(chunk)}\n\n" for chunk in chunks) + "data: [DONE]\n\n").encode()
+
+
+def test_gpt_oss_streaming_completion_splits_reasoning(local_cost_map, fake_aws_env):
+ chunks = (
+ _stream_chunk({"role": "assistant", "content": "plan"}),
+ _stream_chunk({"content": "\n\nHi"}),
+ _stream_chunk({}, finish_reason="stop"),
+ )
+ requests, client = _recording_client(content=_sse(chunks), headers={"content-type": "text/event-stream"})
+ stream = litellm.completion(
+ model="bedrock/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ stream=True,
+ client=client,
+ )
+ deltas = [chunk.choices[0].delta for chunk in stream]
+
+ assert [str(request.url) for request in requests] == [
+ "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
+ ]
+ assert json.loads(requests[0].content)["stream"] is True
+ assert "".join(getattr(delta, "reasoning_content", None) or "" for delta in deltas) == "plan"
+ assert "".join(delta.content or "" for delta in deltas) == "Hi"
+
+
+def test_streaming_handler_keeps_native_reasoning_next_to_the_tagged_split():
+ handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True)
+ parsed = handler.chunk_parser(
+ _stream_chunk({"reasoning": "native ", "content": "taggedHi"}, finish_reason="stop")
+ )
+
+ assert parsed.choices[0].delta.reasoning_content == "native tagged"
+ assert parsed.choices[0].delta.content == "Hi"
diff --git a/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py b/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py
index aa0827c5ae5..a6b8ba1da1d 100644
--- a/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py
+++ b/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py
@@ -138,9 +138,12 @@ def _bedrock_response(model, usage):
@pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id)
-def test_bedrock_gpt_5_6_profiles_route_to_converse(profile, local_model_cost_map):
- """GPT-5.6 is served by Converse on bedrock-runtime, never by Invoke."""
- assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "converse"
+def test_bedrock_gpt_5_6_profiles_never_route_to_invoke(profile, local_model_cost_map):
+ """GPT-5.6 is served by bedrock-runtime's native Chat Completions, and by Converse when
+ the request carries function tools without reasoning_effort "none", never by Invoke."""
+ assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "chat_completions"
+ tools_with_reasoning = {"tools": [{"type": "function", "function": {"name": "f"}}], "reasoning_effort": "low"}
+ assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}", tools_with_reasoning) == "converse"
@pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id)
diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py
index 3336ad6d33a..963bef1114a 100644
--- a/tests/test_litellm/test_utils.py
+++ b/tests/test_litellm/test_utils.py
@@ -878,6 +878,8 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"supports_video_input": {"type": "boolean"},
"supports_vision": {"type": "boolean"},
"supports_web_search": {"type": "boolean"},
+ "use_bedrock_runtime_chat_completions": {"type": "boolean"},
+ "bedrock_runtime_chat_completions_tools_require_reasoning_none": {"type": "boolean"},
"supports_url_context": {"type": "boolean"},
"supports_multimodal": {"type": "boolean"},
"uses_embed_content": {"type": "boolean"},
From a1c089f10771d4cdbdf6fd4920b6eb0c3488cd3d Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Sat, 19 Sep 2026 18:47:54 -0700
Subject: [PATCH 03/18] fix(bedrock): route gpt-oss response_format to Converse
and decide the route once from the raw request
---
ci_cd/generate_model_prices_schema.py | 2 -
.../chat/chat_completions/transformation.py | 5 +-
litellm/llms/bedrock/common_utils.py | 57 ++++++--
litellm/main.py | 7 +-
...odel_prices_and_context_window_backup.json | 42 +++---
litellm/types/completion.py | 3 +-
litellm/utils.py | 9 +-
model_prices_and_context_window.json | 42 +++---
model_prices_and_context_window.schema.json | 15 ++-
...bedrock_chat_completions_transformation.py | 125 +++++++++++++++++-
tests/test_litellm/test_utils.py | 5 +-
tests/test_litellm/types/test_completion.py | 1 +
12 files changed, 248 insertions(+), 65 deletions(-)
diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py
index 648cceb79b0..8eec07dadda 100644
--- a/ci_cd/generate_model_prices_schema.py
+++ b/ci_cd/generate_model_prices_schema.py
@@ -30,8 +30,6 @@ EXTRA_BOOLEAN_KEYS = frozenset(
"gemini_audio_only_live",
"uses_embed_content",
"use_openai_responses_path",
- "use_bedrock_runtime_chat_completions",
- "bedrock_runtime_chat_completions_tools_require_reasoning_none",
"bedrock_converse_supports_strict_tools",
"thinking_always_on",
}
diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py
index 49218270064..18ba06fee0b 100644
--- a/litellm/llms/bedrock/chat/chat_completions/transformation.py
+++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py
@@ -3,7 +3,7 @@ Native OpenAI Chat Completions on Amazon Bedrock Runtime.
AWS serves this surface at
``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions``
-for the models whose price-map entry sets ``use_bedrock_runtime_chat_completions``
+for the models whose price-map entry sets ``supports_bedrock_runtime_chat_completions``
(Grok 4.6, gpt-oss, the GPT-5.6 family): chat completions stay chat completions
instead of being rewritten to Converse.
@@ -19,6 +19,7 @@ from types import MappingProxyType
from typing import TYPE_CHECKING, Final, Literal
import httpx
+from typing_extensions import assert_never
import litellm
from litellm.llms.base_llm.chat.transformation import BaseLLMException
@@ -72,6 +73,8 @@ class ReasoningTagSplitter:
return self._feed_start(self.pending + text)
case "reasoning":
return self._feed_reasoning(self.pending + text)
+ case _:
+ assert_never(self.phase)
def _feed_start(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]:
if buffered.startswith(REASONING_OPEN_TAG):
diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py
index 124ef7c646c..df30eddd856 100644
--- a/litellm/llms/bedrock/common_utils.py
+++ b/litellm/llms/bedrock/common_utils.py
@@ -803,22 +803,33 @@ def _bedrock_price_map_flag(model: str, flag: str) -> bool:
def uses_bedrock_runtime_chat_completions(model: str) -> bool:
"""Whether this Bedrock model should use runtime native Chat Completions.
- Data-driven from the price-map ``use_bedrock_runtime_chat_completions`` flag
+ Data-driven from the price-map ``supports_bedrock_runtime_chat_completions`` flag
so onboarding a model is a JSON change. Explicit ``converse/`` still wins in
``get_bedrock_route`` because prefix routes are checked first, and a request
that needs a Converse-only feature (``bedrock_request_needs_converse``) is
served by Converse even on a flagged model.
"""
- return _bedrock_price_map_flag(model, "use_bedrock_runtime_chat_completions")
+ return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions")
-def bedrock_runtime_chat_completions_tools_require_reasoning_none(model: str) -> bool:
- """Whether AWS's native Chat Completions only serves this model's function tools with ``reasoning_effort="none"``.
+def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool:
+ """Whether AWS's native Chat Completions serves this model's function tools with any ``reasoning_effort``.
- Data-driven from the price-map ``bedrock_runtime_chat_completions_tools_require_reasoning_none``
- flag (the GPT-5.6 family). Converse serves tools with any effort, so those requests fall back to it.
+ Data-driven from the price-map ``supports_bedrock_runtime_chat_completions_tools_with_reasoning``
+ flag (gpt-oss, Grok). Without it AWS only takes tools with ``reasoning_effort="none"``
+ (the GPT-5.6 family), and Converse serves tools with any effort, so those requests fall back to it.
"""
- return _bedrock_price_map_flag(model, "bedrock_runtime_chat_completions_tools_require_reasoning_none")
+ return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_tools_with_reasoning")
+
+
+def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> bool:
+ """Whether AWS's native Chat Completions enforces a ``response_format`` schema for this model.
+
+ Data-driven from the price-map ``supports_bedrock_runtime_chat_completions_response_format`` flag
+ (GPT-5.6, Grok). Without it AWS accepts the field and answers with unconstrained text (gpt-oss), so
+ Converse, which emulates the schema through a forced ``json_tool_call`` tool, serves those requests.
+ """
+ return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_response_format")
BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
@@ -826,26 +837,52 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
)
+def _response_format_constrains_output(response_format: object) -> bool:
+ if response_format is None:
+ return False
+ return not (isinstance(response_format, Mapping) and response_format.get("type") == "text")
+
+
def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool:
"""Whether a request on a runtime-Chat-Completions model must still be served by Converse.
Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``) are rejected as malformed input by
AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body,
- and function tools on a ``bedrock_runtime_chat_completions_tools_require_reasoning_none`` model are
- rejected there unless ``reasoning_effort`` is exactly ``"none"``.
+ function tools on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning``
+ are rejected there unless ``reasoning_effort`` is exactly ``"none"``, and a constraining
+ ``response_format`` on a model without ``supports_bedrock_runtime_chat_completions_response_format``
+ is only honored by Converse.
"""
if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS):
return True
if bedrock_request_metadata_is_owned():
return True
+ if _response_format_constrains_output(
+ request_params.get("response_format")
+ ) and not bedrock_runtime_chat_completions_enforces_response_format(model):
+ return True
if not request_params.get("tools"):
return False
return (
- bedrock_runtime_chat_completions_tools_require_reasoning_none(model)
+ not bedrock_runtime_chat_completions_serves_tools_with_reasoning(model)
and request_params.get("reasoning_effort") != "none"
)
+def bedrock_route_for_request(
+ model: str, request_params: Mapping[str, object], additional_drop_params: Sequence[str] | None
+) -> BedrockRoute:
+ """The route for one request, decided from the caller's raw params before any provider mapping.
+
+ Param mapping and dispatch both call this with the same inputs, so a request that falls back to
+ Converse is mapped with the Converse config and sent to Converse, never one without the other.
+ """
+ dropped: Final = frozenset(additional_drop_params or ())
+ return BedrockModelInfo.get_bedrock_route(
+ model, {key: value for key, value in request_params.items() if key not in dropped}
+ )
+
+
def strip_bedrock_throughput_suffix(model: str) -> str:
"""Strip throughput tier suffixes and context window suffixes from Bedrock model names."""
import re
diff --git a/litellm/main.py b/litellm/main.py
index ed78f209ca2..539479fb224 100644
--- a/litellm/main.py
+++ b/litellm/main.py
@@ -104,7 +104,7 @@ from litellm.llms.base_llm import BaseConfig, BaseImageGenerationConfig
from litellm.llms.base_llm.base_model_iterator import (
convert_model_response_to_streaming,
)
-from litellm.llms.bedrock.common_utils import BedrockModelInfo
+from litellm.llms.bedrock.common_utils import BedrockModelInfo, bedrock_route_for_request
from litellm.llms.cohere.common_utils import CohereModelInfo
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler, http2_enabled
from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config
@@ -4078,7 +4078,9 @@ def _complete_bedrock(ctx: _CompletionDispatchContext) -> _CompletionDispatchRes
if "aws_region_name" not in optional_params or optional_params["aws_region_name"] is None:
optional_params["aws_region_name"] = aws_bedrock_client.meta.region_name
- bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model, optional_params)
+ bedrock_route: Final = bedrock_route_for_request(
+ model, ctx.request_params, ctx.kwargs.get("additional_drop_params")
+ )
if bedrock_route == "claude_platform":
provider_config = ProviderConfigManager.get_provider_chat_config(
model=model,
@@ -5686,6 +5688,7 @@ def completion(
optional_params=optional_params,
organization=organization,
provider_config=provider_config,
+ request_params={**optional_param_args, **non_default_params},
shared_session=shared_session,
stream=stream,
temperature=temperature,
diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json
index 05658ce2d15..6dbf246fd70 100644
--- a/litellm/model_prices_and_context_window_backup.json
+++ b/litellm/model_prices_and_context_window_backup.json
@@ -40870,7 +40870,8 @@
"output_cost_per_token": 0.0
},
"openai.gpt-oss-120b-1:0": {
- "use_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 1.5e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@@ -40884,7 +40885,8 @@
"supports_tool_choice": true
},
"openai.gpt-oss-20b-1:0": {
- "use_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@@ -47234,7 +47236,9 @@
"input_cost_per_token": 2.64e-06,
"output_cost_per_token": 7.92e-06,
"cache_read_input_token_cost": 6.6e-07,
- "use_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@@ -58443,8 +58447,8 @@
"supports_web_search": true
},
"us.openai.gpt-5.6-sol": {
- "use_bedrock_runtime_chat_completions": true,
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4.4e-06,
"input_cost_per_token_above_272k_tokens": 8.8e-06,
"cache_creation_input_token_cost": 5.5e-06,
@@ -58474,8 +58478,8 @@
"supports_vision": true
},
"global.openai.gpt-5.6-sol": {
- "use_bedrock_runtime_chat_completions": true,
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4e-06,
"input_cost_per_token_above_272k_tokens": 8e-06,
"cache_creation_input_token_cost": 5e-06,
@@ -58505,8 +58509,8 @@
"supports_vision": true
},
"us.openai.gpt-5.6-terra": {
- "use_bedrock_runtime_chat_completions": true,
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_above_272k_tokens": 4.4e-06,
"cache_creation_input_token_cost": 2.75e-06,
@@ -58536,8 +58540,8 @@
"supports_vision": true
},
"global.openai.gpt-5.6-terra": {
- "use_bedrock_runtime_chat_completions": true,
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-06,
"input_cost_per_token_above_272k_tokens": 4e-06,
"cache_creation_input_token_cost": 2.5e-06,
@@ -58567,8 +58571,8 @@
"supports_vision": true
},
"us.openai.gpt-5.6-luna": {
- "use_bedrock_runtime_chat_completions": true,
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-07,
"input_cost_per_token_above_272k_tokens": 4.4e-07,
"cache_creation_input_token_cost": 2.75e-07,
@@ -58598,8 +58602,8 @@
"supports_vision": true
},
"global.openai.gpt-5.6-luna": {
- "use_bedrock_runtime_chat_completions": true,
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_272k_tokens": 4e-07,
"cache_creation_input_token_cost": 2.5e-07,
@@ -58909,7 +58913,9 @@
"input_cost_per_token": 2.2e-06,
"output_cost_per_token": 6.6e-06,
"cache_read_input_token_cost": 5.5e-07,
- "use_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@@ -58925,7 +58931,9 @@
"input_cost_per_token": 2e-06,
"output_cost_per_token": 6e-06,
"cache_read_input_token_cost": 5e-07,
- "use_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
diff --git a/litellm/types/completion.py b/litellm/types/completion.py
index c1c6cc9ed1c..1e6cfc0ee33 100644
--- a/litellm/types/completion.py
+++ b/litellm/types/completion.py
@@ -1,6 +1,6 @@
from __future__ import annotations
-from collections.abc import Callable, Coroutine, Iterable
+from collections.abc import Callable, Coroutine, Iterable, Mapping
from dataclasses import dataclass
from typing import TYPE_CHECKING, Any, Literal, Union
@@ -229,6 +229,7 @@ class _CompletionDispatchContext:
optional_params: dict
organization: str | None
provider_config: BaseConfig | None
+ request_params: Mapping[str, object]
shared_session: ClientSession | None
stream: bool | None
temperature: float | None
diff --git a/litellm/utils.py b/litellm/utils.py
index 82c8069770d..7226b9a865c 100644
--- a/litellm/utils.py
+++ b/litellm/utils.py
@@ -401,7 +401,7 @@ if TYPE_CHECKING:
BaseVectorStoreFilesConfig,
)
from litellm.llms.base_llm.videos.transformation import BaseVideoConfig
- from litellm.llms.bedrock.common_utils import BedrockModelInfo, BedrockRoute
+ from litellm.llms.bedrock.common_utils import BedrockRoute
from litellm.llms.bedrock.embed.amazon_nova_transformation import (
AmazonNovaEmbeddingConfig,
)
@@ -3353,12 +3353,9 @@ def _should_drop_param(k, additional_drop_params) -> bool:
def _bedrock_route_for_request(
model: str, passed_params: Mapping[str, object], additional_drop_params: list | None
) -> BedrockRoute:
- from litellm.llms.bedrock.common_utils import BedrockModelInfo
+ from litellm.llms.bedrock.common_utils import bedrock_route_for_request
- return BedrockModelInfo.get_bedrock_route(
- model,
- {k: v for k, v in passed_params.items() if not _should_drop_param(k, additional_drop_params)},
- )
+ return bedrock_route_for_request(model, passed_params, additional_drop_params)
def _get_non_default_params(passed_params: dict, default_params: dict, additional_drop_params: list | None) -> dict:
diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json
index 05658ce2d15..6dbf246fd70 100644
--- a/model_prices_and_context_window.json
+++ b/model_prices_and_context_window.json
@@ -40870,7 +40870,8 @@
"output_cost_per_token": 0.0
},
"openai.gpt-oss-120b-1:0": {
- "use_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 1.5e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@@ -40884,7 +40885,8 @@
"supports_tool_choice": true
},
"openai.gpt-oss-20b-1:0": {
- "use_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@@ -47234,7 +47236,9 @@
"input_cost_per_token": 2.64e-06,
"output_cost_per_token": 7.92e-06,
"cache_read_input_token_cost": 6.6e-07,
- "use_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@@ -58443,8 +58447,8 @@
"supports_web_search": true
},
"us.openai.gpt-5.6-sol": {
- "use_bedrock_runtime_chat_completions": true,
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4.4e-06,
"input_cost_per_token_above_272k_tokens": 8.8e-06,
"cache_creation_input_token_cost": 5.5e-06,
@@ -58474,8 +58478,8 @@
"supports_vision": true
},
"global.openai.gpt-5.6-sol": {
- "use_bedrock_runtime_chat_completions": true,
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4e-06,
"input_cost_per_token_above_272k_tokens": 8e-06,
"cache_creation_input_token_cost": 5e-06,
@@ -58505,8 +58509,8 @@
"supports_vision": true
},
"us.openai.gpt-5.6-terra": {
- "use_bedrock_runtime_chat_completions": true,
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_above_272k_tokens": 4.4e-06,
"cache_creation_input_token_cost": 2.75e-06,
@@ -58536,8 +58540,8 @@
"supports_vision": true
},
"global.openai.gpt-5.6-terra": {
- "use_bedrock_runtime_chat_completions": true,
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-06,
"input_cost_per_token_above_272k_tokens": 4e-06,
"cache_creation_input_token_cost": 2.5e-06,
@@ -58567,8 +58571,8 @@
"supports_vision": true
},
"us.openai.gpt-5.6-luna": {
- "use_bedrock_runtime_chat_completions": true,
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-07,
"input_cost_per_token_above_272k_tokens": 4.4e-07,
"cache_creation_input_token_cost": 2.75e-07,
@@ -58598,8 +58602,8 @@
"supports_vision": true
},
"global.openai.gpt-5.6-luna": {
- "use_bedrock_runtime_chat_completions": true,
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_272k_tokens": 4e-07,
"cache_creation_input_token_cost": 2.5e-07,
@@ -58909,7 +58913,9 @@
"input_cost_per_token": 2.2e-06,
"output_cost_per_token": 6.6e-06,
"cache_read_input_token_cost": 5.5e-07,
- "use_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@@ -58925,7 +58931,9 @@
"input_cost_per_token": 2e-06,
"output_cost_per_token": 6e-06,
"cache_read_input_token_cost": 5e-07,
- "use_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
+ "supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json
index 9db1c8eedc4..1a3523cdff2 100644
--- a/model_prices_and_context_window.schema.json
+++ b/model_prices_and_context_window.schema.json
@@ -74,9 +74,6 @@
"xhigh"
]
},
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": {
- "type": "boolean"
- },
"cache_creation_input_audio_token_cost": {
"type": "number",
"minimum": 0
@@ -830,6 +827,15 @@
"supports_audio_output": {
"type": "boolean"
},
+ "supports_bedrock_runtime_chat_completions": {
+ "type": "boolean"
+ },
+ "supports_bedrock_runtime_chat_completions_response_format": {
+ "type": "boolean"
+ },
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": {
+ "type": "boolean"
+ },
"supports_computer_use": {
"type": "boolean"
},
@@ -1000,9 +1006,6 @@
"minimum": 0,
"description": "Provider default tokens-per-minute limit."
},
- "use_bedrock_runtime_chat_completions": {
- "type": "boolean"
- },
"use_openai_responses_path": {
"type": "boolean"
},
diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
index 230e323ace3..ce2e5b7e8f7 100644
--- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
+++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -4,6 +4,7 @@ import json
import httpx
import pytest
+from pydantic import BaseModel
import litellm
from litellm.llms.bedrock.chat.chat_completions.transformation import (
@@ -17,6 +18,7 @@ from litellm.llms.bedrock.common_utils import (
BEDROCK_CONVERSE_ONLY_REQUEST_KEYS,
BedrockModelInfo,
bedrock_request_needs_converse,
+ bedrock_route_for_request,
get_bedrock_chat_config,
uses_bedrock_runtime_chat_completions,
)
@@ -471,7 +473,7 @@ def test_converse_only_request_keys_go_to_converse(local_cost_map, fake_aws_env,
)
assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse")
- (key, value), = converse_only_param.items()
+ ((key, value),) = converse_only_param.items()
assert json.loads(requests[0].content)[key] == value
@@ -620,3 +622,124 @@ def test_streaming_handler_keeps_native_reasoning_next_to_the_tagged_split():
assert parsed.choices[0].delta.reasoning_content == "native tagged"
assert parsed.choices[0].delta.content == "Hi"
+
+
+RESPONSE_FORMAT_JSON_SCHEMA = {
+ "type": "json_schema",
+ "json_schema": {
+ "name": "answer",
+ "schema": {"type": "object", "properties": {"word": {"type": "string"}}, "required": ["word"]},
+ "strict": True,
+ },
+}
+
+
+class Answer(BaseModel):
+ word: str
+
+
+@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "bedrock/openai.gpt-oss-120b-1:0"])
+@pytest.mark.parametrize(
+ "response_format, expected_route",
+ [
+ (RESPONSE_FORMAT_JSON_SCHEMA, "converse"),
+ ({"type": "json_object"}, "converse"),
+ (Answer, "converse"),
+ ({"type": "text"}, "chat_completions"),
+ (None, "chat_completions"),
+ ],
+ ids=["json_schema", "json_object", "pydantic", "text", "none"],
+)
+def test_gpt_oss_response_format_falls_back_to_converse(local_cost_map, model, response_format, expected_route):
+ params = {"response_format": response_format}
+ assert bedrock_request_needs_converse(model, params) is (expected_route == "converse")
+ assert BedrockModelInfo.get_bedrock_route(model, params) == expected_route
+
+
+@pytest.mark.parametrize("model", ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"])
+def test_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model):
+ params = {"response_format": RESPONSE_FORMAT_JSON_SCHEMA}
+ assert bedrock_request_needs_converse(model, params) is False
+ assert BedrockModelInfo.get_bedrock_route(model, params) == "chat_completions"
+
+
+SYNTHETIC_NATIVE_MODEL = "vendor.native-model-v1:0"
+
+
+@pytest.mark.parametrize(
+ "capability_flags, request_params, needs_converse",
+ [
+ ({}, {"tools": [GET_WEATHER_TOOL], "reasoning_effort": "low"}, True),
+ ({}, {"tools": [GET_WEATHER_TOOL]}, True),
+ ({}, {"tools": [GET_WEATHER_TOOL], "reasoning_effort": "none"}, False),
+ (
+ {"supports_bedrock_runtime_chat_completions_tools_with_reasoning": True},
+ {"tools": [GET_WEATHER_TOOL], "reasoning_effort": "low"},
+ False,
+ ),
+ ({}, {"response_format": RESPONSE_FORMAT_JSON_SCHEMA}, True),
+ (
+ {"supports_bedrock_runtime_chat_completions_response_format": True},
+ {"response_format": RESPONSE_FORMAT_JSON_SCHEMA},
+ False,
+ ),
+ (
+ {"supports_bedrock_runtime_chat_completions_response_format": True},
+ {"response_format": RESPONSE_FORMAT_JSON_SCHEMA, "tools": [GET_WEATHER_TOOL], "reasoning_effort": "low"},
+ True,
+ ),
+ ],
+)
+def test_capability_flags_are_read_from_the_cost_map(monkeypatch, capability_flags, request_params, needs_converse):
+ entry = {
+ "litellm_provider": "bedrock_converse",
+ "supports_bedrock_runtime_chat_completions": True,
+ **capability_flags,
+ }
+ monkeypatch.setattr(litellm, "model_cost", {SYNTHETIC_NATIVE_MODEL: entry})
+ assert bedrock_request_needs_converse(SYNTHETIC_NATIVE_MODEL, request_params) is needs_converse
+ route = bedrock_route_for_request(SYNTHETIC_NATIVE_MODEL, request_params, None)
+ assert (route == "chat_completions") is (not needs_converse)
+
+
+def test_route_for_request_ignores_dropped_params(local_cost_map):
+ params = {"response_format": RESPONSE_FORMAT_JSON_SCHEMA, "guardrailConfig": {"guardrailIdentifier": "gr-1"}}
+ assert bedrock_route_for_request("openai.gpt-oss-20b-1:0", params, None) == "converse"
+ assert bedrock_route_for_request("openai.gpt-oss-20b-1:0", params, ["guardrailConfig"]) == "converse"
+ assert (
+ bedrock_route_for_request("openai.gpt-oss-20b-1:0", params, ["guardrailConfig", "response_format"])
+ == "chat_completions"
+ )
+
+
+def test_gpt_oss_response_format_goes_to_converse_with_json_tool_call(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=CONVERSE_JSON)
+ litellm.completion(
+ model="bedrock/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "Reply with the single word pong."}],
+ response_format=RESPONSE_FORMAT_JSON_SCHEMA,
+ max_tokens=64,
+ client=client,
+ )
+
+ assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse")
+ body = json.loads(requests[0].content)
+ assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "json_tool_call"
+ assert body["toolConfig"]["toolChoice"] == {"tool": {"name": "json_tool_call"}}
+ assert body["inferenceConfig"]["maxTokens"] == 64
+ assert "response_format" not in body
+ assert "max_completion_tokens" not in body
+
+
+def test_gpt56_response_format_is_sent_as_is_on_chat_completions(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=_chat_completion_json('{"word": "pong"}', "global.openai.gpt-5.6-sol"))
+ response = litellm.completion(
+ model="bedrock/global.openai.gpt-5.6-sol",
+ messages=[{"role": "user", "content": "Reply with the single word pong."}],
+ response_format=RESPONSE_FORMAT_JSON_SCHEMA,
+ client=client,
+ )
+
+ assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
+ assert json.loads(requests[0].content)["response_format"] == RESPONSE_FORMAT_JSON_SCHEMA
+ assert response.choices[0].message.content == '{"word": "pong"}'
diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py
index 963bef1114a..d049d83c6a9 100644
--- a/tests/test_litellm/test_utils.py
+++ b/tests/test_litellm/test_utils.py
@@ -878,8 +878,9 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"supports_video_input": {"type": "boolean"},
"supports_vision": {"type": "boolean"},
"supports_web_search": {"type": "boolean"},
- "use_bedrock_runtime_chat_completions": {"type": "boolean"},
- "bedrock_runtime_chat_completions_tools_require_reasoning_none": {"type": "boolean"},
+ "supports_bedrock_runtime_chat_completions": {"type": "boolean"},
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": {"type": "boolean"},
+ "supports_bedrock_runtime_chat_completions_response_format": {"type": "boolean"},
"supports_url_context": {"type": "boolean"},
"supports_multimodal": {"type": "boolean"},
"uses_embed_content": {"type": "boolean"},
diff --git a/tests/test_litellm/types/test_completion.py b/tests/test_litellm/types/test_completion.py
index cd51913c5dd..482dd351ad1 100644
--- a/tests/test_litellm/types/test_completion.py
+++ b/tests/test_litellm/types/test_completion.py
@@ -181,6 +181,7 @@ def _build_dispatch_context() -> _CompletionDispatchContext:
optional_params={},
organization=None,
provider_config=None,
+ request_params={},
shared_session=None,
stream=None,
temperature=None,
From b90c2f113d1091896bb80988c790898320678f00 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Sat, 19 Sep 2026 20:39:18 -0700
Subject: [PATCH 04/18] fix(bedrock): serve region-path and GovCloud gpt-oss
ids on native Chat Completions
The cost-map parity tests require every regional variant of a flagged id to carry the same supports_ flags, so the six us-gov gpt-oss entries now carry the native-route flags too. A region path in the model name (bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0) is routing, not a different model: the route is looked up on the id after the path, the path's region picks the endpoint and the SigV4 scope, an explicit aws_region_name still wins, and the body carries the bare id AWS expects
---
.../chat/chat_completions/transformation.py | 18 ++++++---
litellm/llms/bedrock/common_utils.py | 18 ++++++++-
...odel_prices_and_context_window_backup.json | 12 ++++++
model_prices_and_context_window.json | 12 ++++++
...bedrock_chat_completions_transformation.py | 38 ++++++++++++++++++-
5 files changed, 91 insertions(+), 7 deletions(-)
diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py
index 18ba06fee0b..a7926fb252e 100644
--- a/litellm/llms/bedrock/chat/chat_completions/transformation.py
+++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py
@@ -24,7 +24,7 @@ from typing_extensions import assert_never
import litellm
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
-from litellm.llms.bedrock.common_utils import BedrockError, strip_bedrock_routing_prefix
+from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path
from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler
from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig
from litellm.types.llms.openai import AllMessageValues
@@ -196,7 +196,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
if api_base is not None and "chat/completions" in api_base:
return api_base.rstrip("/")
aws_region_name: Final = self._aws_signer._get_aws_region_name( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public region resolver
- optional_params=optional_params, model=model
+ optional_params=self._params_with_region_from_path(optional_params, model), model=model
)
endpoint_url, _ = self._aws_signer.get_runtime_endpoint(
api_base=api_base,
@@ -210,6 +210,14 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
return f"{base}/chat/completions"
return f"{base}/openai/v1/chat/completions"
+ def _params_with_region_from_path(
+ self, optional_params: dict, model: str | None
+ ) -> dict: # mutable-ok: BaseAWSLLM's region resolver and signer take a plain dict
+ region_from_path, _ = split_bedrock_region_path(model or "")
+ if region_from_path is None or optional_params.get("aws_region_name") is not None:
+ return optional_params
+ return {**optional_params, "aws_region_name": region_from_path} # mutable-ok: BaseAWSLLM takes a plain dict
+
def sign_request(
self,
headers: dict, # mutable-ok: BaseConfig signature
@@ -224,7 +232,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
return self._aws_signer._sign_request( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public signer
service_name="bedrock",
headers=headers,
- optional_params=optional_params,
+ optional_params=self._params_with_region_from_path(optional_params, model),
request_data=request_data,
api_base=api_base,
api_key=api_key,
@@ -268,7 +276,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
headers: dict, # mutable-ok: BaseConfig signature
) -> dict: # mutable-ok: BaseConfig signature
return super().transform_request(
- model=strip_bedrock_routing_prefix(model),
+ model=split_bedrock_region_path(model)[1],
messages=messages,
optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
@@ -284,7 +292,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
headers: dict, # mutable-ok: BaseConfig signature
) -> dict: # mutable-ok: BaseConfig signature
return await super().async_transform_request(
- model=strip_bedrock_routing_prefix(model),
+ model=split_bedrock_region_path(model)[1],
messages=messages,
optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py
index df30eddd856..3b624b4ab02 100644
--- a/litellm/llms/bedrock/common_utils.py
+++ b/litellm/llms/bedrock/common_utils.py
@@ -795,8 +795,24 @@ def strip_bedrock_routing_prefix(model: str) -> str:
return model
+def split_bedrock_region_path(model: str) -> tuple[str | None, str]:
+ """Split a ``/`` routing path into the region and the id AWS receives.
+
+ ``bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0`` -> ``("us-gov-west-1", "openai.gpt-oss-20b-1:0")``;
+ a model without a region path comes back as ``(None, )``.
+ """
+ stripped: Final = strip_bedrock_routing_prefix(model)
+ region, separator, model_id = stripped.partition("/")
+ if separator and region in _get_all_bedrock_regions():
+ return region, model_id
+ return None, stripped
+
+
def _bedrock_price_map_flag(model: str, flag: str) -> bool:
- entries: Final = (litellm.model_cost.get(key) for key in (model, strip_bedrock_routing_prefix(model)))
+ entries: Final = (
+ litellm.model_cost.get(key)
+ for key in (model, strip_bedrock_routing_prefix(model), split_bedrock_region_path(model)[1])
+ )
return any(entry is not None and entry.get(flag) is True for entry in entries)
diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json
index 6dbf246fd70..db0e3a4dc18 100644
--- a/litellm/model_prices_and_context_window_backup.json
+++ b/litellm/model_prices_and_context_window_backup.json
@@ -47214,6 +47214,8 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@@ -47227,6 +47229,8 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@@ -64459,6 +64463,8 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@@ -64472,6 +64478,8 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@@ -64665,6 +64673,8 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@@ -64678,6 +64688,8 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json
index 6dbf246fd70..db0e3a4dc18 100644
--- a/model_prices_and_context_window.json
+++ b/model_prices_and_context_window.json
@@ -47214,6 +47214,8 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@@ -47227,6 +47229,8 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@@ -64459,6 +64463,8 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@@ -64472,6 +64478,8 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@@ -64665,6 +64673,8 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@@ -64678,6 +64688,8 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
+ "supports_bedrock_runtime_chat_completions": true,
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
index ce2e5b7e8f7..83fe7628e0a 100644
--- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
+++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -163,6 +163,33 @@ def test_completion_posts_runtime_chat_completions(local_cost_map, fake_aws_env)
assert "inferenceConfig" not in body
+def test_region_path_sends_the_bare_model_id_to_the_path_region(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0"))
+ litellm.completion(
+ model="bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ client=client,
+ )
+
+ assert str(requests[0].url) == "https://bedrock-runtime.us-gov-west-1.amazonaws.com/openai/v1/chat/completions"
+ assert json.loads(requests[0].content)["model"] == "openai.gpt-oss-20b-1:0"
+ assert "/us-gov-west-1/bedrock/aws4_request" in requests[0].headers["Authorization"]
+
+
+def test_explicit_aws_region_name_wins_over_the_region_path(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0"))
+ litellm.completion(
+ model="bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ aws_region_name="us-gov-east-1",
+ client=client,
+ )
+
+ assert str(requests[0].url) == "https://bedrock-runtime.us-gov-east-1.amazonaws.com/openai/v1/chat/completions"
+ assert json.loads(requests[0].content)["model"] == "openai.gpt-oss-20b-1:0"
+ assert "/us-gov-east-1/bedrock/aws4_request" in requests[0].headers["Authorization"]
+
+
OPENAI_RUNTIME_MODELS = (
"openai.gpt-oss-20b-1:0",
"openai.gpt-oss-120b-1:0",
@@ -182,7 +209,16 @@ GET_WEATHER_TOOL = {
}
-@pytest.mark.parametrize("model", [*OPENAI_RUNTIME_MODELS, "bedrock/openai.gpt-oss-20b-1:0"])
+@pytest.mark.parametrize(
+ "model",
+ [
+ *OPENAI_RUNTIME_MODELS,
+ "bedrock/openai.gpt-oss-20b-1:0",
+ "us-gov.openai.gpt-oss-20b-1:0",
+ "bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0",
+ "us-gov-east-1/openai.gpt-oss-120b-1:0",
+ ],
+)
def test_openai_runtime_models_use_chat_completions_route(local_cost_map, model):
assert uses_bedrock_runtime_chat_completions(model) is True
assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions"
From 22b217bd95bb409d480d6da5a4292578e0899c7f Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Mon, 21 Sep 2026 12:53:43 -0700
Subject: [PATCH 05/18] fix(bedrock): keep params AWS refuses natively off the
chat completions route
Drop the params each family 400s or 503s on runtime Chat Completions from the native config's supported list (GPT-5.6 penalties, stop, and logprobs, Grok penalties, gpt-oss logit_bias) so drop_params drops them as Converse did, gate legacy functions on GPT-5.6 the same way as tools, and send an Anthropic-style thinking block to Converse, the only route that forwards it
---
.../chat/chat_completions/transformation.py | 24 +++-
litellm/llms/bedrock/common_utils.py | 15 +--
...bedrock_chat_completions_transformation.py | 107 ++++++++++++++++++
3 files changed, 138 insertions(+), 8 deletions(-)
diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py
index a7926fb252e..c6bf15b893f 100644
--- a/litellm/llms/bedrock/chat/chat_completions/transformation.py
+++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py
@@ -38,6 +38,27 @@ if TYPE_CHECKING:
REASONING_OPEN_TAG: Final = ""
REASONING_CLOSE_TAG: Final = ""
+CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType(
+ {
+ "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs")),
+ "openai.gpt-oss": frozenset(("logit_bias",)),
+ "xai.": frozenset(("frequency_penalty", "presence_penalty")),
+ }
+)
+
+
+def chat_completions_params_refused_for(model: str) -> frozenset[str]:
+ """The OpenAI params AWS's Chat Completions endpoint rejects for this model whatever else the request says.
+
+ Each family answers them with a 400 (GPT-5.6, gpt-oss) or a 503 (Grok), where Converse dropped the same
+ params under ``drop_params``, so the native config leaves them out of its supported list and the usual
+ drop-or-raise handling applies before the request reaches AWS.
+ """
+ model_id: Final = split_bedrock_region_path(model)[1]
+ return frozenset().union(
+ *(refused for family, refused in CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY.items() if family in model_id)
+ )
+
def _held_close_tag_prefix(text: str) -> int:
return next(
@@ -362,7 +383,8 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
return {**validated, "OpenAI-Project": project_id} # mutable-ok: BaseConfig signature returns a dict
def get_supported_openai_params(self, model: str) -> list: # mutable-ok: BaseConfig signature
- base_params: Final = [param for param in super().get_supported_openai_params(model) if param != "n"]
+ refused: Final = {"n", *chat_completions_params_refused_for(model)}
+ base_params: Final = [param for param in super().get_supported_openai_params(model) if param not in refused]
if "reasoning_effort" in base_params or not litellm.supports_reasoning(
model=model, custom_llm_provider=self.custom_llm_provider
):
diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py
index 3b624b4ab02..8798b05cfa4 100644
--- a/litellm/llms/bedrock/common_utils.py
+++ b/litellm/llms/bedrock/common_utils.py
@@ -849,7 +849,7 @@ def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> boo
BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
- ("guardrailConfig", "performanceConfig", "serviceTier", "requestMetadata", "outputConfig")
+ ("guardrailConfig", "performanceConfig", "serviceTier", "requestMetadata", "outputConfig", "thinking")
)
@@ -862,12 +862,13 @@ def _response_format_constrains_output(response_format: object) -> bool:
def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool:
"""Whether a request on a runtime-Chat-Completions model must still be served by Converse.
- Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``) are rejected as malformed input by
+ Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking``
+ block included, which only Converse forwards as ``additionalModelRequestFields``) have no field on
AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body,
- function tools on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning``
- are rejected there unless ``reasoning_effort`` is exactly ``"none"``, and a constraining
- ``response_format`` on a model without ``supports_bedrock_runtime_chat_completions_response_format``
- is only honored by Converse.
+ function tools (``tools`` or legacy ``functions``) on a model without
+ ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless
+ ``reasoning_effort`` is exactly ``"none"``, and a constraining ``response_format`` on a model without
+ ``supports_bedrock_runtime_chat_completions_response_format`` is only honored by Converse.
"""
if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS):
return True
@@ -877,7 +878,7 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje
request_params.get("response_format")
) and not bedrock_runtime_chat_completions_enforces_response_format(model):
return True
- if not request_params.get("tools"):
+ if not (request_params.get("tools") or request_params.get("functions")):
return False
return (
not bedrock_runtime_chat_completions_serves_tools_with_reasoning(model)
diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
index 83fe7628e0a..2852ed3ae84 100644
--- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
+++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -264,6 +264,26 @@ def test_gpt_oss_tools_with_any_reasoning_effort_stay_on_chat_completions(local_
assert BedrockModelInfo.get_bedrock_route("openai.gpt-oss-120b-1:0", params) == "chat_completions"
+@pytest.mark.parametrize(
+ "request_params, expected_route",
+ [
+ ({"functions": [GET_WEATHER_TOOL["function"]]}, "converse"),
+ ({"functions": [GET_WEATHER_TOOL["function"]], "reasoning_effort": "low"}, "converse"),
+ ({"functions": [GET_WEATHER_TOOL["function"]], "reasoning_effort": "none"}, "chat_completions"),
+ ({"functions": [], "reasoning_effort": "low"}, "chat_completions"),
+ ],
+)
+def test_gpt56_legacy_functions_route_like_tools(local_cost_map, request_params, expected_route):
+ assert BedrockModelInfo.get_bedrock_route("global.openai.gpt-5.6-sol", request_params) == expected_route
+ assert BedrockModelInfo.get_bedrock_route("openai.gpt-oss-120b-1:0", request_params) == "chat_completions"
+
+
+def test_thinking_block_goes_to_converse(local_cost_map):
+ thinking = {"type": "enabled", "budget_tokens": 1024}
+ assert BedrockModelInfo.get_bedrock_route("us.xai.grok-4.6", {"thinking": thinking}) == "converse"
+ assert BedrockModelInfo.get_bedrock_route("us.xai.grok-4.6", {"thinking": None}) == "chat_completions"
+
+
def test_explicit_converse_prefix_wins_for_openai_models(local_cost_map):
assert BedrockModelInfo.get_bedrock_route("bedrock/converse/openai.gpt-oss-20b-1:0") == "converse"
assert BedrockModelInfo.get_bedrock_route("converse/global.openai.gpt-5.6-sol", {}) == "converse"
@@ -301,6 +321,54 @@ def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map):
assert "reasoning_effort" in cfg.get_supported_openai_params("openai.gpt-oss-20b-1:0")
+@pytest.mark.parametrize(
+ "model, refused, kept",
+ [
+ (
+ "bedrock/global.openai.gpt-5.6-sol",
+ ("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs", "n"),
+ ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions"),
+ ),
+ (
+ "us.xai.grok-4.6",
+ ("frequency_penalty", "presence_penalty", "n"),
+ ("stop", "logprobs", "top_p", "logit_bias", "reasoning_effort"),
+ ),
+ (
+ "bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0",
+ ("logit_bias", "n"),
+ ("frequency_penalty", "presence_penalty", "stop", "logprobs", "reasoning_effort"),
+ ),
+ ],
+)
+def test_supported_params_leave_out_what_each_family_refuses(local_cost_map, model, refused, kept):
+ supported = set(AmazonBedrockRuntimeChatCompletionsConfig().get_supported_openai_params(model))
+ assert supported.isdisjoint(refused)
+ assert set(kept) <= supported
+
+
+@pytest.mark.parametrize(
+ "model, param",
+ [
+ ("bedrock/global.openai.gpt-5.6-sol", {"frequency_penalty": 0.5}),
+ ("bedrock/global.openai.gpt-5.6-sol", {"logprobs": True, "top_logprobs": 2}),
+ ("bedrock/us.xai.grok-4.6", {"presence_penalty": 0.5}),
+ ("bedrock/openai.gpt-oss-20b-1:0", {"logit_bias": {"1": 1}}),
+ ],
+ ids=lambda value: value if isinstance(value, str) else next(iter(value)),
+)
+def test_refused_params_are_dropped_or_refused_before_reaching_aws(local_cost_map, fake_aws_env, model, param):
+ requests, client = _recording_client(json=_chat_completion_json("ok", model.removeprefix("bedrock/")))
+ with pytest.raises(litellm.UnsupportedParamsError, match=next(iter(param))):
+ litellm.completion(model=model, messages=[{"role": "user", "content": "hello"}], client=client, **param)
+ litellm.completion(
+ model=model, messages=[{"role": "user", "content": "hello"}], drop_params=True, client=client, **param
+ )
+
+ assert str(requests[0].url).endswith("/openai/v1/chat/completions")
+ assert param.keys().isdisjoint(json.loads(requests[0].content))
+
+
def test_split_reasoning_tag_splits_leading_tag():
assert split_reasoning_tag("plan it\n\n\nHello") == ("plan it\n", "Hello")
@@ -579,6 +647,45 @@ def test_legacy_functions_stay_on_chat_completions(local_cost_map, fake_aws_env)
assert json.loads(requests[0].content)["functions"] == [GET_WEATHER_TOOL["function"]]
+def test_gpt56_legacy_functions_with_reasoning_fall_back_to_converse(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=CONVERSE_JSON)
+ with pytest.raises(litellm.UnsupportedParamsError, match="functions"):
+ litellm.completion(
+ model="bedrock/global.openai.gpt-5.6-sol",
+ messages=[{"role": "user", "content": "hello"}],
+ functions=[GET_WEATHER_TOOL["function"]],
+ reasoning_effort="low",
+ client=client,
+ )
+ litellm.completion(
+ model="bedrock/global.openai.gpt-5.6-sol",
+ messages=[{"role": "user", "content": "hello"}],
+ functions=[GET_WEATHER_TOOL["function"]],
+ reasoning_effort="low",
+ drop_params=True,
+ client=client,
+ )
+
+ assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse")
+ body = json.loads(requests[0].content)
+ assert "functions" not in body
+ assert "toolConfig" not in body
+
+
+def test_grok_thinking_block_is_served_by_converse(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=CONVERSE_JSON)
+ thinking = {"type": "enabled", "budget_tokens": 1024}
+ litellm.completion(
+ model="bedrock/us.xai.grok-4.6",
+ messages=[{"role": "user", "content": "hello"}],
+ thinking=thinking,
+ client=client,
+ )
+
+ assert requests[0].url.raw_path.endswith(b"/model/us.xai.grok-4.6/converse")
+ assert json.loads(requests[0].content)["additionalModelRequestFields"]["thinking"] == thinking
+
+
def test_converse_fallback_validates_against_converse_params(local_cost_map, fake_aws_env):
requests, client = _recording_client(json=CONVERSE_JSON)
guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"}
From 0a85e2799814b4582d110f93c0b9c87ed3f66d8b Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Mon, 21 Sep 2026 13:57:27 -0700
Subject: [PATCH 06/18] fix(bedrock): keep schema-less json_object on Converse
for the chat completions models
---
litellm/llms/bedrock/common_utils.py | 20 +++++----
...bedrock_chat_completions_transformation.py | 42 +++++++++++++++++--
2 files changed, 52 insertions(+), 10 deletions(-)
diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py
index 8798b05cfa4..2fe1af9edf4 100644
--- a/litellm/llms/bedrock/common_utils.py
+++ b/litellm/llms/bedrock/common_utils.py
@@ -853,10 +853,15 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
)
-def _response_format_constrains_output(response_format: object) -> bool:
+def _response_format_needs_converse(model: str, response_format: object) -> bool:
if response_format is None:
return False
- return not (isinstance(response_format, Mapping) and response_format.get("type") == "text")
+ if not isinstance(response_format, Mapping):
+ return not bedrock_runtime_chat_completions_enforces_response_format(model)
+ if response_format.get("type") == "text":
+ return False
+ carries_schema: Final = "json_schema" in response_format or "response_schema" in response_format
+ return not (carries_schema and bedrock_runtime_chat_completions_enforces_response_format(model))
def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool:
@@ -867,16 +872,17 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje
AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body,
function tools (``tools`` or legacy ``functions``) on a model without
``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless
- ``reasoning_effort`` is exactly ``"none"``, and a constraining ``response_format`` on a model without
- ``supports_bedrock_runtime_chat_completions_response_format`` is only honored by Converse.
+ ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as a JSON schema
+ (a ``json_schema`` or ``response_schema`` mapping, or a pydantic model) on a model with
+ ``supports_bedrock_runtime_chat_completions_response_format``: a schema on any other model is only
+ honored by Converse, and a schema-less ``json_object`` keeps Converse's handling everywhere, since AWS's
+ native surface rejects it with a 400 unless the prompt mentions json.
"""
if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS):
return True
if bedrock_request_metadata_is_owned():
return True
- if _response_format_constrains_output(
- request_params.get("response_format")
- ) and not bedrock_runtime_chat_completions_enforces_response_format(model):
+ if _response_format_needs_converse(model, request_params.get("response_format")):
return True
if not (request_params.get("tools") or request_params.get("functions")):
return False
diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
index 2852ed3ae84..e49a2f7f252 100644
--- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
+++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -799,13 +799,32 @@ def test_gpt_oss_response_format_falls_back_to_converse(local_cost_map, model, r
assert BedrockModelInfo.get_bedrock_route(model, params) == expected_route
-@pytest.mark.parametrize("model", ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"])
-def test_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model):
- params = {"response_format": RESPONSE_FORMAT_JSON_SCHEMA}
+RESPONSE_FORMAT_ENFORCING_MODELS = ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"]
+
+
+@pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS)
+@pytest.mark.parametrize(
+ "response_format",
+ [
+ RESPONSE_FORMAT_JSON_SCHEMA,
+ {"type": "json_object", "response_schema": RESPONSE_FORMAT_JSON_SCHEMA["json_schema"]["schema"]},
+ Answer,
+ ],
+ ids=["json_schema", "response_schema", "pydantic"],
+)
+def test_schema_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model, response_format):
+ params = {"response_format": response_format}
assert bedrock_request_needs_converse(model, params) is False
assert BedrockModelInfo.get_bedrock_route(model, params) == "chat_completions"
+@pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS)
+def test_schema_less_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_cost_map, model):
+ params = {"response_format": {"type": "json_object"}}
+ assert bedrock_request_needs_converse(model, params) is True
+ assert BedrockModelInfo.get_bedrock_route(model, params) == "converse"
+
+
SYNTHETIC_NATIVE_MODEL = "vendor.native-model-v1:0"
@@ -886,3 +905,20 @@ def test_gpt56_response_format_is_sent_as_is_on_chat_completions(local_cost_map,
assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
assert json.loads(requests[0].content)["response_format"] == RESPONSE_FORMAT_JSON_SCHEMA
assert response.choices[0].message.content == '{"word": "pong"}'
+
+
+def test_gpt56_schema_less_json_object_goes_to_converse_without_a_schema_tool(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=CONVERSE_JSON)
+ litellm.completion(
+ model="bedrock/global.openai.gpt-5.6-sol",
+ messages=[{"role": "user", "content": "Reply with the single word pong."}],
+ response_format={"type": "json_object"},
+ max_tokens=64,
+ client=client,
+ )
+
+ assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse")
+ body = json.loads(requests[0].content)
+ assert "toolConfig" not in body
+ assert "response_format" not in body
+ assert body["inferenceConfig"]["maxTokens"] == 64
From 0139dd08a635843deaac24256b53439531f3b9a7 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Mon, 21 Sep 2026 14:10:27 -0700
Subject: [PATCH 07/18] fix(bedrock): keep every json_object response_format on
Converse for the chat completions models
---
litellm/llms/bedrock/common_utils.py | 16 ++++---
...bedrock_chat_completions_transformation.py | 46 ++++++++++++++-----
2 files changed, 43 insertions(+), 19 deletions(-)
diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py
index 2fe1af9edf4..486b3b2bd51 100644
--- a/litellm/llms/bedrock/common_utils.py
+++ b/litellm/llms/bedrock/common_utils.py
@@ -858,10 +858,11 @@ def _response_format_needs_converse(model: str, response_format: object) -> bool
return False
if not isinstance(response_format, Mapping):
return not bedrock_runtime_chat_completions_enforces_response_format(model)
- if response_format.get("type") == "text":
+ response_format_type: Final = response_format.get("type")
+ if response_format_type == "text":
return False
- carries_schema: Final = "json_schema" in response_format or "response_schema" in response_format
- return not (carries_schema and bedrock_runtime_chat_completions_enforces_response_format(model))
+ is_json_schema: Final = response_format_type == "json_schema" and "json_schema" in response_format
+ return not (is_json_schema and bedrock_runtime_chat_completions_enforces_response_format(model))
def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool:
@@ -872,11 +873,12 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje
AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body,
function tools (``tools`` or legacy ``functions``) on a model without
``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless
- ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as a JSON schema
- (a ``json_schema`` or ``response_schema`` mapping, or a pydantic model) on a model with
+ ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as
+ ``{"type": "json_schema", "json_schema": ...}`` (a pydantic model is converted to that) on a model with
``supports_bedrock_runtime_chat_completions_response_format``: a schema on any other model is only
- honored by Converse, and a schema-less ``json_object`` keeps Converse's handling everywhere, since AWS's
- native surface rejects it with a 400 unless the prompt mentions json.
+ honored by Converse, and every ``json_object`` form (``response_schema`` included) keeps Converse's
+ handling everywhere, since AWS's native surface rejects that type with a 400 unless the prompt
+ mentions json.
"""
if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS):
return True
diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
index e49a2f7f252..2f0d384fcdd 100644
--- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
+++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -802,25 +802,30 @@ def test_gpt_oss_response_format_falls_back_to_converse(local_cost_map, model, r
RESPONSE_FORMAT_ENFORCING_MODELS = ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"]
+JSON_OBJECT_WITH_RESPONSE_SCHEMA = {
+ "type": "json_object",
+ "response_schema": RESPONSE_FORMAT_JSON_SCHEMA["json_schema"]["schema"],
+}
+
+
@pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS)
-@pytest.mark.parametrize(
- "response_format",
- [
- RESPONSE_FORMAT_JSON_SCHEMA,
- {"type": "json_object", "response_schema": RESPONSE_FORMAT_JSON_SCHEMA["json_schema"]["schema"]},
- Answer,
- ],
- ids=["json_schema", "response_schema", "pydantic"],
-)
-def test_schema_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model, response_format):
+@pytest.mark.parametrize("response_format", [RESPONSE_FORMAT_JSON_SCHEMA, Answer], ids=["json_schema", "pydantic"])
+def test_json_schema_response_format_stays_on_chat_completions_where_aws_enforces_it(
+ local_cost_map, model, response_format
+):
params = {"response_format": response_format}
assert bedrock_request_needs_converse(model, params) is False
assert BedrockModelInfo.get_bedrock_route(model, params) == "chat_completions"
@pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS)
-def test_schema_less_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_cost_map, model):
- params = {"response_format": {"type": "json_object"}}
+@pytest.mark.parametrize(
+ "response_format",
+ [{"type": "json_object"}, JSON_OBJECT_WITH_RESPONSE_SCHEMA],
+ ids=["json_object", "json_object_with_response_schema"],
+)
+def test_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_cost_map, model, response_format):
+ params = {"response_format": response_format}
assert bedrock_request_needs_converse(model, params) is True
assert BedrockModelInfo.get_bedrock_route(model, params) == "converse"
@@ -922,3 +927,20 @@ def test_gpt56_schema_less_json_object_goes_to_converse_without_a_schema_tool(lo
assert "toolConfig" not in body
assert "response_format" not in body
assert body["inferenceConfig"]["maxTokens"] == 64
+
+
+def test_gpt56_json_object_with_response_schema_goes_to_converse_as_a_json_tool(local_cost_map, fake_aws_env):
+ requests, client = _recording_client(json=CONVERSE_JSON)
+ litellm.completion(
+ model="bedrock/global.openai.gpt-5.6-sol",
+ messages=[{"role": "user", "content": "Reply with the single word pong."}],
+ response_format=JSON_OBJECT_WITH_RESPONSE_SCHEMA,
+ max_tokens=64,
+ client=client,
+ )
+
+ assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse")
+ body = json.loads(requests[0].content)
+ assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "json_tool_call"
+ assert body["toolConfig"]["toolChoice"] == {"tool": {"name": "json_tool_call"}}
+ assert "response_format" not in body
From c4c24dd9864866c37b30d6dc973e585e36a428b6 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Thu, 24 Sep 2026 13:31:29 -0700
Subject: [PATCH 08/18] fix(rust): declare the bedrock runtime chat completions
flags on ModelInfo
---
litellm-rust/crates/model-catalog/src/model_info.rs | 6 ++++++
1 file changed, 6 insertions(+)
diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs
index 4a56e1112d1..0b52bea94de 100644
--- a/litellm-rust/crates/model-catalog/src/model_info.rs
+++ b/litellm-rust/crates/model-catalog/src/model_info.rs
@@ -582,6 +582,12 @@ pub struct ModelInfo {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub supports_audio_output: Option,
#[serde(default, skip_serializing_if = "Option::is_none")]
+ pub supports_bedrock_runtime_chat_completions: Option,
+ #[serde(default, skip_serializing_if = "Option::is_none")]
+ pub supports_bedrock_runtime_chat_completions_response_format: Option,
+ #[serde(default, skip_serializing_if = "Option::is_none")]
+ pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option,
+ #[serde(default, skip_serializing_if = "Option::is_none")]
pub supports_computer_use: Option,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub supports_embedding_image_input: Option,
From ff8e15d84460d9970443f0004c6fb07ae45dd01e Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Thu, 24 Sep 2026 15:21:07 -0700
Subject: [PATCH 09/18] fix(bedrock): opt into the native chat completions
route through supported_endpoints
---
.../crates/model-catalog/src/model_info.rs | 2 -
.../chat/chat_completions/transformation.py | 2 +-
litellm/llms/bedrock/common_utils.py | 32 +++++++++--
...odel_prices_and_context_window_backup.json | 56 +++++++++++++------
model_prices_and_context_window.json | 56 +++++++++++++------
model_prices_and_context_window.schema.json | 3 -
...bedrock_chat_completions_transformation.py | 24 +++++++-
tests/test_litellm/test_utils.py | 1 -
8 files changed, 126 insertions(+), 50 deletions(-)
diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs
index 0b52bea94de..23e69f66492 100644
--- a/litellm-rust/crates/model-catalog/src/model_info.rs
+++ b/litellm-rust/crates/model-catalog/src/model_info.rs
@@ -582,8 +582,6 @@ pub struct ModelInfo {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub supports_audio_output: Option,
#[serde(default, skip_serializing_if = "Option::is_none")]
- pub supports_bedrock_runtime_chat_completions: Option,
- #[serde(default, skip_serializing_if = "Option::is_none")]
pub supports_bedrock_runtime_chat_completions_response_format: Option,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option,
diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py
index 11e2467b093..4c5e5768119 100644
--- a/litellm/llms/bedrock/chat/chat_completions/transformation.py
+++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py
@@ -3,7 +3,7 @@ Native OpenAI Chat Completions on Amazon Bedrock Runtime.
AWS serves this surface at
``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions``
-for the models whose price-map entry sets ``supports_bedrock_runtime_chat_completions``
+for the models whose price-map ``supported_endpoints`` lists ``/v1/chat/completions``
(Grok 4.6, gpt-oss, the GPT-5.6 family): chat completions stay chat completions
instead of being rewritten to Converse.
diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py
index 09e62dc393b..4492052b3c4 100644
--- a/litellm/llms/bedrock/common_utils.py
+++ b/litellm/llms/bedrock/common_utils.py
@@ -829,24 +829,44 @@ def split_bedrock_region_path(model: str) -> tuple[str | None, str]:
return None, stripped
-def _bedrock_price_map_flag(model: str, flag: str) -> bool:
- entries: Final = (
+BEDROCK_RUNTIME_PRICE_MAP_PROVIDERS: Final = frozenset(("bedrock", "bedrock_converse"))
+
+
+def _bedrock_price_map_entries(model: str) -> tuple[Mapping[str, object] | None, ...]:
+ return tuple(
litellm.model_cost.get(key)
for key in (model, strip_bedrock_routing_prefix(model), split_bedrock_region_path(model)[1])
)
- return any(entry is not None and entry.get(flag) is True for entry in entries)
+
+
+def _bedrock_price_map_flag(model: str, flag: str) -> bool:
+ return any(entry is not None and entry.get(flag) is True for entry in _bedrock_price_map_entries(model))
+
+
+def _bedrock_runtime_row_lists_chat_completions(entry: Mapping[str, object]) -> bool:
+ endpoints: Final = entry.get("supported_endpoints")
+ return (
+ entry.get("litellm_provider") in BEDROCK_RUNTIME_PRICE_MAP_PROVIDERS
+ and isinstance(endpoints, (list, tuple))
+ and "/v1/chat/completions" in endpoints
+ )
def uses_bedrock_runtime_chat_completions(model: str) -> bool:
"""Whether this Bedrock model should use runtime native Chat Completions.
- Data-driven from the price-map ``supports_bedrock_runtime_chat_completions`` flag
+ Data-driven from ``/v1/chat/completions`` in the price-map row's ``supported_endpoints``,
+ the same per-model signal ``bedrock_supports_openai_responses`` reads for ``/v1/responses``,
so onboarding a model is a JSON change. Explicit ``converse/`` still wins in
``get_bedrock_route`` because prefix routes are checked first, and a request
that needs a Converse-only feature (``bedrock_request_needs_converse``) is
- served by Converse even on a flagged model.
+ served by Converse even on a listed model. Only a bedrock-runtime row counts: a
+ ``bedrock_mantle`` row lists the endpoints of the Mantle host, not this one.
"""
- return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions")
+ return any(
+ entry is not None and _bedrock_runtime_row_lists_chat_completions(entry)
+ for entry in _bedrock_price_map_entries(model)
+ )
def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool:
diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json
index d58dd4d105a..7fdd8c85ddc 100644
--- a/litellm/model_prices_and_context_window_backup.json
+++ b/litellm/model_prices_and_context_window_backup.json
@@ -40834,7 +40834,9 @@
"output_cost_per_token": 0.0
},
"openai.gpt-oss-120b-1:0": {
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 1.5e-07,
"litellm_provider": "bedrock_converse",
@@ -40850,7 +40852,9 @@
"supports_tool_choice": true
},
"openai.gpt-oss-20b-1:0": {
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
@@ -46591,7 +46595,9 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
@@ -46606,7 +46612,9 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
@@ -46617,7 +46625,9 @@
"input_cost_per_token": 2.64e-06,
"output_cost_per_token": 7.92e-06,
"cache_read_input_token_cost": 6.6e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
@@ -57156,7 +57166,6 @@
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-luna.html"
},
"us.openai.gpt-5.6-sol": {
- "supports_bedrock_runtime_chat_completions": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4.4e-06,
"input_cost_per_token_above_272k_tokens": 8.8e-06,
@@ -57188,11 +57197,11 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
+ "/v1/chat/completions",
"/v1/responses"
]
},
"global.openai.gpt-5.6-sol": {
- "supports_bedrock_runtime_chat_completions": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4e-06,
"input_cost_per_token_above_272k_tokens": 8e-06,
@@ -57224,11 +57233,11 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
+ "/v1/chat/completions",
"/v1/responses"
]
},
"us.openai.gpt-5.6-terra": {
- "supports_bedrock_runtime_chat_completions": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_above_272k_tokens": 4.4e-06,
@@ -57260,11 +57269,11 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
+ "/v1/chat/completions",
"/v1/responses"
]
},
"global.openai.gpt-5.6-terra": {
- "supports_bedrock_runtime_chat_completions": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-06,
"input_cost_per_token_above_272k_tokens": 4e-06,
@@ -57296,11 +57305,11 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
+ "/v1/chat/completions",
"/v1/responses"
]
},
"us.openai.gpt-5.6-luna": {
- "supports_bedrock_runtime_chat_completions": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-07,
"input_cost_per_token_above_272k_tokens": 4.4e-07,
@@ -57332,6 +57341,7 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
+ "/v1/chat/completions",
"/v1/responses"
]
},
@@ -57460,7 +57470,6 @@
]
},
"global.openai.gpt-5.6-luna": {
- "supports_bedrock_runtime_chat_completions": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_272k_tokens": 4e-07,
@@ -57492,6 +57501,7 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
+ "/v1/chat/completions",
"/v1/responses"
]
},
@@ -58095,7 +58105,9 @@
"input_cost_per_token": 2.2e-06,
"output_cost_per_token": 6.6e-06,
"cache_read_input_token_cost": 5.5e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
@@ -58114,7 +58126,9 @@
"input_cost_per_token": 2e-06,
"output_cost_per_token": 6e-06,
"cache_read_input_token_cost": 5e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
@@ -63818,7 +63832,9 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
@@ -63833,7 +63849,9 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
@@ -64076,7 +64094,9 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
@@ -64091,7 +64111,9 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json
index d58dd4d105a..7fdd8c85ddc 100644
--- a/model_prices_and_context_window.json
+++ b/model_prices_and_context_window.json
@@ -40834,7 +40834,9 @@
"output_cost_per_token": 0.0
},
"openai.gpt-oss-120b-1:0": {
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 1.5e-07,
"litellm_provider": "bedrock_converse",
@@ -40850,7 +40852,9 @@
"supports_tool_choice": true
},
"openai.gpt-oss-20b-1:0": {
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
@@ -46591,7 +46595,9 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
@@ -46606,7 +46612,9 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
@@ -46617,7 +46625,9 @@
"input_cost_per_token": 2.64e-06,
"output_cost_per_token": 7.92e-06,
"cache_read_input_token_cost": 6.6e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
@@ -57156,7 +57166,6 @@
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-luna.html"
},
"us.openai.gpt-5.6-sol": {
- "supports_bedrock_runtime_chat_completions": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4.4e-06,
"input_cost_per_token_above_272k_tokens": 8.8e-06,
@@ -57188,11 +57197,11 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
+ "/v1/chat/completions",
"/v1/responses"
]
},
"global.openai.gpt-5.6-sol": {
- "supports_bedrock_runtime_chat_completions": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4e-06,
"input_cost_per_token_above_272k_tokens": 8e-06,
@@ -57224,11 +57233,11 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
+ "/v1/chat/completions",
"/v1/responses"
]
},
"us.openai.gpt-5.6-terra": {
- "supports_bedrock_runtime_chat_completions": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_above_272k_tokens": 4.4e-06,
@@ -57260,11 +57269,11 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
+ "/v1/chat/completions",
"/v1/responses"
]
},
"global.openai.gpt-5.6-terra": {
- "supports_bedrock_runtime_chat_completions": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-06,
"input_cost_per_token_above_272k_tokens": 4e-06,
@@ -57296,11 +57305,11 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
+ "/v1/chat/completions",
"/v1/responses"
]
},
"us.openai.gpt-5.6-luna": {
- "supports_bedrock_runtime_chat_completions": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-07,
"input_cost_per_token_above_272k_tokens": 4.4e-07,
@@ -57332,6 +57341,7 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
+ "/v1/chat/completions",
"/v1/responses"
]
},
@@ -57460,7 +57470,6 @@
]
},
"global.openai.gpt-5.6-luna": {
- "supports_bedrock_runtime_chat_completions": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_272k_tokens": 4e-07,
@@ -57492,6 +57501,7 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
+ "/v1/chat/completions",
"/v1/responses"
]
},
@@ -58095,7 +58105,9 @@
"input_cost_per_token": 2.2e-06,
"output_cost_per_token": 6.6e-06,
"cache_read_input_token_cost": 5.5e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
@@ -58114,7 +58126,9 @@
"input_cost_per_token": 2e-06,
"output_cost_per_token": 6e-06,
"cache_read_input_token_cost": 5e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
@@ -63818,7 +63832,9 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
@@ -63833,7 +63849,9 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
@@ -64076,7 +64094,9 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
@@ -64091,7 +64111,9 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
- "supports_bedrock_runtime_chat_completions": true,
+ "supported_endpoints": [
+ "/v1/chat/completions"
+ ],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json
index 89ebedc3a0a..bcb0509f25c 100644
--- a/model_prices_and_context_window.schema.json
+++ b/model_prices_and_context_window.schema.json
@@ -902,9 +902,6 @@
"supports_audio_output": {
"type": "boolean"
},
- "supports_bedrock_runtime_chat_completions": {
- "type": "boolean"
- },
"supports_bedrock_runtime_chat_completions_response_format": {
"type": "boolean"
},
diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
index 2f0d384fcdd..3c347e4bed2 100644
--- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
+++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -59,9 +59,27 @@ def test_claude_stays_on_converse(local_cost_map):
assert BedrockModelInfo.get_bedrock_route("us.anthropic.claude-3-sonnet-20240229-v1:0") == "converse"
-def test_flag_absent_means_no_chat_completions_route(monkeypatch):
- monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": {"litellm_provider": "bedrock_converse"}})
+@pytest.mark.parametrize(
+ "entry",
+ [
+ {"litellm_provider": "bedrock_converse"},
+ {"litellm_provider": "bedrock_converse", "supported_endpoints": ["/v1/responses"]},
+ {"litellm_provider": "bedrock_converse", "supports_bedrock_runtime_chat_completions": True},
+ {"litellm_provider": "bedrock_mantle", "supported_endpoints": ["/v1/chat/completions", "/v1/responses"]},
+ {"litellm_provider": "openai", "supported_endpoints": ["/v1/chat/completions"]},
+ ],
+)
+def test_chat_completions_missing_from_supported_endpoints_means_no_chat_completions_route(monkeypatch, entry):
+ monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": entry})
assert uses_bedrock_runtime_chat_completions("us.xai.grok-4.6") is False
+ assert BedrockModelInfo.get_bedrock_route("us.xai.grok-4.6") == "converse"
+
+
+def test_chat_completions_in_supported_endpoints_opts_into_the_native_route(monkeypatch):
+ entry = {"litellm_provider": "bedrock_converse", "supported_endpoints": ["/v1/chat/completions", "/v1/responses"]}
+ monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": entry})
+ assert uses_bedrock_runtime_chat_completions("us.xai.grok-4.6") is True
+ assert BedrockModelInfo.get_bedrock_route("bedrock/us.xai.grok-4.6") == "chat_completions"
def test_complete_url_is_runtime_openai_chat_completions(monkeypatch):
@@ -860,7 +878,7 @@ SYNTHETIC_NATIVE_MODEL = "vendor.native-model-v1:0"
def test_capability_flags_are_read_from_the_cost_map(monkeypatch, capability_flags, request_params, needs_converse):
entry = {
"litellm_provider": "bedrock_converse",
- "supports_bedrock_runtime_chat_completions": True,
+ "supported_endpoints": ["/v1/chat/completions"],
**capability_flags,
}
monkeypatch.setattr(litellm, "model_cost", {SYNTHETIC_NATIVE_MODEL: entry})
diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py
index 98e46016401..39cb25a0f51 100644
--- a/tests/test_litellm/test_utils.py
+++ b/tests/test_litellm/test_utils.py
@@ -927,7 +927,6 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"supports_video_input": {"type": "boolean"},
"supports_vision": {"type": "boolean"},
"supports_web_search": {"type": "boolean"},
- "supports_bedrock_runtime_chat_completions": {"type": "boolean"},
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": {"type": "boolean"},
"supports_bedrock_runtime_chat_completions_response_format": {"type": "boolean"},
"supports_url_context": {"type": "boolean"},
From 4101c0ceb2e213a177613d89220bb2e64d191aeb Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Thu, 24 Sep 2026 18:08:46 -0700
Subject: [PATCH 10/18] docs(cost-map): describe the bedrock native chat
completions capability flags
---
ci_cd/generate_model_prices_schema.py | 20 ++++++++++++++++++-
.../crates/model-catalog/src/model_info.rs | 2 ++
model_prices_and_context_window.schema.json | 6 ++++--
3 files changed, 25 insertions(+), 3 deletions(-)
diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py
index 8eec07dadda..f54177def8e 100644
--- a/ci_cd/generate_model_prices_schema.py
+++ b/ci_cd/generate_model_prices_schema.py
@@ -211,6 +211,24 @@ NUMBER_KEYS: dict[str, JsonSchema] = {
},
}
+BOOLEAN_KEYS: dict[str, JsonSchema] = {
+ "supports_bedrock_runtime_chat_completions_response_format": {
+ "type": "boolean",
+ "description": (
+ "The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; "
+ "unset means LiteLLM serves those requests through Converse's json_tool_call emulation."
+ ),
+ },
+ "supports_bedrock_runtime_chat_completions_tools_with_reasoning": {
+ "type": "boolean",
+ "description": (
+ "The Bedrock native /v1/chat/completions route serves this model's function tools with any "
+ "reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort "
+ "is exactly 'none'."
+ ),
+ },
+}
+
COST_DESCRIPTIONS: dict[str, str] = {
"input_cost_per_token": "USD per prompt token.",
"output_cost_per_token": "USD per generated token.",
@@ -292,7 +310,7 @@ def string_key_schemas(modes: tuple) -> dict[str, JsonSchema]:
def classify(key: str, modes: tuple) -> Optional[JsonSchema]:
- curated = {**OBJECT_KEYS, **ARRAY_KEYS, **string_key_schemas(modes), **INTEGER_KEYS, **NUMBER_KEYS}
+ curated = {**OBJECT_KEYS, **ARRAY_KEYS, **string_key_schemas(modes), **INTEGER_KEYS, **NUMBER_KEYS, **BOOLEAN_KEYS}
if key in curated:
return curated[key]
if key.startswith("supports_") or key in EXTRA_BOOLEAN_KEYS:
diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs
index 23e69f66492..5dda91b2213 100644
--- a/litellm-rust/crates/model-catalog/src/model_info.rs
+++ b/litellm-rust/crates/model-catalog/src/model_info.rs
@@ -581,8 +581,10 @@ pub struct ModelInfo {
pub supports_audio_input: Option,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub supports_audio_output: Option,
+ /// The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; unset means LiteLLM serves those requests through Converse's json_tool_call emulation.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub supports_bedrock_runtime_chat_completions_response_format: Option,
+ /// The Bedrock native /v1/chat/completions route serves this model's function tools with any reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort is exactly 'none'.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option,
#[serde(default, skip_serializing_if = "Option::is_none")]
diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json
index bcb0509f25c..d9a7dd2fff5 100644
--- a/model_prices_and_context_window.schema.json
+++ b/model_prices_and_context_window.schema.json
@@ -903,10 +903,12 @@
"type": "boolean"
},
"supports_bedrock_runtime_chat_completions_response_format": {
- "type": "boolean"
+ "type": "boolean",
+ "description": "The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; unset means LiteLLM serves those requests through Converse's json_tool_call emulation."
},
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": {
- "type": "boolean"
+ "type": "boolean",
+ "description": "The Bedrock native /v1/chat/completions route serves this model's function tools with any reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort is exactly 'none'."
},
"supports_computer_use": {
"type": "boolean"
From 61a130c9a13a6d4563ba0c26b44a447ae12e98b2 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Thu, 24 Sep 2026 18:20:45 -0700
Subject: [PATCH 11/18] revert: docs(cost-map): describe the bedrock native
chat completions capability flags
This reverts commit 4101c0ceb2e213a177613d89220bb2e64d191aeb.
cost-map-guard runs main's schema generator under pull_request_target and compares
its output to the PR's committed schema, so a PR that changes the generator's output
cannot pass that required check until the generator change lands on main first. The
descriptions move to a follow-up that lands the generator change ahead of the schema
---
ci_cd/generate_model_prices_schema.py | 20 +------------------
.../crates/model-catalog/src/model_info.rs | 2 --
model_prices_and_context_window.schema.json | 6 ++----
3 files changed, 3 insertions(+), 25 deletions(-)
diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py
index f54177def8e..8eec07dadda 100644
--- a/ci_cd/generate_model_prices_schema.py
+++ b/ci_cd/generate_model_prices_schema.py
@@ -211,24 +211,6 @@ NUMBER_KEYS: dict[str, JsonSchema] = {
},
}
-BOOLEAN_KEYS: dict[str, JsonSchema] = {
- "supports_bedrock_runtime_chat_completions_response_format": {
- "type": "boolean",
- "description": (
- "The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; "
- "unset means LiteLLM serves those requests through Converse's json_tool_call emulation."
- ),
- },
- "supports_bedrock_runtime_chat_completions_tools_with_reasoning": {
- "type": "boolean",
- "description": (
- "The Bedrock native /v1/chat/completions route serves this model's function tools with any "
- "reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort "
- "is exactly 'none'."
- ),
- },
-}
-
COST_DESCRIPTIONS: dict[str, str] = {
"input_cost_per_token": "USD per prompt token.",
"output_cost_per_token": "USD per generated token.",
@@ -310,7 +292,7 @@ def string_key_schemas(modes: tuple) -> dict[str, JsonSchema]:
def classify(key: str, modes: tuple) -> Optional[JsonSchema]:
- curated = {**OBJECT_KEYS, **ARRAY_KEYS, **string_key_schemas(modes), **INTEGER_KEYS, **NUMBER_KEYS, **BOOLEAN_KEYS}
+ curated = {**OBJECT_KEYS, **ARRAY_KEYS, **string_key_schemas(modes), **INTEGER_KEYS, **NUMBER_KEYS}
if key in curated:
return curated[key]
if key.startswith("supports_") or key in EXTRA_BOOLEAN_KEYS:
diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs
index 5dda91b2213..23e69f66492 100644
--- a/litellm-rust/crates/model-catalog/src/model_info.rs
+++ b/litellm-rust/crates/model-catalog/src/model_info.rs
@@ -581,10 +581,8 @@ pub struct ModelInfo {
pub supports_audio_input: Option,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub supports_audio_output: Option,
- /// The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; unset means LiteLLM serves those requests through Converse's json_tool_call emulation.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub supports_bedrock_runtime_chat_completions_response_format: Option,
- /// The Bedrock native /v1/chat/completions route serves this model's function tools with any reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort is exactly 'none'.
#[serde(default, skip_serializing_if = "Option::is_none")]
pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option,
#[serde(default, skip_serializing_if = "Option::is_none")]
diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json
index d9a7dd2fff5..bcb0509f25c 100644
--- a/model_prices_and_context_window.schema.json
+++ b/model_prices_and_context_window.schema.json
@@ -903,12 +903,10 @@
"type": "boolean"
},
"supports_bedrock_runtime_chat_completions_response_format": {
- "type": "boolean",
- "description": "The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; unset means LiteLLM serves those requests through Converse's json_tool_call emulation."
+ "type": "boolean"
},
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": {
- "type": "boolean",
- "description": "The Bedrock native /v1/chat/completions route serves this model's function tools with any reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort is exactly 'none'."
+ "type": "boolean"
},
"supports_computer_use": {
"type": "boolean"
From c60249fa882116c4cd8fe784770f5e3a3be3ec30 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Sat, 26 Sep 2026 16:02:49 +0000
Subject: [PATCH 12/18] test(bedrock): move the native chat completions tests
under tests/unit
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
tests/unit/llms/bedrock/chat/chat_completions/__init__.py | 0
.../test_bedrock_chat_completions_transformation.py | 0
2 files changed, 0 insertions(+), 0 deletions(-)
create mode 100644 tests/unit/llms/bedrock/chat/chat_completions/__init__.py
rename tests/{test_litellm => unit}/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py (100%)
diff --git a/tests/unit/llms/bedrock/chat/chat_completions/__init__.py b/tests/unit/llms/bedrock/chat/chat_completions/__init__.py
new file mode 100644
index 00000000000..e69de29bb2d
diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
similarity index 100%
rename from tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
rename to tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
From f4d84f8f5db0c26e0aa3c50bc2ddb77d07122f5b Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Sat, 26 Sep 2026 17:28:06 +0000
Subject: [PATCH 13/18] fix(bedrock): drop reasoning_effort none for grok on
the native chat completions route
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../chat/chat_completions/transformation.py | 29 ++++++++++++-
...bedrock_chat_completions_transformation.py | 42 +++++++++++++++++++
2 files changed, 70 insertions(+), 1 deletion(-)
diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py
index 4c5e5768119..be2eb8c7713 100644
--- a/litellm/llms/bedrock/chat/chat_completions/transformation.py
+++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py
@@ -60,6 +60,31 @@ def chat_completions_params_refused_for(model: str) -> frozenset[str]:
)
+CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY: Final = MappingProxyType({"xai.": frozenset(("none",))})
+
+
+def chat_completions_reasoning_efforts_refused_for(model: str) -> frozenset[str]:
+ """The ``reasoning_effort`` values AWS's Chat Completions endpoint rejects for this model.
+
+ Grok answers ``"none"`` with a 400 (it takes low, medium, high, and xhigh) where Converse dropped every
+ ``reasoning_effort`` for it, so the native config drops the value and AWS applies its default effort as before.
+ """
+ model_id: Final = split_bedrock_region_path(model)[1]
+ return frozenset().union(
+ *(
+ refused
+ for family, refused in CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY.items()
+ if family in model_id
+ )
+ )
+
+
+def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) -> Mapping[str, object]:
+ if params.get("reasoning_effort") not in chat_completions_reasoning_efforts_refused_for(model):
+ return params
+ return MappingProxyType({key: value for key, value in params.items() if key != "reasoning_effort"})
+
+
def _held_close_tag_prefix(text: str) -> int:
return next(
(
@@ -277,7 +302,9 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
drop_params=drop_params,
replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens,
)
- return dict(with_max_completion_tokens(mapped)) # mutable-ok: get_optional_params keeps filling this dict
+ return dict( # mutable-ok: get_optional_params keeps filling this dict
+ without_refused_reasoning_effort(model, with_max_completion_tokens(mapped))
+ )
def _inference_params(
self, optional_params: Mapping[str, object]
diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
index 3c347e4bed2..061df201c15 100644
--- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
+++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -11,6 +11,7 @@ from litellm.llms.bedrock.chat.chat_completions.transformation import (
AmazonBedrockRuntimeChatCompletionsConfig,
BedrockRuntimeChatCompletionsStreamingHandler,
ReasoningTagSplitter,
+ chat_completions_reasoning_efforts_refused_for,
split_reasoning_tag,
with_max_completion_tokens,
)
@@ -333,6 +334,47 @@ def test_with_max_completion_tokens_leaves_other_params_alone():
assert with_max_completion_tokens({"temperature": 0.5}) == {"temperature": 0.5}
+@pytest.mark.parametrize(
+ "model",
+ ["us.xai.grok-4.6", "bedrock/us-gov-west-1/us.xai.grok-4.6"],
+)
+def test_map_openai_params_drops_reasoning_effort_none_for_grok(model):
+ cfg = AmazonBedrockRuntimeChatCompletionsConfig()
+ mapped = cfg.map_openai_params(
+ non_default_params={"reasoning_effort": "none", "max_tokens": 64},
+ optional_params={},
+ model=model,
+ drop_params=False,
+ )
+ assert "reasoning_effort" not in mapped
+
+
+def test_map_openai_params_keeps_reasoning_effort_low_for_grok():
+ cfg = AmazonBedrockRuntimeChatCompletionsConfig()
+ mapped = cfg.map_openai_params(
+ non_default_params={"reasoning_effort": "low", "max_tokens": 64},
+ optional_params={},
+ model="us.xai.grok-4.6",
+ drop_params=False,
+ )
+ assert mapped["reasoning_effort"] == "low"
+
+
+def test_map_openai_params_keeps_reasoning_effort_none_for_gpt56():
+ cfg = AmazonBedrockRuntimeChatCompletionsConfig()
+ mapped = cfg.map_openai_params(
+ non_default_params={"reasoning_effort": "none", "max_tokens": 64},
+ optional_params={},
+ model="global.openai.gpt-5.6-sol",
+ drop_params=False,
+ )
+ assert mapped["reasoning_effort"] == "none"
+
+
+def test_reasoning_efforts_refused_for_is_empty_outside_xai():
+ assert chat_completions_reasoning_efforts_refused_for("openai.gpt-oss-20b-1:0") == frozenset()
+
+
def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map):
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
assert "reasoning_effort" in cfg.get_supported_openai_params("global.openai.gpt-5.6-sol")
From e653217228d191d0c0ecb7c9b275786c19052e94 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Sat, 26 Sep 2026 17:33:01 +0000
Subject: [PATCH 14/18] fix(bedrock): keep converse extension params on the
converse route
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
litellm/llms/bedrock/common_utils.py | 14 ++++++++++++--
...test_bedrock_chat_completions_transformation.py | 12 ++++++++++++
2 files changed, 24 insertions(+), 2 deletions(-)
diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py
index 8fe545a3272..098be7082d3 100644
--- a/litellm/llms/bedrock/common_utils.py
+++ b/litellm/llms/bedrock/common_utils.py
@@ -890,7 +890,16 @@ def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> boo
BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
- ("guardrailConfig", "performanceConfig", "serviceTier", "requestMetadata", "outputConfig", "thinking")
+ (
+ "guardrailConfig",
+ "performanceConfig",
+ "serviceTier",
+ "requestMetadata",
+ "outputConfig",
+ "thinking",
+ "additionalModelRequestFields",
+ "top_k",
+ )
)
@@ -910,7 +919,8 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje
"""Whether a request on a runtime-Chat-Completions model must still be served by Converse.
Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking``
- block included, which only Converse forwards as ``additionalModelRequestFields``) have no field on
+ block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse
+ forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on
AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body,
function tools (``tools`` or legacy ``functions``) on a model without
``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless
diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
index 061df201c15..17c0f8d3c1d 100644
--- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
+++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -258,6 +258,18 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model):
assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": None}) == "chat_completions"
+@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"])
+@pytest.mark.parametrize(
+ "request_params",
+ [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}],
+ ids=["additionalModelRequestFields", "top_k"],
+)
+def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params):
+ assert bedrock_request_needs_converse(model, request_params) is True
+ assert BedrockModelInfo.get_bedrock_route(model, request_params) == "converse"
+ assert BedrockModelInfo.get_bedrock_route(model, {key: None for key in request_params}) == "chat_completions"
+
+
@pytest.mark.parametrize(
"request_params, expected_route",
[
From 093a9d4ddfbb010977ad7a50a0c8c3f7745dbcaf Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Sat, 26 Sep 2026 18:37:36 +0000
Subject: [PATCH 15/18] fix(bedrock): inline http image urls and keep stop on
converse for native chat completions
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../chat/chat_completions/transformation.py | 57 +++++++++++++-
litellm/llms/bedrock/common_utils.py | 4 +-
...bedrock_chat_completions_transformation.py | 75 +++++++++++++++++--
3 files changed, 127 insertions(+), 9 deletions(-)
diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py
index be2eb8c7713..123531cf227 100644
--- a/litellm/llms/bedrock/chat/chat_completions/transformation.py
+++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py
@@ -22,6 +22,11 @@ import httpx
from typing_extensions import assert_never
import litellm
+from litellm.litellm_core_utils.prompt_templates.image_handling import (
+ async_inline_remote_media,
+ convert_url_to_base64,
+ inline_remote_image_urls,
+)
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path
@@ -40,7 +45,7 @@ REASONING_CLOSE_TAG: Final = ""
CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType(
{
- "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs")),
+ "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs")),
"openai.gpt-oss": frozenset(("logit_bias",)),
"xai.": frozenset(("frequency_penalty", "presence_penalty")),
}
@@ -167,6 +172,48 @@ def split_reasoning_tag(content: str) -> tuple[str | None, str]:
return reasoning or None, body
+def _remote_http_url(candidate: object) -> str | None:
+ return candidate if isinstance(candidate, str) and candidate.startswith(("http://", "https://")) else None
+
+
+def _inlined_image_url_part(part: object) -> object:
+ fields: Final = part if isinstance(part, Mapping) else None
+ if fields is None or fields.get("type") != "image_url":
+ return part
+ image_url: Final = fields.get("image_url")
+ image_url_fields: Final = image_url if isinstance(image_url, Mapping) else None
+ url: Final = _remote_http_url(image_url_fields.get("url") if image_url_fields is not None else image_url)
+ if url is None:
+ return part
+ data_url: Final = convert_url_to_base64(url)
+ inlined: Final = {**image_url_fields, "url": data_url} if image_url_fields is not None else data_url
+ return {**fields, "image_url": inlined} # mutable-ok: json-serialized message part
+
+
+def _inlined_image_url_message(message: AllMessageValues) -> AllMessageValues:
+ content: Final = message.get("content")
+ if not isinstance(content, list):
+ return message
+ inlined_message: Final = { # mutable-ok: json-serialized message
+ **message,
+ "content": [_inlined_image_url_part(part) for part in content],
+ }
+ return inlined_message # pyright: ignore[reportReturnType] # the same message with remote image parts inlined
+
+
+def _with_inlined_remote_image_urls(
+ messages: list[AllMessageValues],
+) -> list[AllMessageValues]: # mutable-ok: transform_request takes a list
+ """Inline every remote ``image_url`` so AWS never sees the ``http(s)://`` URLs it rejects.
+
+ AWS's native surface only takes inline ``data:`` URLs and S3 URLs where Converse downloaded
+ remote images itself, so the bytes are fetched and inlined here exactly like Converse did.
+ """
+ return [ # mutable-ok: transform_request takes a list
+ _inlined_image_url_message(message) for message in messages
+ ]
+
+
class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler):
"""OpenAI chunk parsing plus the ```` split, tracked per choice index."""
@@ -222,6 +269,10 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
def custom_llm_provider(self) -> str | None:
return "bedrock"
+ @property
+ def uses_async_transform_request(self) -> bool:
+ return True
+
def get_error_class(
self,
error_message: str,
@@ -325,7 +376,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
) -> dict: # mutable-ok: BaseConfig signature
return super().transform_request(
model=split_bedrock_region_path(model)[1],
- messages=messages,
+ messages=_with_inlined_remote_image_urls(messages),
optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
headers=headers,
@@ -341,7 +392,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
) -> dict: # mutable-ok: BaseConfig signature
return await super().async_transform_request(
model=split_bedrock_region_path(model)[1],
- messages=messages,
+ messages=await async_inline_remote_media(messages, should_inline=inline_remote_image_urls),
optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
headers=headers,
diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py
index 098be7082d3..6bd4a8d634c 100644
--- a/litellm/llms/bedrock/common_utils.py
+++ b/litellm/llms/bedrock/common_utils.py
@@ -899,6 +899,7 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
"thinking",
"additionalModelRequestFields",
"top_k",
+ "stop",
)
)
@@ -921,7 +922,8 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje
Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking``
block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse
forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on
- AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body,
+ AWS's native OpenAI surface, ``stop`` stays on Converse where it fails loudly instead of silently
+ stopping hidden reasoning, operator-owned request metadata is only written onto the Converse body,
function tools (``tools`` or legacy ``functions``) on a model without
``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless
``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as
diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
index 17c0f8d3c1d..4bfa7bf0fc4 100644
--- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
+++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -258,11 +258,13 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model):
assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": None}) == "chat_completions"
-@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"])
+@pytest.mark.parametrize(
+ "model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6", "global.openai.gpt-5.6-sol"]
+)
@pytest.mark.parametrize(
"request_params",
- [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}],
- ids=["additionalModelRequestFields", "top_k"],
+ [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}, {"stop": ["END"]}],
+ ids=["additionalModelRequestFields", "top_k", "stop"],
)
def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params):
assert bedrock_request_needs_converse(model, request_params) is True
@@ -331,6 +333,69 @@ def test_map_openai_params_sends_max_tokens_as_max_completion_tokens():
assert mapped == {"max_completion_tokens": 64, "temperature": 0.1}
+HTTPS_IMAGE_URL = "https://example.com/cat.png"
+IMAGE_MESSAGES = [
+ {
+ "role": "user",
+ "content": [
+ {"type": "text", "text": "what is this"},
+ {"type": "image_url", "image_url": HTTPS_IMAGE_URL},
+ {"type": "image_url", "image_url": {"url": HTTPS_IMAGE_URL, "detail": "high"}},
+ {"type": "image_url", "image_url": {"url": "data:image/png;base64,AAA"}},
+ {"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}},
+ ],
+ }
+]
+
+
+def _assert_remote_images_inlined(content):
+ assert content[0] == {"type": "text", "text": "what is this"}
+ assert content[1]["image_url"]["url"] == f"data:image/png;base64,{HTTPS_IMAGE_URL}"
+ assert content[2] == {
+ "type": "image_url",
+ "image_url": {"url": f"data:image/png;base64,{HTTPS_IMAGE_URL}", "detail": "high"},
+ }
+ assert content[3]["image_url"]["url"] == "data:image/png;base64,AAA"
+ assert content[4]["image_url"]["url"] == "s3://bucket/key.png"
+
+
+def test_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch):
+ import litellm.llms.bedrock.chat.chat_completions.transformation as native_cc
+
+ monkeypatch.setattr(
+ native_cc, "convert_url_to_base64", lambda url: f"data:image/png;base64,{url}"
+ )
+ body = AmazonBedrockRuntimeChatCompletionsConfig().transform_request(
+ model="us.xai.grok-4.6",
+ messages=IMAGE_MESSAGES,
+ optional_params={},
+ litellm_params={},
+ headers={},
+ )
+
+ _assert_remote_images_inlined(body["messages"][0]["content"])
+
+
+async def test_async_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch):
+ import litellm.litellm_core_utils.prompt_templates.image_handling as image_handling
+
+ async def fake_convert(url):
+ return f"data:image/png;base64,{url}"
+
+ monkeypatch.setattr(image_handling, "async_convert_url_to_base64", fake_convert)
+ cfg = AmazonBedrockRuntimeChatCompletionsConfig()
+ assert cfg.uses_async_transform_request is True
+ body = await cfg.async_transform_request(
+ model="us.xai.grok-4.6",
+ messages=IMAGE_MESSAGES,
+ optional_params={},
+ litellm_params={},
+ headers={},
+ )
+
+ _assert_remote_images_inlined(body["messages"][0]["content"])
+
+
def test_map_openai_params_keeps_explicit_max_completion_tokens():
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
mapped = cfg.map_openai_params(
@@ -398,8 +463,8 @@ def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map):
[
(
"bedrock/global.openai.gpt-5.6-sol",
- ("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs", "n"),
- ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions"),
+ ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "n"),
+ ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions", "stop"),
),
(
"us.xai.grok-4.6",
From daba2576f503ab4b313cac6d196f28fa731ef0f9 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Sat, 26 Sep 2026 18:41:15 +0000
Subject: [PATCH 16/18] refactor(bedrock): share the sync remote media inliner
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../prompt_templates/image_handling.py | 20 ++++++++
.../chat/chat_completions/transformation.py | 46 +------------------
.../litellm_core_utils/test_image_handling.py | 45 ++++++++++++++++++
...bedrock_chat_completions_transformation.py | 4 +-
4 files changed, 69 insertions(+), 46 deletions(-)
diff --git a/litellm/litellm_core_utils/prompt_templates/image_handling.py b/litellm/litellm_core_utils/prompt_templates/image_handling.py
index c44c80bc0a0..cb5c02ce10e 100644
--- a/litellm/litellm_core_utils/prompt_templates/image_handling.py
+++ b/litellm/litellm_core_utils/prompt_templates/image_handling.py
@@ -310,6 +310,26 @@ async def _fetch_data_urls(remote_urls: tuple[str, ...]) -> tuple[str, ...]:
raise
+def inline_remote_media(
+ messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues]
+ should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url,
+) -> list[AllMessageValues]: # mutable-ok: every transform_request takes list[AllMessageValues]
+ remote_urls: Final = tuple(
+ dict.fromkeys(
+ remote.url
+ for message in messages
+ for part in _content_parts(message)
+ if (remote := _parse_remote_part(part)) is not None and should_inline(_remote_media(remote))
+ )
+ )
+ if not remote_urls:
+ return messages
+ data_urls: Final = MappingProxyType({url: convert_url_to_base64(url) for url in remote_urls})
+ return [ # mutable-ok: transform_request takes a list
+ _inline_message(message, data_urls, should_inline) for message in messages
+ ]
+
+
async def async_inline_remote_media(
messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues]
should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url,
diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py
index 123531cf227..d907ef613a2 100644
--- a/litellm/llms/bedrock/chat/chat_completions/transformation.py
+++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py
@@ -24,8 +24,8 @@ from typing_extensions import assert_never
import litellm
from litellm.litellm_core_utils.prompt_templates.image_handling import (
async_inline_remote_media,
- convert_url_to_base64,
inline_remote_image_urls,
+ inline_remote_media,
)
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
@@ -172,48 +172,6 @@ def split_reasoning_tag(content: str) -> tuple[str | None, str]:
return reasoning or None, body
-def _remote_http_url(candidate: object) -> str | None:
- return candidate if isinstance(candidate, str) and candidate.startswith(("http://", "https://")) else None
-
-
-def _inlined_image_url_part(part: object) -> object:
- fields: Final = part if isinstance(part, Mapping) else None
- if fields is None or fields.get("type") != "image_url":
- return part
- image_url: Final = fields.get("image_url")
- image_url_fields: Final = image_url if isinstance(image_url, Mapping) else None
- url: Final = _remote_http_url(image_url_fields.get("url") if image_url_fields is not None else image_url)
- if url is None:
- return part
- data_url: Final = convert_url_to_base64(url)
- inlined: Final = {**image_url_fields, "url": data_url} if image_url_fields is not None else data_url
- return {**fields, "image_url": inlined} # mutable-ok: json-serialized message part
-
-
-def _inlined_image_url_message(message: AllMessageValues) -> AllMessageValues:
- content: Final = message.get("content")
- if not isinstance(content, list):
- return message
- inlined_message: Final = { # mutable-ok: json-serialized message
- **message,
- "content": [_inlined_image_url_part(part) for part in content],
- }
- return inlined_message # pyright: ignore[reportReturnType] # the same message with remote image parts inlined
-
-
-def _with_inlined_remote_image_urls(
- messages: list[AllMessageValues],
-) -> list[AllMessageValues]: # mutable-ok: transform_request takes a list
- """Inline every remote ``image_url`` so AWS never sees the ``http(s)://`` URLs it rejects.
-
- AWS's native surface only takes inline ``data:`` URLs and S3 URLs where Converse downloaded
- remote images itself, so the bytes are fetched and inlined here exactly like Converse did.
- """
- return [ # mutable-ok: transform_request takes a list
- _inlined_image_url_message(message) for message in messages
- ]
-
-
class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler):
"""OpenAI chunk parsing plus the ```` split, tracked per choice index."""
@@ -376,7 +334,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
) -> dict: # mutable-ok: BaseConfig signature
return super().transform_request(
model=split_bedrock_region_path(model)[1],
- messages=_with_inlined_remote_image_urls(messages),
+ messages=inline_remote_media(messages, should_inline=inline_remote_image_urls),
optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
headers=headers,
diff --git a/tests/unit/litellm_core_utils/test_image_handling.py b/tests/unit/litellm_core_utils/test_image_handling.py
index 21e97e97357..1fc8d54cecc 100644
--- a/tests/unit/litellm_core_utils/test_image_handling.py
+++ b/tests/unit/litellm_core_utils/test_image_handling.py
@@ -16,6 +16,7 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import (
async_convert_url_to_base64,
async_inline_remote_media,
convert_url_to_base64,
+ inline_remote_media,
)
from litellm.litellm_core_utils.url_utils import SSRFError
@@ -320,6 +321,50 @@ async def test_async_inline_remote_media_inlines_every_remote_part_shape(async_o
assert messages == snapshot
+def test_inline_remote_media_inlines_every_remote_part_shape(monkeypatch):
+ image_url = f"http://img.example/{uuid.uuid4()}.png"
+ pdf_url = f"http://docs.example/{uuid.uuid4()}.pdf"
+ fetched = []
+
+ def fake_convert(url):
+ fetched.append(url)
+ return f"data:image/png;base64,{url}"
+
+ monkeypatch.setattr(image_handling, "convert_url_to_base64", fake_convert)
+ messages = [
+ {"role": "system", "content": "be terse"},
+ {
+ "role": "user",
+ "content": [
+ {"type": "text", "text": "what is this?"},
+ {"type": "image_url", "image_url": {"url": image_url, "detail": "low"}},
+ {"type": "image_url", "image_url": image_url},
+ {"type": "image_url", "image_url": {"url": "data:image/png;base64,iVBORw0KGgo="}},
+ {"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}},
+ {"type": "file", "file": {"file_id": pdf_url}},
+ {"type": "document", "source": {"type": "url", "url": pdf_url}, "title": "the doc"},
+ ],
+ },
+ ]
+ snapshot = copy.deepcopy(messages)
+
+ inlined = inline_remote_media(messages, should_inline=image_handling.inline_remote_image_urls)
+
+ data_url = f"data:image/png;base64,{image_url}"
+ assert inlined[0] == {"role": "system", "content": "be terse"}
+ assert inlined[1]["content"] == [
+ {"type": "text", "text": "what is this?"},
+ {"type": "image_url", "image_url": {"url": data_url, "detail": "low"}},
+ {"type": "image_url", "image_url": data_url},
+ {"type": "image_url", "image_url": {"url": "data:image/png;base64,iVBORw0KGgo="}},
+ {"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}},
+ {"type": "file", "file": {"file_id": pdf_url}},
+ {"type": "document", "source": {"type": "url", "url": pdf_url}, "title": "the doc"},
+ ]
+ assert fetched == [image_url]
+ assert messages == snapshot
+
+
async def test_async_inline_remote_media_inlines_only_the_parts_the_predicate_accepts(async_only_image_fetch):
files_api_prefix = "https://generativelanguage.googleapis.com/v1beta/files/"
files_api_pdf = f"{files_api_prefix}{uuid.uuid4().hex}"
diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
index 4bfa7bf0fc4..cbde3dd03a0 100644
--- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
+++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -360,10 +360,10 @@ def _assert_remote_images_inlined(content):
def test_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch):
- import litellm.llms.bedrock.chat.chat_completions.transformation as native_cc
+ import litellm.litellm_core_utils.prompt_templates.image_handling as image_handling
monkeypatch.setattr(
- native_cc, "convert_url_to_base64", lambda url: f"data:image/png;base64,{url}"
+ image_handling, "convert_url_to_base64", lambda url: f"data:image/png;base64,{url}"
)
body = AmazonBedrockRuntimeChatCompletionsConfig().transform_request(
model="us.xai.grok-4.6",
From cbf01c25babb5e4410e65f57950787f4f76b2f93 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Sat, 26 Sep 2026 19:16:13 +0000
Subject: [PATCH 17/18] fix(image-handling): infer the image mime type when the
server sends a generic content type
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
---
.../prompt_templates/image_handling.py | 28 +++++------
.../litellm_core_utils/test_image_handling.py | 49 +++++++++++++++++++
2 files changed, 60 insertions(+), 17 deletions(-)
diff --git a/litellm/litellm_core_utils/prompt_templates/image_handling.py b/litellm/litellm_core_utils/prompt_templates/image_handling.py
index cb5c02ce10e..57b4f545301 100644
--- a/litellm/litellm_core_utils/prompt_templates/image_handling.py
+++ b/litellm/litellm_core_utils/prompt_templates/image_handling.py
@@ -15,6 +15,7 @@ import litellm
from litellm import verbose_logger
from litellm.caching.caching import InMemoryCache
from litellm.constants import MAX_IMAGE_URL_DOWNLOAD_SIZE_MB
+from litellm.litellm_core_utils.prompt_templates.common_utils import infer_content_type_from_url_and_content
from litellm.litellm_core_utils.url_utils import SSRFError, async_safe_get, safe_get
from litellm.types.llms.openai import AllMessageValues
@@ -55,23 +56,16 @@ def _process_image_response(response: Response, url: str) -> str:
base64_image: Final = base64.b64encode(image_bytes).decode("utf-8")
- image_type: Final = response.headers.get("Content-Type")
- if image_type is None:
- img_type = url.split(".")[-1].lower()
- _img_type: Final = {
- "jpg": "image/jpeg",
- "jpeg": "image/jpeg",
- "png": "image/png",
- "gif": "image/gif",
- "webp": "image/webp",
- }.get(img_type)
- if _img_type is None:
- raise Exception(
- f"Error: Unsupported image format. Format={_img_type}. Supported types = ['image/jpeg', 'image/png', 'image/gif', 'image/webp']"
- )
- img_type = _img_type
- else:
- img_type = image_type
+ try:
+ img_type: Final = infer_content_type_from_url_and_content(
+ url=url,
+ content=bytes(image_bytes),
+ current_content_type=response.headers.get("Content-Type"),
+ )
+ except ValueError as e:
+ raise litellm.ImageFetchError(
+ f"Error: Unable to determine image content type from the server's headers, the URL, or the image bytes. url={url}"
+ ) from e
result: Final = f"data:{img_type};base64,{base64_image}"
in_memory_cache.set_cache(url, result)
diff --git a/tests/unit/litellm_core_utils/test_image_handling.py b/tests/unit/litellm_core_utils/test_image_handling.py
index 1fc8d54cecc..57eb32f98e5 100644
--- a/tests/unit/litellm_core_utils/test_image_handling.py
+++ b/tests/unit/litellm_core_utils/test_image_handling.py
@@ -1,4 +1,5 @@
import asyncio
+import base64
import copy
import time
import uuid
@@ -259,6 +260,54 @@ async def test_async_data_url_is_returned_unchanged_without_fetch(monkeypatch):
assert await async_convert_url_to_base64(data_url) == data_url
+REAL_PNG_BYTES = base64.b64decode(
+ "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="
+)
+
+
+def _stub_image_client(content, content_type):
+ class _Client:
+ def get(self, url, follow_redirects=True):
+ headers = {} if content_type is None else {"Content-Type": content_type}
+ return Response(200, content=content, headers=headers, request=Request("GET", url))
+
+ return _Client()
+
+
+def test_convert_url_to_base64_infers_the_type_when_the_server_sends_octet_stream(monkeypatch):
+ monkeypatch.setattr(
+ litellm, "module_level_client", _stub_image_client(REAL_PNG_BYTES, "application/octet-stream")
+ )
+
+ result = convert_url_to_base64(f"http://img.example/{uuid.uuid4()}")
+
+ assert result.startswith("data:image/png;base64,")
+
+
+def test_convert_url_to_base64_keeps_a_real_content_type(monkeypatch):
+ monkeypatch.setattr(
+ litellm, "module_level_client", _stub_image_client(REAL_PNG_BYTES, "image/jpeg")
+ )
+
+ result = convert_url_to_base64(f"http://img.example/{uuid.uuid4()}.png")
+
+ assert result.startswith("data:image/jpeg;base64,")
+
+
+def test_convert_url_to_base64_raises_when_no_content_type_is_determinable(monkeypatch):
+ monkeypatch.setattr(
+ litellm,
+ "module_level_client",
+ _stub_image_client(b"\x00\x01\x02\x03not-an-image", "application/octet-stream"),
+ )
+ url = f"http://img.example/{uuid.uuid4()}"
+
+ with pytest.raises(litellm.ImageFetchError) as excinfo:
+ convert_url_to_base64(url)
+
+ assert url in str(excinfo.value)
+
+
def test_image_size_limit_disabled(monkeypatch):
"""
Test that setting MAX_IMAGE_URL_DOWNLOAD_SIZE_MB to 0 disables all image URL downloads.
From 4900a1ae85b314e108e6150adfb2d3acef1c0914 Mon Sep 17 00:00:00 2001
From: mateo-berri <277851410+mateo-berri@users.noreply.github.com>
Date: Sat, 26 Sep 2026 14:25:38 -0700
Subject: [PATCH 18/18] fix(bedrock): stop sending aws_bedrock_project_id as
OpenAI-Project on the runtime chat completions route
---
.../chat/chat_completions/transformation.py | 24 -------------------
...bedrock_chat_completions_transformation.py | 13 ++++++++++
2 files changed, 13 insertions(+), 24 deletions(-)
diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py
index d907ef613a2..381ce7922d7 100644
--- a/litellm/llms/bedrock/chat/chat_completions/transformation.py
+++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py
@@ -394,30 +394,6 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
choice.message.content = content
return response
- def validate_environment(
- self,
- headers: dict, # mutable-ok: BaseConfig signature
- model: str,
- messages: list[AllMessageValues], # mutable-ok: BaseConfig signature
- optional_params: dict, # mutable-ok: BaseConfig signature
- litellm_params: dict, # mutable-ok: BaseConfig signature
- api_key: str | None = None,
- api_base: str | None = None,
- ) -> dict: # mutable-ok: BaseConfig signature
- validated: Final = super().validate_environment(
- headers=headers,
- model=model,
- messages=messages,
- optional_params=optional_params,
- litellm_params=litellm_params,
- api_key=api_key,
- api_base=api_base,
- )
- project_id: Final = litellm_params.get("aws_bedrock_project_id")
- if not project_id:
- return validated
- return {**validated, "OpenAI-Project": project_id} # mutable-ok: BaseConfig signature returns a dict
-
def get_supported_openai_params(self, model: str) -> list: # mutable-ok: BaseConfig signature
refused: Final = frozenset(("n", *chat_completions_params_refused_for(model)))
base_params: Final = tuple(
diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
index cbde3dd03a0..f9f330a7340 100644
--- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
+++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py
@@ -109,6 +109,19 @@ def test_complete_url_appends_to_openai_v1_base():
assert url == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
+def test_project_id_is_not_sent_as_openai_project_header():
+ cfg = AmazonBedrockRuntimeChatCompletionsConfig()
+ headers = cfg.validate_environment(
+ headers={},
+ model="bedrock/openai.gpt-oss-20b-1:0",
+ messages=[{"role": "user", "content": "hello"}],
+ optional_params={},
+ litellm_params={"aws_bedrock_project_id": "proj_from_config"},
+ )
+ assert "OpenAI-Project" not in headers
+ assert headers["Content-Type"] == "application/json"
+
+
def test_transform_request_is_openai_chat_body_not_converse():
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
body = cfg.transform_request(