From 84084d9a82548e364bea390ba021ff926e87c468 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 12:22:16 -0700 Subject: [PATCH 01/18] feat(bedrock): send grok chat completions through runtime openai path Unspecified bedrock grok was rewritten to Converse. Chat completions now hit bedrock-runtime /openai/v1/chat/completions, and converse/ still uses Converse --- ci_cd/generate_model_prices_schema.py | 1 + litellm/__init__.py | 3 + litellm/_lazy_imports_registry.py | 5 + .../chat/chat_completions/transformation.py | 174 ++++++++++++++++++ litellm/llms/bedrock/common_utils.py | 21 +++ ...odel_prices_and_context_window_backup.json | 3 + model_prices_and_context_window.json | 3 + model_prices_and_context_window.schema.json | 3 + ...bedrock_chat_completions_transformation.py | 150 +++++++++++++++ 9 files changed, 363 insertions(+) create mode 100644 litellm/llms/bedrock/chat/chat_completions/transformation.py create mode 100644 tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index ab29b70bdd4..b7f7972d907 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -26,6 +26,7 @@ EXTRA_BOOLEAN_KEYS = frozenset( "gemini_audio_only_live", "uses_embed_content", "use_openai_responses_path", + "use_bedrock_runtime_chat_completions", "bedrock_converse_supports_strict_tools", "thinking_always_on", } diff --git a/litellm/__init__.py b/litellm/__init__.py index 1dfd146a00e..fea9150a6c0 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1736,6 +1736,9 @@ if TYPE_CHECKING: from .llms.bedrock.chat.invoke_transformations.amazon_openai_transformation import ( AmazonBedrockOpenAIConfig as AmazonBedrockOpenAIConfig, ) + from .llms.bedrock.chat.chat_completions.transformation import ( + AmazonBedrockRuntimeChatCompletionsConfig as AmazonBedrockRuntimeChatCompletionsConfig, + ) from .llms.bedrock.image_generation.amazon_stability1_transformation import ( AmazonStabilityConfig as AmazonStabilityConfig, ) diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index dc323c8cc15..0880734fd9f 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -204,6 +204,7 @@ LLM_CONFIG_NAMES: Final = ( "AmazonTwelveLabsPegasusConfig", "AmazonInvokeConfig", "AmazonBedrockOpenAIConfig", + "AmazonBedrockRuntimeChatCompletionsConfig", "AmazonStabilityConfig", "AmazonStability3Config", "AmazonNovaCanvasConfig", @@ -847,6 +848,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ".llms.bedrock.chat.invoke_transformations.amazon_openai_transformation", "AmazonBedrockOpenAIConfig", ), + "AmazonBedrockRuntimeChatCompletionsConfig": ( + ".llms.bedrock.chat.chat_completions.transformation", + "AmazonBedrockRuntimeChatCompletionsConfig", + ), "AmazonStabilityConfig": ( ".llms.bedrock.image_generation.amazon_stability1_transformation", "AmazonStabilityConfig", diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py new file mode 100644 index 00000000000..3bf6b2a2ffd --- /dev/null +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -0,0 +1,174 @@ +""" +Native OpenAI Chat Completions on Amazon Bedrock Runtime. + +AWS serves this surface at +``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions``. +Grok 4.6 on runtime is one of the models that uses it: chat completions stay +chat completions instead of being rewritten to Converse. + +Usage: model="us.xai.grok-4.6" or model="bedrock/us.xai.grok-4.6" +Explicit ``bedrock/converse/...`` still uses Converse. +""" + +from collections.abc import AsyncIterator, Iterator +from typing import Any, Final + +import httpx + +import litellm +from litellm._logging import verbose_logger +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM +from litellm.llms.bedrock.common_utils import BedrockError, strip_bedrock_routing_prefix +from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig +from litellm.types.llms.openai import AllMessageValues + + +class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): + def __init__(self, aws_signer: BaseAWSLLM | None = None): + super().__init__() + self._aws_signer: Final = aws_signer or BaseAWSLLM() + + @property + def custom_llm_provider(self) -> str | None: + return "bedrock" + + def get_error_class( + self, error_message: str, status_code: int, headers: dict[str, object] | httpx.Headers + ) -> BaseLLMException: + return BedrockError(status_code=status_code, message=error_message, headers=headers) + + def get_complete_url( + self, + api_base: str | None, + api_key: str | None, + model: str, + optional_params: dict, + litellm_params: dict, + stream: bool | None = None, + ) -> str: + if api_base is not None and "chat/completions" in api_base: + return api_base.rstrip("/") + aws_region_name: Final = self._aws_signer._get_aws_region_name(optional_params=optional_params, model=model) + endpoint_url, _ = self._aws_signer.get_runtime_endpoint( + api_base=api_base, + aws_bedrock_runtime_endpoint=optional_params.get("aws_bedrock_runtime_endpoint"), + aws_region_name=aws_region_name, + ) + base: Final = endpoint_url.rstrip("/") + if base.endswith("/openai/v1/chat/completions"): + return base + if base.endswith("/openai/v1"): + return f"{base}/chat/completions" + return f"{base}/openai/v1/chat/completions" + + def sign_request( + self, + headers: dict, + optional_params: dict, + request_data: dict, + api_base: str, + api_key: str | None = None, + model: str | None = None, + stream: bool | None = None, + fake_stream: bool | None = None, + ) -> tuple[dict, bytes | None]: + return self._aws_signer._sign_request( + service_name="bedrock", + headers=headers, + optional_params=optional_params, + request_data=request_data, + api_base=api_base, + api_key=api_key, + model=model, + stream=stream, + fake_stream=fake_stream, + ) + + def transform_request( + self, + model: str, + messages: list[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + inference_params: Final = { + k: v for k, v in optional_params.items() if k not in self._aws_signer.aws_authentication_params + } + return super().transform_request( + model=strip_bedrock_routing_prefix(model), + messages=messages, + optional_params=inference_params, + litellm_params=litellm_params, + headers=headers, + ) + + async def async_transform_request( + self, + model: str, + messages: list[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + inference_params: Final = { + k: v for k, v in optional_params.items() if k not in self._aws_signer.aws_authentication_params + } + return await super().async_transform_request( + model=strip_bedrock_routing_prefix(model), + messages=messages, + optional_params=inference_params, + litellm_params=litellm_params, + headers=headers, + ) + + def validate_environment( + self, + headers: dict, + model: str, + messages: list[AllMessageValues], + optional_params: dict, + litellm_params: dict, + api_key: str | None = None, + api_base: str | None = None, + ) -> dict: + headers = super().validate_environment( + headers=headers, + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + api_key=api_key, + api_base=api_base, + ) + project_id: Final = litellm_params.get("aws_bedrock_project_id") + if project_id: + headers["OpenAI-Project"] = project_id + return headers + + def get_supported_openai_params(self, model: str) -> list: + base_params: Final = super().get_supported_openai_params(model) + try: + if litellm.supports_reasoning(model=model, custom_llm_provider=self.custom_llm_provider): + if "reasoning_effort" not in base_params: + base_params.append("reasoning_effort") + except Exception as e: + verbose_logger.debug("AmazonBedrockRuntimeChatCompletionsConfig: error checking reasoning support: %s", e) + return base_params + + def get_model_response_iterator( + self, + streaming_response: Iterator[str] | AsyncIterator[str] | Any, + sync_stream: bool, + json_mode: bool | None = False, + ) -> Any: + from litellm.llms.openai.chat.gpt_transformation import ( + OpenAIChatCompletionStreamingHandler, + ) + + return OpenAIChatCompletionStreamingHandler( + streaming_response=streaming_response, + sync_stream=sync_stream, + json_mode=json_mode, + ) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index cb2c70e74c8..366ccadfbb7 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -780,6 +780,20 @@ def strip_bedrock_routing_prefix(model: str) -> str: return model +def uses_bedrock_runtime_chat_completions(model: str) -> bool: + """Whether this Bedrock model should use runtime native Chat Completions. + + Data-driven from the price-map ``use_bedrock_runtime_chat_completions`` flag + so onboarding a model is a JSON change. Explicit ``converse/`` still wins in + ``get_bedrock_route`` because prefix routes are checked first. + """ + stripped: Final = strip_bedrock_routing_prefix(model) + return any( + (litellm.model_cost.get(key) or {}).get("use_bedrock_runtime_chat_completions") is True + for key in (model, stripped) + ) + + def strip_bedrock_throughput_suffix(model: str) -> str: """Strip throughput tier suffixes and context window suffixes from Bedrock model names.""" import re @@ -1107,6 +1121,7 @@ class BedrockModelInfo(BaseLLMModelInfo): "async_invoke", "openai", "mantle", + "chat_completions", ]: """ Get the bedrock route for the given model. @@ -1123,6 +1138,7 @@ class BedrockModelInfo(BaseLLMModelInfo): "async_invoke", "openai", "mantle", + "chat_completions", ], ] = { "invoke/": "invoke", @@ -1152,6 +1168,9 @@ class BedrockModelInfo(BaseLLMModelInfo): if is_bedrock_application_inference_profile_arn(model): return "converse" + if uses_bedrock_runtime_chat_completions(model): + return "chat_completions" + base_model: Final = BedrockModelInfo.get_base_model(model) alt_model: Final = BedrockModelInfo.get_non_litellm_routing_model_name(model=model) if base_model in litellm.bedrock_converse_models or alt_model in litellm.bedrock_converse_models: @@ -1328,6 +1347,8 @@ def get_bedrock_chat_config(model: str): return litellm.AmazonConverseConfig() elif bedrock_route == "openai": return litellm.AmazonBedrockOpenAIConfig() + elif bedrock_route == "chat_completions": + return litellm.AmazonBedrockRuntimeChatCompletionsConfig() elif bedrock_route == "agent": from litellm.llms.bedrock.chat.invoke_agent.transformation import ( AmazonInvokeAgentConfig, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ad992ff92eb..3c6900cb136 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -44462,6 +44462,7 @@ "input_cost_per_token": 2.64e-06, "output_cost_per_token": 7.92e-06, "cache_read_input_token_cost": 6.6e-07, + "use_bedrock_runtime_chat_completions": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -55840,6 +55841,7 @@ "input_cost_per_token": 2.2e-06, "output_cost_per_token": 6.6e-06, "cache_read_input_token_cost": 5.5e-07, + "use_bedrock_runtime_chat_completions": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -55855,6 +55857,7 @@ "input_cost_per_token": 2e-06, "output_cost_per_token": 6e-06, "cache_read_input_token_cost": 5e-07, + "use_bedrock_runtime_chat_completions": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ad992ff92eb..3c6900cb136 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -44462,6 +44462,7 @@ "input_cost_per_token": 2.64e-06, "output_cost_per_token": 7.92e-06, "cache_read_input_token_cost": 6.6e-07, + "use_bedrock_runtime_chat_completions": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -55840,6 +55841,7 @@ "input_cost_per_token": 2.2e-06, "output_cost_per_token": 6.6e-06, "cache_read_input_token_cost": 5.5e-07, + "use_bedrock_runtime_chat_completions": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -55855,6 +55857,7 @@ "input_cost_per_token": 2e-06, "output_cost_per_token": 6e-06, "cache_read_input_token_cost": 5e-07, + "use_bedrock_runtime_chat_completions": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index 7ed1e7e568b..59d6b75a5f6 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -855,6 +855,9 @@ "minimum": 0, "description": "Provider default tokens-per-minute limit." }, + "use_bedrock_runtime_chat_completions": { + "type": "boolean" + }, "use_openai_responses_path": { "type": "boolean" }, diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py new file mode 100644 index 00000000000..7724b3f0a52 --- /dev/null +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -0,0 +1,150 @@ +"""Native Bedrock Runtime Chat Completions: Grok stays on /openai/v1/chat/completions.""" + +import json +from unittest.mock import patch + +import httpx +import pytest + +import litellm +from litellm.llms.bedrock.chat.chat_completions.transformation import ( + AmazonBedrockRuntimeChatCompletionsConfig, +) +from litellm.llms.bedrock.common_utils import ( + BedrockModelInfo, + get_bedrock_chat_config, + uses_bedrock_runtime_chat_completions, +) + + +@pytest.fixture +def local_cost_map(monkeypatch): + original_model_cost = litellm.model_cost + try: + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true") + litellm.model_cost = litellm.get_model_cost_map(url="") + litellm.get_model_info.cache_clear() + yield + finally: + litellm.model_cost = original_model_cost + litellm.get_model_info.cache_clear() + + +@pytest.mark.parametrize( + "model", + [ + "us.xai.grok-4.6", + "global.xai.grok-4.6", + "us-gov.xai.grok-4.6", + "bedrock/us.xai.grok-4.6", + ], +) +def test_grok_runtime_models_use_chat_completions_route(local_cost_map, model): + assert uses_bedrock_runtime_chat_completions(model) is True + assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions" + assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig) + + +def test_explicit_converse_prefix_still_uses_converse(local_cost_map): + assert BedrockModelInfo.get_bedrock_route("bedrock/converse/us.xai.grok-4.6") == "converse" + assert BedrockModelInfo.get_bedrock_route("converse/us.xai.grok-4.6") == "converse" + + +def test_claude_stays_on_converse(local_cost_map): + assert uses_bedrock_runtime_chat_completions("us.anthropic.claude-3-sonnet-20240229-v1:0") is False + assert BedrockModelInfo.get_bedrock_route("us.anthropic.claude-3-sonnet-20240229-v1:0") == "converse" + + +def test_flag_absent_means_no_chat_completions_route(monkeypatch): + monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": {"litellm_provider": "bedrock_converse"}}) + assert uses_bedrock_runtime_chat_completions("us.xai.grok-4.6") is False + + +def test_complete_url_is_runtime_openai_chat_completions(monkeypatch): + monkeypatch.setenv("AWS_REGION_NAME", "us-east-1") + monkeypatch.delenv("AWS_BEDROCK_RUNTIME_ENDPOINT", raising=False) + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + url = cfg.get_complete_url( + api_base=None, + api_key=None, + model="us.xai.grok-4.6", + optional_params={}, + litellm_params={}, + ) + assert url == "https://bedrock-runtime.us-east-1.amazonaws.com/openai/v1/chat/completions" + + +def test_complete_url_appends_to_openai_v1_base(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + url = cfg.get_complete_url( + api_base="https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1", + api_key=None, + model="us.xai.grok-4.6", + optional_params={}, + litellm_params={}, + ) + assert url == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + + +def test_transform_request_is_openai_chat_body_not_converse(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + body = cfg.transform_request( + model="bedrock/us.xai.grok-4.6", + messages=[{"role": "user", "content": "hello"}], + optional_params={"temperature": 0.2, "aws_region_name": "us-east-1"}, + litellm_params={}, + headers={}, + ) + assert body["model"] == "us.xai.grok-4.6" + assert body["messages"] == [{"role": "user", "content": "hello"}] + assert body["temperature"] == 0.2 + assert "aws_region_name" not in body + assert "inferenceConfig" not in body + assert "messages" in body + + +def test_completion_posts_runtime_chat_completions(local_cost_map, monkeypatch): + monkeypatch.setenv("AWS_REGION_NAME", "us-west-2") + monkeypatch.delenv("AWS_BEDROCK_RUNTIME_ENDPOINT", raising=False) + monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False) + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "testing") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing") + monkeypatch.setenv("AWS_SESSION_TOKEN", "testing") + + requests: list[dict] = [] + + def mock_post(self, url, data=None, json=None, headers=None, **kwargs): + requests.append({"url": url, "data": data, "json": json, "headers": headers or {}}) + return httpx.Response( + status_code=200, + json={ + "id": "chatcmpl-test", + "object": "chat.completion", + "created": 1733529600, + "model": "us.xai.grok-4.6", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "ok"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + }, + request=httpx.Request("POST", url), + ) + + with patch("litellm.llms.custom_httpx.http_handler.HTTPHandler.post", mock_post): + response = litellm.completion( + model="us.xai.grok-4.6", + messages=[{"role": "user", "content": "hello"}], + ) + + assert response.choices[0].message.content == "ok" + assert len(requests) == 1 + assert requests[0]["url"] == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + raw = requests[0]["data"] + body = json.loads(raw) if isinstance(raw, (str, bytes, bytearray)) else (requests[0]["json"] or {}) + assert body["model"] == "us.xai.grok-4.6" + assert body["messages"] == [{"role": "user", "content": "hello"}] + assert "inferenceConfig" not in body From 6b9f067c80eea6ce302eec5205aaf7892f1131e9 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:05:46 -0700 Subject: [PATCH 02/18] feat(bedrock): serve gpt-oss and gpt-5.6 chat completions on runtime's native openai path --- ci_cd/generate_model_prices_schema.py | 1 + .../chat/chat_completions/transformation.py | 319 ++++++++-- litellm/llms/bedrock/common_utils.py | 81 ++- litellm/main.py | 2 +- ...odel_prices_and_context_window_backup.json | 14 + litellm/utils.py | 26 +- model_prices_and_context_window.json | 14 + model_prices_and_context_window.schema.json | 3 + ...bedrock_chat_completions_transformation.py | 554 ++++++++++++++++-- ..._cross_region_inference_profile_mapping.py | 9 +- tests/test_litellm/test_utils.py | 2 + 11 files changed, 897 insertions(+), 128 deletions(-) diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index 5a2a56a5b21..648cceb79b0 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -31,6 +31,7 @@ EXTRA_BOOLEAN_KEYS = frozenset( "uses_embed_content", "use_openai_responses_path", "use_bedrock_runtime_chat_completions", + "bedrock_runtime_chat_completions_tools_require_reasoning_none", "bedrock_converse_supports_strict_tools", "thinking_always_on", } diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 3bf6b2a2ffd..49218270064 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -2,30 +2,170 @@ Native OpenAI Chat Completions on Amazon Bedrock Runtime. AWS serves this surface at -``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions``. -Grok 4.6 on runtime is one of the models that uses it: chat completions stay -chat completions instead of being rewritten to Converse. +``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions`` +for the models whose price-map entry sets ``use_bedrock_runtime_chat_completions`` +(Grok 4.6, gpt-oss, the GPT-5.6 family): chat completions stay chat completions +instead of being rewritten to Converse. -Usage: model="us.xai.grok-4.6" or model="bedrock/us.xai.grok-4.6" -Explicit ``bedrock/converse/...`` still uses Converse. +Usage: model="us.xai.grok-4.6", model="bedrock/openai.gpt-oss-20b-1:0" or +model="bedrock/global.openai.gpt-5.6-sol". Explicit ``bedrock/converse/...`` +still uses Converse, and so does a request that needs a Converse-only feature +(``bedrock_request_needs_converse`` in ``common_utils``). """ -from collections.abc import AsyncIterator, Iterator -from typing import Any, Final +from collections.abc import AsyncIterator, Iterator, Mapping +from dataclasses import dataclass, replace +from types import MappingProxyType +from typing import TYPE_CHECKING, Final, Literal import httpx import litellm -from litellm._logging import verbose_logger from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM from litellm.llms.bedrock.common_utils import BedrockError, strip_bedrock_routing_prefix +from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig from litellm.types.llms.openai import AllMessageValues +from litellm.types.utils import Choices, ModelResponse, ModelResponseStream + +if TYPE_CHECKING: + import tiktoken + + from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj + +REASONING_OPEN_TAG: Final = "" +REASONING_CLOSE_TAG: Final = "" + + +def _held_close_tag_prefix(text: str) -> int: + return next( + ( + size + for size in range(min(len(text), len(REASONING_CLOSE_TAG) - 1), 0, -1) + if REASONING_CLOSE_TAG.startswith(text[-size:]) + ), + 0, + ) + + +@dataclass(frozen=True, slots=True) +class ReasoningTagSplitter: + """ + The same split for a stream of content deltas, where a tag can arrive across chunks. + + ``feed`` returns the next state plus the reasoning and content text the delta contributes; + ``flush`` releases what the stream ended on before a tag resolved. + """ + + phase: Literal["start", "reasoning", "after_close", "content"] = "start" + pending: str = "" + + def feed(self, text: str) -> tuple["ReasoningTagSplitter", str, str]: + match self.phase: + case "content": + return self, "", text + case "after_close": + content: Final = text.lstrip() + return (replace(self, phase="content") if content else self), "", content + case "start": + return self._feed_start(self.pending + text) + case "reasoning": + return self._feed_reasoning(self.pending + text) + + def _feed_start(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]: + if buffered.startswith(REASONING_OPEN_TAG): + return replace(self, phase="reasoning", pending="")._feed_reasoning(buffered[len(REASONING_OPEN_TAG) :]) + if REASONING_OPEN_TAG.startswith(buffered): + return replace(self, pending=buffered), "", "" + return replace(self, phase="content", pending=""), "", buffered + + def _feed_reasoning(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]: + close_at: Final = buffered.find(REASONING_CLOSE_TAG) + if close_at >= 0: + after_close: Final = replace(self, phase="after_close", pending="") + next_state, _, content = after_close.feed(buffered[close_at + len(REASONING_CLOSE_TAG) :]) + return next_state, buffered[:close_at], content + held: Final = _held_close_tag_prefix(buffered) + return replace(self, pending=buffered[len(buffered) - held :]), buffered[: len(buffered) - held], "" + + def flush(self) -> tuple["ReasoningTagSplitter", str, str]: + drained: Final = replace(self, phase="content", pending="") + if self.phase == "reasoning": + return drained, self.pending, "" + return drained, "", self.pending + + +def _split_streamed_content( + splitter: ReasoningTagSplitter, content: str | None, finished: bool +) -> tuple[ReasoningTagSplitter, str, str]: + fed_state, fed_reasoning, fed_content = splitter.feed(content or "") + if not finished: + return fed_state, fed_reasoning, fed_content + drained, flushed_reasoning, flushed_content = fed_state.flush() + return drained, fed_reasoning + flushed_reasoning, fed_content + flushed_content + + +def split_reasoning_tag(content: str) -> tuple[str | None, str]: + """ + Split gpt-oss's inline ``...`` prefix out of a complete message. + + Runs the streaming splitter over the whole message, so a streamed and a non-streamed + response to the same completion split identically. Returns ``(None, content)`` when the + message does not start with the tag. + """ + _, reasoning, body = _split_streamed_content(ReasoningTagSplitter(), content, finished=True) + return reasoning or None, body + + +class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler): + """OpenAI chunk parsing plus the ```` split, tracked per choice index.""" + + def __init__( + self, + streaming_response: Iterator[str] | AsyncIterator[str] | ModelResponse, + sync_stream: bool, + json_mode: bool | None = False, + ) -> None: + super().__init__(streaming_response=streaming_response, sync_stream=sync_stream, json_mode=json_mode) + self._splitters: Mapping[int, ReasoningTagSplitter] = MappingProxyType({}) + + def chunk_parser(self, chunk: dict) -> ModelResponseStream: # mutable-ok: BaseModelResponseIterator signature + parsed: Final = super().chunk_parser(chunk) + for choice in parsed.choices: + next_state, reasoning, content = _split_streamed_content( + self._splitters.get(choice.index, ReasoningTagSplitter()), + choice.delta.content, + choice.finish_reason is not None, + ) + self._splitters = MappingProxyType({**self._splitters, choice.index: next_state}) + if reasoning: + choice.delta.reasoning_content = f"{getattr(choice.delta, 'reasoning_content', None) or ''}{reasoning}" + if content or choice.delta.content is not None: + choice.delta.content = content + return parsed + + +def with_max_completion_tokens(params: Mapping[str, object]) -> Mapping[str, object]: + """ + Send the caller's ``max_tokens`` as ``max_completion_tokens``. + + Every model on this surface accepts ``max_completion_tokens`` and the GPT-5.6 family + rejects ``max_tokens``; an explicit ``max_completion_tokens`` wins when both are set. + """ + if "max_tokens" not in params: + return params + return MappingProxyType( + { + key: value + for key, value in (("max_completion_tokens", params["max_tokens"]), *params.items()) + if key != "max_tokens" + } + ) class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): - def __init__(self, aws_signer: BaseAWSLLM | None = None): + def __init__(self, aws_signer: BaseAWSLLM | None = None) -> None: super().__init__() self._aws_signer: Final = aws_signer or BaseAWSLLM() @@ -34,7 +174,10 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): return "bedrock" def get_error_class( - self, error_message: str, status_code: int, headers: dict[str, object] | httpx.Headers + self, + error_message: str, + status_code: int, + headers: dict[str, object] | httpx.Headers, # mutable-ok: BaseConfig signature ) -> BaseLLMException: return BedrockError(status_code=status_code, message=error_message, headers=headers) @@ -43,13 +186,15 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): api_base: str | None, api_key: str | None, model: str, - optional_params: dict, - litellm_params: dict, + optional_params: dict, # mutable-ok: BaseConfig signature + litellm_params: dict, # mutable-ok: BaseConfig signature stream: bool | None = None, ) -> str: if api_base is not None and "chat/completions" in api_base: return api_base.rstrip("/") - aws_region_name: Final = self._aws_signer._get_aws_region_name(optional_params=optional_params, model=model) + aws_region_name: Final = self._aws_signer._get_aws_region_name( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public region resolver + optional_params=optional_params, model=model + ) endpoint_url, _ = self._aws_signer.get_runtime_endpoint( api_base=api_base, aws_bedrock_runtime_endpoint=optional_params.get("aws_bedrock_runtime_endpoint"), @@ -64,16 +209,16 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): def sign_request( self, - headers: dict, - optional_params: dict, - request_data: dict, + headers: dict, # mutable-ok: BaseConfig signature + optional_params: dict, # mutable-ok: BaseConfig signature + request_data: dict, # mutable-ok: BaseConfig signature api_base: str, api_key: str | None = None, model: str | None = None, stream: bool | None = None, fake_stream: bool | None = None, - ) -> tuple[dict, bytes | None]: - return self._aws_signer._sign_request( + ) -> tuple[dict, bytes | None]: # mutable-ok: BaseConfig signature + return self._aws_signer._sign_request( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public signer service_name="bedrock", headers=headers, optional_params=optional_params, @@ -85,21 +230,44 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): fake_stream=fake_stream, ) + def map_openai_params( + self, + non_default_params: dict, # mutable-ok: BaseConfig signature + optional_params: dict, # mutable-ok: BaseConfig signature + model: str, + drop_params: bool, + replace_max_completion_tokens_with_max_tokens: bool = False, + ) -> dict: # mutable-ok: BaseConfig signature + mapped: Final = super().map_openai_params( + non_default_params=non_default_params, + optional_params=optional_params, + model=model, + drop_params=drop_params, + replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens, + ) + return dict(with_max_completion_tokens(mapped)) # mutable-ok: get_optional_params keeps filling this dict + + def _inference_params( + self, optional_params: Mapping[str, object] + ) -> dict[str, object]: # mutable-ok: BaseConfig signature of transform_request + return { # mutable-ok: OpenAILikeChatConfig.transform_request takes a plain dict + key: value + for key, value in optional_params.items() + if key not in self._aws_signer.aws_authentication_params + } + def transform_request( self, model: str, - messages: list[AllMessageValues], - optional_params: dict, - litellm_params: dict, - headers: dict, - ) -> dict: - inference_params: Final = { - k: v for k, v in optional_params.items() if k not in self._aws_signer.aws_authentication_params - } + messages: list[AllMessageValues], # mutable-ok: BaseConfig signature + optional_params: dict, # mutable-ok: BaseConfig signature + litellm_params: dict, # mutable-ok: BaseConfig signature + headers: dict, # mutable-ok: BaseConfig signature + ) -> dict: # mutable-ok: BaseConfig signature return super().transform_request( model=strip_bedrock_routing_prefix(model), messages=messages, - optional_params=inference_params, + optional_params=self._inference_params(optional_params), litellm_params=litellm_params, headers=headers, ) @@ -107,33 +275,68 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): async def async_transform_request( self, model: str, - messages: list[AllMessageValues], - optional_params: dict, - litellm_params: dict, - headers: dict, - ) -> dict: - inference_params: Final = { - k: v for k, v in optional_params.items() if k not in self._aws_signer.aws_authentication_params - } + messages: list[AllMessageValues], # mutable-ok: BaseConfig signature + optional_params: dict, # mutable-ok: BaseConfig signature + litellm_params: dict, # mutable-ok: BaseConfig signature + headers: dict, # mutable-ok: BaseConfig signature + ) -> dict: # mutable-ok: BaseConfig signature return await super().async_transform_request( model=strip_bedrock_routing_prefix(model), messages=messages, - optional_params=inference_params, + optional_params=self._inference_params(optional_params), litellm_params=litellm_params, headers=headers, ) + def transform_response( + self, + model: str, + raw_response: httpx.Response, + model_response: ModelResponse, + logging_obj: "LiteLLMLoggingObj", + request_data: dict, # mutable-ok: BaseConfig signature + messages: list[AllMessageValues], # mutable-ok: BaseConfig signature + optional_params: dict, # mutable-ok: BaseConfig signature + litellm_params: dict, # mutable-ok: BaseConfig signature + encoding: "tiktoken.Encoding | None", + api_key: str | None = None, + json_mode: bool | None = None, + ) -> ModelResponse: + response: Final = super().transform_response( + model=model, + raw_response=raw_response, + model_response=model_response, + logging_obj=logging_obj, + request_data=request_data, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=encoding, + api_key=api_key, + json_mode=json_mode, + ) + for choice in response.choices: + if not isinstance(choice, Choices) or not isinstance(choice.message.content, str): + continue + reasoning, content = split_reasoning_tag(choice.message.content) + if reasoning is not None: + choice.message.reasoning_content = ( + f"{getattr(choice.message, 'reasoning_content', None) or ''}{reasoning}" + ) + choice.message.content = content + return response + def validate_environment( self, - headers: dict, + headers: dict, # mutable-ok: BaseConfig signature model: str, - messages: list[AllMessageValues], - optional_params: dict, - litellm_params: dict, + messages: list[AllMessageValues], # mutable-ok: BaseConfig signature + optional_params: dict, # mutable-ok: BaseConfig signature + litellm_params: dict, # mutable-ok: BaseConfig signature api_key: str | None = None, api_base: str | None = None, - ) -> dict: - headers = super().validate_environment( + ) -> dict: # mutable-ok: BaseConfig signature + validated: Final = super().validate_environment( headers=headers, model=model, messages=messages, @@ -143,31 +346,25 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): api_base=api_base, ) project_id: Final = litellm_params.get("aws_bedrock_project_id") - if project_id: - headers["OpenAI-Project"] = project_id - return headers + if not project_id: + return validated + return {**validated, "OpenAI-Project": project_id} # mutable-ok: BaseConfig signature returns a dict - def get_supported_openai_params(self, model: str) -> list: - base_params: Final = super().get_supported_openai_params(model) - try: - if litellm.supports_reasoning(model=model, custom_llm_provider=self.custom_llm_provider): - if "reasoning_effort" not in base_params: - base_params.append("reasoning_effort") - except Exception as e: - verbose_logger.debug("AmazonBedrockRuntimeChatCompletionsConfig: error checking reasoning support: %s", e) - return base_params + def get_supported_openai_params(self, model: str) -> list: # mutable-ok: BaseConfig signature + base_params: Final = [param for param in super().get_supported_openai_params(model) if param != "n"] + if "reasoning_effort" in base_params or not litellm.supports_reasoning( + model=model, custom_llm_provider=self.custom_llm_provider + ): + return base_params + return [*base_params, "reasoning_effort"] # mutable-ok: BaseConfig signature returns a list def get_model_response_iterator( self, - streaming_response: Iterator[str] | AsyncIterator[str] | Any, + streaming_response: Iterator[str] | AsyncIterator[str] | ModelResponse, sync_stream: bool, json_mode: bool | None = False, - ) -> Any: - from litellm.llms.openai.chat.gpt_transformation import ( - OpenAIChatCompletionStreamingHandler, - ) - - return OpenAIChatCompletionStreamingHandler( + ) -> BedrockRuntimeChatCompletionsStreamingHandler: + return BedrockRuntimeChatCompletionsStreamingHandler( streaming_response=streaming_response, sync_stream=sync_stream, json_mode=json_mode, diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index f6384aa97f1..124ef7c646c 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -28,6 +28,7 @@ from litellm.llms.base_llm.anthropic_messages.transformation import ( ) from litellm.llms.base_llm.base_utils import BaseLLMModelInfo, BaseTokenCounter from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.llms.bedrock.request_metadata import bedrock_request_metadata_is_owned from litellm.secret_managers.main import get_secret, get_secret_str from litellm.types.llms.bedrock import AWS_AUTH_PARAM_KEYS, AwsAuthParams @@ -37,6 +38,18 @@ if TYPE_CHECKING: _ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs" _OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.") +BedrockRoute = Literal[ + "converse", + "invoke", + "claude_platform", + "converse_like", + "agent", + "agentcore", + "async_invoke", + "openai", + "mantle", + "chat_completions", +] def error_response_text(response: httpx.Response) -> str: @@ -782,17 +795,54 @@ def strip_bedrock_routing_prefix(model: str) -> str: return model +def _bedrock_price_map_flag(model: str, flag: str) -> bool: + entries: Final = (litellm.model_cost.get(key) for key in (model, strip_bedrock_routing_prefix(model))) + return any(entry is not None and entry.get(flag) is True for entry in entries) + + def uses_bedrock_runtime_chat_completions(model: str) -> bool: """Whether this Bedrock model should use runtime native Chat Completions. Data-driven from the price-map ``use_bedrock_runtime_chat_completions`` flag so onboarding a model is a JSON change. Explicit ``converse/`` still wins in - ``get_bedrock_route`` because prefix routes are checked first. + ``get_bedrock_route`` because prefix routes are checked first, and a request + that needs a Converse-only feature (``bedrock_request_needs_converse``) is + served by Converse even on a flagged model. """ - stripped: Final = strip_bedrock_routing_prefix(model) - return any( - (litellm.model_cost.get(key) or {}).get("use_bedrock_runtime_chat_completions") is True - for key in (model, stripped) + return _bedrock_price_map_flag(model, "use_bedrock_runtime_chat_completions") + + +def bedrock_runtime_chat_completions_tools_require_reasoning_none(model: str) -> bool: + """Whether AWS's native Chat Completions only serves this model's function tools with ``reasoning_effort="none"``. + + Data-driven from the price-map ``bedrock_runtime_chat_completions_tools_require_reasoning_none`` + flag (the GPT-5.6 family). Converse serves tools with any effort, so those requests fall back to it. + """ + return _bedrock_price_map_flag(model, "bedrock_runtime_chat_completions_tools_require_reasoning_none") + + +BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( + ("guardrailConfig", "performanceConfig", "serviceTier", "requestMetadata", "outputConfig") +) + + +def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: + """Whether a request on a runtime-Chat-Completions model must still be served by Converse. + + Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``) are rejected as malformed input by + AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, + and function tools on a ``bedrock_runtime_chat_completions_tools_require_reasoning_none`` model are + rejected there unless ``reasoning_effort`` is exactly ``"none"``. + """ + if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS): + return True + if bedrock_request_metadata_is_owned(): + return True + if not request_params.get("tools"): + return False + return ( + bedrock_runtime_chat_completions_tools_require_reasoning_none(model) + and request_params.get("reasoning_effort") != "none" ) @@ -1130,20 +1180,13 @@ class BedrockModelInfo(BaseLLMModelInfo): @staticmethod def get_bedrock_route( model: str, - ) -> Literal[ - "converse", - "invoke", - "claude_platform", - "converse_like", - "agent", - "agentcore", - "async_invoke", - "openai", - "mantle", - "chat_completions", - ]: + request_params: Mapping[str, object] | None = None, + ) -> BedrockRoute: """ Get the bedrock route for the given model. + + ``request_params`` (the caller's chat params) lets a runtime Chat Completions + model fall back to Converse for the requests only Converse can serve. """ route_mappings: dict[ str, @@ -1187,7 +1230,9 @@ class BedrockModelInfo(BaseLLMModelInfo): if is_bedrock_application_inference_profile_arn(model): return "converse" - if uses_bedrock_runtime_chat_completions(model): + if uses_bedrock_runtime_chat_completions(model) and not ( + request_params is not None and bedrock_request_needs_converse(model, request_params) + ): return "chat_completions" base_model: Final = BedrockModelInfo.get_base_model(model) diff --git a/litellm/main.py b/litellm/main.py index 34410f9497c..ed78f209ca2 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -4078,7 +4078,7 @@ def _complete_bedrock(ctx: _CompletionDispatchContext) -> _CompletionDispatchRes if "aws_region_name" not in optional_params or optional_params["aws_region_name"] is None: optional_params["aws_region_name"] = aws_bedrock_client.meta.region_name - bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model) + bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model, optional_params) if bedrock_route == "claude_platform": provider_config = ProviderConfigManager.get_provider_chat_config( model=model, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 70314a2e823..05658ce2d15 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -40870,6 +40870,7 @@ "output_cost_per_token": 0.0 }, "openai.gpt-oss-120b-1:0": { + "use_bedrock_runtime_chat_completions": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -40883,6 +40884,7 @@ "supports_tool_choice": true }, "openai.gpt-oss-20b-1:0": { + "use_bedrock_runtime_chat_completions": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -58441,6 +58443,8 @@ "supports_web_search": true }, "us.openai.gpt-5.6-sol": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 4.4e-06, "input_cost_per_token_above_272k_tokens": 8.8e-06, "cache_creation_input_token_cost": 5.5e-06, @@ -58470,6 +58474,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-sol": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 4e-06, "input_cost_per_token_above_272k_tokens": 8e-06, "cache_creation_input_token_cost": 5e-06, @@ -58499,6 +58505,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-terra": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, "cache_creation_input_token_cost": 2.75e-06, @@ -58528,6 +58536,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-terra": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "cache_creation_input_token_cost": 2.5e-06, @@ -58557,6 +58567,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-luna": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2.2e-07, "input_cost_per_token_above_272k_tokens": 4.4e-07, "cache_creation_input_token_cost": 2.75e-07, @@ -58586,6 +58598,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-luna": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, "cache_creation_input_token_cost": 2.5e-07, diff --git a/litellm/utils.py b/litellm/utils.py index b724313641f..82c8069770d 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -401,7 +401,7 @@ if TYPE_CHECKING: BaseVectorStoreFilesConfig, ) from litellm.llms.base_llm.videos.transformation import BaseVideoConfig - from litellm.llms.bedrock.common_utils import BedrockModelInfo + from litellm.llms.bedrock.common_utils import BedrockModelInfo, BedrockRoute from litellm.llms.bedrock.embed.amazon_nova_transformation import ( AmazonNovaEmbeddingConfig, ) @@ -3350,6 +3350,17 @@ def _should_drop_param(k, additional_drop_params) -> bool: return False +def _bedrock_route_for_request( + model: str, passed_params: Mapping[str, object], additional_drop_params: list | None +) -> BedrockRoute: + from litellm.llms.bedrock.common_utils import BedrockModelInfo + + return BedrockModelInfo.get_bedrock_route( + model, + {k: v for k, v in passed_params.items() if not _should_drop_param(k, additional_drop_params)}, + ) + + def _get_non_default_params(passed_params: dict, default_params: dict, additional_drop_params: list | None) -> dict: non_default_params: Final = {} for k, v in passed_params.items(): @@ -4401,9 +4412,17 @@ def get_optional_params( message=f"{custom_llm_provider} does not support parameters: {list(unsupported_params.keys())}, for model={model}. To drop these, set `litellm.drop_params=True` or for proxy:\n\n`litellm_settings:\n drop_params: true`\n. \n If you want to use these params dynamically send allowed_openai_params={list(unsupported_params.keys())} in your request.", ) + bedrock_route: Final = ( + _bedrock_route_for_request(model, passed_params, additional_drop_params) + if custom_llm_provider == "bedrock" + else None + ) get_supported_openai_params: Final = getattr(sys.modules[__name__], "get_supported_openai_params") - supported_params = get_supported_openai_params( - model=model, custom_llm_provider=custom_llm_provider, base_model=base_model + supported_params = ( + litellm.AmazonConverseConfig().get_supported_openai_params(model=model) + if bedrock_route == "converse" + and isinstance(provider_config, litellm.AmazonBedrockRuntimeChatCompletionsConfig) + else get_supported_openai_params(model=model, custom_llm_provider=custom_llm_provider, base_model=base_model) ) if supported_params is None: supported_params = get_supported_openai_params(model=model, custom_llm_provider="openai") @@ -4573,7 +4592,6 @@ def get_optional_params( ) elif custom_llm_provider == "bedrock": BedrockModelInfo: Final = getattr(sys.modules[__name__], "BedrockModelInfo") - bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model) bedrock_base_model: Final = BedrockModelInfo.get_base_model(model) if bedrock_route == "converse" or bedrock_route == "converse_like": optional_params = litellm.AmazonConverseConfig().map_openai_params( diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 70314a2e823..05658ce2d15 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -40870,6 +40870,7 @@ "output_cost_per_token": 0.0 }, "openai.gpt-oss-120b-1:0": { + "use_bedrock_runtime_chat_completions": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -40883,6 +40884,7 @@ "supports_tool_choice": true }, "openai.gpt-oss-20b-1:0": { + "use_bedrock_runtime_chat_completions": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -58441,6 +58443,8 @@ "supports_web_search": true }, "us.openai.gpt-5.6-sol": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 4.4e-06, "input_cost_per_token_above_272k_tokens": 8.8e-06, "cache_creation_input_token_cost": 5.5e-06, @@ -58470,6 +58474,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-sol": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 4e-06, "input_cost_per_token_above_272k_tokens": 8e-06, "cache_creation_input_token_cost": 5e-06, @@ -58499,6 +58505,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-terra": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, "cache_creation_input_token_cost": 2.75e-06, @@ -58528,6 +58536,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-terra": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "cache_creation_input_token_cost": 2.5e-06, @@ -58557,6 +58567,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-luna": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2.2e-07, "input_cost_per_token_above_272k_tokens": 4.4e-07, "cache_creation_input_token_cost": 2.75e-07, @@ -58586,6 +58598,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-luna": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, "cache_creation_input_token_cost": 2.5e-07, diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index eca89e23887..9db1c8eedc4 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -74,6 +74,9 @@ "xhigh" ] }, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": { + "type": "boolean" + }, "cache_creation_input_audio_token_cost": { "type": "number", "minimum": 0 diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 7724b3f0a52..230e323ace3 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -1,7 +1,6 @@ -"""Native Bedrock Runtime Chat Completions: Grok stays on /openai/v1/chat/completions.""" +"""Native Bedrock Runtime Chat Completions: Grok, gpt-oss and GPT-5.6 stay on /openai/v1/chat/completions.""" import json -from unittest.mock import patch import httpx import pytest @@ -9,25 +8,28 @@ import pytest import litellm from litellm.llms.bedrock.chat.chat_completions.transformation import ( AmazonBedrockRuntimeChatCompletionsConfig, + BedrockRuntimeChatCompletionsStreamingHandler, + ReasoningTagSplitter, + split_reasoning_tag, + with_max_completion_tokens, ) from litellm.llms.bedrock.common_utils import ( + BEDROCK_CONVERSE_ONLY_REQUEST_KEYS, BedrockModelInfo, + bedrock_request_needs_converse, get_bedrock_chat_config, uses_bedrock_runtime_chat_completions, ) +from litellm.llms.custom_httpx.http_handler import HTTPHandler @pytest.fixture def local_cost_map(monkeypatch): - original_model_cost = litellm.model_cost - try: - monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true") - litellm.model_cost = litellm.get_model_cost_map(url="") - litellm.get_model_info.cache_clear() - yield - finally: - litellm.model_cost = original_model_cost - litellm.get_model_info.cache_clear() + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + litellm.get_model_info.cache_clear() + yield + litellm.get_model_info.cache_clear() @pytest.mark.parametrize( @@ -103,7 +105,27 @@ def test_transform_request_is_openai_chat_body_not_converse(): assert "messages" in body -def test_completion_posts_runtime_chat_completions(local_cost_map, monkeypatch): +def _chat_completion_json(content, model, tool_calls=None): + message = {"role": "assistant", "content": content, **({"tool_calls": tool_calls} if tool_calls else {})} + return { + "id": "chatcmpl-test", + "object": "chat.completion", + "created": 1733529600, + "model": model, + "choices": [{"index": 0, "message": message, "finish_reason": "tool_calls" if tool_calls else "stop"}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + } + + +CONVERSE_JSON = { + "output": {"message": {"role": "assistant", "content": [{"text": "ok"}]}}, + "stopReason": "end_turn", + "usage": {"inputTokens": 1, "outputTokens": 1, "totalTokens": 2}, +} + + +@pytest.fixture +def fake_aws_env(monkeypatch): monkeypatch.setenv("AWS_REGION_NAME", "us-west-2") monkeypatch.delenv("AWS_BEDROCK_RUNTIME_ENDPOINT", raising=False) monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False) @@ -111,40 +133,490 @@ def test_completion_posts_runtime_chat_completions(local_cost_map, monkeypatch): monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing") monkeypatch.setenv("AWS_SESSION_TOKEN", "testing") - requests: list[dict] = [] - def mock_post(self, url, data=None, json=None, headers=None, **kwargs): - requests.append({"url": url, "data": data, "json": json, "headers": headers or {}}) - return httpx.Response( - status_code=200, - json={ - "id": "chatcmpl-test", - "object": "chat.completion", - "created": 1733529600, - "model": "us.xai.grok-4.6", - "choices": [ - { - "index": 0, - "message": {"role": "assistant", "content": "ok"}, - "finish_reason": "stop", - } - ], - "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, - }, - request=httpx.Request("POST", url), - ) +def _recording_client(**response_kwargs): + requests: list[httpx.Request] = [] - with patch("litellm.llms.custom_httpx.http_handler.HTTPHandler.post", mock_post): - response = litellm.completion( - model="us.xai.grok-4.6", - messages=[{"role": "user", "content": "hello"}], - ) + def handle(request): + requests.append(request) + return httpx.Response(200, **response_kwargs) + + return requests, HTTPHandler(client=httpx.Client(transport=httpx.MockTransport(handle))) + + +def test_completion_posts_runtime_chat_completions(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "us.xai.grok-4.6")) + response = litellm.completion( + model="us.xai.grok-4.6", + messages=[{"role": "user", "content": "hello"}], + client=client, + ) assert response.choices[0].message.content == "ok" assert len(requests) == 1 - assert requests[0]["url"] == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" - raw = requests[0]["data"] - body = json.loads(raw) if isinstance(raw, (str, bytes, bytearray)) else (requests[0]["json"] or {}) + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) assert body["model"] == "us.xai.grok-4.6" assert body["messages"] == [{"role": "user", "content": "hello"}] assert "inferenceConfig" not in body + + +OPENAI_RUNTIME_MODELS = ( + "openai.gpt-oss-20b-1:0", + "openai.gpt-oss-120b-1:0", + "us.openai.gpt-5.6-sol", + "global.openai.gpt-5.6-sol", + "us.openai.gpt-5.6-terra", + "global.openai.gpt-5.6-terra", + "us.openai.gpt-5.6-luna", + "global.openai.gpt-5.6-luna", +) +GET_WEATHER_TOOL = { + "type": "function", + "function": { + "name": "get_weather", + "parameters": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"]}, + }, +} + + +@pytest.mark.parametrize("model", [*OPENAI_RUNTIME_MODELS, "bedrock/openai.gpt-oss-20b-1:0"]) +def test_openai_runtime_models_use_chat_completions_route(local_cost_map, model): + assert uses_bedrock_runtime_chat_completions(model) is True + assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions" + assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig) + + +@pytest.mark.parametrize("model", ["us.amazon.nova-micro-v1:0", "us.anthropic.claude-haiku-4-5-20251001-v1:0"]) +def test_nova_and_claude_stay_on_converse(local_cost_map, model): + assert uses_bedrock_runtime_chat_completions(model) is False + assert BedrockModelInfo.get_bedrock_route(model, {"tools": [GET_WEATHER_TOOL]}) == "converse" + + +@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "global.openai.gpt-5.6-sol"]) +def test_guardrail_config_falls_back_to_converse(local_cost_map, model): + guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"} + assert bedrock_request_needs_converse(model, {"guardrailConfig": guardrail}) is True + assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": guardrail}) == "converse" + assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": None}) == "chat_completions" + + +@pytest.mark.parametrize( + "request_params, expected_route", + [ + ({"tools": [GET_WEATHER_TOOL]}, "converse"), + ({"tools": [GET_WEATHER_TOOL], "reasoning_effort": "low"}, "converse"), + ({"tools": [GET_WEATHER_TOOL], "reasoning_effort": None}, "converse"), + ({"tools": [GET_WEATHER_TOOL], "reasoning_effort": "none"}, "chat_completions"), + ({"reasoning_effort": "low"}, "chat_completions"), + ({"tools": None, "reasoning_effort": "low"}, "chat_completions"), + ({"tools": [], "reasoning_effort": "low"}, "chat_completions"), + ({}, "chat_completions"), + ], +) +def test_gpt56_tools_need_reasoning_none_on_chat_completions(local_cost_map, request_params, expected_route): + assert BedrockModelInfo.get_bedrock_route("global.openai.gpt-5.6-sol", request_params) == expected_route + assert BedrockModelInfo.get_bedrock_route("bedrock/us.openai.gpt-5.6-terra", request_params) == expected_route + + +@pytest.mark.parametrize("reasoning_effort", ["low", "high", None]) +def test_gpt_oss_tools_with_any_reasoning_effort_stay_on_chat_completions(local_cost_map, reasoning_effort): + params = {"tools": [GET_WEATHER_TOOL], "reasoning_effort": reasoning_effort} + assert bedrock_request_needs_converse("openai.gpt-oss-120b-1:0", params) is False + assert BedrockModelInfo.get_bedrock_route("openai.gpt-oss-120b-1:0", params) == "chat_completions" + + +def test_explicit_converse_prefix_wins_for_openai_models(local_cost_map): + assert BedrockModelInfo.get_bedrock_route("bedrock/converse/openai.gpt-oss-20b-1:0") == "converse" + assert BedrockModelInfo.get_bedrock_route("converse/global.openai.gpt-5.6-sol", {}) == "converse" + + +def test_map_openai_params_sends_max_tokens_as_max_completion_tokens(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"max_tokens": 64, "temperature": 0.1}, + optional_params={}, + model="global.openai.gpt-5.6-sol", + drop_params=False, + ) + assert mapped == {"max_completion_tokens": 64, "temperature": 0.1} + + +def test_map_openai_params_keeps_explicit_max_completion_tokens(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"max_tokens": 64, "max_completion_tokens": 32}, + optional_params={}, + model="openai.gpt-oss-20b-1:0", + drop_params=False, + ) + assert mapped == {"max_completion_tokens": 32} + + +def test_with_max_completion_tokens_leaves_other_params_alone(): + assert with_max_completion_tokens({"temperature": 0.5}) == {"temperature": 0.5} + + +def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + assert "reasoning_effort" in cfg.get_supported_openai_params("global.openai.gpt-5.6-sol") + assert "reasoning_effort" in cfg.get_supported_openai_params("openai.gpt-oss-20b-1:0") + + +def test_split_reasoning_tag_splits_leading_tag(): + assert split_reasoning_tag("plan it\n\n\nHello") == ("plan it\n", "Hello") + + +def test_split_reasoning_tag_drops_an_empty_tag(): + assert split_reasoning_tag("Hello") == (None, "Hello") + + +@pytest.mark.parametrize( + "content", + [ + "plan it\n\n\nHello", + "never closed", + "later", + "", + ], +) +@pytest.mark.parametrize("chunk_size", [1, 3, 7]) +def test_split_reasoning_tag_matches_the_streamed_split(content, chunk_size): + chunks = [content[start : start + chunk_size] for start in range(0, len(content), chunk_size)] + streamed_reasoning, streamed_content = _run_splitter(chunks) + + assert split_reasoning_tag(content) == (streamed_reasoning or None, streamed_content) + + +def test_split_reasoning_tag_passes_plain_content_through(): + assert split_reasoning_tag("Hello") == (None, "Hello") + + +def test_split_reasoning_tag_ignores_tag_after_content_starts(): + content = "Hello not mine" + assert split_reasoning_tag(content) == (None, content) + + +def _run_splitter(chunks): + state = ReasoningTagSplitter() + reasoning = "" + content = "" + for chunk in chunks: + state, fed_reasoning, fed_content = state.feed(chunk) + reasoning += fed_reasoning + content += fed_content + state, flushed_reasoning, flushed_content = state.flush() + return reasoning + flushed_reasoning, content + flushed_content + + +def test_reasoning_tag_splitter_handles_tags_split_across_chunks(): + assert _run_splitter(["I think", " so\n\nHel", "lo"]) == ("I think so", "Hello") + + +def test_reasoning_tag_splitter_passes_plain_content_through(): + assert _run_splitter(["Hel", "lo later"]) == ("", "Hello later") + + +def test_reasoning_tag_splitter_flushes_unclosed_reasoning(): + assert _run_splitter(["never clo", "sed"]) == ("never closed", "") + + +def test_reasoning_tag_splitter_releases_a_false_tag_prefix(): + assert _run_splitter(["<", "b>x"]) == ("", "x") + + +def _stream_chunk(delta, finish_reason=None, index=0): + return { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1733529600, + "model": "openai.gpt-oss-20b-1:0", + "choices": [{"index": index, "delta": delta, "finish_reason": finish_reason}], + } + + +def test_streaming_handler_splits_reasoning_deltas_per_choice(): + handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) + + first = handler.chunk_parser(_stream_chunk({"role": "assistant", "content": "I think"})) + assert first.choices[0].delta.reasoning_content == "I think" + assert not first.choices[0].delta.content + + second = handler.chunk_parser(_stream_chunk({"content": " so\n\nHello"})) + assert second.choices[0].delta.reasoning_content == " so" + assert second.choices[0].delta.content == "Hello" + + tool_call = {"index": 0, "id": "call_0", "type": "function", "function": {"name": "get_weather", "arguments": "{}"}} + third = handler.chunk_parser(_stream_chunk({"content": None, "tool_calls": [tool_call]})) + assert third.choices[0].delta.tool_calls[0].function.name == "get_weather" + + last = handler.chunk_parser(_stream_chunk({}, finish_reason="stop")) + assert last.choices[0].finish_reason == "stop" + + +def _reasoning_of(parsed): + return getattr(parsed.choices[0].delta, "reasoning_content", None) + + +def test_streaming_handler_keeps_split_state_per_choice_index(): + handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) + + opened = handler.chunk_parser(_stream_chunk({"content": "first"}, index=0)) + assert _reasoning_of(opened) == "first" + + plain = handler.chunk_parser(_stream_chunk({"content": "plain answer"}, index=1)) + assert _reasoning_of(plain) is None + assert plain.choices[0].delta.content == "plain answer" + + still_reasoning = handler.chunk_parser(_stream_chunk({"content": " more"}, index=0)) + assert _reasoning_of(still_reasoning) == " more" + assert not still_reasoning.choices[0].delta.content + + +def test_streaming_handler_flushes_held_text_on_an_empty_final_delta(): + handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) + + held = handler.chunk_parser(_stream_chunk({"content": "almost doneplan\n\nHi", "openai.gpt-oss-20b-1:0") + ) + response = litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + max_tokens=64, + reasoning_effort="low", + tools=[GET_WEATHER_TOOL], + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) + assert body["model"] == "openai.gpt-oss-20b-1:0" + assert body["max_completion_tokens"] == 64 + assert "max_tokens" not in body + assert body["reasoning_effort"] == "low" + assert body["tools"] == [GET_WEATHER_TOOL] + assert response.choices[0].message.reasoning_content == "plan" + assert response.choices[0].message.content == "Hi" + + +def test_gpt56_tools_with_reasoning_effort_go_to_converse(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + response = litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "hello"}], + tools=[GET_WEATHER_TOOL], + reasoning_effort="low", + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse") + assert json.loads(requests[0].content)["toolConfig"]["tools"][0]["toolSpec"]["name"] == "get_weather" + assert response.choices[0].message.content == "ok" + + +def test_gpt56_tools_with_reasoning_none_stay_on_chat_completions(local_cost_map, fake_aws_env): + tool_calls = [ + {"id": "call_0", "type": "function", "function": {"name": "get_weather", "arguments": '{"city": "Paris"}'}} + ] + requests, client = _recording_client(json=_chat_completion_json(None, "global.openai.gpt-5.6-sol", tool_calls)) + response = litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "weather in Paris"}], + tools=[GET_WEATHER_TOOL], + reasoning_effort="none", + max_tokens=64, + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) + assert body["tools"] == [GET_WEATHER_TOOL] + assert body["reasoning_effort"] == "none" + assert body["max_completion_tokens"] == 64 + assert response.choices[0].message.tool_calls[0].function.name == "get_weather" + + +@pytest.mark.parametrize( + "converse_only_param", + [ + {"guardrailConfig": {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"}}, + {"performanceConfig": {"latency": "optimized"}}, + {"requestMetadata": {"team": "search"}}, + {"serviceTier": {"type": "priority"}}, + ], + ids=lambda param: next(iter(param)), +) +def test_converse_only_request_keys_go_to_converse(local_cost_map, fake_aws_env, converse_only_param): + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + client=client, + **converse_only_param, + ) + + assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse") + (key, value), = converse_only_param.items() + assert json.loads(requests[0].content)[key] == value + + +def test_converse_only_keys_cover_every_converse_config_block(): + assert set(litellm.AmazonConverseConfig.get_config_blocks()) <= BEDROCK_CONVERSE_ONLY_REQUEST_KEYS + + +def test_operator_owned_request_metadata_goes_to_converse(local_cost_map, fake_aws_env, monkeypatch): + monkeypatch.setattr(litellm, "bedrock_request_metadata_fields", ["user_api_key_team_alias"]) + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + metadata={"user_api_key_team_alias": "search"}, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse") + assert json.loads(requests[0].content)["requestMetadata"] == {"user_api_key_team_alias": "search"} + + +def test_dropped_converse_only_key_keeps_the_request_on_chat_completions(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + guardrailConfig={"guardrailIdentifier": "gr-1", "guardrailVersion": "1"}, + additional_drop_params=["guardrailConfig"], + max_tokens=8, + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) + assert "guardrailConfig" not in body + assert body["max_completion_tokens"] == 8 + assert "inferenceConfig" not in body + + +def test_dropped_tools_keep_gpt56_reasoning_request_on_chat_completions(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "global.openai.gpt-5.6-sol")) + litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "hello"}], + tools=[GET_WEATHER_TOOL], + reasoning_effort="low", + additional_drop_params=["tools"], + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) + assert "tools" not in body + assert body["reasoning_effort"] == "low" + + +def test_legacy_functions_stay_on_chat_completions(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + functions=[GET_WEATHER_TOOL["function"]], + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + assert json.loads(requests[0].content)["functions"] == [GET_WEATHER_TOOL["function"]] + + +def test_converse_fallback_validates_against_converse_params(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"} + with pytest.raises(litellm.UnsupportedParamsError, match="seed"): + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + guardrailConfig=guardrail, + seed=7, + client=client, + ) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + guardrailConfig=guardrail, + seed=7, + drop_params=True, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse") + assert "seed" not in json.loads(requests[0].content) + + +def test_n_is_rejected_before_reaching_chat_completions(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) + with pytest.raises(litellm.UnsupportedParamsError, match="'n'"): + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + n=2, + client=client, + ) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + n=2, + drop_params=True, + client=client, + ) + + assert "n" not in json.loads(requests[0].content) + + +def _sse(chunks): + return ("".join(f"data: {json.dumps(chunk)}\n\n" for chunk in chunks) + "data: [DONE]\n\n").encode() + + +def test_gpt_oss_streaming_completion_splits_reasoning(local_cost_map, fake_aws_env): + chunks = ( + _stream_chunk({"role": "assistant", "content": "plan"}), + _stream_chunk({"content": "\n\nHi"}), + _stream_chunk({}, finish_reason="stop"), + ) + requests, client = _recording_client(content=_sse(chunks), headers={"content-type": "text/event-stream"}) + stream = litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + stream=True, + client=client, + ) + deltas = [chunk.choices[0].delta for chunk in stream] + + assert [str(request.url) for request in requests] == [ + "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + ] + assert json.loads(requests[0].content)["stream"] is True + assert "".join(getattr(delta, "reasoning_content", None) or "" for delta in deltas) == "plan" + assert "".join(delta.content or "" for delta in deltas) == "Hi" + + +def test_streaming_handler_keeps_native_reasoning_next_to_the_tagged_split(): + handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) + parsed = handler.chunk_parser( + _stream_chunk({"reasoning": "native ", "content": "taggedHi"}, finish_reason="stop") + ) + + assert parsed.choices[0].delta.reasoning_content == "native tagged" + assert parsed.choices[0].delta.content == "Hi" diff --git a/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py b/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py index aa0827c5ae5..a6b8ba1da1d 100644 --- a/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py +++ b/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py @@ -138,9 +138,12 @@ def _bedrock_response(model, usage): @pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id) -def test_bedrock_gpt_5_6_profiles_route_to_converse(profile, local_model_cost_map): - """GPT-5.6 is served by Converse on bedrock-runtime, never by Invoke.""" - assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "converse" +def test_bedrock_gpt_5_6_profiles_never_route_to_invoke(profile, local_model_cost_map): + """GPT-5.6 is served by bedrock-runtime's native Chat Completions, and by Converse when + the request carries function tools without reasoning_effort "none", never by Invoke.""" + assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "chat_completions" + tools_with_reasoning = {"tools": [{"type": "function", "function": {"name": "f"}}], "reasoning_effort": "low"} + assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}", tools_with_reasoning) == "converse" @pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id) diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 3336ad6d33a..963bef1114a 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -878,6 +878,8 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "supports_video_input": {"type": "boolean"}, "supports_vision": {"type": "boolean"}, "supports_web_search": {"type": "boolean"}, + "use_bedrock_runtime_chat_completions": {"type": "boolean"}, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": {"type": "boolean"}, "supports_url_context": {"type": "boolean"}, "supports_multimodal": {"type": "boolean"}, "uses_embed_content": {"type": "boolean"}, From a1c089f10771d4cdbdf6fd4920b6eb0c3488cd3d Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:47:54 -0700 Subject: [PATCH 03/18] fix(bedrock): route gpt-oss response_format to Converse and decide the route once from the raw request --- ci_cd/generate_model_prices_schema.py | 2 - .../chat/chat_completions/transformation.py | 5 +- litellm/llms/bedrock/common_utils.py | 57 ++++++-- litellm/main.py | 7 +- ...odel_prices_and_context_window_backup.json | 42 +++--- litellm/types/completion.py | 3 +- litellm/utils.py | 9 +- model_prices_and_context_window.json | 42 +++--- model_prices_and_context_window.schema.json | 15 ++- ...bedrock_chat_completions_transformation.py | 125 +++++++++++++++++- tests/test_litellm/test_utils.py | 5 +- tests/test_litellm/types/test_completion.py | 1 + 12 files changed, 248 insertions(+), 65 deletions(-) diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index 648cceb79b0..8eec07dadda 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -30,8 +30,6 @@ EXTRA_BOOLEAN_KEYS = frozenset( "gemini_audio_only_live", "uses_embed_content", "use_openai_responses_path", - "use_bedrock_runtime_chat_completions", - "bedrock_runtime_chat_completions_tools_require_reasoning_none", "bedrock_converse_supports_strict_tools", "thinking_always_on", } diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 49218270064..18ba06fee0b 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -3,7 +3,7 @@ Native OpenAI Chat Completions on Amazon Bedrock Runtime. AWS serves this surface at ``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions`` -for the models whose price-map entry sets ``use_bedrock_runtime_chat_completions`` +for the models whose price-map entry sets ``supports_bedrock_runtime_chat_completions`` (Grok 4.6, gpt-oss, the GPT-5.6 family): chat completions stay chat completions instead of being rewritten to Converse. @@ -19,6 +19,7 @@ from types import MappingProxyType from typing import TYPE_CHECKING, Final, Literal import httpx +from typing_extensions import assert_never import litellm from litellm.llms.base_llm.chat.transformation import BaseLLMException @@ -72,6 +73,8 @@ class ReasoningTagSplitter: return self._feed_start(self.pending + text) case "reasoning": return self._feed_reasoning(self.pending + text) + case _: + assert_never(self.phase) def _feed_start(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]: if buffered.startswith(REASONING_OPEN_TAG): diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 124ef7c646c..df30eddd856 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -803,22 +803,33 @@ def _bedrock_price_map_flag(model: str, flag: str) -> bool: def uses_bedrock_runtime_chat_completions(model: str) -> bool: """Whether this Bedrock model should use runtime native Chat Completions. - Data-driven from the price-map ``use_bedrock_runtime_chat_completions`` flag + Data-driven from the price-map ``supports_bedrock_runtime_chat_completions`` flag so onboarding a model is a JSON change. Explicit ``converse/`` still wins in ``get_bedrock_route`` because prefix routes are checked first, and a request that needs a Converse-only feature (``bedrock_request_needs_converse``) is served by Converse even on a flagged model. """ - return _bedrock_price_map_flag(model, "use_bedrock_runtime_chat_completions") + return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions") -def bedrock_runtime_chat_completions_tools_require_reasoning_none(model: str) -> bool: - """Whether AWS's native Chat Completions only serves this model's function tools with ``reasoning_effort="none"``. +def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool: + """Whether AWS's native Chat Completions serves this model's function tools with any ``reasoning_effort``. - Data-driven from the price-map ``bedrock_runtime_chat_completions_tools_require_reasoning_none`` - flag (the GPT-5.6 family). Converse serves tools with any effort, so those requests fall back to it. + Data-driven from the price-map ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` + flag (gpt-oss, Grok). Without it AWS only takes tools with ``reasoning_effort="none"`` + (the GPT-5.6 family), and Converse serves tools with any effort, so those requests fall back to it. """ - return _bedrock_price_map_flag(model, "bedrock_runtime_chat_completions_tools_require_reasoning_none") + return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_tools_with_reasoning") + + +def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> bool: + """Whether AWS's native Chat Completions enforces a ``response_format`` schema for this model. + + Data-driven from the price-map ``supports_bedrock_runtime_chat_completions_response_format`` flag + (GPT-5.6, Grok). Without it AWS accepts the field and answers with unconstrained text (gpt-oss), so + Converse, which emulates the schema through a forced ``json_tool_call`` tool, serves those requests. + """ + return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_response_format") BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( @@ -826,26 +837,52 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( ) +def _response_format_constrains_output(response_format: object) -> bool: + if response_format is None: + return False + return not (isinstance(response_format, Mapping) and response_format.get("type") == "text") + + def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: """Whether a request on a runtime-Chat-Completions model must still be served by Converse. Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``) are rejected as malformed input by AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, - and function tools on a ``bedrock_runtime_chat_completions_tools_require_reasoning_none`` model are - rejected there unless ``reasoning_effort`` is exactly ``"none"``. + function tools on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` + are rejected there unless ``reasoning_effort`` is exactly ``"none"``, and a constraining + ``response_format`` on a model without ``supports_bedrock_runtime_chat_completions_response_format`` + is only honored by Converse. """ if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS): return True if bedrock_request_metadata_is_owned(): return True + if _response_format_constrains_output( + request_params.get("response_format") + ) and not bedrock_runtime_chat_completions_enforces_response_format(model): + return True if not request_params.get("tools"): return False return ( - bedrock_runtime_chat_completions_tools_require_reasoning_none(model) + not bedrock_runtime_chat_completions_serves_tools_with_reasoning(model) and request_params.get("reasoning_effort") != "none" ) +def bedrock_route_for_request( + model: str, request_params: Mapping[str, object], additional_drop_params: Sequence[str] | None +) -> BedrockRoute: + """The route for one request, decided from the caller's raw params before any provider mapping. + + Param mapping and dispatch both call this with the same inputs, so a request that falls back to + Converse is mapped with the Converse config and sent to Converse, never one without the other. + """ + dropped: Final = frozenset(additional_drop_params or ()) + return BedrockModelInfo.get_bedrock_route( + model, {key: value for key, value in request_params.items() if key not in dropped} + ) + + def strip_bedrock_throughput_suffix(model: str) -> str: """Strip throughput tier suffixes and context window suffixes from Bedrock model names.""" import re diff --git a/litellm/main.py b/litellm/main.py index ed78f209ca2..539479fb224 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -104,7 +104,7 @@ from litellm.llms.base_llm import BaseConfig, BaseImageGenerationConfig from litellm.llms.base_llm.base_model_iterator import ( convert_model_response_to_streaming, ) -from litellm.llms.bedrock.common_utils import BedrockModelInfo +from litellm.llms.bedrock.common_utils import BedrockModelInfo, bedrock_route_for_request from litellm.llms.cohere.common_utils import CohereModelInfo from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler, http2_enabled from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config @@ -4078,7 +4078,9 @@ def _complete_bedrock(ctx: _CompletionDispatchContext) -> _CompletionDispatchRes if "aws_region_name" not in optional_params or optional_params["aws_region_name"] is None: optional_params["aws_region_name"] = aws_bedrock_client.meta.region_name - bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model, optional_params) + bedrock_route: Final = bedrock_route_for_request( + model, ctx.request_params, ctx.kwargs.get("additional_drop_params") + ) if bedrock_route == "claude_platform": provider_config = ProviderConfigManager.get_provider_chat_config( model=model, @@ -5686,6 +5688,7 @@ def completion( optional_params=optional_params, organization=organization, provider_config=provider_config, + request_params={**optional_param_args, **non_default_params}, shared_session=shared_session, stream=stream, temperature=temperature, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 05658ce2d15..6dbf246fd70 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -40870,7 +40870,8 @@ "output_cost_per_token": 0.0 }, "openai.gpt-oss-120b-1:0": { - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -40884,7 +40885,8 @@ "supports_tool_choice": true }, "openai.gpt-oss-20b-1:0": { - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -47234,7 +47236,9 @@ "input_cost_per_token": 2.64e-06, "output_cost_per_token": 7.92e-06, "cache_read_input_token_cost": 6.6e-07, - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -58443,8 +58447,8 @@ "supports_web_search": true }, "us.openai.gpt-5.6-sol": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4.4e-06, "input_cost_per_token_above_272k_tokens": 8.8e-06, "cache_creation_input_token_cost": 5.5e-06, @@ -58474,8 +58478,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-sol": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4e-06, "input_cost_per_token_above_272k_tokens": 8e-06, "cache_creation_input_token_cost": 5e-06, @@ -58505,8 +58509,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-terra": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, "cache_creation_input_token_cost": 2.75e-06, @@ -58536,8 +58540,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-terra": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "cache_creation_input_token_cost": 2.5e-06, @@ -58567,8 +58571,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-luna": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-07, "input_cost_per_token_above_272k_tokens": 4.4e-07, "cache_creation_input_token_cost": 2.75e-07, @@ -58598,8 +58602,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-luna": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, "cache_creation_input_token_cost": 2.5e-07, @@ -58909,7 +58913,9 @@ "input_cost_per_token": 2.2e-06, "output_cost_per_token": 6.6e-06, "cache_read_input_token_cost": 5.5e-07, - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -58925,7 +58931,9 @@ "input_cost_per_token": 2e-06, "output_cost_per_token": 6e-06, "cache_read_input_token_cost": 5e-07, - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, diff --git a/litellm/types/completion.py b/litellm/types/completion.py index c1c6cc9ed1c..1e6cfc0ee33 100644 --- a/litellm/types/completion.py +++ b/litellm/types/completion.py @@ -1,6 +1,6 @@ from __future__ import annotations -from collections.abc import Callable, Coroutine, Iterable +from collections.abc import Callable, Coroutine, Iterable, Mapping from dataclasses import dataclass from typing import TYPE_CHECKING, Any, Literal, Union @@ -229,6 +229,7 @@ class _CompletionDispatchContext: optional_params: dict organization: str | None provider_config: BaseConfig | None + request_params: Mapping[str, object] shared_session: ClientSession | None stream: bool | None temperature: float | None diff --git a/litellm/utils.py b/litellm/utils.py index 82c8069770d..7226b9a865c 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -401,7 +401,7 @@ if TYPE_CHECKING: BaseVectorStoreFilesConfig, ) from litellm.llms.base_llm.videos.transformation import BaseVideoConfig - from litellm.llms.bedrock.common_utils import BedrockModelInfo, BedrockRoute + from litellm.llms.bedrock.common_utils import BedrockRoute from litellm.llms.bedrock.embed.amazon_nova_transformation import ( AmazonNovaEmbeddingConfig, ) @@ -3353,12 +3353,9 @@ def _should_drop_param(k, additional_drop_params) -> bool: def _bedrock_route_for_request( model: str, passed_params: Mapping[str, object], additional_drop_params: list | None ) -> BedrockRoute: - from litellm.llms.bedrock.common_utils import BedrockModelInfo + from litellm.llms.bedrock.common_utils import bedrock_route_for_request - return BedrockModelInfo.get_bedrock_route( - model, - {k: v for k, v in passed_params.items() if not _should_drop_param(k, additional_drop_params)}, - ) + return bedrock_route_for_request(model, passed_params, additional_drop_params) def _get_non_default_params(passed_params: dict, default_params: dict, additional_drop_params: list | None) -> dict: diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 05658ce2d15..6dbf246fd70 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -40870,7 +40870,8 @@ "output_cost_per_token": 0.0 }, "openai.gpt-oss-120b-1:0": { - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -40884,7 +40885,8 @@ "supports_tool_choice": true }, "openai.gpt-oss-20b-1:0": { - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -47234,7 +47236,9 @@ "input_cost_per_token": 2.64e-06, "output_cost_per_token": 7.92e-06, "cache_read_input_token_cost": 6.6e-07, - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -58443,8 +58447,8 @@ "supports_web_search": true }, "us.openai.gpt-5.6-sol": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4.4e-06, "input_cost_per_token_above_272k_tokens": 8.8e-06, "cache_creation_input_token_cost": 5.5e-06, @@ -58474,8 +58478,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-sol": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4e-06, "input_cost_per_token_above_272k_tokens": 8e-06, "cache_creation_input_token_cost": 5e-06, @@ -58505,8 +58509,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-terra": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, "cache_creation_input_token_cost": 2.75e-06, @@ -58536,8 +58540,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-terra": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "cache_creation_input_token_cost": 2.5e-06, @@ -58567,8 +58571,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-luna": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-07, "input_cost_per_token_above_272k_tokens": 4.4e-07, "cache_creation_input_token_cost": 2.75e-07, @@ -58598,8 +58602,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-luna": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, "cache_creation_input_token_cost": 2.5e-07, @@ -58909,7 +58913,9 @@ "input_cost_per_token": 2.2e-06, "output_cost_per_token": 6.6e-06, "cache_read_input_token_cost": 5.5e-07, - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -58925,7 +58931,9 @@ "input_cost_per_token": 2e-06, "output_cost_per_token": 6e-06, "cache_read_input_token_cost": 5e-07, - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index 9db1c8eedc4..1a3523cdff2 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -74,9 +74,6 @@ "xhigh" ] }, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": { - "type": "boolean" - }, "cache_creation_input_audio_token_cost": { "type": "number", "minimum": 0 @@ -830,6 +827,15 @@ "supports_audio_output": { "type": "boolean" }, + "supports_bedrock_runtime_chat_completions": { + "type": "boolean" + }, + "supports_bedrock_runtime_chat_completions_response_format": { + "type": "boolean" + }, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": { + "type": "boolean" + }, "supports_computer_use": { "type": "boolean" }, @@ -1000,9 +1006,6 @@ "minimum": 0, "description": "Provider default tokens-per-minute limit." }, - "use_bedrock_runtime_chat_completions": { - "type": "boolean" - }, "use_openai_responses_path": { "type": "boolean" }, diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 230e323ace3..ce2e5b7e8f7 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -4,6 +4,7 @@ import json import httpx import pytest +from pydantic import BaseModel import litellm from litellm.llms.bedrock.chat.chat_completions.transformation import ( @@ -17,6 +18,7 @@ from litellm.llms.bedrock.common_utils import ( BEDROCK_CONVERSE_ONLY_REQUEST_KEYS, BedrockModelInfo, bedrock_request_needs_converse, + bedrock_route_for_request, get_bedrock_chat_config, uses_bedrock_runtime_chat_completions, ) @@ -471,7 +473,7 @@ def test_converse_only_request_keys_go_to_converse(local_cost_map, fake_aws_env, ) assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse") - (key, value), = converse_only_param.items() + ((key, value),) = converse_only_param.items() assert json.loads(requests[0].content)[key] == value @@ -620,3 +622,124 @@ def test_streaming_handler_keeps_native_reasoning_next_to_the_tagged_split(): assert parsed.choices[0].delta.reasoning_content == "native tagged" assert parsed.choices[0].delta.content == "Hi" + + +RESPONSE_FORMAT_JSON_SCHEMA = { + "type": "json_schema", + "json_schema": { + "name": "answer", + "schema": {"type": "object", "properties": {"word": {"type": "string"}}, "required": ["word"]}, + "strict": True, + }, +} + + +class Answer(BaseModel): + word: str + + +@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "bedrock/openai.gpt-oss-120b-1:0"]) +@pytest.mark.parametrize( + "response_format, expected_route", + [ + (RESPONSE_FORMAT_JSON_SCHEMA, "converse"), + ({"type": "json_object"}, "converse"), + (Answer, "converse"), + ({"type": "text"}, "chat_completions"), + (None, "chat_completions"), + ], + ids=["json_schema", "json_object", "pydantic", "text", "none"], +) +def test_gpt_oss_response_format_falls_back_to_converse(local_cost_map, model, response_format, expected_route): + params = {"response_format": response_format} + assert bedrock_request_needs_converse(model, params) is (expected_route == "converse") + assert BedrockModelInfo.get_bedrock_route(model, params) == expected_route + + +@pytest.mark.parametrize("model", ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"]) +def test_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model): + params = {"response_format": RESPONSE_FORMAT_JSON_SCHEMA} + assert bedrock_request_needs_converse(model, params) is False + assert BedrockModelInfo.get_bedrock_route(model, params) == "chat_completions" + + +SYNTHETIC_NATIVE_MODEL = "vendor.native-model-v1:0" + + +@pytest.mark.parametrize( + "capability_flags, request_params, needs_converse", + [ + ({}, {"tools": [GET_WEATHER_TOOL], "reasoning_effort": "low"}, True), + ({}, {"tools": [GET_WEATHER_TOOL]}, True), + ({}, {"tools": [GET_WEATHER_TOOL], "reasoning_effort": "none"}, False), + ( + {"supports_bedrock_runtime_chat_completions_tools_with_reasoning": True}, + {"tools": [GET_WEATHER_TOOL], "reasoning_effort": "low"}, + False, + ), + ({}, {"response_format": RESPONSE_FORMAT_JSON_SCHEMA}, True), + ( + {"supports_bedrock_runtime_chat_completions_response_format": True}, + {"response_format": RESPONSE_FORMAT_JSON_SCHEMA}, + False, + ), + ( + {"supports_bedrock_runtime_chat_completions_response_format": True}, + {"response_format": RESPONSE_FORMAT_JSON_SCHEMA, "tools": [GET_WEATHER_TOOL], "reasoning_effort": "low"}, + True, + ), + ], +) +def test_capability_flags_are_read_from_the_cost_map(monkeypatch, capability_flags, request_params, needs_converse): + entry = { + "litellm_provider": "bedrock_converse", + "supports_bedrock_runtime_chat_completions": True, + **capability_flags, + } + monkeypatch.setattr(litellm, "model_cost", {SYNTHETIC_NATIVE_MODEL: entry}) + assert bedrock_request_needs_converse(SYNTHETIC_NATIVE_MODEL, request_params) is needs_converse + route = bedrock_route_for_request(SYNTHETIC_NATIVE_MODEL, request_params, None) + assert (route == "chat_completions") is (not needs_converse) + + +def test_route_for_request_ignores_dropped_params(local_cost_map): + params = {"response_format": RESPONSE_FORMAT_JSON_SCHEMA, "guardrailConfig": {"guardrailIdentifier": "gr-1"}} + assert bedrock_route_for_request("openai.gpt-oss-20b-1:0", params, None) == "converse" + assert bedrock_route_for_request("openai.gpt-oss-20b-1:0", params, ["guardrailConfig"]) == "converse" + assert ( + bedrock_route_for_request("openai.gpt-oss-20b-1:0", params, ["guardrailConfig", "response_format"]) + == "chat_completions" + ) + + +def test_gpt_oss_response_format_goes_to_converse_with_json_tool_call(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "Reply with the single word pong."}], + response_format=RESPONSE_FORMAT_JSON_SCHEMA, + max_tokens=64, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse") + body = json.loads(requests[0].content) + assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "json_tool_call" + assert body["toolConfig"]["toolChoice"] == {"tool": {"name": "json_tool_call"}} + assert body["inferenceConfig"]["maxTokens"] == 64 + assert "response_format" not in body + assert "max_completion_tokens" not in body + + +def test_gpt56_response_format_is_sent_as_is_on_chat_completions(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json('{"word": "pong"}', "global.openai.gpt-5.6-sol")) + response = litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "Reply with the single word pong."}], + response_format=RESPONSE_FORMAT_JSON_SCHEMA, + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + assert json.loads(requests[0].content)["response_format"] == RESPONSE_FORMAT_JSON_SCHEMA + assert response.choices[0].message.content == '{"word": "pong"}' diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 963bef1114a..d049d83c6a9 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -878,8 +878,9 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "supports_video_input": {"type": "boolean"}, "supports_vision": {"type": "boolean"}, "supports_web_search": {"type": "boolean"}, - "use_bedrock_runtime_chat_completions": {"type": "boolean"}, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": {"type": "boolean"}, + "supports_bedrock_runtime_chat_completions": {"type": "boolean"}, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": {"type": "boolean"}, + "supports_bedrock_runtime_chat_completions_response_format": {"type": "boolean"}, "supports_url_context": {"type": "boolean"}, "supports_multimodal": {"type": "boolean"}, "uses_embed_content": {"type": "boolean"}, diff --git a/tests/test_litellm/types/test_completion.py b/tests/test_litellm/types/test_completion.py index cd51913c5dd..482dd351ad1 100644 --- a/tests/test_litellm/types/test_completion.py +++ b/tests/test_litellm/types/test_completion.py @@ -181,6 +181,7 @@ def _build_dispatch_context() -> _CompletionDispatchContext: optional_params={}, organization=None, provider_config=None, + request_params={}, shared_session=None, stream=None, temperature=None, From b90c2f113d1091896bb80988c790898320678f00 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 19 Sep 2026 20:39:18 -0700 Subject: [PATCH 04/18] fix(bedrock): serve region-path and GovCloud gpt-oss ids on native Chat Completions The cost-map parity tests require every regional variant of a flagged id to carry the same supports_ flags, so the six us-gov gpt-oss entries now carry the native-route flags too. A region path in the model name (bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0) is routing, not a different model: the route is looked up on the id after the path, the path's region picks the endpoint and the SigV4 scope, an explicit aws_region_name still wins, and the body carries the bare id AWS expects --- .../chat/chat_completions/transformation.py | 18 ++++++--- litellm/llms/bedrock/common_utils.py | 18 ++++++++- ...odel_prices_and_context_window_backup.json | 12 ++++++ model_prices_and_context_window.json | 12 ++++++ ...bedrock_chat_completions_transformation.py | 38 ++++++++++++++++++- 5 files changed, 91 insertions(+), 7 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 18ba06fee0b..a7926fb252e 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -24,7 +24,7 @@ from typing_extensions import assert_never import litellm from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM -from litellm.llms.bedrock.common_utils import BedrockError, strip_bedrock_routing_prefix +from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig from litellm.types.llms.openai import AllMessageValues @@ -196,7 +196,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): if api_base is not None and "chat/completions" in api_base: return api_base.rstrip("/") aws_region_name: Final = self._aws_signer._get_aws_region_name( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public region resolver - optional_params=optional_params, model=model + optional_params=self._params_with_region_from_path(optional_params, model), model=model ) endpoint_url, _ = self._aws_signer.get_runtime_endpoint( api_base=api_base, @@ -210,6 +210,14 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): return f"{base}/chat/completions" return f"{base}/openai/v1/chat/completions" + def _params_with_region_from_path( + self, optional_params: dict, model: str | None + ) -> dict: # mutable-ok: BaseAWSLLM's region resolver and signer take a plain dict + region_from_path, _ = split_bedrock_region_path(model or "") + if region_from_path is None or optional_params.get("aws_region_name") is not None: + return optional_params + return {**optional_params, "aws_region_name": region_from_path} # mutable-ok: BaseAWSLLM takes a plain dict + def sign_request( self, headers: dict, # mutable-ok: BaseConfig signature @@ -224,7 +232,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): return self._aws_signer._sign_request( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public signer service_name="bedrock", headers=headers, - optional_params=optional_params, + optional_params=self._params_with_region_from_path(optional_params, model), request_data=request_data, api_base=api_base, api_key=api_key, @@ -268,7 +276,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): headers: dict, # mutable-ok: BaseConfig signature ) -> dict: # mutable-ok: BaseConfig signature return super().transform_request( - model=strip_bedrock_routing_prefix(model), + model=split_bedrock_region_path(model)[1], messages=messages, optional_params=self._inference_params(optional_params), litellm_params=litellm_params, @@ -284,7 +292,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): headers: dict, # mutable-ok: BaseConfig signature ) -> dict: # mutable-ok: BaseConfig signature return await super().async_transform_request( - model=strip_bedrock_routing_prefix(model), + model=split_bedrock_region_path(model)[1], messages=messages, optional_params=self._inference_params(optional_params), litellm_params=litellm_params, diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index df30eddd856..3b624b4ab02 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -795,8 +795,24 @@ def strip_bedrock_routing_prefix(model: str) -> str: return model +def split_bedrock_region_path(model: str) -> tuple[str | None, str]: + """Split a ``/`` routing path into the region and the id AWS receives. + + ``bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0`` -> ``("us-gov-west-1", "openai.gpt-oss-20b-1:0")``; + a model without a region path comes back as ``(None, )``. + """ + stripped: Final = strip_bedrock_routing_prefix(model) + region, separator, model_id = stripped.partition("/") + if separator and region in _get_all_bedrock_regions(): + return region, model_id + return None, stripped + + def _bedrock_price_map_flag(model: str, flag: str) -> bool: - entries: Final = (litellm.model_cost.get(key) for key in (model, strip_bedrock_routing_prefix(model))) + entries: Final = ( + litellm.model_cost.get(key) + for key in (model, strip_bedrock_routing_prefix(model), split_bedrock_region_path(model)[1]) + ) return any(entry is not None and entry.get(flag) is True for entry in entries) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 6dbf246fd70..db0e3a4dc18 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -47214,6 +47214,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -47227,6 +47229,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64459,6 +64463,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64472,6 +64478,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64665,6 +64673,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64678,6 +64688,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 6dbf246fd70..db0e3a4dc18 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -47214,6 +47214,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -47227,6 +47229,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64459,6 +64463,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64472,6 +64478,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64665,6 +64673,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64678,6 +64688,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index ce2e5b7e8f7..83fe7628e0a 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -163,6 +163,33 @@ def test_completion_posts_runtime_chat_completions(local_cost_map, fake_aws_env) assert "inferenceConfig" not in body +def test_region_path_sends_the_bare_model_id_to_the_path_region(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) + litellm.completion( + model="bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-gov-west-1.amazonaws.com/openai/v1/chat/completions" + assert json.loads(requests[0].content)["model"] == "openai.gpt-oss-20b-1:0" + assert "/us-gov-west-1/bedrock/aws4_request" in requests[0].headers["Authorization"] + + +def test_explicit_aws_region_name_wins_over_the_region_path(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) + litellm.completion( + model="bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + aws_region_name="us-gov-east-1", + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-gov-east-1.amazonaws.com/openai/v1/chat/completions" + assert json.loads(requests[0].content)["model"] == "openai.gpt-oss-20b-1:0" + assert "/us-gov-east-1/bedrock/aws4_request" in requests[0].headers["Authorization"] + + OPENAI_RUNTIME_MODELS = ( "openai.gpt-oss-20b-1:0", "openai.gpt-oss-120b-1:0", @@ -182,7 +209,16 @@ GET_WEATHER_TOOL = { } -@pytest.mark.parametrize("model", [*OPENAI_RUNTIME_MODELS, "bedrock/openai.gpt-oss-20b-1:0"]) +@pytest.mark.parametrize( + "model", + [ + *OPENAI_RUNTIME_MODELS, + "bedrock/openai.gpt-oss-20b-1:0", + "us-gov.openai.gpt-oss-20b-1:0", + "bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", + "us-gov-east-1/openai.gpt-oss-120b-1:0", + ], +) def test_openai_runtime_models_use_chat_completions_route(local_cost_map, model): assert uses_bedrock_runtime_chat_completions(model) is True assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions" From 22b217bd95bb409d480d6da5a4292578e0899c7f Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 21 Sep 2026 12:53:43 -0700 Subject: [PATCH 05/18] fix(bedrock): keep params AWS refuses natively off the chat completions route Drop the params each family 400s or 503s on runtime Chat Completions from the native config's supported list (GPT-5.6 penalties, stop, and logprobs, Grok penalties, gpt-oss logit_bias) so drop_params drops them as Converse did, gate legacy functions on GPT-5.6 the same way as tools, and send an Anthropic-style thinking block to Converse, the only route that forwards it --- .../chat/chat_completions/transformation.py | 24 +++- litellm/llms/bedrock/common_utils.py | 15 +-- ...bedrock_chat_completions_transformation.py | 107 ++++++++++++++++++ 3 files changed, 138 insertions(+), 8 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index a7926fb252e..c6bf15b893f 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -38,6 +38,27 @@ if TYPE_CHECKING: REASONING_OPEN_TAG: Final = "" REASONING_CLOSE_TAG: Final = "" +CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType( + { + "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs")), + "openai.gpt-oss": frozenset(("logit_bias",)), + "xai.": frozenset(("frequency_penalty", "presence_penalty")), + } +) + + +def chat_completions_params_refused_for(model: str) -> frozenset[str]: + """The OpenAI params AWS's Chat Completions endpoint rejects for this model whatever else the request says. + + Each family answers them with a 400 (GPT-5.6, gpt-oss) or a 503 (Grok), where Converse dropped the same + params under ``drop_params``, so the native config leaves them out of its supported list and the usual + drop-or-raise handling applies before the request reaches AWS. + """ + model_id: Final = split_bedrock_region_path(model)[1] + return frozenset().union( + *(refused for family, refused in CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY.items() if family in model_id) + ) + def _held_close_tag_prefix(text: str) -> int: return next( @@ -362,7 +383,8 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): return {**validated, "OpenAI-Project": project_id} # mutable-ok: BaseConfig signature returns a dict def get_supported_openai_params(self, model: str) -> list: # mutable-ok: BaseConfig signature - base_params: Final = [param for param in super().get_supported_openai_params(model) if param != "n"] + refused: Final = {"n", *chat_completions_params_refused_for(model)} + base_params: Final = [param for param in super().get_supported_openai_params(model) if param not in refused] if "reasoning_effort" in base_params or not litellm.supports_reasoning( model=model, custom_llm_provider=self.custom_llm_provider ): diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 3b624b4ab02..8798b05cfa4 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -849,7 +849,7 @@ def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> boo BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( - ("guardrailConfig", "performanceConfig", "serviceTier", "requestMetadata", "outputConfig") + ("guardrailConfig", "performanceConfig", "serviceTier", "requestMetadata", "outputConfig", "thinking") ) @@ -862,12 +862,13 @@ def _response_format_constrains_output(response_format: object) -> bool: def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: """Whether a request on a runtime-Chat-Completions model must still be served by Converse. - Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``) are rejected as malformed input by + Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking`` + block included, which only Converse forwards as ``additionalModelRequestFields``) have no field on AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, - function tools on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` - are rejected there unless ``reasoning_effort`` is exactly ``"none"``, and a constraining - ``response_format`` on a model without ``supports_bedrock_runtime_chat_completions_response_format`` - is only honored by Converse. + function tools (``tools`` or legacy ``functions``) on a model without + ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless + ``reasoning_effort`` is exactly ``"none"``, and a constraining ``response_format`` on a model without + ``supports_bedrock_runtime_chat_completions_response_format`` is only honored by Converse. """ if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS): return True @@ -877,7 +878,7 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje request_params.get("response_format") ) and not bedrock_runtime_chat_completions_enforces_response_format(model): return True - if not request_params.get("tools"): + if not (request_params.get("tools") or request_params.get("functions")): return False return ( not bedrock_runtime_chat_completions_serves_tools_with_reasoning(model) diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 83fe7628e0a..2852ed3ae84 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -264,6 +264,26 @@ def test_gpt_oss_tools_with_any_reasoning_effort_stay_on_chat_completions(local_ assert BedrockModelInfo.get_bedrock_route("openai.gpt-oss-120b-1:0", params) == "chat_completions" +@pytest.mark.parametrize( + "request_params, expected_route", + [ + ({"functions": [GET_WEATHER_TOOL["function"]]}, "converse"), + ({"functions": [GET_WEATHER_TOOL["function"]], "reasoning_effort": "low"}, "converse"), + ({"functions": [GET_WEATHER_TOOL["function"]], "reasoning_effort": "none"}, "chat_completions"), + ({"functions": [], "reasoning_effort": "low"}, "chat_completions"), + ], +) +def test_gpt56_legacy_functions_route_like_tools(local_cost_map, request_params, expected_route): + assert BedrockModelInfo.get_bedrock_route("global.openai.gpt-5.6-sol", request_params) == expected_route + assert BedrockModelInfo.get_bedrock_route("openai.gpt-oss-120b-1:0", request_params) == "chat_completions" + + +def test_thinking_block_goes_to_converse(local_cost_map): + thinking = {"type": "enabled", "budget_tokens": 1024} + assert BedrockModelInfo.get_bedrock_route("us.xai.grok-4.6", {"thinking": thinking}) == "converse" + assert BedrockModelInfo.get_bedrock_route("us.xai.grok-4.6", {"thinking": None}) == "chat_completions" + + def test_explicit_converse_prefix_wins_for_openai_models(local_cost_map): assert BedrockModelInfo.get_bedrock_route("bedrock/converse/openai.gpt-oss-20b-1:0") == "converse" assert BedrockModelInfo.get_bedrock_route("converse/global.openai.gpt-5.6-sol", {}) == "converse" @@ -301,6 +321,54 @@ def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): assert "reasoning_effort" in cfg.get_supported_openai_params("openai.gpt-oss-20b-1:0") +@pytest.mark.parametrize( + "model, refused, kept", + [ + ( + "bedrock/global.openai.gpt-5.6-sol", + ("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs", "n"), + ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions"), + ), + ( + "us.xai.grok-4.6", + ("frequency_penalty", "presence_penalty", "n"), + ("stop", "logprobs", "top_p", "logit_bias", "reasoning_effort"), + ), + ( + "bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", + ("logit_bias", "n"), + ("frequency_penalty", "presence_penalty", "stop", "logprobs", "reasoning_effort"), + ), + ], +) +def test_supported_params_leave_out_what_each_family_refuses(local_cost_map, model, refused, kept): + supported = set(AmazonBedrockRuntimeChatCompletionsConfig().get_supported_openai_params(model)) + assert supported.isdisjoint(refused) + assert set(kept) <= supported + + +@pytest.mark.parametrize( + "model, param", + [ + ("bedrock/global.openai.gpt-5.6-sol", {"frequency_penalty": 0.5}), + ("bedrock/global.openai.gpt-5.6-sol", {"logprobs": True, "top_logprobs": 2}), + ("bedrock/us.xai.grok-4.6", {"presence_penalty": 0.5}), + ("bedrock/openai.gpt-oss-20b-1:0", {"logit_bias": {"1": 1}}), + ], + ids=lambda value: value if isinstance(value, str) else next(iter(value)), +) +def test_refused_params_are_dropped_or_refused_before_reaching_aws(local_cost_map, fake_aws_env, model, param): + requests, client = _recording_client(json=_chat_completion_json("ok", model.removeprefix("bedrock/"))) + with pytest.raises(litellm.UnsupportedParamsError, match=next(iter(param))): + litellm.completion(model=model, messages=[{"role": "user", "content": "hello"}], client=client, **param) + litellm.completion( + model=model, messages=[{"role": "user", "content": "hello"}], drop_params=True, client=client, **param + ) + + assert str(requests[0].url).endswith("/openai/v1/chat/completions") + assert param.keys().isdisjoint(json.loads(requests[0].content)) + + def test_split_reasoning_tag_splits_leading_tag(): assert split_reasoning_tag("plan it\n\n\nHello") == ("plan it\n", "Hello") @@ -579,6 +647,45 @@ def test_legacy_functions_stay_on_chat_completions(local_cost_map, fake_aws_env) assert json.loads(requests[0].content)["functions"] == [GET_WEATHER_TOOL["function"]] +def test_gpt56_legacy_functions_with_reasoning_fall_back_to_converse(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + with pytest.raises(litellm.UnsupportedParamsError, match="functions"): + litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "hello"}], + functions=[GET_WEATHER_TOOL["function"]], + reasoning_effort="low", + client=client, + ) + litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "hello"}], + functions=[GET_WEATHER_TOOL["function"]], + reasoning_effort="low", + drop_params=True, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse") + body = json.loads(requests[0].content) + assert "functions" not in body + assert "toolConfig" not in body + + +def test_grok_thinking_block_is_served_by_converse(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + thinking = {"type": "enabled", "budget_tokens": 1024} + litellm.completion( + model="bedrock/us.xai.grok-4.6", + messages=[{"role": "user", "content": "hello"}], + thinking=thinking, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/us.xai.grok-4.6/converse") + assert json.loads(requests[0].content)["additionalModelRequestFields"]["thinking"] == thinking + + def test_converse_fallback_validates_against_converse_params(local_cost_map, fake_aws_env): requests, client = _recording_client(json=CONVERSE_JSON) guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"} From 0a85e2799814b4582d110f93c0b9c87ed3f66d8b Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 21 Sep 2026 13:57:27 -0700 Subject: [PATCH 06/18] fix(bedrock): keep schema-less json_object on Converse for the chat completions models --- litellm/llms/bedrock/common_utils.py | 20 +++++---- ...bedrock_chat_completions_transformation.py | 42 +++++++++++++++++-- 2 files changed, 52 insertions(+), 10 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 8798b05cfa4..2fe1af9edf4 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -853,10 +853,15 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( ) -def _response_format_constrains_output(response_format: object) -> bool: +def _response_format_needs_converse(model: str, response_format: object) -> bool: if response_format is None: return False - return not (isinstance(response_format, Mapping) and response_format.get("type") == "text") + if not isinstance(response_format, Mapping): + return not bedrock_runtime_chat_completions_enforces_response_format(model) + if response_format.get("type") == "text": + return False + carries_schema: Final = "json_schema" in response_format or "response_schema" in response_format + return not (carries_schema and bedrock_runtime_chat_completions_enforces_response_format(model)) def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: @@ -867,16 +872,17 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, function tools (``tools`` or legacy ``functions``) on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless - ``reasoning_effort`` is exactly ``"none"``, and a constraining ``response_format`` on a model without - ``supports_bedrock_runtime_chat_completions_response_format`` is only honored by Converse. + ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as a JSON schema + (a ``json_schema`` or ``response_schema`` mapping, or a pydantic model) on a model with + ``supports_bedrock_runtime_chat_completions_response_format``: a schema on any other model is only + honored by Converse, and a schema-less ``json_object`` keeps Converse's handling everywhere, since AWS's + native surface rejects it with a 400 unless the prompt mentions json. """ if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS): return True if bedrock_request_metadata_is_owned(): return True - if _response_format_constrains_output( - request_params.get("response_format") - ) and not bedrock_runtime_chat_completions_enforces_response_format(model): + if _response_format_needs_converse(model, request_params.get("response_format")): return True if not (request_params.get("tools") or request_params.get("functions")): return False diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 2852ed3ae84..e49a2f7f252 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -799,13 +799,32 @@ def test_gpt_oss_response_format_falls_back_to_converse(local_cost_map, model, r assert BedrockModelInfo.get_bedrock_route(model, params) == expected_route -@pytest.mark.parametrize("model", ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"]) -def test_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model): - params = {"response_format": RESPONSE_FORMAT_JSON_SCHEMA} +RESPONSE_FORMAT_ENFORCING_MODELS = ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"] + + +@pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS) +@pytest.mark.parametrize( + "response_format", + [ + RESPONSE_FORMAT_JSON_SCHEMA, + {"type": "json_object", "response_schema": RESPONSE_FORMAT_JSON_SCHEMA["json_schema"]["schema"]}, + Answer, + ], + ids=["json_schema", "response_schema", "pydantic"], +) +def test_schema_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model, response_format): + params = {"response_format": response_format} assert bedrock_request_needs_converse(model, params) is False assert BedrockModelInfo.get_bedrock_route(model, params) == "chat_completions" +@pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS) +def test_schema_less_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_cost_map, model): + params = {"response_format": {"type": "json_object"}} + assert bedrock_request_needs_converse(model, params) is True + assert BedrockModelInfo.get_bedrock_route(model, params) == "converse" + + SYNTHETIC_NATIVE_MODEL = "vendor.native-model-v1:0" @@ -886,3 +905,20 @@ def test_gpt56_response_format_is_sent_as_is_on_chat_completions(local_cost_map, assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" assert json.loads(requests[0].content)["response_format"] == RESPONSE_FORMAT_JSON_SCHEMA assert response.choices[0].message.content == '{"word": "pong"}' + + +def test_gpt56_schema_less_json_object_goes_to_converse_without_a_schema_tool(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "Reply with the single word pong."}], + response_format={"type": "json_object"}, + max_tokens=64, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse") + body = json.loads(requests[0].content) + assert "toolConfig" not in body + assert "response_format" not in body + assert body["inferenceConfig"]["maxTokens"] == 64 From 0139dd08a635843deaac24256b53439531f3b9a7 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 21 Sep 2026 14:10:27 -0700 Subject: [PATCH 07/18] fix(bedrock): keep every json_object response_format on Converse for the chat completions models --- litellm/llms/bedrock/common_utils.py | 16 ++++--- ...bedrock_chat_completions_transformation.py | 46 ++++++++++++++----- 2 files changed, 43 insertions(+), 19 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 2fe1af9edf4..486b3b2bd51 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -858,10 +858,11 @@ def _response_format_needs_converse(model: str, response_format: object) -> bool return False if not isinstance(response_format, Mapping): return not bedrock_runtime_chat_completions_enforces_response_format(model) - if response_format.get("type") == "text": + response_format_type: Final = response_format.get("type") + if response_format_type == "text": return False - carries_schema: Final = "json_schema" in response_format or "response_schema" in response_format - return not (carries_schema and bedrock_runtime_chat_completions_enforces_response_format(model)) + is_json_schema: Final = response_format_type == "json_schema" and "json_schema" in response_format + return not (is_json_schema and bedrock_runtime_chat_completions_enforces_response_format(model)) def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: @@ -872,11 +873,12 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, function tools (``tools`` or legacy ``functions``) on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless - ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as a JSON schema - (a ``json_schema`` or ``response_schema`` mapping, or a pydantic model) on a model with + ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as + ``{"type": "json_schema", "json_schema": ...}`` (a pydantic model is converted to that) on a model with ``supports_bedrock_runtime_chat_completions_response_format``: a schema on any other model is only - honored by Converse, and a schema-less ``json_object`` keeps Converse's handling everywhere, since AWS's - native surface rejects it with a 400 unless the prompt mentions json. + honored by Converse, and every ``json_object`` form (``response_schema`` included) keeps Converse's + handling everywhere, since AWS's native surface rejects that type with a 400 unless the prompt + mentions json. """ if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS): return True diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index e49a2f7f252..2f0d384fcdd 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -802,25 +802,30 @@ def test_gpt_oss_response_format_falls_back_to_converse(local_cost_map, model, r RESPONSE_FORMAT_ENFORCING_MODELS = ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"] +JSON_OBJECT_WITH_RESPONSE_SCHEMA = { + "type": "json_object", + "response_schema": RESPONSE_FORMAT_JSON_SCHEMA["json_schema"]["schema"], +} + + @pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS) -@pytest.mark.parametrize( - "response_format", - [ - RESPONSE_FORMAT_JSON_SCHEMA, - {"type": "json_object", "response_schema": RESPONSE_FORMAT_JSON_SCHEMA["json_schema"]["schema"]}, - Answer, - ], - ids=["json_schema", "response_schema", "pydantic"], -) -def test_schema_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model, response_format): +@pytest.mark.parametrize("response_format", [RESPONSE_FORMAT_JSON_SCHEMA, Answer], ids=["json_schema", "pydantic"]) +def test_json_schema_response_format_stays_on_chat_completions_where_aws_enforces_it( + local_cost_map, model, response_format +): params = {"response_format": response_format} assert bedrock_request_needs_converse(model, params) is False assert BedrockModelInfo.get_bedrock_route(model, params) == "chat_completions" @pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS) -def test_schema_less_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_cost_map, model): - params = {"response_format": {"type": "json_object"}} +@pytest.mark.parametrize( + "response_format", + [{"type": "json_object"}, JSON_OBJECT_WITH_RESPONSE_SCHEMA], + ids=["json_object", "json_object_with_response_schema"], +) +def test_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_cost_map, model, response_format): + params = {"response_format": response_format} assert bedrock_request_needs_converse(model, params) is True assert BedrockModelInfo.get_bedrock_route(model, params) == "converse" @@ -922,3 +927,20 @@ def test_gpt56_schema_less_json_object_goes_to_converse_without_a_schema_tool(lo assert "toolConfig" not in body assert "response_format" not in body assert body["inferenceConfig"]["maxTokens"] == 64 + + +def test_gpt56_json_object_with_response_schema_goes_to_converse_as_a_json_tool(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "Reply with the single word pong."}], + response_format=JSON_OBJECT_WITH_RESPONSE_SCHEMA, + max_tokens=64, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse") + body = json.loads(requests[0].content) + assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "json_tool_call" + assert body["toolConfig"]["toolChoice"] == {"tool": {"name": "json_tool_call"}} + assert "response_format" not in body From c4c24dd9864866c37b30d6dc973e585e36a428b6 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 24 Sep 2026 13:31:29 -0700 Subject: [PATCH 08/18] fix(rust): declare the bedrock runtime chat completions flags on ModelInfo --- litellm-rust/crates/model-catalog/src/model_info.rs | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs index 4a56e1112d1..0b52bea94de 100644 --- a/litellm-rust/crates/model-catalog/src/model_info.rs +++ b/litellm-rust/crates/model-catalog/src/model_info.rs @@ -582,6 +582,12 @@ pub struct ModelInfo { #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_audio_output: Option, #[serde(default, skip_serializing_if = "Option::is_none")] + pub supports_bedrock_runtime_chat_completions: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub supports_bedrock_runtime_chat_completions_response_format: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_computer_use: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_embedding_image_input: Option, From ff8e15d84460d9970443f0004c6fb07ae45dd01e Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 24 Sep 2026 15:21:07 -0700 Subject: [PATCH 09/18] fix(bedrock): opt into the native chat completions route through supported_endpoints --- .../crates/model-catalog/src/model_info.rs | 2 - .../chat/chat_completions/transformation.py | 2 +- litellm/llms/bedrock/common_utils.py | 32 +++++++++-- ...odel_prices_and_context_window_backup.json | 56 +++++++++++++------ model_prices_and_context_window.json | 56 +++++++++++++------ model_prices_and_context_window.schema.json | 3 - ...bedrock_chat_completions_transformation.py | 24 +++++++- tests/test_litellm/test_utils.py | 1 - 8 files changed, 126 insertions(+), 50 deletions(-) diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs index 0b52bea94de..23e69f66492 100644 --- a/litellm-rust/crates/model-catalog/src/model_info.rs +++ b/litellm-rust/crates/model-catalog/src/model_info.rs @@ -582,8 +582,6 @@ pub struct ModelInfo { #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_audio_output: Option, #[serde(default, skip_serializing_if = "Option::is_none")] - pub supports_bedrock_runtime_chat_completions: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_bedrock_runtime_chat_completions_response_format: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option, diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 11e2467b093..4c5e5768119 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -3,7 +3,7 @@ Native OpenAI Chat Completions on Amazon Bedrock Runtime. AWS serves this surface at ``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions`` -for the models whose price-map entry sets ``supports_bedrock_runtime_chat_completions`` +for the models whose price-map ``supported_endpoints`` lists ``/v1/chat/completions`` (Grok 4.6, gpt-oss, the GPT-5.6 family): chat completions stay chat completions instead of being rewritten to Converse. diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 09e62dc393b..4492052b3c4 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -829,24 +829,44 @@ def split_bedrock_region_path(model: str) -> tuple[str | None, str]: return None, stripped -def _bedrock_price_map_flag(model: str, flag: str) -> bool: - entries: Final = ( +BEDROCK_RUNTIME_PRICE_MAP_PROVIDERS: Final = frozenset(("bedrock", "bedrock_converse")) + + +def _bedrock_price_map_entries(model: str) -> tuple[Mapping[str, object] | None, ...]: + return tuple( litellm.model_cost.get(key) for key in (model, strip_bedrock_routing_prefix(model), split_bedrock_region_path(model)[1]) ) - return any(entry is not None and entry.get(flag) is True for entry in entries) + + +def _bedrock_price_map_flag(model: str, flag: str) -> bool: + return any(entry is not None and entry.get(flag) is True for entry in _bedrock_price_map_entries(model)) + + +def _bedrock_runtime_row_lists_chat_completions(entry: Mapping[str, object]) -> bool: + endpoints: Final = entry.get("supported_endpoints") + return ( + entry.get("litellm_provider") in BEDROCK_RUNTIME_PRICE_MAP_PROVIDERS + and isinstance(endpoints, (list, tuple)) + and "/v1/chat/completions" in endpoints + ) def uses_bedrock_runtime_chat_completions(model: str) -> bool: """Whether this Bedrock model should use runtime native Chat Completions. - Data-driven from the price-map ``supports_bedrock_runtime_chat_completions`` flag + Data-driven from ``/v1/chat/completions`` in the price-map row's ``supported_endpoints``, + the same per-model signal ``bedrock_supports_openai_responses`` reads for ``/v1/responses``, so onboarding a model is a JSON change. Explicit ``converse/`` still wins in ``get_bedrock_route`` because prefix routes are checked first, and a request that needs a Converse-only feature (``bedrock_request_needs_converse``) is - served by Converse even on a flagged model. + served by Converse even on a listed model. Only a bedrock-runtime row counts: a + ``bedrock_mantle`` row lists the endpoints of the Mantle host, not this one. """ - return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions") + return any( + entry is not None and _bedrock_runtime_row_lists_chat_completions(entry) + for entry in _bedrock_price_map_entries(model) + ) def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool: diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d58dd4d105a..7fdd8c85ddc 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -40834,7 +40834,9 @@ "output_cost_per_token": 0.0 }, "openai.gpt-oss-120b-1:0": { - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", @@ -40850,7 +40852,9 @@ "supports_tool_choice": true }, "openai.gpt-oss-20b-1:0": { - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", @@ -46591,7 +46595,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -46606,7 +46612,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -46617,7 +46625,9 @@ "input_cost_per_token": 2.64e-06, "output_cost_per_token": 7.92e-06, "cache_read_input_token_cost": 6.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", @@ -57156,7 +57166,6 @@ "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-luna.html" }, "us.openai.gpt-5.6-sol": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4.4e-06, "input_cost_per_token_above_272k_tokens": 8.8e-06, @@ -57188,11 +57197,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "global.openai.gpt-5.6-sol": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4e-06, "input_cost_per_token_above_272k_tokens": 8e-06, @@ -57224,11 +57233,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-5.6-terra": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, @@ -57260,11 +57269,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "global.openai.gpt-5.6-terra": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, @@ -57296,11 +57305,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-5.6-luna": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-07, "input_cost_per_token_above_272k_tokens": 4.4e-07, @@ -57332,6 +57341,7 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -57460,7 +57470,6 @@ ] }, "global.openai.gpt-5.6-luna": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, @@ -57492,6 +57501,7 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -58095,7 +58105,9 @@ "input_cost_per_token": 2.2e-06, "output_cost_per_token": 6.6e-06, "cache_read_input_token_cost": 5.5e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", @@ -58114,7 +58126,9 @@ "input_cost_per_token": 2e-06, "output_cost_per_token": 6e-06, "cache_read_input_token_cost": 5e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", @@ -63818,7 +63832,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -63833,7 +63849,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -64076,7 +64094,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -64091,7 +64111,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index d58dd4d105a..7fdd8c85ddc 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -40834,7 +40834,9 @@ "output_cost_per_token": 0.0 }, "openai.gpt-oss-120b-1:0": { - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", @@ -40850,7 +40852,9 @@ "supports_tool_choice": true }, "openai.gpt-oss-20b-1:0": { - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", @@ -46591,7 +46595,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -46606,7 +46612,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -46617,7 +46625,9 @@ "input_cost_per_token": 2.64e-06, "output_cost_per_token": 7.92e-06, "cache_read_input_token_cost": 6.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", @@ -57156,7 +57166,6 @@ "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-luna.html" }, "us.openai.gpt-5.6-sol": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4.4e-06, "input_cost_per_token_above_272k_tokens": 8.8e-06, @@ -57188,11 +57197,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "global.openai.gpt-5.6-sol": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4e-06, "input_cost_per_token_above_272k_tokens": 8e-06, @@ -57224,11 +57233,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-5.6-terra": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, @@ -57260,11 +57269,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "global.openai.gpt-5.6-terra": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, @@ -57296,11 +57305,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-5.6-luna": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-07, "input_cost_per_token_above_272k_tokens": 4.4e-07, @@ -57332,6 +57341,7 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -57460,7 +57470,6 @@ ] }, "global.openai.gpt-5.6-luna": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, @@ -57492,6 +57501,7 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -58095,7 +58105,9 @@ "input_cost_per_token": 2.2e-06, "output_cost_per_token": 6.6e-06, "cache_read_input_token_cost": 5.5e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", @@ -58114,7 +58126,9 @@ "input_cost_per_token": 2e-06, "output_cost_per_token": 6e-06, "cache_read_input_token_cost": 5e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", @@ -63818,7 +63832,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -63833,7 +63849,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -64076,7 +64094,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -64091,7 +64111,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index 89ebedc3a0a..bcb0509f25c 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -902,9 +902,6 @@ "supports_audio_output": { "type": "boolean" }, - "supports_bedrock_runtime_chat_completions": { - "type": "boolean" - }, "supports_bedrock_runtime_chat_completions_response_format": { "type": "boolean" }, diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 2f0d384fcdd..3c347e4bed2 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -59,9 +59,27 @@ def test_claude_stays_on_converse(local_cost_map): assert BedrockModelInfo.get_bedrock_route("us.anthropic.claude-3-sonnet-20240229-v1:0") == "converse" -def test_flag_absent_means_no_chat_completions_route(monkeypatch): - monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": {"litellm_provider": "bedrock_converse"}}) +@pytest.mark.parametrize( + "entry", + [ + {"litellm_provider": "bedrock_converse"}, + {"litellm_provider": "bedrock_converse", "supported_endpoints": ["/v1/responses"]}, + {"litellm_provider": "bedrock_converse", "supports_bedrock_runtime_chat_completions": True}, + {"litellm_provider": "bedrock_mantle", "supported_endpoints": ["/v1/chat/completions", "/v1/responses"]}, + {"litellm_provider": "openai", "supported_endpoints": ["/v1/chat/completions"]}, + ], +) +def test_chat_completions_missing_from_supported_endpoints_means_no_chat_completions_route(monkeypatch, entry): + monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": entry}) assert uses_bedrock_runtime_chat_completions("us.xai.grok-4.6") is False + assert BedrockModelInfo.get_bedrock_route("us.xai.grok-4.6") == "converse" + + +def test_chat_completions_in_supported_endpoints_opts_into_the_native_route(monkeypatch): + entry = {"litellm_provider": "bedrock_converse", "supported_endpoints": ["/v1/chat/completions", "/v1/responses"]} + monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": entry}) + assert uses_bedrock_runtime_chat_completions("us.xai.grok-4.6") is True + assert BedrockModelInfo.get_bedrock_route("bedrock/us.xai.grok-4.6") == "chat_completions" def test_complete_url_is_runtime_openai_chat_completions(monkeypatch): @@ -860,7 +878,7 @@ SYNTHETIC_NATIVE_MODEL = "vendor.native-model-v1:0" def test_capability_flags_are_read_from_the_cost_map(monkeypatch, capability_flags, request_params, needs_converse): entry = { "litellm_provider": "bedrock_converse", - "supports_bedrock_runtime_chat_completions": True, + "supported_endpoints": ["/v1/chat/completions"], **capability_flags, } monkeypatch.setattr(litellm, "model_cost", {SYNTHETIC_NATIVE_MODEL: entry}) diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 98e46016401..39cb25a0f51 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -927,7 +927,6 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "supports_video_input": {"type": "boolean"}, "supports_vision": {"type": "boolean"}, "supports_web_search": {"type": "boolean"}, - "supports_bedrock_runtime_chat_completions": {"type": "boolean"}, "supports_bedrock_runtime_chat_completions_tools_with_reasoning": {"type": "boolean"}, "supports_bedrock_runtime_chat_completions_response_format": {"type": "boolean"}, "supports_url_context": {"type": "boolean"}, From 4101c0ceb2e213a177613d89220bb2e64d191aeb Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 24 Sep 2026 18:08:46 -0700 Subject: [PATCH 10/18] docs(cost-map): describe the bedrock native chat completions capability flags --- ci_cd/generate_model_prices_schema.py | 20 ++++++++++++++++++- .../crates/model-catalog/src/model_info.rs | 2 ++ model_prices_and_context_window.schema.json | 6 ++++-- 3 files changed, 25 insertions(+), 3 deletions(-) diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index 8eec07dadda..f54177def8e 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -211,6 +211,24 @@ NUMBER_KEYS: dict[str, JsonSchema] = { }, } +BOOLEAN_KEYS: dict[str, JsonSchema] = { + "supports_bedrock_runtime_chat_completions_response_format": { + "type": "boolean", + "description": ( + "The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; " + "unset means LiteLLM serves those requests through Converse's json_tool_call emulation." + ), + }, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": { + "type": "boolean", + "description": ( + "The Bedrock native /v1/chat/completions route serves this model's function tools with any " + "reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort " + "is exactly 'none'." + ), + }, +} + COST_DESCRIPTIONS: dict[str, str] = { "input_cost_per_token": "USD per prompt token.", "output_cost_per_token": "USD per generated token.", @@ -292,7 +310,7 @@ def string_key_schemas(modes: tuple) -> dict[str, JsonSchema]: def classify(key: str, modes: tuple) -> Optional[JsonSchema]: - curated = {**OBJECT_KEYS, **ARRAY_KEYS, **string_key_schemas(modes), **INTEGER_KEYS, **NUMBER_KEYS} + curated = {**OBJECT_KEYS, **ARRAY_KEYS, **string_key_schemas(modes), **INTEGER_KEYS, **NUMBER_KEYS, **BOOLEAN_KEYS} if key in curated: return curated[key] if key.startswith("supports_") or key in EXTRA_BOOLEAN_KEYS: diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs index 23e69f66492..5dda91b2213 100644 --- a/litellm-rust/crates/model-catalog/src/model_info.rs +++ b/litellm-rust/crates/model-catalog/src/model_info.rs @@ -581,8 +581,10 @@ pub struct ModelInfo { pub supports_audio_input: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_audio_output: Option, + /// The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; unset means LiteLLM serves those requests through Converse's json_tool_call emulation. #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_bedrock_runtime_chat_completions_response_format: Option, + /// The Bedrock native /v1/chat/completions route serves this model's function tools with any reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort is exactly 'none'. #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option, #[serde(default, skip_serializing_if = "Option::is_none")] diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index bcb0509f25c..d9a7dd2fff5 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -903,10 +903,12 @@ "type": "boolean" }, "supports_bedrock_runtime_chat_completions_response_format": { - "type": "boolean" + "type": "boolean", + "description": "The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; unset means LiteLLM serves those requests through Converse's json_tool_call emulation." }, "supports_bedrock_runtime_chat_completions_tools_with_reasoning": { - "type": "boolean" + "type": "boolean", + "description": "The Bedrock native /v1/chat/completions route serves this model's function tools with any reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort is exactly 'none'." }, "supports_computer_use": { "type": "boolean" From 61a130c9a13a6d4563ba0c26b44a447ae12e98b2 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 24 Sep 2026 18:20:45 -0700 Subject: [PATCH 11/18] revert: docs(cost-map): describe the bedrock native chat completions capability flags This reverts commit 4101c0ceb2e213a177613d89220bb2e64d191aeb. cost-map-guard runs main's schema generator under pull_request_target and compares its output to the PR's committed schema, so a PR that changes the generator's output cannot pass that required check until the generator change lands on main first. The descriptions move to a follow-up that lands the generator change ahead of the schema --- ci_cd/generate_model_prices_schema.py | 20 +------------------ .../crates/model-catalog/src/model_info.rs | 2 -- model_prices_and_context_window.schema.json | 6 ++---- 3 files changed, 3 insertions(+), 25 deletions(-) diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index f54177def8e..8eec07dadda 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -211,24 +211,6 @@ NUMBER_KEYS: dict[str, JsonSchema] = { }, } -BOOLEAN_KEYS: dict[str, JsonSchema] = { - "supports_bedrock_runtime_chat_completions_response_format": { - "type": "boolean", - "description": ( - "The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; " - "unset means LiteLLM serves those requests through Converse's json_tool_call emulation." - ), - }, - "supports_bedrock_runtime_chat_completions_tools_with_reasoning": { - "type": "boolean", - "description": ( - "The Bedrock native /v1/chat/completions route serves this model's function tools with any " - "reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort " - "is exactly 'none'." - ), - }, -} - COST_DESCRIPTIONS: dict[str, str] = { "input_cost_per_token": "USD per prompt token.", "output_cost_per_token": "USD per generated token.", @@ -310,7 +292,7 @@ def string_key_schemas(modes: tuple) -> dict[str, JsonSchema]: def classify(key: str, modes: tuple) -> Optional[JsonSchema]: - curated = {**OBJECT_KEYS, **ARRAY_KEYS, **string_key_schemas(modes), **INTEGER_KEYS, **NUMBER_KEYS, **BOOLEAN_KEYS} + curated = {**OBJECT_KEYS, **ARRAY_KEYS, **string_key_schemas(modes), **INTEGER_KEYS, **NUMBER_KEYS} if key in curated: return curated[key] if key.startswith("supports_") or key in EXTRA_BOOLEAN_KEYS: diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs index 5dda91b2213..23e69f66492 100644 --- a/litellm-rust/crates/model-catalog/src/model_info.rs +++ b/litellm-rust/crates/model-catalog/src/model_info.rs @@ -581,10 +581,8 @@ pub struct ModelInfo { pub supports_audio_input: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_audio_output: Option, - /// The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; unset means LiteLLM serves those requests through Converse's json_tool_call emulation. #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_bedrock_runtime_chat_completions_response_format: Option, - /// The Bedrock native /v1/chat/completions route serves this model's function tools with any reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort is exactly 'none'. #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option, #[serde(default, skip_serializing_if = "Option::is_none")] diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index d9a7dd2fff5..bcb0509f25c 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -903,12 +903,10 @@ "type": "boolean" }, "supports_bedrock_runtime_chat_completions_response_format": { - "type": "boolean", - "description": "The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; unset means LiteLLM serves those requests through Converse's json_tool_call emulation." + "type": "boolean" }, "supports_bedrock_runtime_chat_completions_tools_with_reasoning": { - "type": "boolean", - "description": "The Bedrock native /v1/chat/completions route serves this model's function tools with any reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort is exactly 'none'." + "type": "boolean" }, "supports_computer_use": { "type": "boolean" From c60249fa882116c4cd8fe784770f5e3a3be3ec30 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 16:02:49 +0000 Subject: [PATCH 12/18] test(bedrock): move the native chat completions tests under tests/unit Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/unit/llms/bedrock/chat/chat_completions/__init__.py | 0 .../test_bedrock_chat_completions_transformation.py | 0 2 files changed, 0 insertions(+), 0 deletions(-) create mode 100644 tests/unit/llms/bedrock/chat/chat_completions/__init__.py rename tests/{test_litellm => unit}/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py (100%) diff --git a/tests/unit/llms/bedrock/chat/chat_completions/__init__.py b/tests/unit/llms/bedrock/chat/chat_completions/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py similarity index 100% rename from tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py rename to tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py From f4d84f8f5db0c26e0aa3c50bc2ddb77d07122f5b Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:28:06 +0000 Subject: [PATCH 13/18] fix(bedrock): drop reasoning_effort none for grok on the native chat completions route Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../chat/chat_completions/transformation.py | 29 ++++++++++++- ...bedrock_chat_completions_transformation.py | 42 +++++++++++++++++++ 2 files changed, 70 insertions(+), 1 deletion(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 4c5e5768119..be2eb8c7713 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -60,6 +60,31 @@ def chat_completions_params_refused_for(model: str) -> frozenset[str]: ) +CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY: Final = MappingProxyType({"xai.": frozenset(("none",))}) + + +def chat_completions_reasoning_efforts_refused_for(model: str) -> frozenset[str]: + """The ``reasoning_effort`` values AWS's Chat Completions endpoint rejects for this model. + + Grok answers ``"none"`` with a 400 (it takes low, medium, high, and xhigh) where Converse dropped every + ``reasoning_effort`` for it, so the native config drops the value and AWS applies its default effort as before. + """ + model_id: Final = split_bedrock_region_path(model)[1] + return frozenset().union( + *( + refused + for family, refused in CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY.items() + if family in model_id + ) + ) + + +def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) -> Mapping[str, object]: + if params.get("reasoning_effort") not in chat_completions_reasoning_efforts_refused_for(model): + return params + return MappingProxyType({key: value for key, value in params.items() if key != "reasoning_effort"}) + + def _held_close_tag_prefix(text: str) -> int: return next( ( @@ -277,7 +302,9 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): drop_params=drop_params, replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens, ) - return dict(with_max_completion_tokens(mapped)) # mutable-ok: get_optional_params keeps filling this dict + return dict( # mutable-ok: get_optional_params keeps filling this dict + without_refused_reasoning_effort(model, with_max_completion_tokens(mapped)) + ) def _inference_params( self, optional_params: Mapping[str, object] diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 3c347e4bed2..061df201c15 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -11,6 +11,7 @@ from litellm.llms.bedrock.chat.chat_completions.transformation import ( AmazonBedrockRuntimeChatCompletionsConfig, BedrockRuntimeChatCompletionsStreamingHandler, ReasoningTagSplitter, + chat_completions_reasoning_efforts_refused_for, split_reasoning_tag, with_max_completion_tokens, ) @@ -333,6 +334,47 @@ def test_with_max_completion_tokens_leaves_other_params_alone(): assert with_max_completion_tokens({"temperature": 0.5}) == {"temperature": 0.5} +@pytest.mark.parametrize( + "model", + ["us.xai.grok-4.6", "bedrock/us-gov-west-1/us.xai.grok-4.6"], +) +def test_map_openai_params_drops_reasoning_effort_none_for_grok(model): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"reasoning_effort": "none", "max_tokens": 64}, + optional_params={}, + model=model, + drop_params=False, + ) + assert "reasoning_effort" not in mapped + + +def test_map_openai_params_keeps_reasoning_effort_low_for_grok(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"reasoning_effort": "low", "max_tokens": 64}, + optional_params={}, + model="us.xai.grok-4.6", + drop_params=False, + ) + assert mapped["reasoning_effort"] == "low" + + +def test_map_openai_params_keeps_reasoning_effort_none_for_gpt56(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"reasoning_effort": "none", "max_tokens": 64}, + optional_params={}, + model="global.openai.gpt-5.6-sol", + drop_params=False, + ) + assert mapped["reasoning_effort"] == "none" + + +def test_reasoning_efforts_refused_for_is_empty_outside_xai(): + assert chat_completions_reasoning_efforts_refused_for("openai.gpt-oss-20b-1:0") == frozenset() + + def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): cfg = AmazonBedrockRuntimeChatCompletionsConfig() assert "reasoning_effort" in cfg.get_supported_openai_params("global.openai.gpt-5.6-sol") From e653217228d191d0c0ecb7c9b275786c19052e94 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:33:01 +0000 Subject: [PATCH 14/18] fix(bedrock): keep converse extension params on the converse route Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/bedrock/common_utils.py | 14 ++++++++++++-- ...test_bedrock_chat_completions_transformation.py | 12 ++++++++++++ 2 files changed, 24 insertions(+), 2 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 8fe545a3272..098be7082d3 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -890,7 +890,16 @@ def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> boo BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( - ("guardrailConfig", "performanceConfig", "serviceTier", "requestMetadata", "outputConfig", "thinking") + ( + "guardrailConfig", + "performanceConfig", + "serviceTier", + "requestMetadata", + "outputConfig", + "thinking", + "additionalModelRequestFields", + "top_k", + ) ) @@ -910,7 +919,8 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje """Whether a request on a runtime-Chat-Completions model must still be served by Converse. Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking`` - block included, which only Converse forwards as ``additionalModelRequestFields``) have no field on + block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse + forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, function tools (``tools`` or legacy ``functions``) on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 061df201c15..17c0f8d3c1d 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -258,6 +258,18 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model): assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": None}) == "chat_completions" +@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"]) +@pytest.mark.parametrize( + "request_params", + [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}], + ids=["additionalModelRequestFields", "top_k"], +) +def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params): + assert bedrock_request_needs_converse(model, request_params) is True + assert BedrockModelInfo.get_bedrock_route(model, request_params) == "converse" + assert BedrockModelInfo.get_bedrock_route(model, {key: None for key in request_params}) == "chat_completions" + + @pytest.mark.parametrize( "request_params, expected_route", [ From 093a9d4ddfbb010977ad7a50a0c8c3f7745dbcaf Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:37:36 +0000 Subject: [PATCH 15/18] fix(bedrock): inline http image urls and keep stop on converse for native chat completions Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../chat/chat_completions/transformation.py | 57 +++++++++++++- litellm/llms/bedrock/common_utils.py | 4 +- ...bedrock_chat_completions_transformation.py | 75 +++++++++++++++++-- 3 files changed, 127 insertions(+), 9 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index be2eb8c7713..123531cf227 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -22,6 +22,11 @@ import httpx from typing_extensions import assert_never import litellm +from litellm.litellm_core_utils.prompt_templates.image_handling import ( + async_inline_remote_media, + convert_url_to_base64, + inline_remote_image_urls, +) from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path @@ -40,7 +45,7 @@ REASONING_CLOSE_TAG: Final = "" CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType( { - "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs")), + "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs")), "openai.gpt-oss": frozenset(("logit_bias",)), "xai.": frozenset(("frequency_penalty", "presence_penalty")), } @@ -167,6 +172,48 @@ def split_reasoning_tag(content: str) -> tuple[str | None, str]: return reasoning or None, body +def _remote_http_url(candidate: object) -> str | None: + return candidate if isinstance(candidate, str) and candidate.startswith(("http://", "https://")) else None + + +def _inlined_image_url_part(part: object) -> object: + fields: Final = part if isinstance(part, Mapping) else None + if fields is None or fields.get("type") != "image_url": + return part + image_url: Final = fields.get("image_url") + image_url_fields: Final = image_url if isinstance(image_url, Mapping) else None + url: Final = _remote_http_url(image_url_fields.get("url") if image_url_fields is not None else image_url) + if url is None: + return part + data_url: Final = convert_url_to_base64(url) + inlined: Final = {**image_url_fields, "url": data_url} if image_url_fields is not None else data_url + return {**fields, "image_url": inlined} # mutable-ok: json-serialized message part + + +def _inlined_image_url_message(message: AllMessageValues) -> AllMessageValues: + content: Final = message.get("content") + if not isinstance(content, list): + return message + inlined_message: Final = { # mutable-ok: json-serialized message + **message, + "content": [_inlined_image_url_part(part) for part in content], + } + return inlined_message # pyright: ignore[reportReturnType] # the same message with remote image parts inlined + + +def _with_inlined_remote_image_urls( + messages: list[AllMessageValues], +) -> list[AllMessageValues]: # mutable-ok: transform_request takes a list + """Inline every remote ``image_url`` so AWS never sees the ``http(s)://`` URLs it rejects. + + AWS's native surface only takes inline ``data:`` URLs and S3 URLs where Converse downloaded + remote images itself, so the bytes are fetched and inlined here exactly like Converse did. + """ + return [ # mutable-ok: transform_request takes a list + _inlined_image_url_message(message) for message in messages + ] + + class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler): """OpenAI chunk parsing plus the ```` split, tracked per choice index.""" @@ -222,6 +269,10 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): def custom_llm_provider(self) -> str | None: return "bedrock" + @property + def uses_async_transform_request(self) -> bool: + return True + def get_error_class( self, error_message: str, @@ -325,7 +376,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): ) -> dict: # mutable-ok: BaseConfig signature return super().transform_request( model=split_bedrock_region_path(model)[1], - messages=messages, + messages=_with_inlined_remote_image_urls(messages), optional_params=self._inference_params(optional_params), litellm_params=litellm_params, headers=headers, @@ -341,7 +392,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): ) -> dict: # mutable-ok: BaseConfig signature return await super().async_transform_request( model=split_bedrock_region_path(model)[1], - messages=messages, + messages=await async_inline_remote_media(messages, should_inline=inline_remote_image_urls), optional_params=self._inference_params(optional_params), litellm_params=litellm_params, headers=headers, diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 098be7082d3..6bd4a8d634c 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -899,6 +899,7 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( "thinking", "additionalModelRequestFields", "top_k", + "stop", ) ) @@ -921,7 +922,8 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking`` block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on - AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, + AWS's native OpenAI surface, ``stop`` stays on Converse where it fails loudly instead of silently + stopping hidden reasoning, operator-owned request metadata is only written onto the Converse body, function tools (``tools`` or legacy ``functions``) on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 17c0f8d3c1d..4bfa7bf0fc4 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -258,11 +258,13 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model): assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": None}) == "chat_completions" -@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"]) +@pytest.mark.parametrize( + "model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6", "global.openai.gpt-5.6-sol"] +) @pytest.mark.parametrize( "request_params", - [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}], - ids=["additionalModelRequestFields", "top_k"], + [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}, {"stop": ["END"]}], + ids=["additionalModelRequestFields", "top_k", "stop"], ) def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params): assert bedrock_request_needs_converse(model, request_params) is True @@ -331,6 +333,69 @@ def test_map_openai_params_sends_max_tokens_as_max_completion_tokens(): assert mapped == {"max_completion_tokens": 64, "temperature": 0.1} +HTTPS_IMAGE_URL = "https://example.com/cat.png" +IMAGE_MESSAGES = [ + { + "role": "user", + "content": [ + {"type": "text", "text": "what is this"}, + {"type": "image_url", "image_url": HTTPS_IMAGE_URL}, + {"type": "image_url", "image_url": {"url": HTTPS_IMAGE_URL, "detail": "high"}}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,AAA"}}, + {"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}}, + ], + } +] + + +def _assert_remote_images_inlined(content): + assert content[0] == {"type": "text", "text": "what is this"} + assert content[1]["image_url"]["url"] == f"data:image/png;base64,{HTTPS_IMAGE_URL}" + assert content[2] == { + "type": "image_url", + "image_url": {"url": f"data:image/png;base64,{HTTPS_IMAGE_URL}", "detail": "high"}, + } + assert content[3]["image_url"]["url"] == "data:image/png;base64,AAA" + assert content[4]["image_url"]["url"] == "s3://bucket/key.png" + + +def test_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch): + import litellm.llms.bedrock.chat.chat_completions.transformation as native_cc + + monkeypatch.setattr( + native_cc, "convert_url_to_base64", lambda url: f"data:image/png;base64,{url}" + ) + body = AmazonBedrockRuntimeChatCompletionsConfig().transform_request( + model="us.xai.grok-4.6", + messages=IMAGE_MESSAGES, + optional_params={}, + litellm_params={}, + headers={}, + ) + + _assert_remote_images_inlined(body["messages"][0]["content"]) + + +async def test_async_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch): + import litellm.litellm_core_utils.prompt_templates.image_handling as image_handling + + async def fake_convert(url): + return f"data:image/png;base64,{url}" + + monkeypatch.setattr(image_handling, "async_convert_url_to_base64", fake_convert) + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + assert cfg.uses_async_transform_request is True + body = await cfg.async_transform_request( + model="us.xai.grok-4.6", + messages=IMAGE_MESSAGES, + optional_params={}, + litellm_params={}, + headers={}, + ) + + _assert_remote_images_inlined(body["messages"][0]["content"]) + + def test_map_openai_params_keeps_explicit_max_completion_tokens(): cfg = AmazonBedrockRuntimeChatCompletionsConfig() mapped = cfg.map_openai_params( @@ -398,8 +463,8 @@ def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): [ ( "bedrock/global.openai.gpt-5.6-sol", - ("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs", "n"), - ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions"), + ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "n"), + ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions", "stop"), ), ( "us.xai.grok-4.6", From daba2576f503ab4b313cac6d196f28fa731ef0f9 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:41:15 +0000 Subject: [PATCH 16/18] refactor(bedrock): share the sync remote media inliner Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../prompt_templates/image_handling.py | 20 ++++++++ .../chat/chat_completions/transformation.py | 46 +------------------ .../litellm_core_utils/test_image_handling.py | 45 ++++++++++++++++++ ...bedrock_chat_completions_transformation.py | 4 +- 4 files changed, 69 insertions(+), 46 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/image_handling.py b/litellm/litellm_core_utils/prompt_templates/image_handling.py index c44c80bc0a0..cb5c02ce10e 100644 --- a/litellm/litellm_core_utils/prompt_templates/image_handling.py +++ b/litellm/litellm_core_utils/prompt_templates/image_handling.py @@ -310,6 +310,26 @@ async def _fetch_data_urls(remote_urls: tuple[str, ...]) -> tuple[str, ...]: raise +def inline_remote_media( + messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues] + should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url, +) -> list[AllMessageValues]: # mutable-ok: every transform_request takes list[AllMessageValues] + remote_urls: Final = tuple( + dict.fromkeys( + remote.url + for message in messages + for part in _content_parts(message) + if (remote := _parse_remote_part(part)) is not None and should_inline(_remote_media(remote)) + ) + ) + if not remote_urls: + return messages + data_urls: Final = MappingProxyType({url: convert_url_to_base64(url) for url in remote_urls}) + return [ # mutable-ok: transform_request takes a list + _inline_message(message, data_urls, should_inline) for message in messages + ] + + async def async_inline_remote_media( messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues] should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url, diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 123531cf227..d907ef613a2 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -24,8 +24,8 @@ from typing_extensions import assert_never import litellm from litellm.litellm_core_utils.prompt_templates.image_handling import ( async_inline_remote_media, - convert_url_to_base64, inline_remote_image_urls, + inline_remote_media, ) from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM @@ -172,48 +172,6 @@ def split_reasoning_tag(content: str) -> tuple[str | None, str]: return reasoning or None, body -def _remote_http_url(candidate: object) -> str | None: - return candidate if isinstance(candidate, str) and candidate.startswith(("http://", "https://")) else None - - -def _inlined_image_url_part(part: object) -> object: - fields: Final = part if isinstance(part, Mapping) else None - if fields is None or fields.get("type") != "image_url": - return part - image_url: Final = fields.get("image_url") - image_url_fields: Final = image_url if isinstance(image_url, Mapping) else None - url: Final = _remote_http_url(image_url_fields.get("url") if image_url_fields is not None else image_url) - if url is None: - return part - data_url: Final = convert_url_to_base64(url) - inlined: Final = {**image_url_fields, "url": data_url} if image_url_fields is not None else data_url - return {**fields, "image_url": inlined} # mutable-ok: json-serialized message part - - -def _inlined_image_url_message(message: AllMessageValues) -> AllMessageValues: - content: Final = message.get("content") - if not isinstance(content, list): - return message - inlined_message: Final = { # mutable-ok: json-serialized message - **message, - "content": [_inlined_image_url_part(part) for part in content], - } - return inlined_message # pyright: ignore[reportReturnType] # the same message with remote image parts inlined - - -def _with_inlined_remote_image_urls( - messages: list[AllMessageValues], -) -> list[AllMessageValues]: # mutable-ok: transform_request takes a list - """Inline every remote ``image_url`` so AWS never sees the ``http(s)://`` URLs it rejects. - - AWS's native surface only takes inline ``data:`` URLs and S3 URLs where Converse downloaded - remote images itself, so the bytes are fetched and inlined here exactly like Converse did. - """ - return [ # mutable-ok: transform_request takes a list - _inlined_image_url_message(message) for message in messages - ] - - class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler): """OpenAI chunk parsing plus the ```` split, tracked per choice index.""" @@ -376,7 +334,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): ) -> dict: # mutable-ok: BaseConfig signature return super().transform_request( model=split_bedrock_region_path(model)[1], - messages=_with_inlined_remote_image_urls(messages), + messages=inline_remote_media(messages, should_inline=inline_remote_image_urls), optional_params=self._inference_params(optional_params), litellm_params=litellm_params, headers=headers, diff --git a/tests/unit/litellm_core_utils/test_image_handling.py b/tests/unit/litellm_core_utils/test_image_handling.py index 21e97e97357..1fc8d54cecc 100644 --- a/tests/unit/litellm_core_utils/test_image_handling.py +++ b/tests/unit/litellm_core_utils/test_image_handling.py @@ -16,6 +16,7 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import ( async_convert_url_to_base64, async_inline_remote_media, convert_url_to_base64, + inline_remote_media, ) from litellm.litellm_core_utils.url_utils import SSRFError @@ -320,6 +321,50 @@ async def test_async_inline_remote_media_inlines_every_remote_part_shape(async_o assert messages == snapshot +def test_inline_remote_media_inlines_every_remote_part_shape(monkeypatch): + image_url = f"http://img.example/{uuid.uuid4()}.png" + pdf_url = f"http://docs.example/{uuid.uuid4()}.pdf" + fetched = [] + + def fake_convert(url): + fetched.append(url) + return f"data:image/png;base64,{url}" + + monkeypatch.setattr(image_handling, "convert_url_to_base64", fake_convert) + messages = [ + {"role": "system", "content": "be terse"}, + { + "role": "user", + "content": [ + {"type": "text", "text": "what is this?"}, + {"type": "image_url", "image_url": {"url": image_url, "detail": "low"}}, + {"type": "image_url", "image_url": image_url}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,iVBORw0KGgo="}}, + {"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}}, + {"type": "file", "file": {"file_id": pdf_url}}, + {"type": "document", "source": {"type": "url", "url": pdf_url}, "title": "the doc"}, + ], + }, + ] + snapshot = copy.deepcopy(messages) + + inlined = inline_remote_media(messages, should_inline=image_handling.inline_remote_image_urls) + + data_url = f"data:image/png;base64,{image_url}" + assert inlined[0] == {"role": "system", "content": "be terse"} + assert inlined[1]["content"] == [ + {"type": "text", "text": "what is this?"}, + {"type": "image_url", "image_url": {"url": data_url, "detail": "low"}}, + {"type": "image_url", "image_url": data_url}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,iVBORw0KGgo="}}, + {"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}}, + {"type": "file", "file": {"file_id": pdf_url}}, + {"type": "document", "source": {"type": "url", "url": pdf_url}, "title": "the doc"}, + ] + assert fetched == [image_url] + assert messages == snapshot + + async def test_async_inline_remote_media_inlines_only_the_parts_the_predicate_accepts(async_only_image_fetch): files_api_prefix = "https://generativelanguage.googleapis.com/v1beta/files/" files_api_pdf = f"{files_api_prefix}{uuid.uuid4().hex}" diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 4bfa7bf0fc4..cbde3dd03a0 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -360,10 +360,10 @@ def _assert_remote_images_inlined(content): def test_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch): - import litellm.llms.bedrock.chat.chat_completions.transformation as native_cc + import litellm.litellm_core_utils.prompt_templates.image_handling as image_handling monkeypatch.setattr( - native_cc, "convert_url_to_base64", lambda url: f"data:image/png;base64,{url}" + image_handling, "convert_url_to_base64", lambda url: f"data:image/png;base64,{url}" ) body = AmazonBedrockRuntimeChatCompletionsConfig().transform_request( model="us.xai.grok-4.6", From cbf01c25babb5e4410e65f57950787f4f76b2f93 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:16:13 +0000 Subject: [PATCH 17/18] fix(image-handling): infer the image mime type when the server sends a generic content type Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../prompt_templates/image_handling.py | 28 +++++------ .../litellm_core_utils/test_image_handling.py | 49 +++++++++++++++++++ 2 files changed, 60 insertions(+), 17 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/image_handling.py b/litellm/litellm_core_utils/prompt_templates/image_handling.py index cb5c02ce10e..57b4f545301 100644 --- a/litellm/litellm_core_utils/prompt_templates/image_handling.py +++ b/litellm/litellm_core_utils/prompt_templates/image_handling.py @@ -15,6 +15,7 @@ import litellm from litellm import verbose_logger from litellm.caching.caching import InMemoryCache from litellm.constants import MAX_IMAGE_URL_DOWNLOAD_SIZE_MB +from litellm.litellm_core_utils.prompt_templates.common_utils import infer_content_type_from_url_and_content from litellm.litellm_core_utils.url_utils import SSRFError, async_safe_get, safe_get from litellm.types.llms.openai import AllMessageValues @@ -55,23 +56,16 @@ def _process_image_response(response: Response, url: str) -> str: base64_image: Final = base64.b64encode(image_bytes).decode("utf-8") - image_type: Final = response.headers.get("Content-Type") - if image_type is None: - img_type = url.split(".")[-1].lower() - _img_type: Final = { - "jpg": "image/jpeg", - "jpeg": "image/jpeg", - "png": "image/png", - "gif": "image/gif", - "webp": "image/webp", - }.get(img_type) - if _img_type is None: - raise Exception( - f"Error: Unsupported image format. Format={_img_type}. Supported types = ['image/jpeg', 'image/png', 'image/gif', 'image/webp']" - ) - img_type = _img_type - else: - img_type = image_type + try: + img_type: Final = infer_content_type_from_url_and_content( + url=url, + content=bytes(image_bytes), + current_content_type=response.headers.get("Content-Type"), + ) + except ValueError as e: + raise litellm.ImageFetchError( + f"Error: Unable to determine image content type from the server's headers, the URL, or the image bytes. url={url}" + ) from e result: Final = f"data:{img_type};base64,{base64_image}" in_memory_cache.set_cache(url, result) diff --git a/tests/unit/litellm_core_utils/test_image_handling.py b/tests/unit/litellm_core_utils/test_image_handling.py index 1fc8d54cecc..57eb32f98e5 100644 --- a/tests/unit/litellm_core_utils/test_image_handling.py +++ b/tests/unit/litellm_core_utils/test_image_handling.py @@ -1,4 +1,5 @@ import asyncio +import base64 import copy import time import uuid @@ -259,6 +260,54 @@ async def test_async_data_url_is_returned_unchanged_without_fetch(monkeypatch): assert await async_convert_url_to_base64(data_url) == data_url +REAL_PNG_BYTES = base64.b64decode( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg==" +) + + +def _stub_image_client(content, content_type): + class _Client: + def get(self, url, follow_redirects=True): + headers = {} if content_type is None else {"Content-Type": content_type} + return Response(200, content=content, headers=headers, request=Request("GET", url)) + + return _Client() + + +def test_convert_url_to_base64_infers_the_type_when_the_server_sends_octet_stream(monkeypatch): + monkeypatch.setattr( + litellm, "module_level_client", _stub_image_client(REAL_PNG_BYTES, "application/octet-stream") + ) + + result = convert_url_to_base64(f"http://img.example/{uuid.uuid4()}") + + assert result.startswith("data:image/png;base64,") + + +def test_convert_url_to_base64_keeps_a_real_content_type(monkeypatch): + monkeypatch.setattr( + litellm, "module_level_client", _stub_image_client(REAL_PNG_BYTES, "image/jpeg") + ) + + result = convert_url_to_base64(f"http://img.example/{uuid.uuid4()}.png") + + assert result.startswith("data:image/jpeg;base64,") + + +def test_convert_url_to_base64_raises_when_no_content_type_is_determinable(monkeypatch): + monkeypatch.setattr( + litellm, + "module_level_client", + _stub_image_client(b"\x00\x01\x02\x03not-an-image", "application/octet-stream"), + ) + url = f"http://img.example/{uuid.uuid4()}" + + with pytest.raises(litellm.ImageFetchError) as excinfo: + convert_url_to_base64(url) + + assert url in str(excinfo.value) + + def test_image_size_limit_disabled(monkeypatch): """ Test that setting MAX_IMAGE_URL_DOWNLOAD_SIZE_MB to 0 disables all image URL downloads. From 4900a1ae85b314e108e6150adfb2d3acef1c0914 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 14:25:38 -0700 Subject: [PATCH 18/18] fix(bedrock): stop sending aws_bedrock_project_id as OpenAI-Project on the runtime chat completions route --- .../chat/chat_completions/transformation.py | 24 ------------------- ...bedrock_chat_completions_transformation.py | 13 ++++++++++ 2 files changed, 13 insertions(+), 24 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index d907ef613a2..381ce7922d7 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -394,30 +394,6 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): choice.message.content = content return response - def validate_environment( - self, - headers: dict, # mutable-ok: BaseConfig signature - model: str, - messages: list[AllMessageValues], # mutable-ok: BaseConfig signature - optional_params: dict, # mutable-ok: BaseConfig signature - litellm_params: dict, # mutable-ok: BaseConfig signature - api_key: str | None = None, - api_base: str | None = None, - ) -> dict: # mutable-ok: BaseConfig signature - validated: Final = super().validate_environment( - headers=headers, - model=model, - messages=messages, - optional_params=optional_params, - litellm_params=litellm_params, - api_key=api_key, - api_base=api_base, - ) - project_id: Final = litellm_params.get("aws_bedrock_project_id") - if not project_id: - return validated - return {**validated, "OpenAI-Project": project_id} # mutable-ok: BaseConfig signature returns a dict - def get_supported_openai_params(self, model: str) -> list: # mutable-ok: BaseConfig signature refused: Final = frozenset(("n", *chat_completions_params_refused_for(model))) base_params: Final = tuple( diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index cbde3dd03a0..f9f330a7340 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -109,6 +109,19 @@ def test_complete_url_appends_to_openai_v1_base(): assert url == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" +def test_project_id_is_not_sent_as_openai_project_header(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + headers = cfg.validate_environment( + headers={}, + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + optional_params={}, + litellm_params={"aws_bedrock_project_id": "proj_from_config"}, + ) + assert "OpenAI-Project" not in headers + assert headers["Content-Type"] == "application/json" + + def test_transform_request_is_openai_chat_body_not_converse(): cfg = AmazonBedrockRuntimeChatCompletionsConfig() body = cfg.transform_request(