From 84084d9a82548e364bea390ba021ff926e87c468 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Fri, 11 Sep 2026 12:22:16 -0700 Subject: [PATCH 01/31] feat(bedrock): send grok chat completions through runtime openai path Unspecified bedrock grok was rewritten to Converse. Chat completions now hit bedrock-runtime /openai/v1/chat/completions, and converse/ still uses Converse --- ci_cd/generate_model_prices_schema.py | 1 + litellm/__init__.py | 3 + litellm/_lazy_imports_registry.py | 5 + .../chat/chat_completions/transformation.py | 174 ++++++++++++++++++ litellm/llms/bedrock/common_utils.py | 21 +++ ...odel_prices_and_context_window_backup.json | 3 + model_prices_and_context_window.json | 3 + model_prices_and_context_window.schema.json | 3 + ...bedrock_chat_completions_transformation.py | 150 +++++++++++++++ 9 files changed, 363 insertions(+) create mode 100644 litellm/llms/bedrock/chat/chat_completions/transformation.py create mode 100644 tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index ab29b70bdd4..b7f7972d907 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -26,6 +26,7 @@ EXTRA_BOOLEAN_KEYS = frozenset( "gemini_audio_only_live", "uses_embed_content", "use_openai_responses_path", + "use_bedrock_runtime_chat_completions", "bedrock_converse_supports_strict_tools", "thinking_always_on", } diff --git a/litellm/__init__.py b/litellm/__init__.py index 1dfd146a00e..fea9150a6c0 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -1736,6 +1736,9 @@ if TYPE_CHECKING: from .llms.bedrock.chat.invoke_transformations.amazon_openai_transformation import ( AmazonBedrockOpenAIConfig as AmazonBedrockOpenAIConfig, ) + from .llms.bedrock.chat.chat_completions.transformation import ( + AmazonBedrockRuntimeChatCompletionsConfig as AmazonBedrockRuntimeChatCompletionsConfig, + ) from .llms.bedrock.image_generation.amazon_stability1_transformation import ( AmazonStabilityConfig as AmazonStabilityConfig, ) diff --git a/litellm/_lazy_imports_registry.py b/litellm/_lazy_imports_registry.py index dc323c8cc15..0880734fd9f 100644 --- a/litellm/_lazy_imports_registry.py +++ b/litellm/_lazy_imports_registry.py @@ -204,6 +204,7 @@ LLM_CONFIG_NAMES: Final = ( "AmazonTwelveLabsPegasusConfig", "AmazonInvokeConfig", "AmazonBedrockOpenAIConfig", + "AmazonBedrockRuntimeChatCompletionsConfig", "AmazonStabilityConfig", "AmazonStability3Config", "AmazonNovaCanvasConfig", @@ -847,6 +848,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = { ".llms.bedrock.chat.invoke_transformations.amazon_openai_transformation", "AmazonBedrockOpenAIConfig", ), + "AmazonBedrockRuntimeChatCompletionsConfig": ( + ".llms.bedrock.chat.chat_completions.transformation", + "AmazonBedrockRuntimeChatCompletionsConfig", + ), "AmazonStabilityConfig": ( ".llms.bedrock.image_generation.amazon_stability1_transformation", "AmazonStabilityConfig", diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py new file mode 100644 index 00000000000..3bf6b2a2ffd --- /dev/null +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -0,0 +1,174 @@ +""" +Native OpenAI Chat Completions on Amazon Bedrock Runtime. + +AWS serves this surface at +``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions``. +Grok 4.6 on runtime is one of the models that uses it: chat completions stay +chat completions instead of being rewritten to Converse. + +Usage: model="us.xai.grok-4.6" or model="bedrock/us.xai.grok-4.6" +Explicit ``bedrock/converse/...`` still uses Converse. +""" + +from collections.abc import AsyncIterator, Iterator +from typing import Any, Final + +import httpx + +import litellm +from litellm._logging import verbose_logger +from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM +from litellm.llms.bedrock.common_utils import BedrockError, strip_bedrock_routing_prefix +from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig +from litellm.types.llms.openai import AllMessageValues + + +class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): + def __init__(self, aws_signer: BaseAWSLLM | None = None): + super().__init__() + self._aws_signer: Final = aws_signer or BaseAWSLLM() + + @property + def custom_llm_provider(self) -> str | None: + return "bedrock" + + def get_error_class( + self, error_message: str, status_code: int, headers: dict[str, object] | httpx.Headers + ) -> BaseLLMException: + return BedrockError(status_code=status_code, message=error_message, headers=headers) + + def get_complete_url( + self, + api_base: str | None, + api_key: str | None, + model: str, + optional_params: dict, + litellm_params: dict, + stream: bool | None = None, + ) -> str: + if api_base is not None and "chat/completions" in api_base: + return api_base.rstrip("/") + aws_region_name: Final = self._aws_signer._get_aws_region_name(optional_params=optional_params, model=model) + endpoint_url, _ = self._aws_signer.get_runtime_endpoint( + api_base=api_base, + aws_bedrock_runtime_endpoint=optional_params.get("aws_bedrock_runtime_endpoint"), + aws_region_name=aws_region_name, + ) + base: Final = endpoint_url.rstrip("/") + if base.endswith("/openai/v1/chat/completions"): + return base + if base.endswith("/openai/v1"): + return f"{base}/chat/completions" + return f"{base}/openai/v1/chat/completions" + + def sign_request( + self, + headers: dict, + optional_params: dict, + request_data: dict, + api_base: str, + api_key: str | None = None, + model: str | None = None, + stream: bool | None = None, + fake_stream: bool | None = None, + ) -> tuple[dict, bytes | None]: + return self._aws_signer._sign_request( + service_name="bedrock", + headers=headers, + optional_params=optional_params, + request_data=request_data, + api_base=api_base, + api_key=api_key, + model=model, + stream=stream, + fake_stream=fake_stream, + ) + + def transform_request( + self, + model: str, + messages: list[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + inference_params: Final = { + k: v for k, v in optional_params.items() if k not in self._aws_signer.aws_authentication_params + } + return super().transform_request( + model=strip_bedrock_routing_prefix(model), + messages=messages, + optional_params=inference_params, + litellm_params=litellm_params, + headers=headers, + ) + + async def async_transform_request( + self, + model: str, + messages: list[AllMessageValues], + optional_params: dict, + litellm_params: dict, + headers: dict, + ) -> dict: + inference_params: Final = { + k: v for k, v in optional_params.items() if k not in self._aws_signer.aws_authentication_params + } + return await super().async_transform_request( + model=strip_bedrock_routing_prefix(model), + messages=messages, + optional_params=inference_params, + litellm_params=litellm_params, + headers=headers, + ) + + def validate_environment( + self, + headers: dict, + model: str, + messages: list[AllMessageValues], + optional_params: dict, + litellm_params: dict, + api_key: str | None = None, + api_base: str | None = None, + ) -> dict: + headers = super().validate_environment( + headers=headers, + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + api_key=api_key, + api_base=api_base, + ) + project_id: Final = litellm_params.get("aws_bedrock_project_id") + if project_id: + headers["OpenAI-Project"] = project_id + return headers + + def get_supported_openai_params(self, model: str) -> list: + base_params: Final = super().get_supported_openai_params(model) + try: + if litellm.supports_reasoning(model=model, custom_llm_provider=self.custom_llm_provider): + if "reasoning_effort" not in base_params: + base_params.append("reasoning_effort") + except Exception as e: + verbose_logger.debug("AmazonBedrockRuntimeChatCompletionsConfig: error checking reasoning support: %s", e) + return base_params + + def get_model_response_iterator( + self, + streaming_response: Iterator[str] | AsyncIterator[str] | Any, + sync_stream: bool, + json_mode: bool | None = False, + ) -> Any: + from litellm.llms.openai.chat.gpt_transformation import ( + OpenAIChatCompletionStreamingHandler, + ) + + return OpenAIChatCompletionStreamingHandler( + streaming_response=streaming_response, + sync_stream=sync_stream, + json_mode=json_mode, + ) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index cb2c70e74c8..366ccadfbb7 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -780,6 +780,20 @@ def strip_bedrock_routing_prefix(model: str) -> str: return model +def uses_bedrock_runtime_chat_completions(model: str) -> bool: + """Whether this Bedrock model should use runtime native Chat Completions. + + Data-driven from the price-map ``use_bedrock_runtime_chat_completions`` flag + so onboarding a model is a JSON change. Explicit ``converse/`` still wins in + ``get_bedrock_route`` because prefix routes are checked first. + """ + stripped: Final = strip_bedrock_routing_prefix(model) + return any( + (litellm.model_cost.get(key) or {}).get("use_bedrock_runtime_chat_completions") is True + for key in (model, stripped) + ) + + def strip_bedrock_throughput_suffix(model: str) -> str: """Strip throughput tier suffixes and context window suffixes from Bedrock model names.""" import re @@ -1107,6 +1121,7 @@ class BedrockModelInfo(BaseLLMModelInfo): "async_invoke", "openai", "mantle", + "chat_completions", ]: """ Get the bedrock route for the given model. @@ -1123,6 +1138,7 @@ class BedrockModelInfo(BaseLLMModelInfo): "async_invoke", "openai", "mantle", + "chat_completions", ], ] = { "invoke/": "invoke", @@ -1152,6 +1168,9 @@ class BedrockModelInfo(BaseLLMModelInfo): if is_bedrock_application_inference_profile_arn(model): return "converse" + if uses_bedrock_runtime_chat_completions(model): + return "chat_completions" + base_model: Final = BedrockModelInfo.get_base_model(model) alt_model: Final = BedrockModelInfo.get_non_litellm_routing_model_name(model=model) if base_model in litellm.bedrock_converse_models or alt_model in litellm.bedrock_converse_models: @@ -1328,6 +1347,8 @@ def get_bedrock_chat_config(model: str): return litellm.AmazonConverseConfig() elif bedrock_route == "openai": return litellm.AmazonBedrockOpenAIConfig() + elif bedrock_route == "chat_completions": + return litellm.AmazonBedrockRuntimeChatCompletionsConfig() elif bedrock_route == "agent": from litellm.llms.bedrock.chat.invoke_agent.transformation import ( AmazonInvokeAgentConfig, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index ad992ff92eb..3c6900cb136 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -44462,6 +44462,7 @@ "input_cost_per_token": 2.64e-06, "output_cost_per_token": 7.92e-06, "cache_read_input_token_cost": 6.6e-07, + "use_bedrock_runtime_chat_completions": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -55840,6 +55841,7 @@ "input_cost_per_token": 2.2e-06, "output_cost_per_token": 6.6e-06, "cache_read_input_token_cost": 5.5e-07, + "use_bedrock_runtime_chat_completions": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -55855,6 +55857,7 @@ "input_cost_per_token": 2e-06, "output_cost_per_token": 6e-06, "cache_read_input_token_cost": 5e-07, + "use_bedrock_runtime_chat_completions": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ad992ff92eb..3c6900cb136 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -44462,6 +44462,7 @@ "input_cost_per_token": 2.64e-06, "output_cost_per_token": 7.92e-06, "cache_read_input_token_cost": 6.6e-07, + "use_bedrock_runtime_chat_completions": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -55840,6 +55841,7 @@ "input_cost_per_token": 2.2e-06, "output_cost_per_token": 6.6e-06, "cache_read_input_token_cost": 5.5e-07, + "use_bedrock_runtime_chat_completions": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -55855,6 +55857,7 @@ "input_cost_per_token": 2e-06, "output_cost_per_token": 6e-06, "cache_read_input_token_cost": 5e-07, + "use_bedrock_runtime_chat_completions": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index 7ed1e7e568b..59d6b75a5f6 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -855,6 +855,9 @@ "minimum": 0, "description": "Provider default tokens-per-minute limit." }, + "use_bedrock_runtime_chat_completions": { + "type": "boolean" + }, "use_openai_responses_path": { "type": "boolean" }, diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py new file mode 100644 index 00000000000..7724b3f0a52 --- /dev/null +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -0,0 +1,150 @@ +"""Native Bedrock Runtime Chat Completions: Grok stays on /openai/v1/chat/completions.""" + +import json +from unittest.mock import patch + +import httpx +import pytest + +import litellm +from litellm.llms.bedrock.chat.chat_completions.transformation import ( + AmazonBedrockRuntimeChatCompletionsConfig, +) +from litellm.llms.bedrock.common_utils import ( + BedrockModelInfo, + get_bedrock_chat_config, + uses_bedrock_runtime_chat_completions, +) + + +@pytest.fixture +def local_cost_map(monkeypatch): + original_model_cost = litellm.model_cost + try: + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true") + litellm.model_cost = litellm.get_model_cost_map(url="") + litellm.get_model_info.cache_clear() + yield + finally: + litellm.model_cost = original_model_cost + litellm.get_model_info.cache_clear() + + +@pytest.mark.parametrize( + "model", + [ + "us.xai.grok-4.6", + "global.xai.grok-4.6", + "us-gov.xai.grok-4.6", + "bedrock/us.xai.grok-4.6", + ], +) +def test_grok_runtime_models_use_chat_completions_route(local_cost_map, model): + assert uses_bedrock_runtime_chat_completions(model) is True + assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions" + assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig) + + +def test_explicit_converse_prefix_still_uses_converse(local_cost_map): + assert BedrockModelInfo.get_bedrock_route("bedrock/converse/us.xai.grok-4.6") == "converse" + assert BedrockModelInfo.get_bedrock_route("converse/us.xai.grok-4.6") == "converse" + + +def test_claude_stays_on_converse(local_cost_map): + assert uses_bedrock_runtime_chat_completions("us.anthropic.claude-3-sonnet-20240229-v1:0") is False + assert BedrockModelInfo.get_bedrock_route("us.anthropic.claude-3-sonnet-20240229-v1:0") == "converse" + + +def test_flag_absent_means_no_chat_completions_route(monkeypatch): + monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": {"litellm_provider": "bedrock_converse"}}) + assert uses_bedrock_runtime_chat_completions("us.xai.grok-4.6") is False + + +def test_complete_url_is_runtime_openai_chat_completions(monkeypatch): + monkeypatch.setenv("AWS_REGION_NAME", "us-east-1") + monkeypatch.delenv("AWS_BEDROCK_RUNTIME_ENDPOINT", raising=False) + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + url = cfg.get_complete_url( + api_base=None, + api_key=None, + model="us.xai.grok-4.6", + optional_params={}, + litellm_params={}, + ) + assert url == "https://bedrock-runtime.us-east-1.amazonaws.com/openai/v1/chat/completions" + + +def test_complete_url_appends_to_openai_v1_base(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + url = cfg.get_complete_url( + api_base="https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1", + api_key=None, + model="us.xai.grok-4.6", + optional_params={}, + litellm_params={}, + ) + assert url == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + + +def test_transform_request_is_openai_chat_body_not_converse(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + body = cfg.transform_request( + model="bedrock/us.xai.grok-4.6", + messages=[{"role": "user", "content": "hello"}], + optional_params={"temperature": 0.2, "aws_region_name": "us-east-1"}, + litellm_params={}, + headers={}, + ) + assert body["model"] == "us.xai.grok-4.6" + assert body["messages"] == [{"role": "user", "content": "hello"}] + assert body["temperature"] == 0.2 + assert "aws_region_name" not in body + assert "inferenceConfig" not in body + assert "messages" in body + + +def test_completion_posts_runtime_chat_completions(local_cost_map, monkeypatch): + monkeypatch.setenv("AWS_REGION_NAME", "us-west-2") + monkeypatch.delenv("AWS_BEDROCK_RUNTIME_ENDPOINT", raising=False) + monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False) + monkeypatch.setenv("AWS_ACCESS_KEY_ID", "testing") + monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing") + monkeypatch.setenv("AWS_SESSION_TOKEN", "testing") + + requests: list[dict] = [] + + def mock_post(self, url, data=None, json=None, headers=None, **kwargs): + requests.append({"url": url, "data": data, "json": json, "headers": headers or {}}) + return httpx.Response( + status_code=200, + json={ + "id": "chatcmpl-test", + "object": "chat.completion", + "created": 1733529600, + "model": "us.xai.grok-4.6", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "ok"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + }, + request=httpx.Request("POST", url), + ) + + with patch("litellm.llms.custom_httpx.http_handler.HTTPHandler.post", mock_post): + response = litellm.completion( + model="us.xai.grok-4.6", + messages=[{"role": "user", "content": "hello"}], + ) + + assert response.choices[0].message.content == "ok" + assert len(requests) == 1 + assert requests[0]["url"] == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + raw = requests[0]["data"] + body = json.loads(raw) if isinstance(raw, (str, bytes, bytearray)) else (requests[0]["json"] or {}) + assert body["model"] == "us.xai.grok-4.6" + assert body["messages"] == [{"role": "user", "content": "hello"}] + assert "inferenceConfig" not in body From 6b9f067c80eea6ce302eec5205aaf7892f1131e9 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:05:46 -0700 Subject: [PATCH 02/31] feat(bedrock): serve gpt-oss and gpt-5.6 chat completions on runtime's native openai path --- ci_cd/generate_model_prices_schema.py | 1 + .../chat/chat_completions/transformation.py | 319 ++++++++-- litellm/llms/bedrock/common_utils.py | 81 ++- litellm/main.py | 2 +- ...odel_prices_and_context_window_backup.json | 14 + litellm/utils.py | 26 +- model_prices_and_context_window.json | 14 + model_prices_and_context_window.schema.json | 3 + ...bedrock_chat_completions_transformation.py | 554 ++++++++++++++++-- ..._cross_region_inference_profile_mapping.py | 9 +- tests/test_litellm/test_utils.py | 2 + 11 files changed, 897 insertions(+), 128 deletions(-) diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index 5a2a56a5b21..648cceb79b0 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -31,6 +31,7 @@ EXTRA_BOOLEAN_KEYS = frozenset( "uses_embed_content", "use_openai_responses_path", "use_bedrock_runtime_chat_completions", + "bedrock_runtime_chat_completions_tools_require_reasoning_none", "bedrock_converse_supports_strict_tools", "thinking_always_on", } diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 3bf6b2a2ffd..49218270064 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -2,30 +2,170 @@ Native OpenAI Chat Completions on Amazon Bedrock Runtime. AWS serves this surface at -``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions``. -Grok 4.6 on runtime is one of the models that uses it: chat completions stay -chat completions instead of being rewritten to Converse. +``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions`` +for the models whose price-map entry sets ``use_bedrock_runtime_chat_completions`` +(Grok 4.6, gpt-oss, the GPT-5.6 family): chat completions stay chat completions +instead of being rewritten to Converse. -Usage: model="us.xai.grok-4.6" or model="bedrock/us.xai.grok-4.6" -Explicit ``bedrock/converse/...`` still uses Converse. +Usage: model="us.xai.grok-4.6", model="bedrock/openai.gpt-oss-20b-1:0" or +model="bedrock/global.openai.gpt-5.6-sol". Explicit ``bedrock/converse/...`` +still uses Converse, and so does a request that needs a Converse-only feature +(``bedrock_request_needs_converse`` in ``common_utils``). """ -from collections.abc import AsyncIterator, Iterator -from typing import Any, Final +from collections.abc import AsyncIterator, Iterator, Mapping +from dataclasses import dataclass, replace +from types import MappingProxyType +from typing import TYPE_CHECKING, Final, Literal import httpx import litellm -from litellm._logging import verbose_logger from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM from litellm.llms.bedrock.common_utils import BedrockError, strip_bedrock_routing_prefix +from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig from litellm.types.llms.openai import AllMessageValues +from litellm.types.utils import Choices, ModelResponse, ModelResponseStream + +if TYPE_CHECKING: + import tiktoken + + from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj + +REASONING_OPEN_TAG: Final = "" +REASONING_CLOSE_TAG: Final = "" + + +def _held_close_tag_prefix(text: str) -> int: + return next( + ( + size + for size in range(min(len(text), len(REASONING_CLOSE_TAG) - 1), 0, -1) + if REASONING_CLOSE_TAG.startswith(text[-size:]) + ), + 0, + ) + + +@dataclass(frozen=True, slots=True) +class ReasoningTagSplitter: + """ + The same split for a stream of content deltas, where a tag can arrive across chunks. + + ``feed`` returns the next state plus the reasoning and content text the delta contributes; + ``flush`` releases what the stream ended on before a tag resolved. + """ + + phase: Literal["start", "reasoning", "after_close", "content"] = "start" + pending: str = "" + + def feed(self, text: str) -> tuple["ReasoningTagSplitter", str, str]: + match self.phase: + case "content": + return self, "", text + case "after_close": + content: Final = text.lstrip() + return (replace(self, phase="content") if content else self), "", content + case "start": + return self._feed_start(self.pending + text) + case "reasoning": + return self._feed_reasoning(self.pending + text) + + def _feed_start(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]: + if buffered.startswith(REASONING_OPEN_TAG): + return replace(self, phase="reasoning", pending="")._feed_reasoning(buffered[len(REASONING_OPEN_TAG) :]) + if REASONING_OPEN_TAG.startswith(buffered): + return replace(self, pending=buffered), "", "" + return replace(self, phase="content", pending=""), "", buffered + + def _feed_reasoning(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]: + close_at: Final = buffered.find(REASONING_CLOSE_TAG) + if close_at >= 0: + after_close: Final = replace(self, phase="after_close", pending="") + next_state, _, content = after_close.feed(buffered[close_at + len(REASONING_CLOSE_TAG) :]) + return next_state, buffered[:close_at], content + held: Final = _held_close_tag_prefix(buffered) + return replace(self, pending=buffered[len(buffered) - held :]), buffered[: len(buffered) - held], "" + + def flush(self) -> tuple["ReasoningTagSplitter", str, str]: + drained: Final = replace(self, phase="content", pending="") + if self.phase == "reasoning": + return drained, self.pending, "" + return drained, "", self.pending + + +def _split_streamed_content( + splitter: ReasoningTagSplitter, content: str | None, finished: bool +) -> tuple[ReasoningTagSplitter, str, str]: + fed_state, fed_reasoning, fed_content = splitter.feed(content or "") + if not finished: + return fed_state, fed_reasoning, fed_content + drained, flushed_reasoning, flushed_content = fed_state.flush() + return drained, fed_reasoning + flushed_reasoning, fed_content + flushed_content + + +def split_reasoning_tag(content: str) -> tuple[str | None, str]: + """ + Split gpt-oss's inline ``...`` prefix out of a complete message. + + Runs the streaming splitter over the whole message, so a streamed and a non-streamed + response to the same completion split identically. Returns ``(None, content)`` when the + message does not start with the tag. + """ + _, reasoning, body = _split_streamed_content(ReasoningTagSplitter(), content, finished=True) + return reasoning or None, body + + +class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler): + """OpenAI chunk parsing plus the ```` split, tracked per choice index.""" + + def __init__( + self, + streaming_response: Iterator[str] | AsyncIterator[str] | ModelResponse, + sync_stream: bool, + json_mode: bool | None = False, + ) -> None: + super().__init__(streaming_response=streaming_response, sync_stream=sync_stream, json_mode=json_mode) + self._splitters: Mapping[int, ReasoningTagSplitter] = MappingProxyType({}) + + def chunk_parser(self, chunk: dict) -> ModelResponseStream: # mutable-ok: BaseModelResponseIterator signature + parsed: Final = super().chunk_parser(chunk) + for choice in parsed.choices: + next_state, reasoning, content = _split_streamed_content( + self._splitters.get(choice.index, ReasoningTagSplitter()), + choice.delta.content, + choice.finish_reason is not None, + ) + self._splitters = MappingProxyType({**self._splitters, choice.index: next_state}) + if reasoning: + choice.delta.reasoning_content = f"{getattr(choice.delta, 'reasoning_content', None) or ''}{reasoning}" + if content or choice.delta.content is not None: + choice.delta.content = content + return parsed + + +def with_max_completion_tokens(params: Mapping[str, object]) -> Mapping[str, object]: + """ + Send the caller's ``max_tokens`` as ``max_completion_tokens``. + + Every model on this surface accepts ``max_completion_tokens`` and the GPT-5.6 family + rejects ``max_tokens``; an explicit ``max_completion_tokens`` wins when both are set. + """ + if "max_tokens" not in params: + return params + return MappingProxyType( + { + key: value + for key, value in (("max_completion_tokens", params["max_tokens"]), *params.items()) + if key != "max_tokens" + } + ) class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): - def __init__(self, aws_signer: BaseAWSLLM | None = None): + def __init__(self, aws_signer: BaseAWSLLM | None = None) -> None: super().__init__() self._aws_signer: Final = aws_signer or BaseAWSLLM() @@ -34,7 +174,10 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): return "bedrock" def get_error_class( - self, error_message: str, status_code: int, headers: dict[str, object] | httpx.Headers + self, + error_message: str, + status_code: int, + headers: dict[str, object] | httpx.Headers, # mutable-ok: BaseConfig signature ) -> BaseLLMException: return BedrockError(status_code=status_code, message=error_message, headers=headers) @@ -43,13 +186,15 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): api_base: str | None, api_key: str | None, model: str, - optional_params: dict, - litellm_params: dict, + optional_params: dict, # mutable-ok: BaseConfig signature + litellm_params: dict, # mutable-ok: BaseConfig signature stream: bool | None = None, ) -> str: if api_base is not None and "chat/completions" in api_base: return api_base.rstrip("/") - aws_region_name: Final = self._aws_signer._get_aws_region_name(optional_params=optional_params, model=model) + aws_region_name: Final = self._aws_signer._get_aws_region_name( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public region resolver + optional_params=optional_params, model=model + ) endpoint_url, _ = self._aws_signer.get_runtime_endpoint( api_base=api_base, aws_bedrock_runtime_endpoint=optional_params.get("aws_bedrock_runtime_endpoint"), @@ -64,16 +209,16 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): def sign_request( self, - headers: dict, - optional_params: dict, - request_data: dict, + headers: dict, # mutable-ok: BaseConfig signature + optional_params: dict, # mutable-ok: BaseConfig signature + request_data: dict, # mutable-ok: BaseConfig signature api_base: str, api_key: str | None = None, model: str | None = None, stream: bool | None = None, fake_stream: bool | None = None, - ) -> tuple[dict, bytes | None]: - return self._aws_signer._sign_request( + ) -> tuple[dict, bytes | None]: # mutable-ok: BaseConfig signature + return self._aws_signer._sign_request( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public signer service_name="bedrock", headers=headers, optional_params=optional_params, @@ -85,21 +230,44 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): fake_stream=fake_stream, ) + def map_openai_params( + self, + non_default_params: dict, # mutable-ok: BaseConfig signature + optional_params: dict, # mutable-ok: BaseConfig signature + model: str, + drop_params: bool, + replace_max_completion_tokens_with_max_tokens: bool = False, + ) -> dict: # mutable-ok: BaseConfig signature + mapped: Final = super().map_openai_params( + non_default_params=non_default_params, + optional_params=optional_params, + model=model, + drop_params=drop_params, + replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens, + ) + return dict(with_max_completion_tokens(mapped)) # mutable-ok: get_optional_params keeps filling this dict + + def _inference_params( + self, optional_params: Mapping[str, object] + ) -> dict[str, object]: # mutable-ok: BaseConfig signature of transform_request + return { # mutable-ok: OpenAILikeChatConfig.transform_request takes a plain dict + key: value + for key, value in optional_params.items() + if key not in self._aws_signer.aws_authentication_params + } + def transform_request( self, model: str, - messages: list[AllMessageValues], - optional_params: dict, - litellm_params: dict, - headers: dict, - ) -> dict: - inference_params: Final = { - k: v for k, v in optional_params.items() if k not in self._aws_signer.aws_authentication_params - } + messages: list[AllMessageValues], # mutable-ok: BaseConfig signature + optional_params: dict, # mutable-ok: BaseConfig signature + litellm_params: dict, # mutable-ok: BaseConfig signature + headers: dict, # mutable-ok: BaseConfig signature + ) -> dict: # mutable-ok: BaseConfig signature return super().transform_request( model=strip_bedrock_routing_prefix(model), messages=messages, - optional_params=inference_params, + optional_params=self._inference_params(optional_params), litellm_params=litellm_params, headers=headers, ) @@ -107,33 +275,68 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): async def async_transform_request( self, model: str, - messages: list[AllMessageValues], - optional_params: dict, - litellm_params: dict, - headers: dict, - ) -> dict: - inference_params: Final = { - k: v for k, v in optional_params.items() if k not in self._aws_signer.aws_authentication_params - } + messages: list[AllMessageValues], # mutable-ok: BaseConfig signature + optional_params: dict, # mutable-ok: BaseConfig signature + litellm_params: dict, # mutable-ok: BaseConfig signature + headers: dict, # mutable-ok: BaseConfig signature + ) -> dict: # mutable-ok: BaseConfig signature return await super().async_transform_request( model=strip_bedrock_routing_prefix(model), messages=messages, - optional_params=inference_params, + optional_params=self._inference_params(optional_params), litellm_params=litellm_params, headers=headers, ) + def transform_response( + self, + model: str, + raw_response: httpx.Response, + model_response: ModelResponse, + logging_obj: "LiteLLMLoggingObj", + request_data: dict, # mutable-ok: BaseConfig signature + messages: list[AllMessageValues], # mutable-ok: BaseConfig signature + optional_params: dict, # mutable-ok: BaseConfig signature + litellm_params: dict, # mutable-ok: BaseConfig signature + encoding: "tiktoken.Encoding | None", + api_key: str | None = None, + json_mode: bool | None = None, + ) -> ModelResponse: + response: Final = super().transform_response( + model=model, + raw_response=raw_response, + model_response=model_response, + logging_obj=logging_obj, + request_data=request_data, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + encoding=encoding, + api_key=api_key, + json_mode=json_mode, + ) + for choice in response.choices: + if not isinstance(choice, Choices) or not isinstance(choice.message.content, str): + continue + reasoning, content = split_reasoning_tag(choice.message.content) + if reasoning is not None: + choice.message.reasoning_content = ( + f"{getattr(choice.message, 'reasoning_content', None) or ''}{reasoning}" + ) + choice.message.content = content + return response + def validate_environment( self, - headers: dict, + headers: dict, # mutable-ok: BaseConfig signature model: str, - messages: list[AllMessageValues], - optional_params: dict, - litellm_params: dict, + messages: list[AllMessageValues], # mutable-ok: BaseConfig signature + optional_params: dict, # mutable-ok: BaseConfig signature + litellm_params: dict, # mutable-ok: BaseConfig signature api_key: str | None = None, api_base: str | None = None, - ) -> dict: - headers = super().validate_environment( + ) -> dict: # mutable-ok: BaseConfig signature + validated: Final = super().validate_environment( headers=headers, model=model, messages=messages, @@ -143,31 +346,25 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): api_base=api_base, ) project_id: Final = litellm_params.get("aws_bedrock_project_id") - if project_id: - headers["OpenAI-Project"] = project_id - return headers + if not project_id: + return validated + return {**validated, "OpenAI-Project": project_id} # mutable-ok: BaseConfig signature returns a dict - def get_supported_openai_params(self, model: str) -> list: - base_params: Final = super().get_supported_openai_params(model) - try: - if litellm.supports_reasoning(model=model, custom_llm_provider=self.custom_llm_provider): - if "reasoning_effort" not in base_params: - base_params.append("reasoning_effort") - except Exception as e: - verbose_logger.debug("AmazonBedrockRuntimeChatCompletionsConfig: error checking reasoning support: %s", e) - return base_params + def get_supported_openai_params(self, model: str) -> list: # mutable-ok: BaseConfig signature + base_params: Final = [param for param in super().get_supported_openai_params(model) if param != "n"] + if "reasoning_effort" in base_params or not litellm.supports_reasoning( + model=model, custom_llm_provider=self.custom_llm_provider + ): + return base_params + return [*base_params, "reasoning_effort"] # mutable-ok: BaseConfig signature returns a list def get_model_response_iterator( self, - streaming_response: Iterator[str] | AsyncIterator[str] | Any, + streaming_response: Iterator[str] | AsyncIterator[str] | ModelResponse, sync_stream: bool, json_mode: bool | None = False, - ) -> Any: - from litellm.llms.openai.chat.gpt_transformation import ( - OpenAIChatCompletionStreamingHandler, - ) - - return OpenAIChatCompletionStreamingHandler( + ) -> BedrockRuntimeChatCompletionsStreamingHandler: + return BedrockRuntimeChatCompletionsStreamingHandler( streaming_response=streaming_response, sync_stream=sync_stream, json_mode=json_mode, diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index f6384aa97f1..124ef7c646c 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -28,6 +28,7 @@ from litellm.llms.base_llm.anthropic_messages.transformation import ( ) from litellm.llms.base_llm.base_utils import BaseLLMModelInfo, BaseTokenCounter from litellm.llms.base_llm.chat.transformation import BaseLLMException +from litellm.llms.bedrock.request_metadata import bedrock_request_metadata_is_owned from litellm.secret_managers.main import get_secret, get_secret_str from litellm.types.llms.bedrock import AWS_AUTH_PARAM_KEYS, AwsAuthParams @@ -37,6 +38,18 @@ if TYPE_CHECKING: _ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs" _OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.") +BedrockRoute = Literal[ + "converse", + "invoke", + "claude_platform", + "converse_like", + "agent", + "agentcore", + "async_invoke", + "openai", + "mantle", + "chat_completions", +] def error_response_text(response: httpx.Response) -> str: @@ -782,17 +795,54 @@ def strip_bedrock_routing_prefix(model: str) -> str: return model +def _bedrock_price_map_flag(model: str, flag: str) -> bool: + entries: Final = (litellm.model_cost.get(key) for key in (model, strip_bedrock_routing_prefix(model))) + return any(entry is not None and entry.get(flag) is True for entry in entries) + + def uses_bedrock_runtime_chat_completions(model: str) -> bool: """Whether this Bedrock model should use runtime native Chat Completions. Data-driven from the price-map ``use_bedrock_runtime_chat_completions`` flag so onboarding a model is a JSON change. Explicit ``converse/`` still wins in - ``get_bedrock_route`` because prefix routes are checked first. + ``get_bedrock_route`` because prefix routes are checked first, and a request + that needs a Converse-only feature (``bedrock_request_needs_converse``) is + served by Converse even on a flagged model. """ - stripped: Final = strip_bedrock_routing_prefix(model) - return any( - (litellm.model_cost.get(key) or {}).get("use_bedrock_runtime_chat_completions") is True - for key in (model, stripped) + return _bedrock_price_map_flag(model, "use_bedrock_runtime_chat_completions") + + +def bedrock_runtime_chat_completions_tools_require_reasoning_none(model: str) -> bool: + """Whether AWS's native Chat Completions only serves this model's function tools with ``reasoning_effort="none"``. + + Data-driven from the price-map ``bedrock_runtime_chat_completions_tools_require_reasoning_none`` + flag (the GPT-5.6 family). Converse serves tools with any effort, so those requests fall back to it. + """ + return _bedrock_price_map_flag(model, "bedrock_runtime_chat_completions_tools_require_reasoning_none") + + +BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( + ("guardrailConfig", "performanceConfig", "serviceTier", "requestMetadata", "outputConfig") +) + + +def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: + """Whether a request on a runtime-Chat-Completions model must still be served by Converse. + + Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``) are rejected as malformed input by + AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, + and function tools on a ``bedrock_runtime_chat_completions_tools_require_reasoning_none`` model are + rejected there unless ``reasoning_effort`` is exactly ``"none"``. + """ + if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS): + return True + if bedrock_request_metadata_is_owned(): + return True + if not request_params.get("tools"): + return False + return ( + bedrock_runtime_chat_completions_tools_require_reasoning_none(model) + and request_params.get("reasoning_effort") != "none" ) @@ -1130,20 +1180,13 @@ class BedrockModelInfo(BaseLLMModelInfo): @staticmethod def get_bedrock_route( model: str, - ) -> Literal[ - "converse", - "invoke", - "claude_platform", - "converse_like", - "agent", - "agentcore", - "async_invoke", - "openai", - "mantle", - "chat_completions", - ]: + request_params: Mapping[str, object] | None = None, + ) -> BedrockRoute: """ Get the bedrock route for the given model. + + ``request_params`` (the caller's chat params) lets a runtime Chat Completions + model fall back to Converse for the requests only Converse can serve. """ route_mappings: dict[ str, @@ -1187,7 +1230,9 @@ class BedrockModelInfo(BaseLLMModelInfo): if is_bedrock_application_inference_profile_arn(model): return "converse" - if uses_bedrock_runtime_chat_completions(model): + if uses_bedrock_runtime_chat_completions(model) and not ( + request_params is not None and bedrock_request_needs_converse(model, request_params) + ): return "chat_completions" base_model: Final = BedrockModelInfo.get_base_model(model) diff --git a/litellm/main.py b/litellm/main.py index 34410f9497c..ed78f209ca2 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -4078,7 +4078,7 @@ def _complete_bedrock(ctx: _CompletionDispatchContext) -> _CompletionDispatchRes if "aws_region_name" not in optional_params or optional_params["aws_region_name"] is None: optional_params["aws_region_name"] = aws_bedrock_client.meta.region_name - bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model) + bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model, optional_params) if bedrock_route == "claude_platform": provider_config = ProviderConfigManager.get_provider_chat_config( model=model, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 70314a2e823..05658ce2d15 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -40870,6 +40870,7 @@ "output_cost_per_token": 0.0 }, "openai.gpt-oss-120b-1:0": { + "use_bedrock_runtime_chat_completions": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -40883,6 +40884,7 @@ "supports_tool_choice": true }, "openai.gpt-oss-20b-1:0": { + "use_bedrock_runtime_chat_completions": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -58441,6 +58443,8 @@ "supports_web_search": true }, "us.openai.gpt-5.6-sol": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 4.4e-06, "input_cost_per_token_above_272k_tokens": 8.8e-06, "cache_creation_input_token_cost": 5.5e-06, @@ -58470,6 +58474,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-sol": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 4e-06, "input_cost_per_token_above_272k_tokens": 8e-06, "cache_creation_input_token_cost": 5e-06, @@ -58499,6 +58505,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-terra": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, "cache_creation_input_token_cost": 2.75e-06, @@ -58528,6 +58536,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-terra": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "cache_creation_input_token_cost": 2.5e-06, @@ -58557,6 +58567,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-luna": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2.2e-07, "input_cost_per_token_above_272k_tokens": 4.4e-07, "cache_creation_input_token_cost": 2.75e-07, @@ -58586,6 +58598,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-luna": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, "cache_creation_input_token_cost": 2.5e-07, diff --git a/litellm/utils.py b/litellm/utils.py index b724313641f..82c8069770d 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -401,7 +401,7 @@ if TYPE_CHECKING: BaseVectorStoreFilesConfig, ) from litellm.llms.base_llm.videos.transformation import BaseVideoConfig - from litellm.llms.bedrock.common_utils import BedrockModelInfo + from litellm.llms.bedrock.common_utils import BedrockModelInfo, BedrockRoute from litellm.llms.bedrock.embed.amazon_nova_transformation import ( AmazonNovaEmbeddingConfig, ) @@ -3350,6 +3350,17 @@ def _should_drop_param(k, additional_drop_params) -> bool: return False +def _bedrock_route_for_request( + model: str, passed_params: Mapping[str, object], additional_drop_params: list | None +) -> BedrockRoute: + from litellm.llms.bedrock.common_utils import BedrockModelInfo + + return BedrockModelInfo.get_bedrock_route( + model, + {k: v for k, v in passed_params.items() if not _should_drop_param(k, additional_drop_params)}, + ) + + def _get_non_default_params(passed_params: dict, default_params: dict, additional_drop_params: list | None) -> dict: non_default_params: Final = {} for k, v in passed_params.items(): @@ -4401,9 +4412,17 @@ def get_optional_params( message=f"{custom_llm_provider} does not support parameters: {list(unsupported_params.keys())}, for model={model}. To drop these, set `litellm.drop_params=True` or for proxy:\n\n`litellm_settings:\n drop_params: true`\n. \n If you want to use these params dynamically send allowed_openai_params={list(unsupported_params.keys())} in your request.", ) + bedrock_route: Final = ( + _bedrock_route_for_request(model, passed_params, additional_drop_params) + if custom_llm_provider == "bedrock" + else None + ) get_supported_openai_params: Final = getattr(sys.modules[__name__], "get_supported_openai_params") - supported_params = get_supported_openai_params( - model=model, custom_llm_provider=custom_llm_provider, base_model=base_model + supported_params = ( + litellm.AmazonConverseConfig().get_supported_openai_params(model=model) + if bedrock_route == "converse" + and isinstance(provider_config, litellm.AmazonBedrockRuntimeChatCompletionsConfig) + else get_supported_openai_params(model=model, custom_llm_provider=custom_llm_provider, base_model=base_model) ) if supported_params is None: supported_params = get_supported_openai_params(model=model, custom_llm_provider="openai") @@ -4573,7 +4592,6 @@ def get_optional_params( ) elif custom_llm_provider == "bedrock": BedrockModelInfo: Final = getattr(sys.modules[__name__], "BedrockModelInfo") - bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model) bedrock_base_model: Final = BedrockModelInfo.get_base_model(model) if bedrock_route == "converse" or bedrock_route == "converse_like": optional_params = litellm.AmazonConverseConfig().map_openai_params( diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 70314a2e823..05658ce2d15 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -40870,6 +40870,7 @@ "output_cost_per_token": 0.0 }, "openai.gpt-oss-120b-1:0": { + "use_bedrock_runtime_chat_completions": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -40883,6 +40884,7 @@ "supports_tool_choice": true }, "openai.gpt-oss-20b-1:0": { + "use_bedrock_runtime_chat_completions": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -58441,6 +58443,8 @@ "supports_web_search": true }, "us.openai.gpt-5.6-sol": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 4.4e-06, "input_cost_per_token_above_272k_tokens": 8.8e-06, "cache_creation_input_token_cost": 5.5e-06, @@ -58470,6 +58474,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-sol": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 4e-06, "input_cost_per_token_above_272k_tokens": 8e-06, "cache_creation_input_token_cost": 5e-06, @@ -58499,6 +58505,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-terra": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, "cache_creation_input_token_cost": 2.75e-06, @@ -58528,6 +58536,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-terra": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "cache_creation_input_token_cost": 2.5e-06, @@ -58557,6 +58567,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-luna": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2.2e-07, "input_cost_per_token_above_272k_tokens": 4.4e-07, "cache_creation_input_token_cost": 2.75e-07, @@ -58586,6 +58598,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-luna": { + "use_bedrock_runtime_chat_completions": true, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, "cache_creation_input_token_cost": 2.5e-07, diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index eca89e23887..9db1c8eedc4 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -74,6 +74,9 @@ "xhigh" ] }, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": { + "type": "boolean" + }, "cache_creation_input_audio_token_cost": { "type": "number", "minimum": 0 diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 7724b3f0a52..230e323ace3 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -1,7 +1,6 @@ -"""Native Bedrock Runtime Chat Completions: Grok stays on /openai/v1/chat/completions.""" +"""Native Bedrock Runtime Chat Completions: Grok, gpt-oss and GPT-5.6 stay on /openai/v1/chat/completions.""" import json -from unittest.mock import patch import httpx import pytest @@ -9,25 +8,28 @@ import pytest import litellm from litellm.llms.bedrock.chat.chat_completions.transformation import ( AmazonBedrockRuntimeChatCompletionsConfig, + BedrockRuntimeChatCompletionsStreamingHandler, + ReasoningTagSplitter, + split_reasoning_tag, + with_max_completion_tokens, ) from litellm.llms.bedrock.common_utils import ( + BEDROCK_CONVERSE_ONLY_REQUEST_KEYS, BedrockModelInfo, + bedrock_request_needs_converse, get_bedrock_chat_config, uses_bedrock_runtime_chat_completions, ) +from litellm.llms.custom_httpx.http_handler import HTTPHandler @pytest.fixture def local_cost_map(monkeypatch): - original_model_cost = litellm.model_cost - try: - monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true") - litellm.model_cost = litellm.get_model_cost_map(url="") - litellm.get_model_info.cache_clear() - yield - finally: - litellm.model_cost = original_model_cost - litellm.get_model_info.cache_clear() + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "true") + monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url="")) + litellm.get_model_info.cache_clear() + yield + litellm.get_model_info.cache_clear() @pytest.mark.parametrize( @@ -103,7 +105,27 @@ def test_transform_request_is_openai_chat_body_not_converse(): assert "messages" in body -def test_completion_posts_runtime_chat_completions(local_cost_map, monkeypatch): +def _chat_completion_json(content, model, tool_calls=None): + message = {"role": "assistant", "content": content, **({"tool_calls": tool_calls} if tool_calls else {})} + return { + "id": "chatcmpl-test", + "object": "chat.completion", + "created": 1733529600, + "model": model, + "choices": [{"index": 0, "message": message, "finish_reason": "tool_calls" if tool_calls else "stop"}], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + } + + +CONVERSE_JSON = { + "output": {"message": {"role": "assistant", "content": [{"text": "ok"}]}}, + "stopReason": "end_turn", + "usage": {"inputTokens": 1, "outputTokens": 1, "totalTokens": 2}, +} + + +@pytest.fixture +def fake_aws_env(monkeypatch): monkeypatch.setenv("AWS_REGION_NAME", "us-west-2") monkeypatch.delenv("AWS_BEDROCK_RUNTIME_ENDPOINT", raising=False) monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False) @@ -111,40 +133,490 @@ def test_completion_posts_runtime_chat_completions(local_cost_map, monkeypatch): monkeypatch.setenv("AWS_SECRET_ACCESS_KEY", "testing") monkeypatch.setenv("AWS_SESSION_TOKEN", "testing") - requests: list[dict] = [] - def mock_post(self, url, data=None, json=None, headers=None, **kwargs): - requests.append({"url": url, "data": data, "json": json, "headers": headers or {}}) - return httpx.Response( - status_code=200, - json={ - "id": "chatcmpl-test", - "object": "chat.completion", - "created": 1733529600, - "model": "us.xai.grok-4.6", - "choices": [ - { - "index": 0, - "message": {"role": "assistant", "content": "ok"}, - "finish_reason": "stop", - } - ], - "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, - }, - request=httpx.Request("POST", url), - ) +def _recording_client(**response_kwargs): + requests: list[httpx.Request] = [] - with patch("litellm.llms.custom_httpx.http_handler.HTTPHandler.post", mock_post): - response = litellm.completion( - model="us.xai.grok-4.6", - messages=[{"role": "user", "content": "hello"}], - ) + def handle(request): + requests.append(request) + return httpx.Response(200, **response_kwargs) + + return requests, HTTPHandler(client=httpx.Client(transport=httpx.MockTransport(handle))) + + +def test_completion_posts_runtime_chat_completions(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "us.xai.grok-4.6")) + response = litellm.completion( + model="us.xai.grok-4.6", + messages=[{"role": "user", "content": "hello"}], + client=client, + ) assert response.choices[0].message.content == "ok" assert len(requests) == 1 - assert requests[0]["url"] == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" - raw = requests[0]["data"] - body = json.loads(raw) if isinstance(raw, (str, bytes, bytearray)) else (requests[0]["json"] or {}) + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) assert body["model"] == "us.xai.grok-4.6" assert body["messages"] == [{"role": "user", "content": "hello"}] assert "inferenceConfig" not in body + + +OPENAI_RUNTIME_MODELS = ( + "openai.gpt-oss-20b-1:0", + "openai.gpt-oss-120b-1:0", + "us.openai.gpt-5.6-sol", + "global.openai.gpt-5.6-sol", + "us.openai.gpt-5.6-terra", + "global.openai.gpt-5.6-terra", + "us.openai.gpt-5.6-luna", + "global.openai.gpt-5.6-luna", +) +GET_WEATHER_TOOL = { + "type": "function", + "function": { + "name": "get_weather", + "parameters": {"type": "object", "properties": {"city": {"type": "string"}}, "required": ["city"]}, + }, +} + + +@pytest.mark.parametrize("model", [*OPENAI_RUNTIME_MODELS, "bedrock/openai.gpt-oss-20b-1:0"]) +def test_openai_runtime_models_use_chat_completions_route(local_cost_map, model): + assert uses_bedrock_runtime_chat_completions(model) is True + assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions" + assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig) + + +@pytest.mark.parametrize("model", ["us.amazon.nova-micro-v1:0", "us.anthropic.claude-haiku-4-5-20251001-v1:0"]) +def test_nova_and_claude_stay_on_converse(local_cost_map, model): + assert uses_bedrock_runtime_chat_completions(model) is False + assert BedrockModelInfo.get_bedrock_route(model, {"tools": [GET_WEATHER_TOOL]}) == "converse" + + +@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "global.openai.gpt-5.6-sol"]) +def test_guardrail_config_falls_back_to_converse(local_cost_map, model): + guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"} + assert bedrock_request_needs_converse(model, {"guardrailConfig": guardrail}) is True + assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": guardrail}) == "converse" + assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": None}) == "chat_completions" + + +@pytest.mark.parametrize( + "request_params, expected_route", + [ + ({"tools": [GET_WEATHER_TOOL]}, "converse"), + ({"tools": [GET_WEATHER_TOOL], "reasoning_effort": "low"}, "converse"), + ({"tools": [GET_WEATHER_TOOL], "reasoning_effort": None}, "converse"), + ({"tools": [GET_WEATHER_TOOL], "reasoning_effort": "none"}, "chat_completions"), + ({"reasoning_effort": "low"}, "chat_completions"), + ({"tools": None, "reasoning_effort": "low"}, "chat_completions"), + ({"tools": [], "reasoning_effort": "low"}, "chat_completions"), + ({}, "chat_completions"), + ], +) +def test_gpt56_tools_need_reasoning_none_on_chat_completions(local_cost_map, request_params, expected_route): + assert BedrockModelInfo.get_bedrock_route("global.openai.gpt-5.6-sol", request_params) == expected_route + assert BedrockModelInfo.get_bedrock_route("bedrock/us.openai.gpt-5.6-terra", request_params) == expected_route + + +@pytest.mark.parametrize("reasoning_effort", ["low", "high", None]) +def test_gpt_oss_tools_with_any_reasoning_effort_stay_on_chat_completions(local_cost_map, reasoning_effort): + params = {"tools": [GET_WEATHER_TOOL], "reasoning_effort": reasoning_effort} + assert bedrock_request_needs_converse("openai.gpt-oss-120b-1:0", params) is False + assert BedrockModelInfo.get_bedrock_route("openai.gpt-oss-120b-1:0", params) == "chat_completions" + + +def test_explicit_converse_prefix_wins_for_openai_models(local_cost_map): + assert BedrockModelInfo.get_bedrock_route("bedrock/converse/openai.gpt-oss-20b-1:0") == "converse" + assert BedrockModelInfo.get_bedrock_route("converse/global.openai.gpt-5.6-sol", {}) == "converse" + + +def test_map_openai_params_sends_max_tokens_as_max_completion_tokens(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"max_tokens": 64, "temperature": 0.1}, + optional_params={}, + model="global.openai.gpt-5.6-sol", + drop_params=False, + ) + assert mapped == {"max_completion_tokens": 64, "temperature": 0.1} + + +def test_map_openai_params_keeps_explicit_max_completion_tokens(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"max_tokens": 64, "max_completion_tokens": 32}, + optional_params={}, + model="openai.gpt-oss-20b-1:0", + drop_params=False, + ) + assert mapped == {"max_completion_tokens": 32} + + +def test_with_max_completion_tokens_leaves_other_params_alone(): + assert with_max_completion_tokens({"temperature": 0.5}) == {"temperature": 0.5} + + +def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + assert "reasoning_effort" in cfg.get_supported_openai_params("global.openai.gpt-5.6-sol") + assert "reasoning_effort" in cfg.get_supported_openai_params("openai.gpt-oss-20b-1:0") + + +def test_split_reasoning_tag_splits_leading_tag(): + assert split_reasoning_tag("plan it\n\n\nHello") == ("plan it\n", "Hello") + + +def test_split_reasoning_tag_drops_an_empty_tag(): + assert split_reasoning_tag("Hello") == (None, "Hello") + + +@pytest.mark.parametrize( + "content", + [ + "plan it\n\n\nHello", + "never closed", + "later", + "", + ], +) +@pytest.mark.parametrize("chunk_size", [1, 3, 7]) +def test_split_reasoning_tag_matches_the_streamed_split(content, chunk_size): + chunks = [content[start : start + chunk_size] for start in range(0, len(content), chunk_size)] + streamed_reasoning, streamed_content = _run_splitter(chunks) + + assert split_reasoning_tag(content) == (streamed_reasoning or None, streamed_content) + + +def test_split_reasoning_tag_passes_plain_content_through(): + assert split_reasoning_tag("Hello") == (None, "Hello") + + +def test_split_reasoning_tag_ignores_tag_after_content_starts(): + content = "Hello not mine" + assert split_reasoning_tag(content) == (None, content) + + +def _run_splitter(chunks): + state = ReasoningTagSplitter() + reasoning = "" + content = "" + for chunk in chunks: + state, fed_reasoning, fed_content = state.feed(chunk) + reasoning += fed_reasoning + content += fed_content + state, flushed_reasoning, flushed_content = state.flush() + return reasoning + flushed_reasoning, content + flushed_content + + +def test_reasoning_tag_splitter_handles_tags_split_across_chunks(): + assert _run_splitter(["I think", " so\n\nHel", "lo"]) == ("I think so", "Hello") + + +def test_reasoning_tag_splitter_passes_plain_content_through(): + assert _run_splitter(["Hel", "lo later"]) == ("", "Hello later") + + +def test_reasoning_tag_splitter_flushes_unclosed_reasoning(): + assert _run_splitter(["never clo", "sed"]) == ("never closed", "") + + +def test_reasoning_tag_splitter_releases_a_false_tag_prefix(): + assert _run_splitter(["<", "b>x"]) == ("", "x") + + +def _stream_chunk(delta, finish_reason=None, index=0): + return { + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "created": 1733529600, + "model": "openai.gpt-oss-20b-1:0", + "choices": [{"index": index, "delta": delta, "finish_reason": finish_reason}], + } + + +def test_streaming_handler_splits_reasoning_deltas_per_choice(): + handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) + + first = handler.chunk_parser(_stream_chunk({"role": "assistant", "content": "I think"})) + assert first.choices[0].delta.reasoning_content == "I think" + assert not first.choices[0].delta.content + + second = handler.chunk_parser(_stream_chunk({"content": " so\n\nHello"})) + assert second.choices[0].delta.reasoning_content == " so" + assert second.choices[0].delta.content == "Hello" + + tool_call = {"index": 0, "id": "call_0", "type": "function", "function": {"name": "get_weather", "arguments": "{}"}} + third = handler.chunk_parser(_stream_chunk({"content": None, "tool_calls": [tool_call]})) + assert third.choices[0].delta.tool_calls[0].function.name == "get_weather" + + last = handler.chunk_parser(_stream_chunk({}, finish_reason="stop")) + assert last.choices[0].finish_reason == "stop" + + +def _reasoning_of(parsed): + return getattr(parsed.choices[0].delta, "reasoning_content", None) + + +def test_streaming_handler_keeps_split_state_per_choice_index(): + handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) + + opened = handler.chunk_parser(_stream_chunk({"content": "first"}, index=0)) + assert _reasoning_of(opened) == "first" + + plain = handler.chunk_parser(_stream_chunk({"content": "plain answer"}, index=1)) + assert _reasoning_of(plain) is None + assert plain.choices[0].delta.content == "plain answer" + + still_reasoning = handler.chunk_parser(_stream_chunk({"content": " more"}, index=0)) + assert _reasoning_of(still_reasoning) == " more" + assert not still_reasoning.choices[0].delta.content + + +def test_streaming_handler_flushes_held_text_on_an_empty_final_delta(): + handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) + + held = handler.chunk_parser(_stream_chunk({"content": "almost doneplan\n\nHi", "openai.gpt-oss-20b-1:0") + ) + response = litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + max_tokens=64, + reasoning_effort="low", + tools=[GET_WEATHER_TOOL], + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) + assert body["model"] == "openai.gpt-oss-20b-1:0" + assert body["max_completion_tokens"] == 64 + assert "max_tokens" not in body + assert body["reasoning_effort"] == "low" + assert body["tools"] == [GET_WEATHER_TOOL] + assert response.choices[0].message.reasoning_content == "plan" + assert response.choices[0].message.content == "Hi" + + +def test_gpt56_tools_with_reasoning_effort_go_to_converse(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + response = litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "hello"}], + tools=[GET_WEATHER_TOOL], + reasoning_effort="low", + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse") + assert json.loads(requests[0].content)["toolConfig"]["tools"][0]["toolSpec"]["name"] == "get_weather" + assert response.choices[0].message.content == "ok" + + +def test_gpt56_tools_with_reasoning_none_stay_on_chat_completions(local_cost_map, fake_aws_env): + tool_calls = [ + {"id": "call_0", "type": "function", "function": {"name": "get_weather", "arguments": '{"city": "Paris"}'}} + ] + requests, client = _recording_client(json=_chat_completion_json(None, "global.openai.gpt-5.6-sol", tool_calls)) + response = litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "weather in Paris"}], + tools=[GET_WEATHER_TOOL], + reasoning_effort="none", + max_tokens=64, + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) + assert body["tools"] == [GET_WEATHER_TOOL] + assert body["reasoning_effort"] == "none" + assert body["max_completion_tokens"] == 64 + assert response.choices[0].message.tool_calls[0].function.name == "get_weather" + + +@pytest.mark.parametrize( + "converse_only_param", + [ + {"guardrailConfig": {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"}}, + {"performanceConfig": {"latency": "optimized"}}, + {"requestMetadata": {"team": "search"}}, + {"serviceTier": {"type": "priority"}}, + ], + ids=lambda param: next(iter(param)), +) +def test_converse_only_request_keys_go_to_converse(local_cost_map, fake_aws_env, converse_only_param): + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + client=client, + **converse_only_param, + ) + + assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse") + (key, value), = converse_only_param.items() + assert json.loads(requests[0].content)[key] == value + + +def test_converse_only_keys_cover_every_converse_config_block(): + assert set(litellm.AmazonConverseConfig.get_config_blocks()) <= BEDROCK_CONVERSE_ONLY_REQUEST_KEYS + + +def test_operator_owned_request_metadata_goes_to_converse(local_cost_map, fake_aws_env, monkeypatch): + monkeypatch.setattr(litellm, "bedrock_request_metadata_fields", ["user_api_key_team_alias"]) + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + metadata={"user_api_key_team_alias": "search"}, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse") + assert json.loads(requests[0].content)["requestMetadata"] == {"user_api_key_team_alias": "search"} + + +def test_dropped_converse_only_key_keeps_the_request_on_chat_completions(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + guardrailConfig={"guardrailIdentifier": "gr-1", "guardrailVersion": "1"}, + additional_drop_params=["guardrailConfig"], + max_tokens=8, + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) + assert "guardrailConfig" not in body + assert body["max_completion_tokens"] == 8 + assert "inferenceConfig" not in body + + +def test_dropped_tools_keep_gpt56_reasoning_request_on_chat_completions(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "global.openai.gpt-5.6-sol")) + litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "hello"}], + tools=[GET_WEATHER_TOOL], + reasoning_effort="low", + additional_drop_params=["tools"], + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) + assert "tools" not in body + assert body["reasoning_effort"] == "low" + + +def test_legacy_functions_stay_on_chat_completions(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + functions=[GET_WEATHER_TOOL["function"]], + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + assert json.loads(requests[0].content)["functions"] == [GET_WEATHER_TOOL["function"]] + + +def test_converse_fallback_validates_against_converse_params(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"} + with pytest.raises(litellm.UnsupportedParamsError, match="seed"): + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + guardrailConfig=guardrail, + seed=7, + client=client, + ) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + guardrailConfig=guardrail, + seed=7, + drop_params=True, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse") + assert "seed" not in json.loads(requests[0].content) + + +def test_n_is_rejected_before_reaching_chat_completions(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) + with pytest.raises(litellm.UnsupportedParamsError, match="'n'"): + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + n=2, + client=client, + ) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + n=2, + drop_params=True, + client=client, + ) + + assert "n" not in json.loads(requests[0].content) + + +def _sse(chunks): + return ("".join(f"data: {json.dumps(chunk)}\n\n" for chunk in chunks) + "data: [DONE]\n\n").encode() + + +def test_gpt_oss_streaming_completion_splits_reasoning(local_cost_map, fake_aws_env): + chunks = ( + _stream_chunk({"role": "assistant", "content": "plan"}), + _stream_chunk({"content": "\n\nHi"}), + _stream_chunk({}, finish_reason="stop"), + ) + requests, client = _recording_client(content=_sse(chunks), headers={"content-type": "text/event-stream"}) + stream = litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + stream=True, + client=client, + ) + deltas = [chunk.choices[0].delta for chunk in stream] + + assert [str(request.url) for request in requests] == [ + "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + ] + assert json.loads(requests[0].content)["stream"] is True + assert "".join(getattr(delta, "reasoning_content", None) or "" for delta in deltas) == "plan" + assert "".join(delta.content or "" for delta in deltas) == "Hi" + + +def test_streaming_handler_keeps_native_reasoning_next_to_the_tagged_split(): + handler = BedrockRuntimeChatCompletionsStreamingHandler(streaming_response=iter(()), sync_stream=True) + parsed = handler.chunk_parser( + _stream_chunk({"reasoning": "native ", "content": "taggedHi"}, finish_reason="stop") + ) + + assert parsed.choices[0].delta.reasoning_content == "native tagged" + assert parsed.choices[0].delta.content == "Hi" diff --git a/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py b/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py index aa0827c5ae5..a6b8ba1da1d 100644 --- a/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py +++ b/tests/test_litellm/llms/bedrock/test_cross_region_inference_profile_mapping.py @@ -138,9 +138,12 @@ def _bedrock_response(model, usage): @pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id) -def test_bedrock_gpt_5_6_profiles_route_to_converse(profile, local_model_cost_map): - """GPT-5.6 is served by Converse on bedrock-runtime, never by Invoke.""" - assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "converse" +def test_bedrock_gpt_5_6_profiles_never_route_to_invoke(profile, local_model_cost_map): + """GPT-5.6 is served by bedrock-runtime's native Chat Completions, and by Converse when + the request carries function tools without reasoning_effort "none", never by Invoke.""" + assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "chat_completions" + tools_with_reasoning = {"tools": [{"type": "function", "function": {"name": "f"}}], "reasoning_effort": "low"} + assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}", tools_with_reasoning) == "converse" @pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id) diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 3336ad6d33a..963bef1114a 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -878,6 +878,8 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "supports_video_input": {"type": "boolean"}, "supports_vision": {"type": "boolean"}, "supports_web_search": {"type": "boolean"}, + "use_bedrock_runtime_chat_completions": {"type": "boolean"}, + "bedrock_runtime_chat_completions_tools_require_reasoning_none": {"type": "boolean"}, "supports_url_context": {"type": "boolean"}, "supports_multimodal": {"type": "boolean"}, "uses_embed_content": {"type": "boolean"}, From a1c089f10771d4cdbdf6fd4920b6eb0c3488cd3d Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 19 Sep 2026 18:47:54 -0700 Subject: [PATCH 03/31] fix(bedrock): route gpt-oss response_format to Converse and decide the route once from the raw request --- ci_cd/generate_model_prices_schema.py | 2 - .../chat/chat_completions/transformation.py | 5 +- litellm/llms/bedrock/common_utils.py | 57 ++++++-- litellm/main.py | 7 +- ...odel_prices_and_context_window_backup.json | 42 +++--- litellm/types/completion.py | 3 +- litellm/utils.py | 9 +- model_prices_and_context_window.json | 42 +++--- model_prices_and_context_window.schema.json | 15 ++- ...bedrock_chat_completions_transformation.py | 125 +++++++++++++++++- tests/test_litellm/test_utils.py | 5 +- tests/test_litellm/types/test_completion.py | 1 + 12 files changed, 248 insertions(+), 65 deletions(-) diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index 648cceb79b0..8eec07dadda 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -30,8 +30,6 @@ EXTRA_BOOLEAN_KEYS = frozenset( "gemini_audio_only_live", "uses_embed_content", "use_openai_responses_path", - "use_bedrock_runtime_chat_completions", - "bedrock_runtime_chat_completions_tools_require_reasoning_none", "bedrock_converse_supports_strict_tools", "thinking_always_on", } diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 49218270064..18ba06fee0b 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -3,7 +3,7 @@ Native OpenAI Chat Completions on Amazon Bedrock Runtime. AWS serves this surface at ``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions`` -for the models whose price-map entry sets ``use_bedrock_runtime_chat_completions`` +for the models whose price-map entry sets ``supports_bedrock_runtime_chat_completions`` (Grok 4.6, gpt-oss, the GPT-5.6 family): chat completions stay chat completions instead of being rewritten to Converse. @@ -19,6 +19,7 @@ from types import MappingProxyType from typing import TYPE_CHECKING, Final, Literal import httpx +from typing_extensions import assert_never import litellm from litellm.llms.base_llm.chat.transformation import BaseLLMException @@ -72,6 +73,8 @@ class ReasoningTagSplitter: return self._feed_start(self.pending + text) case "reasoning": return self._feed_reasoning(self.pending + text) + case _: + assert_never(self.phase) def _feed_start(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]: if buffered.startswith(REASONING_OPEN_TAG): diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 124ef7c646c..df30eddd856 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -803,22 +803,33 @@ def _bedrock_price_map_flag(model: str, flag: str) -> bool: def uses_bedrock_runtime_chat_completions(model: str) -> bool: """Whether this Bedrock model should use runtime native Chat Completions. - Data-driven from the price-map ``use_bedrock_runtime_chat_completions`` flag + Data-driven from the price-map ``supports_bedrock_runtime_chat_completions`` flag so onboarding a model is a JSON change. Explicit ``converse/`` still wins in ``get_bedrock_route`` because prefix routes are checked first, and a request that needs a Converse-only feature (``bedrock_request_needs_converse``) is served by Converse even on a flagged model. """ - return _bedrock_price_map_flag(model, "use_bedrock_runtime_chat_completions") + return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions") -def bedrock_runtime_chat_completions_tools_require_reasoning_none(model: str) -> bool: - """Whether AWS's native Chat Completions only serves this model's function tools with ``reasoning_effort="none"``. +def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool: + """Whether AWS's native Chat Completions serves this model's function tools with any ``reasoning_effort``. - Data-driven from the price-map ``bedrock_runtime_chat_completions_tools_require_reasoning_none`` - flag (the GPT-5.6 family). Converse serves tools with any effort, so those requests fall back to it. + Data-driven from the price-map ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` + flag (gpt-oss, Grok). Without it AWS only takes tools with ``reasoning_effort="none"`` + (the GPT-5.6 family), and Converse serves tools with any effort, so those requests fall back to it. """ - return _bedrock_price_map_flag(model, "bedrock_runtime_chat_completions_tools_require_reasoning_none") + return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_tools_with_reasoning") + + +def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> bool: + """Whether AWS's native Chat Completions enforces a ``response_format`` schema for this model. + + Data-driven from the price-map ``supports_bedrock_runtime_chat_completions_response_format`` flag + (GPT-5.6, Grok). Without it AWS accepts the field and answers with unconstrained text (gpt-oss), so + Converse, which emulates the schema through a forced ``json_tool_call`` tool, serves those requests. + """ + return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_response_format") BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( @@ -826,26 +837,52 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( ) +def _response_format_constrains_output(response_format: object) -> bool: + if response_format is None: + return False + return not (isinstance(response_format, Mapping) and response_format.get("type") == "text") + + def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: """Whether a request on a runtime-Chat-Completions model must still be served by Converse. Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``) are rejected as malformed input by AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, - and function tools on a ``bedrock_runtime_chat_completions_tools_require_reasoning_none`` model are - rejected there unless ``reasoning_effort`` is exactly ``"none"``. + function tools on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` + are rejected there unless ``reasoning_effort`` is exactly ``"none"``, and a constraining + ``response_format`` on a model without ``supports_bedrock_runtime_chat_completions_response_format`` + is only honored by Converse. """ if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS): return True if bedrock_request_metadata_is_owned(): return True + if _response_format_constrains_output( + request_params.get("response_format") + ) and not bedrock_runtime_chat_completions_enforces_response_format(model): + return True if not request_params.get("tools"): return False return ( - bedrock_runtime_chat_completions_tools_require_reasoning_none(model) + not bedrock_runtime_chat_completions_serves_tools_with_reasoning(model) and request_params.get("reasoning_effort") != "none" ) +def bedrock_route_for_request( + model: str, request_params: Mapping[str, object], additional_drop_params: Sequence[str] | None +) -> BedrockRoute: + """The route for one request, decided from the caller's raw params before any provider mapping. + + Param mapping and dispatch both call this with the same inputs, so a request that falls back to + Converse is mapped with the Converse config and sent to Converse, never one without the other. + """ + dropped: Final = frozenset(additional_drop_params or ()) + return BedrockModelInfo.get_bedrock_route( + model, {key: value for key, value in request_params.items() if key not in dropped} + ) + + def strip_bedrock_throughput_suffix(model: str) -> str: """Strip throughput tier suffixes and context window suffixes from Bedrock model names.""" import re diff --git a/litellm/main.py b/litellm/main.py index ed78f209ca2..539479fb224 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -104,7 +104,7 @@ from litellm.llms.base_llm import BaseConfig, BaseImageGenerationConfig from litellm.llms.base_llm.base_model_iterator import ( convert_model_response_to_streaming, ) -from litellm.llms.bedrock.common_utils import BedrockModelInfo +from litellm.llms.bedrock.common_utils import BedrockModelInfo, bedrock_route_for_request from litellm.llms.cohere.common_utils import CohereModelInfo from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler, http2_enabled from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config @@ -4078,7 +4078,9 @@ def _complete_bedrock(ctx: _CompletionDispatchContext) -> _CompletionDispatchRes if "aws_region_name" not in optional_params or optional_params["aws_region_name"] is None: optional_params["aws_region_name"] = aws_bedrock_client.meta.region_name - bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model, optional_params) + bedrock_route: Final = bedrock_route_for_request( + model, ctx.request_params, ctx.kwargs.get("additional_drop_params") + ) if bedrock_route == "claude_platform": provider_config = ProviderConfigManager.get_provider_chat_config( model=model, @@ -5686,6 +5688,7 @@ def completion( optional_params=optional_params, organization=organization, provider_config=provider_config, + request_params={**optional_param_args, **non_default_params}, shared_session=shared_session, stream=stream, temperature=temperature, diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 05658ce2d15..6dbf246fd70 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -40870,7 +40870,8 @@ "output_cost_per_token": 0.0 }, "openai.gpt-oss-120b-1:0": { - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -40884,7 +40885,8 @@ "supports_tool_choice": true }, "openai.gpt-oss-20b-1:0": { - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -47234,7 +47236,9 @@ "input_cost_per_token": 2.64e-06, "output_cost_per_token": 7.92e-06, "cache_read_input_token_cost": 6.6e-07, - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -58443,8 +58447,8 @@ "supports_web_search": true }, "us.openai.gpt-5.6-sol": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4.4e-06, "input_cost_per_token_above_272k_tokens": 8.8e-06, "cache_creation_input_token_cost": 5.5e-06, @@ -58474,8 +58478,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-sol": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4e-06, "input_cost_per_token_above_272k_tokens": 8e-06, "cache_creation_input_token_cost": 5e-06, @@ -58505,8 +58509,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-terra": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, "cache_creation_input_token_cost": 2.75e-06, @@ -58536,8 +58540,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-terra": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "cache_creation_input_token_cost": 2.5e-06, @@ -58567,8 +58571,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-luna": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-07, "input_cost_per_token_above_272k_tokens": 4.4e-07, "cache_creation_input_token_cost": 2.75e-07, @@ -58598,8 +58602,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-luna": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, "cache_creation_input_token_cost": 2.5e-07, @@ -58909,7 +58913,9 @@ "input_cost_per_token": 2.2e-06, "output_cost_per_token": 6.6e-06, "cache_read_input_token_cost": 5.5e-07, - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -58925,7 +58931,9 @@ "input_cost_per_token": 2e-06, "output_cost_per_token": 6e-06, "cache_read_input_token_cost": 5e-07, - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, diff --git a/litellm/types/completion.py b/litellm/types/completion.py index c1c6cc9ed1c..1e6cfc0ee33 100644 --- a/litellm/types/completion.py +++ b/litellm/types/completion.py @@ -1,6 +1,6 @@ from __future__ import annotations -from collections.abc import Callable, Coroutine, Iterable +from collections.abc import Callable, Coroutine, Iterable, Mapping from dataclasses import dataclass from typing import TYPE_CHECKING, Any, Literal, Union @@ -229,6 +229,7 @@ class _CompletionDispatchContext: optional_params: dict organization: str | None provider_config: BaseConfig | None + request_params: Mapping[str, object] shared_session: ClientSession | None stream: bool | None temperature: float | None diff --git a/litellm/utils.py b/litellm/utils.py index 82c8069770d..7226b9a865c 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -401,7 +401,7 @@ if TYPE_CHECKING: BaseVectorStoreFilesConfig, ) from litellm.llms.base_llm.videos.transformation import BaseVideoConfig - from litellm.llms.bedrock.common_utils import BedrockModelInfo, BedrockRoute + from litellm.llms.bedrock.common_utils import BedrockRoute from litellm.llms.bedrock.embed.amazon_nova_transformation import ( AmazonNovaEmbeddingConfig, ) @@ -3353,12 +3353,9 @@ def _should_drop_param(k, additional_drop_params) -> bool: def _bedrock_route_for_request( model: str, passed_params: Mapping[str, object], additional_drop_params: list | None ) -> BedrockRoute: - from litellm.llms.bedrock.common_utils import BedrockModelInfo + from litellm.llms.bedrock.common_utils import bedrock_route_for_request - return BedrockModelInfo.get_bedrock_route( - model, - {k: v for k, v in passed_params.items() if not _should_drop_param(k, additional_drop_params)}, - ) + return bedrock_route_for_request(model, passed_params, additional_drop_params) def _get_non_default_params(passed_params: dict, default_params: dict, additional_drop_params: list | None) -> dict: diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 05658ce2d15..6dbf246fd70 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -40870,7 +40870,8 @@ "output_cost_per_token": 0.0 }, "openai.gpt-oss-120b-1:0": { - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -40884,7 +40885,8 @@ "supports_tool_choice": true }, "openai.gpt-oss-20b-1:0": { - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", "max_input_tokens": 128000, @@ -47234,7 +47236,9 @@ "input_cost_per_token": 2.64e-06, "output_cost_per_token": 7.92e-06, "cache_read_input_token_cost": 6.6e-07, - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -58443,8 +58447,8 @@ "supports_web_search": true }, "us.openai.gpt-5.6-sol": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4.4e-06, "input_cost_per_token_above_272k_tokens": 8.8e-06, "cache_creation_input_token_cost": 5.5e-06, @@ -58474,8 +58478,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-sol": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4e-06, "input_cost_per_token_above_272k_tokens": 8e-06, "cache_creation_input_token_cost": 5e-06, @@ -58505,8 +58509,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-terra": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, "cache_creation_input_token_cost": 2.75e-06, @@ -58536,8 +58540,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-terra": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "cache_creation_input_token_cost": 2.5e-06, @@ -58567,8 +58571,8 @@ "supports_vision": true }, "us.openai.gpt-5.6-luna": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-07, "input_cost_per_token_above_272k_tokens": 4.4e-07, "cache_creation_input_token_cost": 2.75e-07, @@ -58598,8 +58602,8 @@ "supports_vision": true }, "global.openai.gpt-5.6-luna": { - "use_bedrock_runtime_chat_completions": true, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, "cache_creation_input_token_cost": 2.5e-07, @@ -58909,7 +58913,9 @@ "input_cost_per_token": 2.2e-06, "output_cost_per_token": 6.6e-06, "cache_read_input_token_cost": 5.5e-07, - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, @@ -58925,7 +58931,9 @@ "input_cost_per_token": 2e-06, "output_cost_per_token": 6e-06, "cache_read_input_token_cost": 5e-07, - "use_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, + "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", "max_input_tokens": 500000, "max_output_tokens": 500000, diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index 9db1c8eedc4..1a3523cdff2 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -74,9 +74,6 @@ "xhigh" ] }, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": { - "type": "boolean" - }, "cache_creation_input_audio_token_cost": { "type": "number", "minimum": 0 @@ -830,6 +827,15 @@ "supports_audio_output": { "type": "boolean" }, + "supports_bedrock_runtime_chat_completions": { + "type": "boolean" + }, + "supports_bedrock_runtime_chat_completions_response_format": { + "type": "boolean" + }, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": { + "type": "boolean" + }, "supports_computer_use": { "type": "boolean" }, @@ -1000,9 +1006,6 @@ "minimum": 0, "description": "Provider default tokens-per-minute limit." }, - "use_bedrock_runtime_chat_completions": { - "type": "boolean" - }, "use_openai_responses_path": { "type": "boolean" }, diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 230e323ace3..ce2e5b7e8f7 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -4,6 +4,7 @@ import json import httpx import pytest +from pydantic import BaseModel import litellm from litellm.llms.bedrock.chat.chat_completions.transformation import ( @@ -17,6 +18,7 @@ from litellm.llms.bedrock.common_utils import ( BEDROCK_CONVERSE_ONLY_REQUEST_KEYS, BedrockModelInfo, bedrock_request_needs_converse, + bedrock_route_for_request, get_bedrock_chat_config, uses_bedrock_runtime_chat_completions, ) @@ -471,7 +473,7 @@ def test_converse_only_request_keys_go_to_converse(local_cost_map, fake_aws_env, ) assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse") - (key, value), = converse_only_param.items() + ((key, value),) = converse_only_param.items() assert json.loads(requests[0].content)[key] == value @@ -620,3 +622,124 @@ def test_streaming_handler_keeps_native_reasoning_next_to_the_tagged_split(): assert parsed.choices[0].delta.reasoning_content == "native tagged" assert parsed.choices[0].delta.content == "Hi" + + +RESPONSE_FORMAT_JSON_SCHEMA = { + "type": "json_schema", + "json_schema": { + "name": "answer", + "schema": {"type": "object", "properties": {"word": {"type": "string"}}, "required": ["word"]}, + "strict": True, + }, +} + + +class Answer(BaseModel): + word: str + + +@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "bedrock/openai.gpt-oss-120b-1:0"]) +@pytest.mark.parametrize( + "response_format, expected_route", + [ + (RESPONSE_FORMAT_JSON_SCHEMA, "converse"), + ({"type": "json_object"}, "converse"), + (Answer, "converse"), + ({"type": "text"}, "chat_completions"), + (None, "chat_completions"), + ], + ids=["json_schema", "json_object", "pydantic", "text", "none"], +) +def test_gpt_oss_response_format_falls_back_to_converse(local_cost_map, model, response_format, expected_route): + params = {"response_format": response_format} + assert bedrock_request_needs_converse(model, params) is (expected_route == "converse") + assert BedrockModelInfo.get_bedrock_route(model, params) == expected_route + + +@pytest.mark.parametrize("model", ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"]) +def test_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model): + params = {"response_format": RESPONSE_FORMAT_JSON_SCHEMA} + assert bedrock_request_needs_converse(model, params) is False + assert BedrockModelInfo.get_bedrock_route(model, params) == "chat_completions" + + +SYNTHETIC_NATIVE_MODEL = "vendor.native-model-v1:0" + + +@pytest.mark.parametrize( + "capability_flags, request_params, needs_converse", + [ + ({}, {"tools": [GET_WEATHER_TOOL], "reasoning_effort": "low"}, True), + ({}, {"tools": [GET_WEATHER_TOOL]}, True), + ({}, {"tools": [GET_WEATHER_TOOL], "reasoning_effort": "none"}, False), + ( + {"supports_bedrock_runtime_chat_completions_tools_with_reasoning": True}, + {"tools": [GET_WEATHER_TOOL], "reasoning_effort": "low"}, + False, + ), + ({}, {"response_format": RESPONSE_FORMAT_JSON_SCHEMA}, True), + ( + {"supports_bedrock_runtime_chat_completions_response_format": True}, + {"response_format": RESPONSE_FORMAT_JSON_SCHEMA}, + False, + ), + ( + {"supports_bedrock_runtime_chat_completions_response_format": True}, + {"response_format": RESPONSE_FORMAT_JSON_SCHEMA, "tools": [GET_WEATHER_TOOL], "reasoning_effort": "low"}, + True, + ), + ], +) +def test_capability_flags_are_read_from_the_cost_map(monkeypatch, capability_flags, request_params, needs_converse): + entry = { + "litellm_provider": "bedrock_converse", + "supports_bedrock_runtime_chat_completions": True, + **capability_flags, + } + monkeypatch.setattr(litellm, "model_cost", {SYNTHETIC_NATIVE_MODEL: entry}) + assert bedrock_request_needs_converse(SYNTHETIC_NATIVE_MODEL, request_params) is needs_converse + route = bedrock_route_for_request(SYNTHETIC_NATIVE_MODEL, request_params, None) + assert (route == "chat_completions") is (not needs_converse) + + +def test_route_for_request_ignores_dropped_params(local_cost_map): + params = {"response_format": RESPONSE_FORMAT_JSON_SCHEMA, "guardrailConfig": {"guardrailIdentifier": "gr-1"}} + assert bedrock_route_for_request("openai.gpt-oss-20b-1:0", params, None) == "converse" + assert bedrock_route_for_request("openai.gpt-oss-20b-1:0", params, ["guardrailConfig"]) == "converse" + assert ( + bedrock_route_for_request("openai.gpt-oss-20b-1:0", params, ["guardrailConfig", "response_format"]) + == "chat_completions" + ) + + +def test_gpt_oss_response_format_goes_to_converse_with_json_tool_call(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "Reply with the single word pong."}], + response_format=RESPONSE_FORMAT_JSON_SCHEMA, + max_tokens=64, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/openai.gpt-oss-20b-1%3A0/converse") + body = json.loads(requests[0].content) + assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "json_tool_call" + assert body["toolConfig"]["toolChoice"] == {"tool": {"name": "json_tool_call"}} + assert body["inferenceConfig"]["maxTokens"] == 64 + assert "response_format" not in body + assert "max_completion_tokens" not in body + + +def test_gpt56_response_format_is_sent_as_is_on_chat_completions(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json('{"word": "pong"}', "global.openai.gpt-5.6-sol")) + response = litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "Reply with the single word pong."}], + response_format=RESPONSE_FORMAT_JSON_SCHEMA, + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + assert json.loads(requests[0].content)["response_format"] == RESPONSE_FORMAT_JSON_SCHEMA + assert response.choices[0].message.content == '{"word": "pong"}' diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 963bef1114a..d049d83c6a9 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -878,8 +878,9 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "supports_video_input": {"type": "boolean"}, "supports_vision": {"type": "boolean"}, "supports_web_search": {"type": "boolean"}, - "use_bedrock_runtime_chat_completions": {"type": "boolean"}, - "bedrock_runtime_chat_completions_tools_require_reasoning_none": {"type": "boolean"}, + "supports_bedrock_runtime_chat_completions": {"type": "boolean"}, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": {"type": "boolean"}, + "supports_bedrock_runtime_chat_completions_response_format": {"type": "boolean"}, "supports_url_context": {"type": "boolean"}, "supports_multimodal": {"type": "boolean"}, "uses_embed_content": {"type": "boolean"}, diff --git a/tests/test_litellm/types/test_completion.py b/tests/test_litellm/types/test_completion.py index cd51913c5dd..482dd351ad1 100644 --- a/tests/test_litellm/types/test_completion.py +++ b/tests/test_litellm/types/test_completion.py @@ -181,6 +181,7 @@ def _build_dispatch_context() -> _CompletionDispatchContext: optional_params={}, organization=None, provider_config=None, + request_params={}, shared_session=None, stream=None, temperature=None, From b90c2f113d1091896bb80988c790898320678f00 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 19 Sep 2026 20:39:18 -0700 Subject: [PATCH 04/31] fix(bedrock): serve region-path and GovCloud gpt-oss ids on native Chat Completions The cost-map parity tests require every regional variant of a flagged id to carry the same supports_ flags, so the six us-gov gpt-oss entries now carry the native-route flags too. A region path in the model name (bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0) is routing, not a different model: the route is looked up on the id after the path, the path's region picks the endpoint and the SigV4 scope, an explicit aws_region_name still wins, and the body carries the bare id AWS expects --- .../chat/chat_completions/transformation.py | 18 ++++++--- litellm/llms/bedrock/common_utils.py | 18 ++++++++- ...odel_prices_and_context_window_backup.json | 12 ++++++ model_prices_and_context_window.json | 12 ++++++ ...bedrock_chat_completions_transformation.py | 38 ++++++++++++++++++- 5 files changed, 91 insertions(+), 7 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 18ba06fee0b..a7926fb252e 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -24,7 +24,7 @@ from typing_extensions import assert_never import litellm from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM -from litellm.llms.bedrock.common_utils import BedrockError, strip_bedrock_routing_prefix +from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig from litellm.types.llms.openai import AllMessageValues @@ -196,7 +196,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): if api_base is not None and "chat/completions" in api_base: return api_base.rstrip("/") aws_region_name: Final = self._aws_signer._get_aws_region_name( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public region resolver - optional_params=optional_params, model=model + optional_params=self._params_with_region_from_path(optional_params, model), model=model ) endpoint_url, _ = self._aws_signer.get_runtime_endpoint( api_base=api_base, @@ -210,6 +210,14 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): return f"{base}/chat/completions" return f"{base}/openai/v1/chat/completions" + def _params_with_region_from_path( + self, optional_params: dict, model: str | None + ) -> dict: # mutable-ok: BaseAWSLLM's region resolver and signer take a plain dict + region_from_path, _ = split_bedrock_region_path(model or "") + if region_from_path is None or optional_params.get("aws_region_name") is not None: + return optional_params + return {**optional_params, "aws_region_name": region_from_path} # mutable-ok: BaseAWSLLM takes a plain dict + def sign_request( self, headers: dict, # mutable-ok: BaseConfig signature @@ -224,7 +232,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): return self._aws_signer._sign_request( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public signer service_name="bedrock", headers=headers, - optional_params=optional_params, + optional_params=self._params_with_region_from_path(optional_params, model), request_data=request_data, api_base=api_base, api_key=api_key, @@ -268,7 +276,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): headers: dict, # mutable-ok: BaseConfig signature ) -> dict: # mutable-ok: BaseConfig signature return super().transform_request( - model=strip_bedrock_routing_prefix(model), + model=split_bedrock_region_path(model)[1], messages=messages, optional_params=self._inference_params(optional_params), litellm_params=litellm_params, @@ -284,7 +292,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): headers: dict, # mutable-ok: BaseConfig signature ) -> dict: # mutable-ok: BaseConfig signature return await super().async_transform_request( - model=strip_bedrock_routing_prefix(model), + model=split_bedrock_region_path(model)[1], messages=messages, optional_params=self._inference_params(optional_params), litellm_params=litellm_params, diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index df30eddd856..3b624b4ab02 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -795,8 +795,24 @@ def strip_bedrock_routing_prefix(model: str) -> str: return model +def split_bedrock_region_path(model: str) -> tuple[str | None, str]: + """Split a ``/`` routing path into the region and the id AWS receives. + + ``bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0`` -> ``("us-gov-west-1", "openai.gpt-oss-20b-1:0")``; + a model without a region path comes back as ``(None, )``. + """ + stripped: Final = strip_bedrock_routing_prefix(model) + region, separator, model_id = stripped.partition("/") + if separator and region in _get_all_bedrock_regions(): + return region, model_id + return None, stripped + + def _bedrock_price_map_flag(model: str, flag: str) -> bool: - entries: Final = (litellm.model_cost.get(key) for key in (model, strip_bedrock_routing_prefix(model))) + entries: Final = ( + litellm.model_cost.get(key) + for key in (model, strip_bedrock_routing_prefix(model), split_bedrock_region_path(model)[1]) + ) return any(entry is not None and entry.get(flag) is True for entry in entries) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 6dbf246fd70..db0e3a4dc18 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -47214,6 +47214,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -47227,6 +47229,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64459,6 +64463,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64472,6 +64478,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64665,6 +64673,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64678,6 +64688,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 6dbf246fd70..db0e3a4dc18 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -47214,6 +47214,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -47227,6 +47229,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64459,6 +64463,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64472,6 +64478,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64665,6 +64673,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, @@ -64678,6 +64688,8 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, + "supports_bedrock_runtime_chat_completions": true, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, "supports_response_schema": true, diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index ce2e5b7e8f7..83fe7628e0a 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -163,6 +163,33 @@ def test_completion_posts_runtime_chat_completions(local_cost_map, fake_aws_env) assert "inferenceConfig" not in body +def test_region_path_sends_the_bare_model_id_to_the_path_region(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) + litellm.completion( + model="bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-gov-west-1.amazonaws.com/openai/v1/chat/completions" + assert json.loads(requests[0].content)["model"] == "openai.gpt-oss-20b-1:0" + assert "/us-gov-west-1/bedrock/aws4_request" in requests[0].headers["Authorization"] + + +def test_explicit_aws_region_name_wins_over_the_region_path(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) + litellm.completion( + model="bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + aws_region_name="us-gov-east-1", + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-gov-east-1.amazonaws.com/openai/v1/chat/completions" + assert json.loads(requests[0].content)["model"] == "openai.gpt-oss-20b-1:0" + assert "/us-gov-east-1/bedrock/aws4_request" in requests[0].headers["Authorization"] + + OPENAI_RUNTIME_MODELS = ( "openai.gpt-oss-20b-1:0", "openai.gpt-oss-120b-1:0", @@ -182,7 +209,16 @@ GET_WEATHER_TOOL = { } -@pytest.mark.parametrize("model", [*OPENAI_RUNTIME_MODELS, "bedrock/openai.gpt-oss-20b-1:0"]) +@pytest.mark.parametrize( + "model", + [ + *OPENAI_RUNTIME_MODELS, + "bedrock/openai.gpt-oss-20b-1:0", + "us-gov.openai.gpt-oss-20b-1:0", + "bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", + "us-gov-east-1/openai.gpt-oss-120b-1:0", + ], +) def test_openai_runtime_models_use_chat_completions_route(local_cost_map, model): assert uses_bedrock_runtime_chat_completions(model) is True assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions" From 22b217bd95bb409d480d6da5a4292578e0899c7f Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 21 Sep 2026 12:53:43 -0700 Subject: [PATCH 05/31] fix(bedrock): keep params AWS refuses natively off the chat completions route Drop the params each family 400s or 503s on runtime Chat Completions from the native config's supported list (GPT-5.6 penalties, stop, and logprobs, Grok penalties, gpt-oss logit_bias) so drop_params drops them as Converse did, gate legacy functions on GPT-5.6 the same way as tools, and send an Anthropic-style thinking block to Converse, the only route that forwards it --- .../chat/chat_completions/transformation.py | 24 +++- litellm/llms/bedrock/common_utils.py | 15 +-- ...bedrock_chat_completions_transformation.py | 107 ++++++++++++++++++ 3 files changed, 138 insertions(+), 8 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index a7926fb252e..c6bf15b893f 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -38,6 +38,27 @@ if TYPE_CHECKING: REASONING_OPEN_TAG: Final = "" REASONING_CLOSE_TAG: Final = "" +CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType( + { + "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs")), + "openai.gpt-oss": frozenset(("logit_bias",)), + "xai.": frozenset(("frequency_penalty", "presence_penalty")), + } +) + + +def chat_completions_params_refused_for(model: str) -> frozenset[str]: + """The OpenAI params AWS's Chat Completions endpoint rejects for this model whatever else the request says. + + Each family answers them with a 400 (GPT-5.6, gpt-oss) or a 503 (Grok), where Converse dropped the same + params under ``drop_params``, so the native config leaves them out of its supported list and the usual + drop-or-raise handling applies before the request reaches AWS. + """ + model_id: Final = split_bedrock_region_path(model)[1] + return frozenset().union( + *(refused for family, refused in CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY.items() if family in model_id) + ) + def _held_close_tag_prefix(text: str) -> int: return next( @@ -362,7 +383,8 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): return {**validated, "OpenAI-Project": project_id} # mutable-ok: BaseConfig signature returns a dict def get_supported_openai_params(self, model: str) -> list: # mutable-ok: BaseConfig signature - base_params: Final = [param for param in super().get_supported_openai_params(model) if param != "n"] + refused: Final = {"n", *chat_completions_params_refused_for(model)} + base_params: Final = [param for param in super().get_supported_openai_params(model) if param not in refused] if "reasoning_effort" in base_params or not litellm.supports_reasoning( model=model, custom_llm_provider=self.custom_llm_provider ): diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 3b624b4ab02..8798b05cfa4 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -849,7 +849,7 @@ def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> boo BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( - ("guardrailConfig", "performanceConfig", "serviceTier", "requestMetadata", "outputConfig") + ("guardrailConfig", "performanceConfig", "serviceTier", "requestMetadata", "outputConfig", "thinking") ) @@ -862,12 +862,13 @@ def _response_format_constrains_output(response_format: object) -> bool: def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: """Whether a request on a runtime-Chat-Completions model must still be served by Converse. - Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``) are rejected as malformed input by + Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking`` + block included, which only Converse forwards as ``additionalModelRequestFields``) have no field on AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, - function tools on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` - are rejected there unless ``reasoning_effort`` is exactly ``"none"``, and a constraining - ``response_format`` on a model without ``supports_bedrock_runtime_chat_completions_response_format`` - is only honored by Converse. + function tools (``tools`` or legacy ``functions``) on a model without + ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless + ``reasoning_effort`` is exactly ``"none"``, and a constraining ``response_format`` on a model without + ``supports_bedrock_runtime_chat_completions_response_format`` is only honored by Converse. """ if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS): return True @@ -877,7 +878,7 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje request_params.get("response_format") ) and not bedrock_runtime_chat_completions_enforces_response_format(model): return True - if not request_params.get("tools"): + if not (request_params.get("tools") or request_params.get("functions")): return False return ( not bedrock_runtime_chat_completions_serves_tools_with_reasoning(model) diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 83fe7628e0a..2852ed3ae84 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -264,6 +264,26 @@ def test_gpt_oss_tools_with_any_reasoning_effort_stay_on_chat_completions(local_ assert BedrockModelInfo.get_bedrock_route("openai.gpt-oss-120b-1:0", params) == "chat_completions" +@pytest.mark.parametrize( + "request_params, expected_route", + [ + ({"functions": [GET_WEATHER_TOOL["function"]]}, "converse"), + ({"functions": [GET_WEATHER_TOOL["function"]], "reasoning_effort": "low"}, "converse"), + ({"functions": [GET_WEATHER_TOOL["function"]], "reasoning_effort": "none"}, "chat_completions"), + ({"functions": [], "reasoning_effort": "low"}, "chat_completions"), + ], +) +def test_gpt56_legacy_functions_route_like_tools(local_cost_map, request_params, expected_route): + assert BedrockModelInfo.get_bedrock_route("global.openai.gpt-5.6-sol", request_params) == expected_route + assert BedrockModelInfo.get_bedrock_route("openai.gpt-oss-120b-1:0", request_params) == "chat_completions" + + +def test_thinking_block_goes_to_converse(local_cost_map): + thinking = {"type": "enabled", "budget_tokens": 1024} + assert BedrockModelInfo.get_bedrock_route("us.xai.grok-4.6", {"thinking": thinking}) == "converse" + assert BedrockModelInfo.get_bedrock_route("us.xai.grok-4.6", {"thinking": None}) == "chat_completions" + + def test_explicit_converse_prefix_wins_for_openai_models(local_cost_map): assert BedrockModelInfo.get_bedrock_route("bedrock/converse/openai.gpt-oss-20b-1:0") == "converse" assert BedrockModelInfo.get_bedrock_route("converse/global.openai.gpt-5.6-sol", {}) == "converse" @@ -301,6 +321,54 @@ def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): assert "reasoning_effort" in cfg.get_supported_openai_params("openai.gpt-oss-20b-1:0") +@pytest.mark.parametrize( + "model, refused, kept", + [ + ( + "bedrock/global.openai.gpt-5.6-sol", + ("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs", "n"), + ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions"), + ), + ( + "us.xai.grok-4.6", + ("frequency_penalty", "presence_penalty", "n"), + ("stop", "logprobs", "top_p", "logit_bias", "reasoning_effort"), + ), + ( + "bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", + ("logit_bias", "n"), + ("frequency_penalty", "presence_penalty", "stop", "logprobs", "reasoning_effort"), + ), + ], +) +def test_supported_params_leave_out_what_each_family_refuses(local_cost_map, model, refused, kept): + supported = set(AmazonBedrockRuntimeChatCompletionsConfig().get_supported_openai_params(model)) + assert supported.isdisjoint(refused) + assert set(kept) <= supported + + +@pytest.mark.parametrize( + "model, param", + [ + ("bedrock/global.openai.gpt-5.6-sol", {"frequency_penalty": 0.5}), + ("bedrock/global.openai.gpt-5.6-sol", {"logprobs": True, "top_logprobs": 2}), + ("bedrock/us.xai.grok-4.6", {"presence_penalty": 0.5}), + ("bedrock/openai.gpt-oss-20b-1:0", {"logit_bias": {"1": 1}}), + ], + ids=lambda value: value if isinstance(value, str) else next(iter(value)), +) +def test_refused_params_are_dropped_or_refused_before_reaching_aws(local_cost_map, fake_aws_env, model, param): + requests, client = _recording_client(json=_chat_completion_json("ok", model.removeprefix("bedrock/"))) + with pytest.raises(litellm.UnsupportedParamsError, match=next(iter(param))): + litellm.completion(model=model, messages=[{"role": "user", "content": "hello"}], client=client, **param) + litellm.completion( + model=model, messages=[{"role": "user", "content": "hello"}], drop_params=True, client=client, **param + ) + + assert str(requests[0].url).endswith("/openai/v1/chat/completions") + assert param.keys().isdisjoint(json.loads(requests[0].content)) + + def test_split_reasoning_tag_splits_leading_tag(): assert split_reasoning_tag("plan it\n\n\nHello") == ("plan it\n", "Hello") @@ -579,6 +647,45 @@ def test_legacy_functions_stay_on_chat_completions(local_cost_map, fake_aws_env) assert json.loads(requests[0].content)["functions"] == [GET_WEATHER_TOOL["function"]] +def test_gpt56_legacy_functions_with_reasoning_fall_back_to_converse(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + with pytest.raises(litellm.UnsupportedParamsError, match="functions"): + litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "hello"}], + functions=[GET_WEATHER_TOOL["function"]], + reasoning_effort="low", + client=client, + ) + litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "hello"}], + functions=[GET_WEATHER_TOOL["function"]], + reasoning_effort="low", + drop_params=True, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse") + body = json.loads(requests[0].content) + assert "functions" not in body + assert "toolConfig" not in body + + +def test_grok_thinking_block_is_served_by_converse(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + thinking = {"type": "enabled", "budget_tokens": 1024} + litellm.completion( + model="bedrock/us.xai.grok-4.6", + messages=[{"role": "user", "content": "hello"}], + thinking=thinking, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/us.xai.grok-4.6/converse") + assert json.loads(requests[0].content)["additionalModelRequestFields"]["thinking"] == thinking + + def test_converse_fallback_validates_against_converse_params(local_cost_map, fake_aws_env): requests, client = _recording_client(json=CONVERSE_JSON) guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"} From 0a85e2799814b4582d110f93c0b9c87ed3f66d8b Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 21 Sep 2026 13:57:27 -0700 Subject: [PATCH 06/31] fix(bedrock): keep schema-less json_object on Converse for the chat completions models --- litellm/llms/bedrock/common_utils.py | 20 +++++---- ...bedrock_chat_completions_transformation.py | 42 +++++++++++++++++-- 2 files changed, 52 insertions(+), 10 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 8798b05cfa4..2fe1af9edf4 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -853,10 +853,15 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( ) -def _response_format_constrains_output(response_format: object) -> bool: +def _response_format_needs_converse(model: str, response_format: object) -> bool: if response_format is None: return False - return not (isinstance(response_format, Mapping) and response_format.get("type") == "text") + if not isinstance(response_format, Mapping): + return not bedrock_runtime_chat_completions_enforces_response_format(model) + if response_format.get("type") == "text": + return False + carries_schema: Final = "json_schema" in response_format or "response_schema" in response_format + return not (carries_schema and bedrock_runtime_chat_completions_enforces_response_format(model)) def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: @@ -867,16 +872,17 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, function tools (``tools`` or legacy ``functions``) on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless - ``reasoning_effort`` is exactly ``"none"``, and a constraining ``response_format`` on a model without - ``supports_bedrock_runtime_chat_completions_response_format`` is only honored by Converse. + ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as a JSON schema + (a ``json_schema`` or ``response_schema`` mapping, or a pydantic model) on a model with + ``supports_bedrock_runtime_chat_completions_response_format``: a schema on any other model is only + honored by Converse, and a schema-less ``json_object`` keeps Converse's handling everywhere, since AWS's + native surface rejects it with a 400 unless the prompt mentions json. """ if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS): return True if bedrock_request_metadata_is_owned(): return True - if _response_format_constrains_output( - request_params.get("response_format") - ) and not bedrock_runtime_chat_completions_enforces_response_format(model): + if _response_format_needs_converse(model, request_params.get("response_format")): return True if not (request_params.get("tools") or request_params.get("functions")): return False diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 2852ed3ae84..e49a2f7f252 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -799,13 +799,32 @@ def test_gpt_oss_response_format_falls_back_to_converse(local_cost_map, model, r assert BedrockModelInfo.get_bedrock_route(model, params) == expected_route -@pytest.mark.parametrize("model", ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"]) -def test_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model): - params = {"response_format": RESPONSE_FORMAT_JSON_SCHEMA} +RESPONSE_FORMAT_ENFORCING_MODELS = ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"] + + +@pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS) +@pytest.mark.parametrize( + "response_format", + [ + RESPONSE_FORMAT_JSON_SCHEMA, + {"type": "json_object", "response_schema": RESPONSE_FORMAT_JSON_SCHEMA["json_schema"]["schema"]}, + Answer, + ], + ids=["json_schema", "response_schema", "pydantic"], +) +def test_schema_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model, response_format): + params = {"response_format": response_format} assert bedrock_request_needs_converse(model, params) is False assert BedrockModelInfo.get_bedrock_route(model, params) == "chat_completions" +@pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS) +def test_schema_less_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_cost_map, model): + params = {"response_format": {"type": "json_object"}} + assert bedrock_request_needs_converse(model, params) is True + assert BedrockModelInfo.get_bedrock_route(model, params) == "converse" + + SYNTHETIC_NATIVE_MODEL = "vendor.native-model-v1:0" @@ -886,3 +905,20 @@ def test_gpt56_response_format_is_sent_as_is_on_chat_completions(local_cost_map, assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" assert json.loads(requests[0].content)["response_format"] == RESPONSE_FORMAT_JSON_SCHEMA assert response.choices[0].message.content == '{"word": "pong"}' + + +def test_gpt56_schema_less_json_object_goes_to_converse_without_a_schema_tool(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "Reply with the single word pong."}], + response_format={"type": "json_object"}, + max_tokens=64, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse") + body = json.loads(requests[0].content) + assert "toolConfig" not in body + assert "response_format" not in body + assert body["inferenceConfig"]["maxTokens"] == 64 From 0139dd08a635843deaac24256b53439531f3b9a7 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 21 Sep 2026 14:10:27 -0700 Subject: [PATCH 07/31] fix(bedrock): keep every json_object response_format on Converse for the chat completions models --- litellm/llms/bedrock/common_utils.py | 16 ++++--- ...bedrock_chat_completions_transformation.py | 46 ++++++++++++++----- 2 files changed, 43 insertions(+), 19 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 2fe1af9edf4..486b3b2bd51 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -858,10 +858,11 @@ def _response_format_needs_converse(model: str, response_format: object) -> bool return False if not isinstance(response_format, Mapping): return not bedrock_runtime_chat_completions_enforces_response_format(model) - if response_format.get("type") == "text": + response_format_type: Final = response_format.get("type") + if response_format_type == "text": return False - carries_schema: Final = "json_schema" in response_format or "response_schema" in response_format - return not (carries_schema and bedrock_runtime_chat_completions_enforces_response_format(model)) + is_json_schema: Final = response_format_type == "json_schema" and "json_schema" in response_format + return not (is_json_schema and bedrock_runtime_chat_completions_enforces_response_format(model)) def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: @@ -872,11 +873,12 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, function tools (``tools`` or legacy ``functions``) on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless - ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as a JSON schema - (a ``json_schema`` or ``response_schema`` mapping, or a pydantic model) on a model with + ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as + ``{"type": "json_schema", "json_schema": ...}`` (a pydantic model is converted to that) on a model with ``supports_bedrock_runtime_chat_completions_response_format``: a schema on any other model is only - honored by Converse, and a schema-less ``json_object`` keeps Converse's handling everywhere, since AWS's - native surface rejects it with a 400 unless the prompt mentions json. + honored by Converse, and every ``json_object`` form (``response_schema`` included) keeps Converse's + handling everywhere, since AWS's native surface rejects that type with a 400 unless the prompt + mentions json. """ if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS): return True diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index e49a2f7f252..2f0d384fcdd 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -802,25 +802,30 @@ def test_gpt_oss_response_format_falls_back_to_converse(local_cost_map, model, r RESPONSE_FORMAT_ENFORCING_MODELS = ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"] +JSON_OBJECT_WITH_RESPONSE_SCHEMA = { + "type": "json_object", + "response_schema": RESPONSE_FORMAT_JSON_SCHEMA["json_schema"]["schema"], +} + + @pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS) -@pytest.mark.parametrize( - "response_format", - [ - RESPONSE_FORMAT_JSON_SCHEMA, - {"type": "json_object", "response_schema": RESPONSE_FORMAT_JSON_SCHEMA["json_schema"]["schema"]}, - Answer, - ], - ids=["json_schema", "response_schema", "pydantic"], -) -def test_schema_response_format_stays_on_chat_completions_where_aws_enforces_it(local_cost_map, model, response_format): +@pytest.mark.parametrize("response_format", [RESPONSE_FORMAT_JSON_SCHEMA, Answer], ids=["json_schema", "pydantic"]) +def test_json_schema_response_format_stays_on_chat_completions_where_aws_enforces_it( + local_cost_map, model, response_format +): params = {"response_format": response_format} assert bedrock_request_needs_converse(model, params) is False assert BedrockModelInfo.get_bedrock_route(model, params) == "chat_completions" @pytest.mark.parametrize("model", RESPONSE_FORMAT_ENFORCING_MODELS) -def test_schema_less_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_cost_map, model): - params = {"response_format": {"type": "json_object"}} +@pytest.mark.parametrize( + "response_format", + [{"type": "json_object"}, JSON_OBJECT_WITH_RESPONSE_SCHEMA], + ids=["json_object", "json_object_with_response_schema"], +) +def test_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_cost_map, model, response_format): + params = {"response_format": response_format} assert bedrock_request_needs_converse(model, params) is True assert BedrockModelInfo.get_bedrock_route(model, params) == "converse" @@ -922,3 +927,20 @@ def test_gpt56_schema_less_json_object_goes_to_converse_without_a_schema_tool(lo assert "toolConfig" not in body assert "response_format" not in body assert body["inferenceConfig"]["maxTokens"] == 64 + + +def test_gpt56_json_object_with_response_schema_goes_to_converse_as_a_json_tool(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/global.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "Reply with the single word pong."}], + response_format=JSON_OBJECT_WITH_RESPONSE_SCHEMA, + max_tokens=64, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-5.6-sol/converse") + body = json.loads(requests[0].content) + assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "json_tool_call" + assert body["toolConfig"]["toolChoice"] == {"tool": {"name": "json_tool_call"}} + assert "response_format" not in body From c4c24dd9864866c37b30d6dc973e585e36a428b6 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 24 Sep 2026 13:31:29 -0700 Subject: [PATCH 08/31] fix(rust): declare the bedrock runtime chat completions flags on ModelInfo --- litellm-rust/crates/model-catalog/src/model_info.rs | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs index 4a56e1112d1..0b52bea94de 100644 --- a/litellm-rust/crates/model-catalog/src/model_info.rs +++ b/litellm-rust/crates/model-catalog/src/model_info.rs @@ -582,6 +582,12 @@ pub struct ModelInfo { #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_audio_output: Option, #[serde(default, skip_serializing_if = "Option::is_none")] + pub supports_bedrock_runtime_chat_completions: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub supports_bedrock_runtime_chat_completions_response_format: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_computer_use: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_embedding_image_input: Option, From ff8e15d84460d9970443f0004c6fb07ae45dd01e Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 24 Sep 2026 15:21:07 -0700 Subject: [PATCH 09/31] fix(bedrock): opt into the native chat completions route through supported_endpoints --- .../crates/model-catalog/src/model_info.rs | 2 - .../chat/chat_completions/transformation.py | 2 +- litellm/llms/bedrock/common_utils.py | 32 +++++++++-- ...odel_prices_and_context_window_backup.json | 56 +++++++++++++------ model_prices_and_context_window.json | 56 +++++++++++++------ model_prices_and_context_window.schema.json | 3 - ...bedrock_chat_completions_transformation.py | 24 +++++++- tests/test_litellm/test_utils.py | 1 - 8 files changed, 126 insertions(+), 50 deletions(-) diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs index 0b52bea94de..23e69f66492 100644 --- a/litellm-rust/crates/model-catalog/src/model_info.rs +++ b/litellm-rust/crates/model-catalog/src/model_info.rs @@ -582,8 +582,6 @@ pub struct ModelInfo { #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_audio_output: Option, #[serde(default, skip_serializing_if = "Option::is_none")] - pub supports_bedrock_runtime_chat_completions: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_bedrock_runtime_chat_completions_response_format: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option, diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 11e2467b093..4c5e5768119 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -3,7 +3,7 @@ Native OpenAI Chat Completions on Amazon Bedrock Runtime. AWS serves this surface at ``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions`` -for the models whose price-map entry sets ``supports_bedrock_runtime_chat_completions`` +for the models whose price-map ``supported_endpoints`` lists ``/v1/chat/completions`` (Grok 4.6, gpt-oss, the GPT-5.6 family): chat completions stay chat completions instead of being rewritten to Converse. diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 09e62dc393b..4492052b3c4 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -829,24 +829,44 @@ def split_bedrock_region_path(model: str) -> tuple[str | None, str]: return None, stripped -def _bedrock_price_map_flag(model: str, flag: str) -> bool: - entries: Final = ( +BEDROCK_RUNTIME_PRICE_MAP_PROVIDERS: Final = frozenset(("bedrock", "bedrock_converse")) + + +def _bedrock_price_map_entries(model: str) -> tuple[Mapping[str, object] | None, ...]: + return tuple( litellm.model_cost.get(key) for key in (model, strip_bedrock_routing_prefix(model), split_bedrock_region_path(model)[1]) ) - return any(entry is not None and entry.get(flag) is True for entry in entries) + + +def _bedrock_price_map_flag(model: str, flag: str) -> bool: + return any(entry is not None and entry.get(flag) is True for entry in _bedrock_price_map_entries(model)) + + +def _bedrock_runtime_row_lists_chat_completions(entry: Mapping[str, object]) -> bool: + endpoints: Final = entry.get("supported_endpoints") + return ( + entry.get("litellm_provider") in BEDROCK_RUNTIME_PRICE_MAP_PROVIDERS + and isinstance(endpoints, (list, tuple)) + and "/v1/chat/completions" in endpoints + ) def uses_bedrock_runtime_chat_completions(model: str) -> bool: """Whether this Bedrock model should use runtime native Chat Completions. - Data-driven from the price-map ``supports_bedrock_runtime_chat_completions`` flag + Data-driven from ``/v1/chat/completions`` in the price-map row's ``supported_endpoints``, + the same per-model signal ``bedrock_supports_openai_responses`` reads for ``/v1/responses``, so onboarding a model is a JSON change. Explicit ``converse/`` still wins in ``get_bedrock_route`` because prefix routes are checked first, and a request that needs a Converse-only feature (``bedrock_request_needs_converse``) is - served by Converse even on a flagged model. + served by Converse even on a listed model. Only a bedrock-runtime row counts: a + ``bedrock_mantle`` row lists the endpoints of the Mantle host, not this one. """ - return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions") + return any( + entry is not None and _bedrock_runtime_row_lists_chat_completions(entry) + for entry in _bedrock_price_map_entries(model) + ) def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool: diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d58dd4d105a..7fdd8c85ddc 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -40834,7 +40834,9 @@ "output_cost_per_token": 0.0 }, "openai.gpt-oss-120b-1:0": { - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", @@ -40850,7 +40852,9 @@ "supports_tool_choice": true }, "openai.gpt-oss-20b-1:0": { - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", @@ -46591,7 +46595,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -46606,7 +46612,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -46617,7 +46625,9 @@ "input_cost_per_token": 2.64e-06, "output_cost_per_token": 7.92e-06, "cache_read_input_token_cost": 6.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", @@ -57156,7 +57166,6 @@ "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-luna.html" }, "us.openai.gpt-5.6-sol": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4.4e-06, "input_cost_per_token_above_272k_tokens": 8.8e-06, @@ -57188,11 +57197,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "global.openai.gpt-5.6-sol": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4e-06, "input_cost_per_token_above_272k_tokens": 8e-06, @@ -57224,11 +57233,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-5.6-terra": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, @@ -57260,11 +57269,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "global.openai.gpt-5.6-terra": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, @@ -57296,11 +57305,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-5.6-luna": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-07, "input_cost_per_token_above_272k_tokens": 4.4e-07, @@ -57332,6 +57341,7 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -57460,7 +57470,6 @@ ] }, "global.openai.gpt-5.6-luna": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, @@ -57492,6 +57501,7 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -58095,7 +58105,9 @@ "input_cost_per_token": 2.2e-06, "output_cost_per_token": 6.6e-06, "cache_read_input_token_cost": 5.5e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", @@ -58114,7 +58126,9 @@ "input_cost_per_token": 2e-06, "output_cost_per_token": 6e-06, "cache_read_input_token_cost": 5e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", @@ -63818,7 +63832,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -63833,7 +63849,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -64076,7 +64094,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -64091,7 +64111,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index d58dd4d105a..7fdd8c85ddc 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -40834,7 +40834,9 @@ "output_cost_per_token": 0.0 }, "openai.gpt-oss-120b-1:0": { - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 1.5e-07, "litellm_provider": "bedrock_converse", @@ -40850,7 +40852,9 @@ "supports_tool_choice": true }, "openai.gpt-oss-20b-1:0": { - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "input_cost_per_token": 7e-08, "litellm_provider": "bedrock_converse", @@ -46591,7 +46595,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -46606,7 +46612,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -46617,7 +46625,9 @@ "input_cost_per_token": 2.64e-06, "output_cost_per_token": 7.92e-06, "cache_read_input_token_cost": 6.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", @@ -57156,7 +57166,6 @@ "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-luna.html" }, "us.openai.gpt-5.6-sol": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4.4e-06, "input_cost_per_token_above_272k_tokens": 8.8e-06, @@ -57188,11 +57197,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "global.openai.gpt-5.6-sol": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 4e-06, "input_cost_per_token_above_272k_tokens": 8e-06, @@ -57224,11 +57233,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-5.6-terra": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, @@ -57260,11 +57269,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "global.openai.gpt-5.6-terra": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, @@ -57296,11 +57305,11 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-5.6-luna": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-07, "input_cost_per_token_above_272k_tokens": 4.4e-07, @@ -57332,6 +57341,7 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -57460,7 +57470,6 @@ ] }, "global.openai.gpt-5.6-luna": { - "supports_bedrock_runtime_chat_completions": true, "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-07, "input_cost_per_token_above_272k_tokens": 4e-07, @@ -57492,6 +57501,7 @@ "supports_vision": true, "supports_sampling_params": false, "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -58095,7 +58105,9 @@ "input_cost_per_token": 2.2e-06, "output_cost_per_token": 6.6e-06, "cache_read_input_token_cost": 5.5e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", @@ -58114,7 +58126,9 @@ "input_cost_per_token": 2e-06, "output_cost_per_token": 6e-06, "cache_read_input_token_cost": 5e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_bedrock_runtime_chat_completions_response_format": true, "litellm_provider": "bedrock_converse", @@ -63818,7 +63832,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -63833,7 +63849,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -64076,7 +64094,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 3.6e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, @@ -64091,7 +64111,9 @@ "max_tokens": 128000, "mode": "chat", "output_cost_per_token": 7.2e-07, - "supports_bedrock_runtime_chat_completions": true, + "supported_endpoints": [ + "/v1/chat/completions" + ], "supports_bedrock_runtime_chat_completions_tools_with_reasoning": true, "supports_function_calling": true, "supports_reasoning": true, diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index 89ebedc3a0a..bcb0509f25c 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -902,9 +902,6 @@ "supports_audio_output": { "type": "boolean" }, - "supports_bedrock_runtime_chat_completions": { - "type": "boolean" - }, "supports_bedrock_runtime_chat_completions_response_format": { "type": "boolean" }, diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 2f0d384fcdd..3c347e4bed2 100644 --- a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -59,9 +59,27 @@ def test_claude_stays_on_converse(local_cost_map): assert BedrockModelInfo.get_bedrock_route("us.anthropic.claude-3-sonnet-20240229-v1:0") == "converse" -def test_flag_absent_means_no_chat_completions_route(monkeypatch): - monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": {"litellm_provider": "bedrock_converse"}}) +@pytest.mark.parametrize( + "entry", + [ + {"litellm_provider": "bedrock_converse"}, + {"litellm_provider": "bedrock_converse", "supported_endpoints": ["/v1/responses"]}, + {"litellm_provider": "bedrock_converse", "supports_bedrock_runtime_chat_completions": True}, + {"litellm_provider": "bedrock_mantle", "supported_endpoints": ["/v1/chat/completions", "/v1/responses"]}, + {"litellm_provider": "openai", "supported_endpoints": ["/v1/chat/completions"]}, + ], +) +def test_chat_completions_missing_from_supported_endpoints_means_no_chat_completions_route(monkeypatch, entry): + monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": entry}) assert uses_bedrock_runtime_chat_completions("us.xai.grok-4.6") is False + assert BedrockModelInfo.get_bedrock_route("us.xai.grok-4.6") == "converse" + + +def test_chat_completions_in_supported_endpoints_opts_into_the_native_route(monkeypatch): + entry = {"litellm_provider": "bedrock_converse", "supported_endpoints": ["/v1/chat/completions", "/v1/responses"]} + monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": entry}) + assert uses_bedrock_runtime_chat_completions("us.xai.grok-4.6") is True + assert BedrockModelInfo.get_bedrock_route("bedrock/us.xai.grok-4.6") == "chat_completions" def test_complete_url_is_runtime_openai_chat_completions(monkeypatch): @@ -860,7 +878,7 @@ SYNTHETIC_NATIVE_MODEL = "vendor.native-model-v1:0" def test_capability_flags_are_read_from_the_cost_map(monkeypatch, capability_flags, request_params, needs_converse): entry = { "litellm_provider": "bedrock_converse", - "supports_bedrock_runtime_chat_completions": True, + "supported_endpoints": ["/v1/chat/completions"], **capability_flags, } monkeypatch.setattr(litellm, "model_cost", {SYNTHETIC_NATIVE_MODEL: entry}) diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 98e46016401..39cb25a0f51 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -927,7 +927,6 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "supports_video_input": {"type": "boolean"}, "supports_vision": {"type": "boolean"}, "supports_web_search": {"type": "boolean"}, - "supports_bedrock_runtime_chat_completions": {"type": "boolean"}, "supports_bedrock_runtime_chat_completions_tools_with_reasoning": {"type": "boolean"}, "supports_bedrock_runtime_chat_completions_response_format": {"type": "boolean"}, "supports_url_context": {"type": "boolean"}, From 4101c0ceb2e213a177613d89220bb2e64d191aeb Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 24 Sep 2026 18:08:46 -0700 Subject: [PATCH 10/31] docs(cost-map): describe the bedrock native chat completions capability flags --- ci_cd/generate_model_prices_schema.py | 20 ++++++++++++++++++- .../crates/model-catalog/src/model_info.rs | 2 ++ model_prices_and_context_window.schema.json | 6 ++++-- 3 files changed, 25 insertions(+), 3 deletions(-) diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index 8eec07dadda..f54177def8e 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -211,6 +211,24 @@ NUMBER_KEYS: dict[str, JsonSchema] = { }, } +BOOLEAN_KEYS: dict[str, JsonSchema] = { + "supports_bedrock_runtime_chat_completions_response_format": { + "type": "boolean", + "description": ( + "The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; " + "unset means LiteLLM serves those requests through Converse's json_tool_call emulation." + ), + }, + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": { + "type": "boolean", + "description": ( + "The Bedrock native /v1/chat/completions route serves this model's function tools with any " + "reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort " + "is exactly 'none'." + ), + }, +} + COST_DESCRIPTIONS: dict[str, str] = { "input_cost_per_token": "USD per prompt token.", "output_cost_per_token": "USD per generated token.", @@ -292,7 +310,7 @@ def string_key_schemas(modes: tuple) -> dict[str, JsonSchema]: def classify(key: str, modes: tuple) -> Optional[JsonSchema]: - curated = {**OBJECT_KEYS, **ARRAY_KEYS, **string_key_schemas(modes), **INTEGER_KEYS, **NUMBER_KEYS} + curated = {**OBJECT_KEYS, **ARRAY_KEYS, **string_key_schemas(modes), **INTEGER_KEYS, **NUMBER_KEYS, **BOOLEAN_KEYS} if key in curated: return curated[key] if key.startswith("supports_") or key in EXTRA_BOOLEAN_KEYS: diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs index 23e69f66492..5dda91b2213 100644 --- a/litellm-rust/crates/model-catalog/src/model_info.rs +++ b/litellm-rust/crates/model-catalog/src/model_info.rs @@ -581,8 +581,10 @@ pub struct ModelInfo { pub supports_audio_input: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_audio_output: Option, + /// The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; unset means LiteLLM serves those requests through Converse's json_tool_call emulation. #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_bedrock_runtime_chat_completions_response_format: Option, + /// The Bedrock native /v1/chat/completions route serves this model's function tools with any reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort is exactly 'none'. #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option, #[serde(default, skip_serializing_if = "Option::is_none")] diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index bcb0509f25c..d9a7dd2fff5 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -903,10 +903,12 @@ "type": "boolean" }, "supports_bedrock_runtime_chat_completions_response_format": { - "type": "boolean" + "type": "boolean", + "description": "The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; unset means LiteLLM serves those requests through Converse's json_tool_call emulation." }, "supports_bedrock_runtime_chat_completions_tools_with_reasoning": { - "type": "boolean" + "type": "boolean", + "description": "The Bedrock native /v1/chat/completions route serves this model's function tools with any reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort is exactly 'none'." }, "supports_computer_use": { "type": "boolean" From 61a130c9a13a6d4563ba0c26b44a447ae12e98b2 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 24 Sep 2026 18:20:45 -0700 Subject: [PATCH 11/31] revert: docs(cost-map): describe the bedrock native chat completions capability flags This reverts commit 4101c0ceb2e213a177613d89220bb2e64d191aeb. cost-map-guard runs main's schema generator under pull_request_target and compares its output to the PR's committed schema, so a PR that changes the generator's output cannot pass that required check until the generator change lands on main first. The descriptions move to a follow-up that lands the generator change ahead of the schema --- ci_cd/generate_model_prices_schema.py | 20 +------------------ .../crates/model-catalog/src/model_info.rs | 2 -- model_prices_and_context_window.schema.json | 6 ++---- 3 files changed, 3 insertions(+), 25 deletions(-) diff --git a/ci_cd/generate_model_prices_schema.py b/ci_cd/generate_model_prices_schema.py index f54177def8e..8eec07dadda 100644 --- a/ci_cd/generate_model_prices_schema.py +++ b/ci_cd/generate_model_prices_schema.py @@ -211,24 +211,6 @@ NUMBER_KEYS: dict[str, JsonSchema] = { }, } -BOOLEAN_KEYS: dict[str, JsonSchema] = { - "supports_bedrock_runtime_chat_completions_response_format": { - "type": "boolean", - "description": ( - "The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; " - "unset means LiteLLM serves those requests through Converse's json_tool_call emulation." - ), - }, - "supports_bedrock_runtime_chat_completions_tools_with_reasoning": { - "type": "boolean", - "description": ( - "The Bedrock native /v1/chat/completions route serves this model's function tools with any " - "reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort " - "is exactly 'none'." - ), - }, -} - COST_DESCRIPTIONS: dict[str, str] = { "input_cost_per_token": "USD per prompt token.", "output_cost_per_token": "USD per generated token.", @@ -310,7 +292,7 @@ def string_key_schemas(modes: tuple) -> dict[str, JsonSchema]: def classify(key: str, modes: tuple) -> Optional[JsonSchema]: - curated = {**OBJECT_KEYS, **ARRAY_KEYS, **string_key_schemas(modes), **INTEGER_KEYS, **NUMBER_KEYS, **BOOLEAN_KEYS} + curated = {**OBJECT_KEYS, **ARRAY_KEYS, **string_key_schemas(modes), **INTEGER_KEYS, **NUMBER_KEYS} if key in curated: return curated[key] if key.startswith("supports_") or key in EXTRA_BOOLEAN_KEYS: diff --git a/litellm-rust/crates/model-catalog/src/model_info.rs b/litellm-rust/crates/model-catalog/src/model_info.rs index 5dda91b2213..23e69f66492 100644 --- a/litellm-rust/crates/model-catalog/src/model_info.rs +++ b/litellm-rust/crates/model-catalog/src/model_info.rs @@ -581,10 +581,8 @@ pub struct ModelInfo { pub supports_audio_input: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_audio_output: Option, - /// The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; unset means LiteLLM serves those requests through Converse's json_tool_call emulation. #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_bedrock_runtime_chat_completions_response_format: Option, - /// The Bedrock native /v1/chat/completions route serves this model's function tools with any reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort is exactly 'none'. #[serde(default, skip_serializing_if = "Option::is_none")] pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option, #[serde(default, skip_serializing_if = "Option::is_none")] diff --git a/model_prices_and_context_window.schema.json b/model_prices_and_context_window.schema.json index d9a7dd2fff5..bcb0509f25c 100644 --- a/model_prices_and_context_window.schema.json +++ b/model_prices_and_context_window.schema.json @@ -903,12 +903,10 @@ "type": "boolean" }, "supports_bedrock_runtime_chat_completions_response_format": { - "type": "boolean", - "description": "The Bedrock native /v1/chat/completions route enforces a json_schema response_format for this model; unset means LiteLLM serves those requests through Converse's json_tool_call emulation." + "type": "boolean" }, "supports_bedrock_runtime_chat_completions_tools_with_reasoning": { - "type": "boolean", - "description": "The Bedrock native /v1/chat/completions route serves this model's function tools with any reasoning_effort; unset means LiteLLM serves a tools request through Converse unless reasoning_effort is exactly 'none'." + "type": "boolean" }, "supports_computer_use": { "type": "boolean" From c60249fa882116c4cd8fe784770f5e3a3be3ec30 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 16:02:49 +0000 Subject: [PATCH 12/31] test(bedrock): move the native chat completions tests under tests/unit Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- tests/unit/llms/bedrock/chat/chat_completions/__init__.py | 0 .../test_bedrock_chat_completions_transformation.py | 0 2 files changed, 0 insertions(+), 0 deletions(-) create mode 100644 tests/unit/llms/bedrock/chat/chat_completions/__init__.py rename tests/{test_litellm => unit}/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py (100%) diff --git a/tests/unit/llms/bedrock/chat/chat_completions/__init__.py b/tests/unit/llms/bedrock/chat/chat_completions/__init__.py new file mode 100644 index 00000000000..e69de29bb2d diff --git a/tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py similarity index 100% rename from tests/test_litellm/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py rename to tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py From f4d84f8f5db0c26e0aa3c50bc2ddb77d07122f5b Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:28:06 +0000 Subject: [PATCH 13/31] fix(bedrock): drop reasoning_effort none for grok on the native chat completions route Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../chat/chat_completions/transformation.py | 29 ++++++++++++- ...bedrock_chat_completions_transformation.py | 42 +++++++++++++++++++ 2 files changed, 70 insertions(+), 1 deletion(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 4c5e5768119..be2eb8c7713 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -60,6 +60,31 @@ def chat_completions_params_refused_for(model: str) -> frozenset[str]: ) +CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY: Final = MappingProxyType({"xai.": frozenset(("none",))}) + + +def chat_completions_reasoning_efforts_refused_for(model: str) -> frozenset[str]: + """The ``reasoning_effort`` values AWS's Chat Completions endpoint rejects for this model. + + Grok answers ``"none"`` with a 400 (it takes low, medium, high, and xhigh) where Converse dropped every + ``reasoning_effort`` for it, so the native config drops the value and AWS applies its default effort as before. + """ + model_id: Final = split_bedrock_region_path(model)[1] + return frozenset().union( + *( + refused + for family, refused in CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY.items() + if family in model_id + ) + ) + + +def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) -> Mapping[str, object]: + if params.get("reasoning_effort") not in chat_completions_reasoning_efforts_refused_for(model): + return params + return MappingProxyType({key: value for key, value in params.items() if key != "reasoning_effort"}) + + def _held_close_tag_prefix(text: str) -> int: return next( ( @@ -277,7 +302,9 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): drop_params=drop_params, replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens, ) - return dict(with_max_completion_tokens(mapped)) # mutable-ok: get_optional_params keeps filling this dict + return dict( # mutable-ok: get_optional_params keeps filling this dict + without_refused_reasoning_effort(model, with_max_completion_tokens(mapped)) + ) def _inference_params( self, optional_params: Mapping[str, object] diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 3c347e4bed2..061df201c15 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -11,6 +11,7 @@ from litellm.llms.bedrock.chat.chat_completions.transformation import ( AmazonBedrockRuntimeChatCompletionsConfig, BedrockRuntimeChatCompletionsStreamingHandler, ReasoningTagSplitter, + chat_completions_reasoning_efforts_refused_for, split_reasoning_tag, with_max_completion_tokens, ) @@ -333,6 +334,47 @@ def test_with_max_completion_tokens_leaves_other_params_alone(): assert with_max_completion_tokens({"temperature": 0.5}) == {"temperature": 0.5} +@pytest.mark.parametrize( + "model", + ["us.xai.grok-4.6", "bedrock/us-gov-west-1/us.xai.grok-4.6"], +) +def test_map_openai_params_drops_reasoning_effort_none_for_grok(model): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"reasoning_effort": "none", "max_tokens": 64}, + optional_params={}, + model=model, + drop_params=False, + ) + assert "reasoning_effort" not in mapped + + +def test_map_openai_params_keeps_reasoning_effort_low_for_grok(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"reasoning_effort": "low", "max_tokens": 64}, + optional_params={}, + model="us.xai.grok-4.6", + drop_params=False, + ) + assert mapped["reasoning_effort"] == "low" + + +def test_map_openai_params_keeps_reasoning_effort_none_for_gpt56(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"reasoning_effort": "none", "max_tokens": 64}, + optional_params={}, + model="global.openai.gpt-5.6-sol", + drop_params=False, + ) + assert mapped["reasoning_effort"] == "none" + + +def test_reasoning_efforts_refused_for_is_empty_outside_xai(): + assert chat_completions_reasoning_efforts_refused_for("openai.gpt-oss-20b-1:0") == frozenset() + + def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): cfg = AmazonBedrockRuntimeChatCompletionsConfig() assert "reasoning_effort" in cfg.get_supported_openai_params("global.openai.gpt-5.6-sol") From e653217228d191d0c0ecb7c9b275786c19052e94 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 17:33:01 +0000 Subject: [PATCH 14/31] fix(bedrock): keep converse extension params on the converse route Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- litellm/llms/bedrock/common_utils.py | 14 ++++++++++++-- ...test_bedrock_chat_completions_transformation.py | 12 ++++++++++++ 2 files changed, 24 insertions(+), 2 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 8fe545a3272..098be7082d3 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -890,7 +890,16 @@ def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> boo BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( - ("guardrailConfig", "performanceConfig", "serviceTier", "requestMetadata", "outputConfig", "thinking") + ( + "guardrailConfig", + "performanceConfig", + "serviceTier", + "requestMetadata", + "outputConfig", + "thinking", + "additionalModelRequestFields", + "top_k", + ) ) @@ -910,7 +919,8 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje """Whether a request on a runtime-Chat-Completions model must still be served by Converse. Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking`` - block included, which only Converse forwards as ``additionalModelRequestFields``) have no field on + block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse + forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, function tools (``tools`` or legacy ``functions``) on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 061df201c15..17c0f8d3c1d 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -258,6 +258,18 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model): assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": None}) == "chat_completions" +@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"]) +@pytest.mark.parametrize( + "request_params", + [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}], + ids=["additionalModelRequestFields", "top_k"], +) +def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params): + assert bedrock_request_needs_converse(model, request_params) is True + assert BedrockModelInfo.get_bedrock_route(model, request_params) == "converse" + assert BedrockModelInfo.get_bedrock_route(model, {key: None for key in request_params}) == "chat_completions" + + @pytest.mark.parametrize( "request_params, expected_route", [ From 093a9d4ddfbb010977ad7a50a0c8c3f7745dbcaf Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:37:36 +0000 Subject: [PATCH 15/31] fix(bedrock): inline http image urls and keep stop on converse for native chat completions Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../chat/chat_completions/transformation.py | 57 +++++++++++++- litellm/llms/bedrock/common_utils.py | 4 +- ...bedrock_chat_completions_transformation.py | 75 +++++++++++++++++-- 3 files changed, 127 insertions(+), 9 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index be2eb8c7713..123531cf227 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -22,6 +22,11 @@ import httpx from typing_extensions import assert_never import litellm +from litellm.litellm_core_utils.prompt_templates.image_handling import ( + async_inline_remote_media, + convert_url_to_base64, + inline_remote_image_urls, +) from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path @@ -40,7 +45,7 @@ REASONING_CLOSE_TAG: Final = "" CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType( { - "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs")), + "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs")), "openai.gpt-oss": frozenset(("logit_bias",)), "xai.": frozenset(("frequency_penalty", "presence_penalty")), } @@ -167,6 +172,48 @@ def split_reasoning_tag(content: str) -> tuple[str | None, str]: return reasoning or None, body +def _remote_http_url(candidate: object) -> str | None: + return candidate if isinstance(candidate, str) and candidate.startswith(("http://", "https://")) else None + + +def _inlined_image_url_part(part: object) -> object: + fields: Final = part if isinstance(part, Mapping) else None + if fields is None or fields.get("type") != "image_url": + return part + image_url: Final = fields.get("image_url") + image_url_fields: Final = image_url if isinstance(image_url, Mapping) else None + url: Final = _remote_http_url(image_url_fields.get("url") if image_url_fields is not None else image_url) + if url is None: + return part + data_url: Final = convert_url_to_base64(url) + inlined: Final = {**image_url_fields, "url": data_url} if image_url_fields is not None else data_url + return {**fields, "image_url": inlined} # mutable-ok: json-serialized message part + + +def _inlined_image_url_message(message: AllMessageValues) -> AllMessageValues: + content: Final = message.get("content") + if not isinstance(content, list): + return message + inlined_message: Final = { # mutable-ok: json-serialized message + **message, + "content": [_inlined_image_url_part(part) for part in content], + } + return inlined_message # pyright: ignore[reportReturnType] # the same message with remote image parts inlined + + +def _with_inlined_remote_image_urls( + messages: list[AllMessageValues], +) -> list[AllMessageValues]: # mutable-ok: transform_request takes a list + """Inline every remote ``image_url`` so AWS never sees the ``http(s)://`` URLs it rejects. + + AWS's native surface only takes inline ``data:`` URLs and S3 URLs where Converse downloaded + remote images itself, so the bytes are fetched and inlined here exactly like Converse did. + """ + return [ # mutable-ok: transform_request takes a list + _inlined_image_url_message(message) for message in messages + ] + + class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler): """OpenAI chunk parsing plus the ```` split, tracked per choice index.""" @@ -222,6 +269,10 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): def custom_llm_provider(self) -> str | None: return "bedrock" + @property + def uses_async_transform_request(self) -> bool: + return True + def get_error_class( self, error_message: str, @@ -325,7 +376,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): ) -> dict: # mutable-ok: BaseConfig signature return super().transform_request( model=split_bedrock_region_path(model)[1], - messages=messages, + messages=_with_inlined_remote_image_urls(messages), optional_params=self._inference_params(optional_params), litellm_params=litellm_params, headers=headers, @@ -341,7 +392,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): ) -> dict: # mutable-ok: BaseConfig signature return await super().async_transform_request( model=split_bedrock_region_path(model)[1], - messages=messages, + messages=await async_inline_remote_media(messages, should_inline=inline_remote_image_urls), optional_params=self._inference_params(optional_params), litellm_params=litellm_params, headers=headers, diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 098be7082d3..6bd4a8d634c 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -899,6 +899,7 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( "thinking", "additionalModelRequestFields", "top_k", + "stop", ) ) @@ -921,7 +922,8 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking`` block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on - AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, + AWS's native OpenAI surface, ``stop`` stays on Converse where it fails loudly instead of silently + stopping hidden reasoning, operator-owned request metadata is only written onto the Converse body, function tools (``tools`` or legacy ``functions``) on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 17c0f8d3c1d..4bfa7bf0fc4 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -258,11 +258,13 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model): assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": None}) == "chat_completions" -@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"]) +@pytest.mark.parametrize( + "model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6", "global.openai.gpt-5.6-sol"] +) @pytest.mark.parametrize( "request_params", - [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}], - ids=["additionalModelRequestFields", "top_k"], + [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}, {"stop": ["END"]}], + ids=["additionalModelRequestFields", "top_k", "stop"], ) def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params): assert bedrock_request_needs_converse(model, request_params) is True @@ -331,6 +333,69 @@ def test_map_openai_params_sends_max_tokens_as_max_completion_tokens(): assert mapped == {"max_completion_tokens": 64, "temperature": 0.1} +HTTPS_IMAGE_URL = "https://example.com/cat.png" +IMAGE_MESSAGES = [ + { + "role": "user", + "content": [ + {"type": "text", "text": "what is this"}, + {"type": "image_url", "image_url": HTTPS_IMAGE_URL}, + {"type": "image_url", "image_url": {"url": HTTPS_IMAGE_URL, "detail": "high"}}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,AAA"}}, + {"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}}, + ], + } +] + + +def _assert_remote_images_inlined(content): + assert content[0] == {"type": "text", "text": "what is this"} + assert content[1]["image_url"]["url"] == f"data:image/png;base64,{HTTPS_IMAGE_URL}" + assert content[2] == { + "type": "image_url", + "image_url": {"url": f"data:image/png;base64,{HTTPS_IMAGE_URL}", "detail": "high"}, + } + assert content[3]["image_url"]["url"] == "data:image/png;base64,AAA" + assert content[4]["image_url"]["url"] == "s3://bucket/key.png" + + +def test_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch): + import litellm.llms.bedrock.chat.chat_completions.transformation as native_cc + + monkeypatch.setattr( + native_cc, "convert_url_to_base64", lambda url: f"data:image/png;base64,{url}" + ) + body = AmazonBedrockRuntimeChatCompletionsConfig().transform_request( + model="us.xai.grok-4.6", + messages=IMAGE_MESSAGES, + optional_params={}, + litellm_params={}, + headers={}, + ) + + _assert_remote_images_inlined(body["messages"][0]["content"]) + + +async def test_async_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch): + import litellm.litellm_core_utils.prompt_templates.image_handling as image_handling + + async def fake_convert(url): + return f"data:image/png;base64,{url}" + + monkeypatch.setattr(image_handling, "async_convert_url_to_base64", fake_convert) + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + assert cfg.uses_async_transform_request is True + body = await cfg.async_transform_request( + model="us.xai.grok-4.6", + messages=IMAGE_MESSAGES, + optional_params={}, + litellm_params={}, + headers={}, + ) + + _assert_remote_images_inlined(body["messages"][0]["content"]) + + def test_map_openai_params_keeps_explicit_max_completion_tokens(): cfg = AmazonBedrockRuntimeChatCompletionsConfig() mapped = cfg.map_openai_params( @@ -398,8 +463,8 @@ def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): [ ( "bedrock/global.openai.gpt-5.6-sol", - ("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs", "n"), - ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions"), + ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "n"), + ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions", "stop"), ), ( "us.xai.grok-4.6", From daba2576f503ab4b313cac6d196f28fa731ef0f9 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:41:15 +0000 Subject: [PATCH 16/31] refactor(bedrock): share the sync remote media inliner Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../prompt_templates/image_handling.py | 20 ++++++++ .../chat/chat_completions/transformation.py | 46 +------------------ .../litellm_core_utils/test_image_handling.py | 45 ++++++++++++++++++ ...bedrock_chat_completions_transformation.py | 4 +- 4 files changed, 69 insertions(+), 46 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/image_handling.py b/litellm/litellm_core_utils/prompt_templates/image_handling.py index c44c80bc0a0..cb5c02ce10e 100644 --- a/litellm/litellm_core_utils/prompt_templates/image_handling.py +++ b/litellm/litellm_core_utils/prompt_templates/image_handling.py @@ -310,6 +310,26 @@ async def _fetch_data_urls(remote_urls: tuple[str, ...]) -> tuple[str, ...]: raise +def inline_remote_media( + messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues] + should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url, +) -> list[AllMessageValues]: # mutable-ok: every transform_request takes list[AllMessageValues] + remote_urls: Final = tuple( + dict.fromkeys( + remote.url + for message in messages + for part in _content_parts(message) + if (remote := _parse_remote_part(part)) is not None and should_inline(_remote_media(remote)) + ) + ) + if not remote_urls: + return messages + data_urls: Final = MappingProxyType({url: convert_url_to_base64(url) for url in remote_urls}) + return [ # mutable-ok: transform_request takes a list + _inline_message(message, data_urls, should_inline) for message in messages + ] + + async def async_inline_remote_media( messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues] should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url, diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 123531cf227..d907ef613a2 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -24,8 +24,8 @@ from typing_extensions import assert_never import litellm from litellm.litellm_core_utils.prompt_templates.image_handling import ( async_inline_remote_media, - convert_url_to_base64, inline_remote_image_urls, + inline_remote_media, ) from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM @@ -172,48 +172,6 @@ def split_reasoning_tag(content: str) -> tuple[str | None, str]: return reasoning or None, body -def _remote_http_url(candidate: object) -> str | None: - return candidate if isinstance(candidate, str) and candidate.startswith(("http://", "https://")) else None - - -def _inlined_image_url_part(part: object) -> object: - fields: Final = part if isinstance(part, Mapping) else None - if fields is None or fields.get("type") != "image_url": - return part - image_url: Final = fields.get("image_url") - image_url_fields: Final = image_url if isinstance(image_url, Mapping) else None - url: Final = _remote_http_url(image_url_fields.get("url") if image_url_fields is not None else image_url) - if url is None: - return part - data_url: Final = convert_url_to_base64(url) - inlined: Final = {**image_url_fields, "url": data_url} if image_url_fields is not None else data_url - return {**fields, "image_url": inlined} # mutable-ok: json-serialized message part - - -def _inlined_image_url_message(message: AllMessageValues) -> AllMessageValues: - content: Final = message.get("content") - if not isinstance(content, list): - return message - inlined_message: Final = { # mutable-ok: json-serialized message - **message, - "content": [_inlined_image_url_part(part) for part in content], - } - return inlined_message # pyright: ignore[reportReturnType] # the same message with remote image parts inlined - - -def _with_inlined_remote_image_urls( - messages: list[AllMessageValues], -) -> list[AllMessageValues]: # mutable-ok: transform_request takes a list - """Inline every remote ``image_url`` so AWS never sees the ``http(s)://`` URLs it rejects. - - AWS's native surface only takes inline ``data:`` URLs and S3 URLs where Converse downloaded - remote images itself, so the bytes are fetched and inlined here exactly like Converse did. - """ - return [ # mutable-ok: transform_request takes a list - _inlined_image_url_message(message) for message in messages - ] - - class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler): """OpenAI chunk parsing plus the ```` split, tracked per choice index.""" @@ -376,7 +334,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): ) -> dict: # mutable-ok: BaseConfig signature return super().transform_request( model=split_bedrock_region_path(model)[1], - messages=_with_inlined_remote_image_urls(messages), + messages=inline_remote_media(messages, should_inline=inline_remote_image_urls), optional_params=self._inference_params(optional_params), litellm_params=litellm_params, headers=headers, diff --git a/tests/unit/litellm_core_utils/test_image_handling.py b/tests/unit/litellm_core_utils/test_image_handling.py index 21e97e97357..1fc8d54cecc 100644 --- a/tests/unit/litellm_core_utils/test_image_handling.py +++ b/tests/unit/litellm_core_utils/test_image_handling.py @@ -16,6 +16,7 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import ( async_convert_url_to_base64, async_inline_remote_media, convert_url_to_base64, + inline_remote_media, ) from litellm.litellm_core_utils.url_utils import SSRFError @@ -320,6 +321,50 @@ async def test_async_inline_remote_media_inlines_every_remote_part_shape(async_o assert messages == snapshot +def test_inline_remote_media_inlines_every_remote_part_shape(monkeypatch): + image_url = f"http://img.example/{uuid.uuid4()}.png" + pdf_url = f"http://docs.example/{uuid.uuid4()}.pdf" + fetched = [] + + def fake_convert(url): + fetched.append(url) + return f"data:image/png;base64,{url}" + + monkeypatch.setattr(image_handling, "convert_url_to_base64", fake_convert) + messages = [ + {"role": "system", "content": "be terse"}, + { + "role": "user", + "content": [ + {"type": "text", "text": "what is this?"}, + {"type": "image_url", "image_url": {"url": image_url, "detail": "low"}}, + {"type": "image_url", "image_url": image_url}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,iVBORw0KGgo="}}, + {"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}}, + {"type": "file", "file": {"file_id": pdf_url}}, + {"type": "document", "source": {"type": "url", "url": pdf_url}, "title": "the doc"}, + ], + }, + ] + snapshot = copy.deepcopy(messages) + + inlined = inline_remote_media(messages, should_inline=image_handling.inline_remote_image_urls) + + data_url = f"data:image/png;base64,{image_url}" + assert inlined[0] == {"role": "system", "content": "be terse"} + assert inlined[1]["content"] == [ + {"type": "text", "text": "what is this?"}, + {"type": "image_url", "image_url": {"url": data_url, "detail": "low"}}, + {"type": "image_url", "image_url": data_url}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,iVBORw0KGgo="}}, + {"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}}, + {"type": "file", "file": {"file_id": pdf_url}}, + {"type": "document", "source": {"type": "url", "url": pdf_url}, "title": "the doc"}, + ] + assert fetched == [image_url] + assert messages == snapshot + + async def test_async_inline_remote_media_inlines_only_the_parts_the_predicate_accepts(async_only_image_fetch): files_api_prefix = "https://generativelanguage.googleapis.com/v1beta/files/" files_api_pdf = f"{files_api_prefix}{uuid.uuid4().hex}" diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 4bfa7bf0fc4..cbde3dd03a0 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -360,10 +360,10 @@ def _assert_remote_images_inlined(content): def test_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch): - import litellm.llms.bedrock.chat.chat_completions.transformation as native_cc + import litellm.litellm_core_utils.prompt_templates.image_handling as image_handling monkeypatch.setattr( - native_cc, "convert_url_to_base64", lambda url: f"data:image/png;base64,{url}" + image_handling, "convert_url_to_base64", lambda url: f"data:image/png;base64,{url}" ) body = AmazonBedrockRuntimeChatCompletionsConfig().transform_request( model="us.xai.grok-4.6", From cbf01c25babb5e4410e65f57950787f4f76b2f93 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 19:16:13 +0000 Subject: [PATCH 17/31] fix(image-handling): infer the image mime type when the server sends a generic content type Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../prompt_templates/image_handling.py | 28 +++++------ .../litellm_core_utils/test_image_handling.py | 49 +++++++++++++++++++ 2 files changed, 60 insertions(+), 17 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/image_handling.py b/litellm/litellm_core_utils/prompt_templates/image_handling.py index cb5c02ce10e..57b4f545301 100644 --- a/litellm/litellm_core_utils/prompt_templates/image_handling.py +++ b/litellm/litellm_core_utils/prompt_templates/image_handling.py @@ -15,6 +15,7 @@ import litellm from litellm import verbose_logger from litellm.caching.caching import InMemoryCache from litellm.constants import MAX_IMAGE_URL_DOWNLOAD_SIZE_MB +from litellm.litellm_core_utils.prompt_templates.common_utils import infer_content_type_from_url_and_content from litellm.litellm_core_utils.url_utils import SSRFError, async_safe_get, safe_get from litellm.types.llms.openai import AllMessageValues @@ -55,23 +56,16 @@ def _process_image_response(response: Response, url: str) -> str: base64_image: Final = base64.b64encode(image_bytes).decode("utf-8") - image_type: Final = response.headers.get("Content-Type") - if image_type is None: - img_type = url.split(".")[-1].lower() - _img_type: Final = { - "jpg": "image/jpeg", - "jpeg": "image/jpeg", - "png": "image/png", - "gif": "image/gif", - "webp": "image/webp", - }.get(img_type) - if _img_type is None: - raise Exception( - f"Error: Unsupported image format. Format={_img_type}. Supported types = ['image/jpeg', 'image/png', 'image/gif', 'image/webp']" - ) - img_type = _img_type - else: - img_type = image_type + try: + img_type: Final = infer_content_type_from_url_and_content( + url=url, + content=bytes(image_bytes), + current_content_type=response.headers.get("Content-Type"), + ) + except ValueError as e: + raise litellm.ImageFetchError( + f"Error: Unable to determine image content type from the server's headers, the URL, or the image bytes. url={url}" + ) from e result: Final = f"data:{img_type};base64,{base64_image}" in_memory_cache.set_cache(url, result) diff --git a/tests/unit/litellm_core_utils/test_image_handling.py b/tests/unit/litellm_core_utils/test_image_handling.py index 1fc8d54cecc..57eb32f98e5 100644 --- a/tests/unit/litellm_core_utils/test_image_handling.py +++ b/tests/unit/litellm_core_utils/test_image_handling.py @@ -1,4 +1,5 @@ import asyncio +import base64 import copy import time import uuid @@ -259,6 +260,54 @@ async def test_async_data_url_is_returned_unchanged_without_fetch(monkeypatch): assert await async_convert_url_to_base64(data_url) == data_url +REAL_PNG_BYTES = base64.b64decode( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg==" +) + + +def _stub_image_client(content, content_type): + class _Client: + def get(self, url, follow_redirects=True): + headers = {} if content_type is None else {"Content-Type": content_type} + return Response(200, content=content, headers=headers, request=Request("GET", url)) + + return _Client() + + +def test_convert_url_to_base64_infers_the_type_when_the_server_sends_octet_stream(monkeypatch): + monkeypatch.setattr( + litellm, "module_level_client", _stub_image_client(REAL_PNG_BYTES, "application/octet-stream") + ) + + result = convert_url_to_base64(f"http://img.example/{uuid.uuid4()}") + + assert result.startswith("data:image/png;base64,") + + +def test_convert_url_to_base64_keeps_a_real_content_type(monkeypatch): + monkeypatch.setattr( + litellm, "module_level_client", _stub_image_client(REAL_PNG_BYTES, "image/jpeg") + ) + + result = convert_url_to_base64(f"http://img.example/{uuid.uuid4()}.png") + + assert result.startswith("data:image/jpeg;base64,") + + +def test_convert_url_to_base64_raises_when_no_content_type_is_determinable(monkeypatch): + monkeypatch.setattr( + litellm, + "module_level_client", + _stub_image_client(b"\x00\x01\x02\x03not-an-image", "application/octet-stream"), + ) + url = f"http://img.example/{uuid.uuid4()}" + + with pytest.raises(litellm.ImageFetchError) as excinfo: + convert_url_to_base64(url) + + assert url in str(excinfo.value) + + def test_image_size_limit_disabled(monkeypatch): """ Test that setting MAX_IMAGE_URL_DOWNLOAD_SIZE_MB to 0 disables all image URL downloads. From 4900a1ae85b314e108e6150adfb2d3acef1c0914 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 14:25:38 -0700 Subject: [PATCH 18/31] fix(bedrock): stop sending aws_bedrock_project_id as OpenAI-Project on the runtime chat completions route --- .../chat/chat_completions/transformation.py | 24 ------------------- ...bedrock_chat_completions_transformation.py | 13 ++++++++++ 2 files changed, 13 insertions(+), 24 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index d907ef613a2..381ce7922d7 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -394,30 +394,6 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): choice.message.content = content return response - def validate_environment( - self, - headers: dict, # mutable-ok: BaseConfig signature - model: str, - messages: list[AllMessageValues], # mutable-ok: BaseConfig signature - optional_params: dict, # mutable-ok: BaseConfig signature - litellm_params: dict, # mutable-ok: BaseConfig signature - api_key: str | None = None, - api_base: str | None = None, - ) -> dict: # mutable-ok: BaseConfig signature - validated: Final = super().validate_environment( - headers=headers, - model=model, - messages=messages, - optional_params=optional_params, - litellm_params=litellm_params, - api_key=api_key, - api_base=api_base, - ) - project_id: Final = litellm_params.get("aws_bedrock_project_id") - if not project_id: - return validated - return {**validated, "OpenAI-Project": project_id} # mutable-ok: BaseConfig signature returns a dict - def get_supported_openai_params(self, model: str) -> list: # mutable-ok: BaseConfig signature refused: Final = frozenset(("n", *chat_completions_params_refused_for(model))) base_params: Final = tuple( diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index cbde3dd03a0..f9f330a7340 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -109,6 +109,19 @@ def test_complete_url_appends_to_openai_v1_base(): assert url == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" +def test_project_id_is_not_sent_as_openai_project_header(): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + headers = cfg.validate_environment( + headers={}, + model="bedrock/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + optional_params={}, + litellm_params={"aws_bedrock_project_id": "proj_from_config"}, + ) + assert "OpenAI-Project" not in headers + assert headers["Content-Type"] == "application/json" + + def test_transform_request_is_openai_chat_body_not_converse(): cfg = AmazonBedrockRuntimeChatCompletionsConfig() body = cfg.transform_request( From 3a390e620645251b3be19a635c06bd538428ad81 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 28 Sep 2026 21:29:04 -0700 Subject: [PATCH 19/31] feat(bedrock): make native chat completions an opt-in bedrock/chat_completions/ route Bare Bedrock OpenAI and Grok model ids stay on Converse as on main. The bedrock/chat_completions/ prefix opts a deployment into bedrock-runtime's /openai/v1/chat/completions, and a request carrying a Converse-only param still falls back to Converse. The cost map no longer decides the route. --- .../chat/chat_completions/transformation.py | 14 +- litellm/llms/bedrock/common_utils.py | 48 +--- litellm/main.py | 2 +- ...bedrock_chat_completions_transformation.py | 220 +++++++++++------- ..._cross_region_inference_profile_mapping.py | 9 +- 5 files changed, 156 insertions(+), 137 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 381ce7922d7..2025927807b 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -3,14 +3,14 @@ Native OpenAI Chat Completions on Amazon Bedrock Runtime. AWS serves this surface at ``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions`` -for the models whose price-map ``supported_endpoints`` lists ``/v1/chat/completions`` -(Grok 4.6, gpt-oss, the GPT-5.6 family): chat completions stay chat completions -instead of being rewritten to Converse. +for Grok 4.6, gpt-oss and the GPT-5.6 family. The ``chat_completions/`` route prefix +opts a model in, so chat completions stay chat completions instead of being rewritten +to Converse; without it these models stay on Converse. -Usage: model="us.xai.grok-4.6", model="bedrock/openai.gpt-oss-20b-1:0" or -model="bedrock/global.openai.gpt-5.6-sol". Explicit ``bedrock/converse/...`` -still uses Converse, and so does a request that needs a Converse-only feature -(``bedrock_request_needs_converse`` in ``common_utils``). +Usage: model="bedrock/chat_completions/openai.gpt-oss-20b-1:0" or +model="bedrock/chat_completions/global.openai.gpt-5.6-sol". A request that needs a +Converse-only feature (``bedrock_request_needs_converse`` in ``common_utils``) is +still served by Converse. """ from collections.abc import AsyncIterator, Iterator, Mapping diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 91b9509eb95..b349f996192 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -801,7 +801,7 @@ def is_bedrock_application_inference_profile_arn(model: str) -> bool: def strip_bedrock_routing_prefix(model: str) -> str: """Strip LiteLLM routing prefixes from model name.""" - for prefix in ["bedrock/", "converse/", "invoke/", "openai/", "mantle/", "nova-2/", "nova/"]: + for prefix in ["bedrock/", "chat_completions/", "converse/", "invoke/", "openai/", "mantle/", "nova-2/", "nova/"]: if model.startswith(prefix): model = model.split("/", 1)[1] return model @@ -820,9 +820,6 @@ def split_bedrock_region_path(model: str) -> tuple[str | None, str]: return None, stripped -BEDROCK_RUNTIME_PRICE_MAP_PROVIDERS: Final = frozenset(("bedrock", "bedrock_converse")) - - def _bedrock_price_map_entries(model: str) -> tuple[Mapping[str, object] | None, ...]: return tuple( litellm.model_cost.get(key) @@ -834,32 +831,6 @@ def _bedrock_price_map_flag(model: str, flag: str) -> bool: return any(entry is not None and entry.get(flag) is True for entry in _bedrock_price_map_entries(model)) -def _bedrock_runtime_row_lists_chat_completions(entry: Mapping[str, object]) -> bool: - endpoints: Final = entry.get("supported_endpoints") - return ( - entry.get("litellm_provider") in BEDROCK_RUNTIME_PRICE_MAP_PROVIDERS - and isinstance(endpoints, (list, tuple)) - and "/v1/chat/completions" in endpoints - ) - - -def uses_bedrock_runtime_chat_completions(model: str) -> bool: - """Whether this Bedrock model should use runtime native Chat Completions. - - Data-driven from ``/v1/chat/completions`` in the price-map row's ``supported_endpoints``, - the same per-model signal ``bedrock_supports_openai_responses`` reads for ``/v1/responses``, - so onboarding a model is a JSON change. Explicit ``converse/`` still wins in - ``get_bedrock_route`` because prefix routes are checked first, and a request - that needs a Converse-only feature (``bedrock_request_needs_converse``) is - served by Converse even on a listed model. Only a bedrock-runtime row counts: a - ``bedrock_mantle`` row lists the endpoints of the Mantle host, not this one. - """ - return any( - entry is not None and _bedrock_runtime_row_lists_chat_completions(entry) - for entry in _bedrock_price_map_entries(model) - ) - - def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool: """Whether AWS's native Chat Completions serves this model's function tools with any ``reasoning_effort``. @@ -908,7 +879,7 @@ def _response_format_needs_converse(model: str, response_format: object) -> bool def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: - """Whether a request on a runtime-Chat-Completions model must still be served by Converse. + """Whether a request on the opt-in ``chat_completions/`` route must still be served by Converse. Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking`` block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse @@ -1314,8 +1285,9 @@ class BedrockModelInfo(BaseLLMModelInfo): """ Get the bedrock route for the given model. - ``request_params`` (the caller's chat params) lets a runtime Chat Completions - model fall back to Converse for the requests only Converse can serve. + ``chat_completions/`` opts a model into bedrock-runtime's native OpenAI Chat Completions; + ``request_params`` (the caller's chat params) sends such a request to Converse when it + needs a feature only Converse serves. Without the prefix, OpenAI-family models stay on Converse. """ route_mappings: dict[ str, @@ -1351,6 +1323,11 @@ class BedrockModelInfo(BaseLLMModelInfo): if BedrockModelInfo._model_has_route_prefix(model, prefix): return route_type + if BedrockModelInfo._model_has_route_prefix(model, "chat_completions/"): + if request_params is not None and bedrock_request_needs_converse(model, request_params): + return "converse" + return "chat_completions" + # Check for nova spec prefixes (nova/ and nova-2/) _model_after_bedrock: Final = model.replace("bedrock/", "", 1) if _model_after_bedrock.startswith("nova-2/") or _model_after_bedrock.startswith("nova/"): @@ -1359,11 +1336,6 @@ class BedrockModelInfo(BaseLLMModelInfo): if is_bedrock_application_inference_profile_arn(model): return "converse" - if uses_bedrock_runtime_chat_completions(model) and not ( - request_params is not None and bedrock_request_needs_converse(model, request_params) - ): - return "chat_completions" - base_model: Final = BedrockModelInfo.get_base_model(model) alt_model: Final = BedrockModelInfo.get_non_litellm_routing_model_name(model=model) if base_model in litellm.bedrock_converse_models or alt_model in litellm.bedrock_converse_models: diff --git a/litellm/main.py b/litellm/main.py index 5b7c06550e8..b287a81605c 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -4234,7 +4234,7 @@ def _complete_bedrock(ctx: _CompletionDispatchContext) -> _CompletionDispatchRes provider_config=provider_config, ) elif bedrock_route == "converse": - model = model.replace("converse/", "") + model = model.replace("converse/", "").replace("chat_completions/", "") response = bedrock_converse_chat_completion.completion( model=model, messages=messages, diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index f9f330a7340..26392921c47 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -1,4 +1,4 @@ -"""Native Bedrock Runtime Chat Completions: Grok, gpt-oss and GPT-5.6 stay on /openai/v1/chat/completions.""" +"""Opt-in Bedrock Runtime Chat Completions: ``bedrock/chat_completions/`` posts to /openai/v1/chat/completions.""" import json @@ -21,7 +21,6 @@ from litellm.llms.bedrock.common_utils import ( bedrock_request_needs_converse, bedrock_route_for_request, get_bedrock_chat_config, - uses_bedrock_runtime_chat_completions, ) from litellm.llms.custom_httpx.http_handler import HTTPHandler @@ -38,14 +37,13 @@ def local_cost_map(monkeypatch): @pytest.mark.parametrize( "model", [ - "us.xai.grok-4.6", - "global.xai.grok-4.6", - "us-gov.xai.grok-4.6", - "bedrock/us.xai.grok-4.6", + "chat_completions/us.xai.grok-4.6", + "chat_completions/global.xai.grok-4.6", + "chat_completions/us-gov.xai.grok-4.6", + "bedrock/chat_completions/us.xai.grok-4.6", ], ) -def test_grok_runtime_models_use_chat_completions_route(local_cost_map, model): - assert uses_bedrock_runtime_chat_completions(model) is True +def test_chat_completions_prefix_opts_grok_into_the_native_route(local_cost_map, model): assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions" assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig) @@ -56,31 +54,44 @@ def test_explicit_converse_prefix_still_uses_converse(local_cost_map): def test_claude_stays_on_converse(local_cost_map): - assert uses_bedrock_runtime_chat_completions("us.anthropic.claude-3-sonnet-20240229-v1:0") is False assert BedrockModelInfo.get_bedrock_route("us.anthropic.claude-3-sonnet-20240229-v1:0") == "converse" @pytest.mark.parametrize( - "entry", + "model", [ - {"litellm_provider": "bedrock_converse"}, - {"litellm_provider": "bedrock_converse", "supported_endpoints": ["/v1/responses"]}, - {"litellm_provider": "bedrock_converse", "supports_bedrock_runtime_chat_completions": True}, - {"litellm_provider": "bedrock_mantle", "supported_endpoints": ["/v1/chat/completions", "/v1/responses"]}, - {"litellm_provider": "openai", "supported_endpoints": ["/v1/chat/completions"]}, + "us.xai.grok-4.6", + "bedrock/openai.gpt-oss-20b-1:0", + "openai.gpt-oss-120b-1:0", + "global.openai.gpt-5.6-sol", + "bedrock/us.openai.gpt-5.6-terra", + "bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", ], ) -def test_chat_completions_missing_from_supported_endpoints_means_no_chat_completions_route(monkeypatch, entry): - monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": entry}) - assert uses_bedrock_runtime_chat_completions("us.xai.grok-4.6") is False - assert BedrockModelInfo.get_bedrock_route("us.xai.grok-4.6") == "converse" +def test_models_without_the_prefix_stay_on_converse(local_cost_map, model): + assert BedrockModelInfo.get_bedrock_route(model) == "converse" + assert BedrockModelInfo.get_bedrock_route(model, {}) == "converse" + assert isinstance(get_bedrock_chat_config(model), litellm.AmazonConverseConfig) -def test_chat_completions_in_supported_endpoints_opts_into_the_native_route(monkeypatch): - entry = {"litellm_provider": "bedrock_converse", "supported_endpoints": ["/v1/chat/completions", "/v1/responses"]} - monkeypatch.setattr(litellm, "model_cost", {"us.xai.grok-4.6": entry}) - assert uses_bedrock_runtime_chat_completions("us.xai.grok-4.6") is True - assert BedrockModelInfo.get_bedrock_route("bedrock/us.xai.grok-4.6") == "chat_completions" +def test_cost_map_row_listing_chat_completions_leaves_the_default_route_alone(monkeypatch): + entry = { + "litellm_provider": "bedrock_converse", + "supported_endpoints": ["/v1/chat/completions", "/v1/responses"], + "supports_bedrock_runtime_chat_completions_tools_with_reasoning": True, + "supports_bedrock_runtime_chat_completions_response_format": True, + } + monkeypatch.setattr(litellm, "model_cost", {"openai.gpt-oss-20b-1:0": entry}) + assert BedrockModelInfo.get_bedrock_route("bedrock/openai.gpt-oss-20b-1:0", {}) == "converse" + assert BedrockModelInfo.get_bedrock_route("bedrock/chat_completions/openai.gpt-oss-20b-1:0", {}) == "chat_completions" + + +@pytest.mark.parametrize("model", ["global.openai.gpt-5.6-sol", "openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"]) +def test_chat_completions_prefix_prices_like_the_bare_model(local_cost_map, model): + prefixed = litellm.get_model_info(model=f"bedrock/chat_completions/{model}") + bare = litellm.get_model_info(model=f"bedrock/{model}") + assert prefixed["input_cost_per_token"] == bare["input_cost_per_token"] > 0 + assert prefixed["output_cost_per_token"] == bare["output_cost_per_token"] > 0 def test_complete_url_is_runtime_openai_chat_completions(monkeypatch): @@ -113,7 +124,7 @@ def test_project_id_is_not_sent_as_openai_project_header(): cfg = AmazonBedrockRuntimeChatCompletionsConfig() headers = cfg.validate_environment( headers={}, - model="bedrock/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], optional_params={}, litellm_params={"aws_bedrock_project_id": "proj_from_config"}, @@ -125,7 +136,7 @@ def test_project_id_is_not_sent_as_openai_project_header(): def test_transform_request_is_openai_chat_body_not_converse(): cfg = AmazonBedrockRuntimeChatCompletionsConfig() body = cfg.transform_request( - model="bedrock/us.xai.grok-4.6", + model="bedrock/chat_completions/us.xai.grok-4.6", messages=[{"role": "user", "content": "hello"}], optional_params={"temperature": 0.2, "aws_region_name": "us-east-1"}, litellm_params={}, @@ -178,10 +189,26 @@ def _recording_client(**response_kwargs): return requests, HTTPHandler(client=httpx.Client(transport=httpx.MockTransport(handle))) +@pytest.mark.parametrize( + "model, model_path", + [ + ("bedrock/us.xai.grok-4.6", b"/model/us.xai.grok-4.6/converse"), + ("bedrock/openai.gpt-oss-20b-1:0", b"/model/openai.gpt-oss-20b-1%3A0/converse"), + ("bedrock/global.openai.gpt-5.6-sol", b"/model/global.openai.gpt-5.6-sol/converse"), + ], +) +def test_completion_without_the_prefix_posts_converse(local_cost_map, fake_aws_env, model, model_path): + requests, client = _recording_client(json=CONVERSE_JSON) + response = litellm.completion(model=model, messages=[{"role": "user", "content": "hello"}], client=client) + + assert response.choices[0].message.content == "ok" + assert [request.url.raw_path for request in requests] == [model_path] + + def test_completion_posts_runtime_chat_completions(local_cost_map, fake_aws_env): requests, client = _recording_client(json=_chat_completion_json("ok", "us.xai.grok-4.6")) response = litellm.completion( - model="us.xai.grok-4.6", + model="bedrock/chat_completions/us.xai.grok-4.6", messages=[{"role": "user", "content": "hello"}], client=client, ) @@ -198,7 +225,7 @@ def test_completion_posts_runtime_chat_completions(local_cost_map, fake_aws_env) def test_region_path_sends_the_bare_model_id_to_the_path_region(local_cost_map, fake_aws_env): requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) litellm.completion( - model="bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/us-gov-west-1/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], client=client, ) @@ -211,7 +238,7 @@ def test_region_path_sends_the_bare_model_id_to_the_path_region(local_cost_map, def test_explicit_aws_region_name_wins_over_the_region_path(local_cost_map, fake_aws_env): requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) litellm.completion( - model="bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/us-gov-west-1/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], aws_region_name="us-gov-east-1", client=client, @@ -222,6 +249,21 @@ def test_explicit_aws_region_name_wins_over_the_region_path(local_cost_map, fake assert "/us-gov-east-1/bedrock/aws4_request" in requests[0].headers["Authorization"] +def test_region_path_falls_back_to_converse_in_the_path_region(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/chat_completions/us-gov-west-1/openai.gpt-oss-20b-1:0", + messages=[{"role": "user", "content": "hello"}], + stop=["END"], + client=client, + ) + + assert requests[0].url.host == "bedrock-runtime.us-gov-west-1.amazonaws.com" + assert requests[0].url.raw_path == b"/model/openai.gpt-oss-20b-1%3A0/converse" + assert json.loads(requests[0].content)["inferenceConfig"]["stopSequences"] == ["END"] + assert "/us-gov-west-1/bedrock/aws4_request" in requests[0].headers["Authorization"] + + OPENAI_RUNTIME_MODELS = ( "openai.gpt-oss-20b-1:0", "openai.gpt-oss-120b-1:0", @@ -244,26 +286,26 @@ GET_WEATHER_TOOL = { @pytest.mark.parametrize( "model", [ - *OPENAI_RUNTIME_MODELS, - "bedrock/openai.gpt-oss-20b-1:0", - "us-gov.openai.gpt-oss-20b-1:0", - "bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", - "us-gov-east-1/openai.gpt-oss-120b-1:0", + *(f"chat_completions/{model}" for model in OPENAI_RUNTIME_MODELS), + "bedrock/chat_completions/openai.gpt-oss-20b-1:0", + "chat_completions/us-gov.openai.gpt-oss-20b-1:0", + "bedrock/chat_completions/us-gov-west-1/openai.gpt-oss-20b-1:0", + "chat_completions/us-gov-east-1/openai.gpt-oss-120b-1:0", ], ) def test_openai_runtime_models_use_chat_completions_route(local_cost_map, model): - assert uses_bedrock_runtime_chat_completions(model) is True assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions" assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig) @pytest.mark.parametrize("model", ["us.amazon.nova-micro-v1:0", "us.anthropic.claude-haiku-4-5-20251001-v1:0"]) def test_nova_and_claude_stay_on_converse(local_cost_map, model): - assert uses_bedrock_runtime_chat_completions(model) is False assert BedrockModelInfo.get_bedrock_route(model, {"tools": [GET_WEATHER_TOOL]}) == "converse" -@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "global.openai.gpt-5.6-sol"]) +@pytest.mark.parametrize( + "model", ["chat_completions/openai.gpt-oss-20b-1:0", "bedrock/chat_completions/global.openai.gpt-5.6-sol"] +) def test_guardrail_config_falls_back_to_converse(local_cost_map, model): guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"} assert bedrock_request_needs_converse(model, {"guardrailConfig": guardrail}) is True @@ -272,7 +314,12 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model): @pytest.mark.parametrize( - "model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6", "global.openai.gpt-5.6-sol"] + "model", + [ + "chat_completions/openai.gpt-oss-20b-1:0", + "chat_completions/us.xai.grok-4.6", + "bedrock/chat_completions/global.openai.gpt-5.6-sol", + ], ) @pytest.mark.parametrize( "request_params", @@ -299,15 +346,18 @@ def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, ], ) def test_gpt56_tools_need_reasoning_none_on_chat_completions(local_cost_map, request_params, expected_route): - assert BedrockModelInfo.get_bedrock_route("global.openai.gpt-5.6-sol", request_params) == expected_route - assert BedrockModelInfo.get_bedrock_route("bedrock/us.openai.gpt-5.6-terra", request_params) == expected_route + assert BedrockModelInfo.get_bedrock_route("chat_completions/global.openai.gpt-5.6-sol", request_params) == expected_route + assert ( + BedrockModelInfo.get_bedrock_route("bedrock/chat_completions/us.openai.gpt-5.6-terra", request_params) + == expected_route + ) @pytest.mark.parametrize("reasoning_effort", ["low", "high", None]) def test_gpt_oss_tools_with_any_reasoning_effort_stay_on_chat_completions(local_cost_map, reasoning_effort): params = {"tools": [GET_WEATHER_TOOL], "reasoning_effort": reasoning_effort} assert bedrock_request_needs_converse("openai.gpt-oss-120b-1:0", params) is False - assert BedrockModelInfo.get_bedrock_route("openai.gpt-oss-120b-1:0", params) == "chat_completions" + assert BedrockModelInfo.get_bedrock_route("chat_completions/openai.gpt-oss-120b-1:0", params) == "chat_completions" @pytest.mark.parametrize( @@ -320,14 +370,14 @@ def test_gpt_oss_tools_with_any_reasoning_effort_stay_on_chat_completions(local_ ], ) def test_gpt56_legacy_functions_route_like_tools(local_cost_map, request_params, expected_route): - assert BedrockModelInfo.get_bedrock_route("global.openai.gpt-5.6-sol", request_params) == expected_route - assert BedrockModelInfo.get_bedrock_route("openai.gpt-oss-120b-1:0", request_params) == "chat_completions" + assert BedrockModelInfo.get_bedrock_route("chat_completions/global.openai.gpt-5.6-sol", request_params) == expected_route + assert BedrockModelInfo.get_bedrock_route("chat_completions/openai.gpt-oss-120b-1:0", request_params) == "chat_completions" def test_thinking_block_goes_to_converse(local_cost_map): thinking = {"type": "enabled", "budget_tokens": 1024} - assert BedrockModelInfo.get_bedrock_route("us.xai.grok-4.6", {"thinking": thinking}) == "converse" - assert BedrockModelInfo.get_bedrock_route("us.xai.grok-4.6", {"thinking": None}) == "chat_completions" + assert BedrockModelInfo.get_bedrock_route("chat_completions/us.xai.grok-4.6", {"thinking": thinking}) == "converse" + assert BedrockModelInfo.get_bedrock_route("chat_completions/us.xai.grok-4.6", {"thinking": None}) == "chat_completions" def test_explicit_converse_prefix_wins_for_openai_models(local_cost_map): @@ -500,15 +550,15 @@ def test_supported_params_leave_out_what_each_family_refuses(local_cost_map, mod @pytest.mark.parametrize( "model, param", [ - ("bedrock/global.openai.gpt-5.6-sol", {"frequency_penalty": 0.5}), - ("bedrock/global.openai.gpt-5.6-sol", {"logprobs": True, "top_logprobs": 2}), - ("bedrock/us.xai.grok-4.6", {"presence_penalty": 0.5}), - ("bedrock/openai.gpt-oss-20b-1:0", {"logit_bias": {"1": 1}}), + ("bedrock/chat_completions/global.openai.gpt-5.6-sol", {"frequency_penalty": 0.5}), + ("bedrock/chat_completions/global.openai.gpt-5.6-sol", {"logprobs": True, "top_logprobs": 2}), + ("bedrock/chat_completions/us.xai.grok-4.6", {"presence_penalty": 0.5}), + ("bedrock/chat_completions/openai.gpt-oss-20b-1:0", {"logit_bias": {"1": 1}}), ], ids=lambda value: value if isinstance(value, str) else next(iter(value)), ) def test_refused_params_are_dropped_or_refused_before_reaching_aws(local_cost_map, fake_aws_env, model, param): - requests, client = _recording_client(json=_chat_completion_json("ok", model.removeprefix("bedrock/"))) + requests, client = _recording_client(json=_chat_completion_json("ok", model.removeprefix("bedrock/chat_completions/"))) with pytest.raises(litellm.UnsupportedParamsError, match=next(iter(param))): litellm.completion(model=model, messages=[{"role": "user", "content": "hello"}], client=client, **param) litellm.completion( @@ -651,7 +701,7 @@ def test_gpt_oss_completion_hits_chat_completions_and_splits_reasoning(local_cos json=_chat_completion_json("plan\n\nHi", "openai.gpt-oss-20b-1:0") ) response = litellm.completion( - model="bedrock/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], max_tokens=64, reasoning_effort="low", @@ -673,7 +723,7 @@ def test_gpt_oss_completion_hits_chat_completions_and_splits_reasoning(local_cos def test_gpt56_tools_with_reasoning_effort_go_to_converse(local_cost_map, fake_aws_env): requests, client = _recording_client(json=CONVERSE_JSON) response = litellm.completion( - model="bedrock/global.openai.gpt-5.6-sol", + model="bedrock/chat_completions/global.openai.gpt-5.6-sol", messages=[{"role": "user", "content": "hello"}], tools=[GET_WEATHER_TOOL], reasoning_effort="low", @@ -691,7 +741,7 @@ def test_gpt56_tools_with_reasoning_none_stay_on_chat_completions(local_cost_map ] requests, client = _recording_client(json=_chat_completion_json(None, "global.openai.gpt-5.6-sol", tool_calls)) response = litellm.completion( - model="bedrock/global.openai.gpt-5.6-sol", + model="bedrock/chat_completions/global.openai.gpt-5.6-sol", messages=[{"role": "user", "content": "weather in Paris"}], tools=[GET_WEATHER_TOOL], reasoning_effort="none", @@ -720,7 +770,7 @@ def test_gpt56_tools_with_reasoning_none_stay_on_chat_completions(local_cost_map def test_converse_only_request_keys_go_to_converse(local_cost_map, fake_aws_env, converse_only_param): requests, client = _recording_client(json=CONVERSE_JSON) litellm.completion( - model="bedrock/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], client=client, **converse_only_param, @@ -739,7 +789,7 @@ def test_operator_owned_request_metadata_goes_to_converse(local_cost_map, fake_a monkeypatch.setattr(litellm, "bedrock_request_metadata_fields", ["user_api_key_team_alias"]) requests, client = _recording_client(json=CONVERSE_JSON) litellm.completion( - model="bedrock/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], metadata={"user_api_key_team_alias": "search"}, client=client, @@ -752,7 +802,7 @@ def test_operator_owned_request_metadata_goes_to_converse(local_cost_map, fake_a def test_dropped_converse_only_key_keeps_the_request_on_chat_completions(local_cost_map, fake_aws_env): requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) litellm.completion( - model="bedrock/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], guardrailConfig={"guardrailIdentifier": "gr-1", "guardrailVersion": "1"}, additional_drop_params=["guardrailConfig"], @@ -770,7 +820,7 @@ def test_dropped_converse_only_key_keeps_the_request_on_chat_completions(local_c def test_dropped_tools_keep_gpt56_reasoning_request_on_chat_completions(local_cost_map, fake_aws_env): requests, client = _recording_client(json=_chat_completion_json("ok", "global.openai.gpt-5.6-sol")) litellm.completion( - model="bedrock/global.openai.gpt-5.6-sol", + model="bedrock/chat_completions/global.openai.gpt-5.6-sol", messages=[{"role": "user", "content": "hello"}], tools=[GET_WEATHER_TOOL], reasoning_effort="low", @@ -787,7 +837,7 @@ def test_dropped_tools_keep_gpt56_reasoning_request_on_chat_completions(local_co def test_legacy_functions_stay_on_chat_completions(local_cost_map, fake_aws_env): requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) litellm.completion( - model="bedrock/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], functions=[GET_WEATHER_TOOL["function"]], client=client, @@ -801,14 +851,14 @@ def test_gpt56_legacy_functions_with_reasoning_fall_back_to_converse(local_cost_ requests, client = _recording_client(json=CONVERSE_JSON) with pytest.raises(litellm.UnsupportedParamsError, match="functions"): litellm.completion( - model="bedrock/global.openai.gpt-5.6-sol", + model="bedrock/chat_completions/global.openai.gpt-5.6-sol", messages=[{"role": "user", "content": "hello"}], functions=[GET_WEATHER_TOOL["function"]], reasoning_effort="low", client=client, ) litellm.completion( - model="bedrock/global.openai.gpt-5.6-sol", + model="bedrock/chat_completions/global.openai.gpt-5.6-sol", messages=[{"role": "user", "content": "hello"}], functions=[GET_WEATHER_TOOL["function"]], reasoning_effort="low", @@ -826,7 +876,7 @@ def test_grok_thinking_block_is_served_by_converse(local_cost_map, fake_aws_env) requests, client = _recording_client(json=CONVERSE_JSON) thinking = {"type": "enabled", "budget_tokens": 1024} litellm.completion( - model="bedrock/us.xai.grok-4.6", + model="bedrock/chat_completions/us.xai.grok-4.6", messages=[{"role": "user", "content": "hello"}], thinking=thinking, client=client, @@ -841,14 +891,14 @@ def test_converse_fallback_validates_against_converse_params(local_cost_map, fak guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"} with pytest.raises(litellm.UnsupportedParamsError, match="seed"): litellm.completion( - model="bedrock/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], guardrailConfig=guardrail, seed=7, client=client, ) litellm.completion( - model="bedrock/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], guardrailConfig=guardrail, seed=7, @@ -864,13 +914,13 @@ def test_n_is_rejected_before_reaching_chat_completions(local_cost_map, fake_aws requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) with pytest.raises(litellm.UnsupportedParamsError, match="'n'"): litellm.completion( - model="bedrock/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], n=2, client=client, ) litellm.completion( - model="bedrock/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], n=2, drop_params=True, @@ -892,7 +942,7 @@ def test_gpt_oss_streaming_completion_splits_reasoning(local_cost_map, fake_aws_ ) requests, client = _recording_client(content=_sse(chunks), headers={"content-type": "text/event-stream"}) stream = litellm.completion( - model="bedrock/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "hello"}], stream=True, client=client, @@ -931,7 +981,9 @@ class Answer(BaseModel): word: str -@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "bedrock/openai.gpt-oss-120b-1:0"]) +@pytest.mark.parametrize( + "model", ["chat_completions/openai.gpt-oss-20b-1:0", "bedrock/chat_completions/openai.gpt-oss-120b-1:0"] +) @pytest.mark.parametrize( "response_format, expected_route", [ @@ -949,7 +1001,11 @@ def test_gpt_oss_response_format_falls_back_to_converse(local_cost_map, model, r assert BedrockModelInfo.get_bedrock_route(model, params) == expected_route -RESPONSE_FORMAT_ENFORCING_MODELS = ["global.openai.gpt-5.6-sol", "us.xai.grok-4.6", "bedrock/us-gov.xai.grok-4.6"] +RESPONSE_FORMAT_ENFORCING_MODELS = [ + "chat_completions/global.openai.gpt-5.6-sol", + "chat_completions/us.xai.grok-4.6", + "bedrock/chat_completions/us-gov.xai.grok-4.6", +] JSON_OBJECT_WITH_RESPONSE_SCHEMA = { @@ -980,7 +1036,7 @@ def test_json_object_keeps_converse_where_aws_would_demand_the_word_json(local_c assert BedrockModelInfo.get_bedrock_route(model, params) == "converse" -SYNTHETIC_NATIVE_MODEL = "vendor.native-model-v1:0" +SYNTHETIC_NATIVE_MODEL = "chat_completions/vendor.native-model-v1:0" @pytest.mark.parametrize( @@ -1008,12 +1064,8 @@ SYNTHETIC_NATIVE_MODEL = "vendor.native-model-v1:0" ], ) def test_capability_flags_are_read_from_the_cost_map(monkeypatch, capability_flags, request_params, needs_converse): - entry = { - "litellm_provider": "bedrock_converse", - "supported_endpoints": ["/v1/chat/completions"], - **capability_flags, - } - monkeypatch.setattr(litellm, "model_cost", {SYNTHETIC_NATIVE_MODEL: entry}) + entry = {"litellm_provider": "bedrock_converse", **capability_flags} + monkeypatch.setattr(litellm, "model_cost", {"vendor.native-model-v1:0": entry}) assert bedrock_request_needs_converse(SYNTHETIC_NATIVE_MODEL, request_params) is needs_converse route = bedrock_route_for_request(SYNTHETIC_NATIVE_MODEL, request_params, None) assert (route == "chat_completions") is (not needs_converse) @@ -1021,18 +1073,16 @@ def test_capability_flags_are_read_from_the_cost_map(monkeypatch, capability_fla def test_route_for_request_ignores_dropped_params(local_cost_map): params = {"response_format": RESPONSE_FORMAT_JSON_SCHEMA, "guardrailConfig": {"guardrailIdentifier": "gr-1"}} - assert bedrock_route_for_request("openai.gpt-oss-20b-1:0", params, None) == "converse" - assert bedrock_route_for_request("openai.gpt-oss-20b-1:0", params, ["guardrailConfig"]) == "converse" - assert ( - bedrock_route_for_request("openai.gpt-oss-20b-1:0", params, ["guardrailConfig", "response_format"]) - == "chat_completions" - ) + model = "chat_completions/openai.gpt-oss-20b-1:0" + assert bedrock_route_for_request(model, params, None) == "converse" + assert bedrock_route_for_request(model, params, ["guardrailConfig"]) == "converse" + assert bedrock_route_for_request(model, params, ["guardrailConfig", "response_format"]) == "chat_completions" def test_gpt_oss_response_format_goes_to_converse_with_json_tool_call(local_cost_map, fake_aws_env): requests, client = _recording_client(json=CONVERSE_JSON) litellm.completion( - model="bedrock/openai.gpt-oss-20b-1:0", + model="bedrock/chat_completions/openai.gpt-oss-20b-1:0", messages=[{"role": "user", "content": "Reply with the single word pong."}], response_format=RESPONSE_FORMAT_JSON_SCHEMA, max_tokens=64, @@ -1051,7 +1101,7 @@ def test_gpt_oss_response_format_goes_to_converse_with_json_tool_call(local_cost def test_gpt56_response_format_is_sent_as_is_on_chat_completions(local_cost_map, fake_aws_env): requests, client = _recording_client(json=_chat_completion_json('{"word": "pong"}', "global.openai.gpt-5.6-sol")) response = litellm.completion( - model="bedrock/global.openai.gpt-5.6-sol", + model="bedrock/chat_completions/global.openai.gpt-5.6-sol", messages=[{"role": "user", "content": "Reply with the single word pong."}], response_format=RESPONSE_FORMAT_JSON_SCHEMA, client=client, @@ -1065,7 +1115,7 @@ def test_gpt56_response_format_is_sent_as_is_on_chat_completions(local_cost_map, def test_gpt56_schema_less_json_object_goes_to_converse_without_a_schema_tool(local_cost_map, fake_aws_env): requests, client = _recording_client(json=CONVERSE_JSON) litellm.completion( - model="bedrock/global.openai.gpt-5.6-sol", + model="bedrock/chat_completions/global.openai.gpt-5.6-sol", messages=[{"role": "user", "content": "Reply with the single word pong."}], response_format={"type": "json_object"}, max_tokens=64, @@ -1082,7 +1132,7 @@ def test_gpt56_schema_less_json_object_goes_to_converse_without_a_schema_tool(lo def test_gpt56_json_object_with_response_schema_goes_to_converse_as_a_json_tool(local_cost_map, fake_aws_env): requests, client = _recording_client(json=CONVERSE_JSON) litellm.completion( - model="bedrock/global.openai.gpt-5.6-sol", + model="bedrock/chat_completions/global.openai.gpt-5.6-sol", messages=[{"role": "user", "content": "Reply with the single word pong."}], response_format=JSON_OBJECT_WITH_RESPONSE_SCHEMA, max_tokens=64, diff --git a/tests/unit/llms/bedrock/test_cross_region_inference_profile_mapping.py b/tests/unit/llms/bedrock/test_cross_region_inference_profile_mapping.py index a6b8ba1da1d..aa0827c5ae5 100644 --- a/tests/unit/llms/bedrock/test_cross_region_inference_profile_mapping.py +++ b/tests/unit/llms/bedrock/test_cross_region_inference_profile_mapping.py @@ -138,12 +138,9 @@ def _bedrock_response(model, usage): @pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id) -def test_bedrock_gpt_5_6_profiles_never_route_to_invoke(profile, local_model_cost_map): - """GPT-5.6 is served by bedrock-runtime's native Chat Completions, and by Converse when - the request carries function tools without reasoning_effort "none", never by Invoke.""" - assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "chat_completions" - tools_with_reasoning = {"tools": [{"type": "function", "function": {"name": "f"}}], "reasoning_effort": "low"} - assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}", tools_with_reasoning) == "converse" +def test_bedrock_gpt_5_6_profiles_route_to_converse(profile, local_model_cost_map): + """GPT-5.6 is served by Converse on bedrock-runtime, never by Invoke.""" + assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "converse" @pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id) From 84ae4596b5bf935b298adbddf21d7b60b9aa0765 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 28 Sep 2026 21:46:35 -0700 Subject: [PATCH 20/31] fix(bedrock): keep chat_completions/-prefixed deployments on the native Responses surface --- .../llms/bedrock/responses/transformation.py | 15 ++++++++++--- .../test_bedrock_openai_responses.py | 21 +++++++++++++++++++ 2 files changed, 33 insertions(+), 3 deletions(-) diff --git a/litellm/llms/bedrock/responses/transformation.py b/litellm/llms/bedrock/responses/transformation.py index e2221b64f62..e6e8f853ef8 100644 --- a/litellm/llms/bedrock/responses/transformation.py +++ b/litellm/llms/bedrock/responses/transformation.py @@ -71,11 +71,16 @@ BEDROCK_RUNTIME_SUPPORTED_RESPONSE_TOOL_TYPES: Final = frozenset( {"function", "mcp", "custom", "apply_patch", "namespace", "tool_search", "computer"} ) BEDROCK_RUNTIME_UNSUPPORTED_RESPONSE_PARAMS: Final = frozenset({"background"}) +BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX: Final = "chat_completions/" REMOTE_IMAGE_URL_SCHEMES: Final = ("http://", "https://") IMAGE_BLOCK_KEYS: Final = ("content", "output") IMAGE_BLOCK_TYPES: Final = frozenset({"input_image", "computer_screenshot"}) +def _without_chat_completions_route(model: str) -> str: + return model.removeprefix(BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX) + + def resolve_bedrock_bearer_token(api_key: str | None) -> str | None: return api_key or get_secret_str("AWS_BEARER_TOKEN_BEDROCK") @@ -168,9 +173,13 @@ class BedrockOpenAIResponsesConfig(BaseAWSLLM, OpenAIResponsesAPIConfig): The capability decision lives here rather than in the shared dispatch so that onboarding a model, or changing how the signal is read, stays inside the Bedrock adapter. ``None`` leaves the caller's existing behaviour untouched -- - chat-only Bedrock models keep the Chat Completions bridge. + chat-only Bedrock models keep the Chat Completions bridge. The ``chat_completions/`` + opt-in only moves Chat Completions calls off Converse, so a Responses call on such a + deployment still takes this surface instead of being bridged. """ - if not bedrock_supports_openai_responses(model, litellm.model_cost): + if not model or not bedrock_supports_openai_responses( + _without_chat_completions_route(model), litellm.model_cost + ): return None return cls() @@ -330,7 +339,7 @@ class BedrockOpenAIResponsesConfig(BaseAWSLLM, OpenAIResponsesAPIConfig): rewritten_types, ) return super().transform_responses_api_request( - model=model, + model=_without_chat_completions_route(model), input=normalized_input, response_api_optional_request_params=response_api_optional_request_params, litellm_params=litellm_params, diff --git a/tests/unit/llms/bedrock/responses/test_bedrock_openai_responses.py b/tests/unit/llms/bedrock/responses/test_bedrock_openai_responses.py index de09879a96a..c0803f6636b 100644 --- a/tests/unit/llms/bedrock/responses/test_bedrock_openai_responses.py +++ b/tests/unit/llms/bedrock/responses/test_bedrock_openai_responses.py @@ -162,6 +162,27 @@ class TestForModelGate: ): assert BedrockOpenAIResponsesConfig.for_model(None) is None + def test_chat_completions_route_keeps_the_native_responses_surface(self): + with patch.object( # test-quality-ok: the gate reads the global cost map by design; no injection point exists + litellm, "model_cost", {MODEL: {"supported_endpoints": ["/v1/responses"]}} + ): + cfg = BedrockOpenAIResponsesConfig.for_model(f"chat_completions/{MODEL}") + assert isinstance(cfg, BedrockOpenAIResponsesConfig) + body = cfg.transform_responses_api_request( + model=f"chat_completions/{MODEL}", + input="hi", + response_api_optional_request_params={}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + assert body["model"] == MODEL + + def test_converse_route_keeps_the_chat_completions_bridge(self): + with patch.object( # test-quality-ok: the gate reads the global cost map by design; no injection point exists + litellm, "model_cost", {MODEL: {"supported_endpoints": ["/v1/responses"]}} + ): + assert BedrockOpenAIResponsesConfig.for_model(f"converse/{MODEL}") is None + class TestProviderResolution: """model_cost is patched explicitly: it is populated at import time from a GitHub From 77204f84bcae164d7c2f63b9db1b7b85300a8a54 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Mon, 28 Sep 2026 22:21:54 -0700 Subject: [PATCH 21/31] fix(bedrock): keep provider response headers on the runtime chat completions route --- .../bedrock/chat/chat_completions/transformation.py | 2 ++ .../test_bedrock_chat_completions_transformation.py | 12 ++++++++++++ 2 files changed, 14 insertions(+) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 2025927807b..c424ca2a069 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -22,6 +22,7 @@ import httpx from typing_extensions import assert_never import litellm +from litellm.litellm_core_utils.core_helpers import set_provider_response_headers_in_hidden_params from litellm.litellm_core_utils.prompt_templates.image_handling import ( async_inline_remote_media, inline_remote_image_urls, @@ -383,6 +384,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): api_key=api_key, json_mode=json_mode, ) + set_provider_response_headers_in_hidden_params(response, raw_response.headers) for choice in response.choices: if not isinstance(choice, Choices) or not isinstance(choice.message.content, str): continue diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 26392921c47..e40ef13e25a 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -222,6 +222,18 @@ def test_completion_posts_runtime_chat_completions(local_cost_map, fake_aws_env) assert "inferenceConfig" not in body +def test_completion_keeps_the_aws_request_id_as_a_provider_header(local_cost_map, fake_aws_env): + _, client = _recording_client( + json=_chat_completion_json("ok", "us.xai.grok-4.6"), headers={"x-amzn-requestid": "req-native-1"} + ) + response = litellm.completion( + model="bedrock/chat_completions/us.xai.grok-4.6", + messages=[{"role": "user", "content": "hello"}], + client=client, + ) + + assert response._hidden_params["additional_headers"]["llm_provider-x-amzn-requestid"] == "req-native-1" + def test_region_path_sends_the_bare_model_id_to_the_path_region(local_cost_map, fake_aws_env): requests, client = _recording_client(json=_chat_completion_json("ok", "openai.gpt-oss-20b-1:0")) litellm.completion( From 3e3b9cdfc7096980cfe0cf13312c7121347cec69 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 30 Sep 2026 16:04:09 -0700 Subject: [PATCH 22/31] feat(bedrock): serve gpt-5.6 and newer on runtime chat completions by default Unprefixed bedrock/ models whose cost-map row lists /v1/chat/completions now route to the native OpenAI-compatible endpoint; converse/ pins Converse and chat_completions/ still opts gpt-oss and Grok in. Guardrails, application inference profile ARNs, and tools with reasoning keep falling back to Converse per request. Hoist the remote-media url comprehension into a single-clause helper. --- .../prompt_templates/image_handling.py | 29 ++-- .../chat/chat_completions/transformation.py | 12 +- litellm/llms/bedrock/common_utils.py | 59 ++++++++- ...odel_prices_and_context_window_backup.json | 16 +++ model_prices_and_context_window.json | 16 +++ ...bedrock_chat_completions_transformation.py | 125 +++++++++++++++++- ..._cross_region_inference_profile_mapping.py | 7 +- 7 files changed, 227 insertions(+), 37 deletions(-) diff --git a/litellm/litellm_core_utils/prompt_templates/image_handling.py b/litellm/litellm_core_utils/prompt_templates/image_handling.py index 57b4f545301..705a7b93381 100644 --- a/litellm/litellm_core_utils/prompt_templates/image_handling.py +++ b/litellm/litellm_core_utils/prompt_templates/image_handling.py @@ -4,8 +4,9 @@ Helper functions to handle images passed in messages import asyncio import base64 -from collections.abc import Callable, Mapping +from collections.abc import Callable, Iterable, Mapping from dataclasses import dataclass +from itertools import chain from types import MappingProxyType from typing import Final @@ -304,18 +305,19 @@ async def _fetch_data_urls(remote_urls: tuple[str, ...]) -> tuple[str, ...]: raise +def _remote_urls_to_inline( + messages: Iterable[AllMessageValues], should_inline: Callable[[RemoteMedia], bool] +) -> tuple[str, ...]: + parts: Final = chain.from_iterable(_content_parts(message) for message in messages) + remotes: Final = (remote for part in parts if (remote := _parse_remote_part(part)) is not None) + return tuple(dict.fromkeys(remote.url for remote in remotes if should_inline(_remote_media(remote)))) + + def inline_remote_media( messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues] should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url, ) -> list[AllMessageValues]: # mutable-ok: every transform_request takes list[AllMessageValues] - remote_urls: Final = tuple( - dict.fromkeys( - remote.url - for message in messages - for part in _content_parts(message) - if (remote := _parse_remote_part(part)) is not None and should_inline(_remote_media(remote)) - ) - ) + remote_urls: Final = _remote_urls_to_inline(messages, should_inline) if not remote_urls: return messages data_urls: Final = MappingProxyType({url: convert_url_to_base64(url) for url in remote_urls}) @@ -328,14 +330,7 @@ async def async_inline_remote_media( messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues] should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url, ) -> list[AllMessageValues]: # mutable-ok: every transform_request takes list[AllMessageValues] - remote_urls: Final = tuple( - dict.fromkeys( - remote.url - for message in messages - for part in _content_parts(message) - if (remote := _parse_remote_part(part)) is not None and should_inline(_remote_media(remote)) - ) - ) + remote_urls: Final = _remote_urls_to_inline(messages, should_inline) if not remote_urls: return messages data_urls: Final = await _fetch_data_urls(remote_urls) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index c424ca2a069..91a643fcd7b 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -3,12 +3,14 @@ Native OpenAI Chat Completions on Amazon Bedrock Runtime. AWS serves this surface at ``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions`` -for Grok 4.6, gpt-oss and the GPT-5.6 family. The ``chat_completions/`` route prefix -opts a model in, so chat completions stay chat completions instead of being rewritten -to Converse; without it these models stay on Converse. +for Grok 4.6, gpt-oss and GPT 5.6 and newer. GPT 5.6 and newer take it by default +(``bedrock_runtime_chat_completions_is_default`` in ``common_utils``), so their chat +completions stay chat completions instead of being rewritten to Converse; the +``chat_completions/`` route prefix opts any other model in, and ``converse/`` pins a +model to Converse. -Usage: model="bedrock/chat_completions/openai.gpt-oss-20b-1:0" or -model="bedrock/chat_completions/global.openai.gpt-5.6-sol". A request that needs a +Usage: model="bedrock/global.openai.gpt-6-sol" or +model="bedrock/chat_completions/openai.gpt-oss-20b-1:0". A request that needs a Converse-only feature (``bedrock_request_needs_converse`` in ``common_utils``) is still served by Converse. """ diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index b349f996192..14ab71bd63a 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -39,6 +39,9 @@ if TYPE_CHECKING: _ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs" _OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.") +_OPENAI_GPT_VERSION_RE: Final = re.compile(r"(^|[./])openai\.gpt-(\d+)(?:\.(\d+))?") +_BEDROCK_RUNTIME_CHAT_COMPLETIONS_DEFAULT_SINCE: Final = (5, 6) +_BEDROCK_RUNTIME_CHAT_COMPLETIONS_ENDPOINT: Final = "/v1/chat/completions" BedrockRoute = Literal[ "converse", "invoke", @@ -831,6 +834,34 @@ def _bedrock_price_map_flag(model: str, flag: str) -> bool: return any(entry is not None and entry.get(flag) is True for entry in _bedrock_price_map_entries(model)) +def _price_map_entry_lists_endpoint(entry: Mapping[str, object] | None, endpoint: str) -> bool: + endpoints: Final = None if entry is None else entry.get("supported_endpoints") + return isinstance(endpoints, (list, tuple)) and endpoint in endpoints + + +def _openai_gpt_version(model: str) -> tuple[int, int] | None: + match: Final = _OPENAI_GPT_VERSION_RE.search(model) + if match is None: + return None + return int(match.group(2)), int(match.group(3) or 0) + + +def bedrock_runtime_chat_completions_is_default(model: str) -> bool: + """Whether a model with no route prefix goes to bedrock-runtime's native Chat Completions by default. + + GPT 5.6 and newer (``openai.gpt-[.]`` at or above 5.6, which gpt-oss never matches) whose + price-map row lists ``/v1/chat/completions`` in ``supported_endpoints``. Older GPT rows, gpt-oss and Grok + stay on Converse unless the ``chat_completions/`` prefix opts them in. + """ + version: Final = _openai_gpt_version(model) + if version is None or version < _BEDROCK_RUNTIME_CHAT_COMPLETIONS_DEFAULT_SINCE: + return False + return any( + _price_map_entry_lists_endpoint(entry, _BEDROCK_RUNTIME_CHAT_COMPLETIONS_ENDPOINT) + for entry in _bedrock_price_map_entries(model) + ) + + def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool: """Whether AWS's native Chat Completions serves this model's function tools with any ``reasoning_effort``. @@ -879,7 +910,10 @@ def _response_format_needs_converse(model: str, response_format: object) -> bool def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: - """Whether a request on the opt-in ``chat_completions/`` route must still be served by Converse. + """Whether a request on the native Chat Completions route must still be served by Converse. + + The route is the default for GPT 5.6 and newer (``bedrock_runtime_chat_completions_is_default``) and the + ``chat_completions/`` prefix's opt-in for the rest; this decides the fallback for both alike. Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking`` block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse @@ -909,6 +943,14 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje ) +def _chat_completions_unless_converse_needed( + model: str, request_params: Mapping[str, object] | None +) -> Literal["converse", "chat_completions"]: + if request_params is not None and bedrock_request_needs_converse(model, request_params): + return "converse" + return "chat_completions" + + def bedrock_route_for_request( model: str, request_params: Mapping[str, object], additional_drop_params: Sequence[str] | None ) -> BedrockRoute: @@ -1285,9 +1327,11 @@ class BedrockModelInfo(BaseLLMModelInfo): """ Get the bedrock route for the given model. - ``chat_completions/`` opts a model into bedrock-runtime's native OpenAI Chat Completions; - ``request_params`` (the caller's chat params) sends such a request to Converse when it - needs a feature only Converse serves. Without the prefix, OpenAI-family models stay on Converse. + GPT 5.6 and newer go to bedrock-runtime's native OpenAI Chat Completions by default + (``bedrock_runtime_chat_completions_is_default``) and ``chat_completions/`` opts any other model in; + ``request_params`` (the caller's chat params) sends such a request to Converse when it needs a + feature only Converse serves, and ``converse/`` pins a model to Converse. Every other OpenAI-family + model stays on Converse without the prefix. """ route_mappings: dict[ str, @@ -1324,9 +1368,7 @@ class BedrockModelInfo(BaseLLMModelInfo): return route_type if BedrockModelInfo._model_has_route_prefix(model, "chat_completions/"): - if request_params is not None and bedrock_request_needs_converse(model, request_params): - return "converse" - return "chat_completions" + return _chat_completions_unless_converse_needed(model, request_params) # Check for nova spec prefixes (nova/ and nova-2/) _model_after_bedrock: Final = model.replace("bedrock/", "", 1) @@ -1336,6 +1378,9 @@ class BedrockModelInfo(BaseLLMModelInfo): if is_bedrock_application_inference_profile_arn(model): return "converse" + if bedrock_runtime_chat_completions_is_default(model): + return _chat_completions_unless_converse_needed(model, request_params) + base_model: Final = BedrockModelInfo.get_base_model(model) alt_model: Final = BedrockModelInfo.get_non_litellm_routing_model_name(model=model) if base_model in litellm.bedrock_converse_models or alt_model in litellm.bedrock_converse_models: diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d847b9712ef..7c70f07329e 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -58449,6 +58449,7 @@ "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html" }, "us.openai.gpt-6-astra": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1.1e-05, "input_cost_per_token_above_272k_tokens": 2.2e-05, "cache_creation_input_token_cost": 1.375e-05, @@ -58480,10 +58481,12 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-6-sol": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, "cache_creation_input_token_cost": 2.75e-06, @@ -58515,10 +58518,12 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-6-luna": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1.1e-07, "input_cost_per_token_above_272k_tokens": 2.2e-07, "cache_creation_input_token_cost": 1.375e-07, @@ -58550,10 +58555,12 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "global.openai.gpt-6-astra": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1e-05, "input_cost_per_token_above_272k_tokens": 2e-05, "cache_creation_input_token_cost": 1.25e-05, @@ -58585,6 +58592,7 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -58621,6 +58629,7 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-sol": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "cache_creation_input_token_cost": 2.5e-06, @@ -58652,6 +58661,7 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -58688,6 +58698,7 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-luna": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1e-07, "input_cost_per_token_above_272k_tokens": 2e-07, "cache_creation_input_token_cost": 1.25e-07, @@ -58719,6 +58730,7 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -79328,6 +79340,7 @@ "output_cost_per_token_above_272k_tokens": 1.5e-05, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-1-sol.html", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ], "supported_modalities": [ @@ -79337,6 +79350,7 @@ "supported_output_modalities": [ "text" ], + "supports_bedrock_runtime_chat_completions_response_format": true, "supports_function_calling": true, "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, @@ -79433,6 +79447,7 @@ "output_cost_per_token_above_272k_tokens": 1.65e-05, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-1-sol.html", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ], "supported_modalities": [ @@ -79442,6 +79457,7 @@ "supported_output_modalities": [ "text" ], + "supports_bedrock_runtime_chat_completions_response_format": true, "supports_function_calling": true, "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index d847b9712ef..7c70f07329e 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -58449,6 +58449,7 @@ "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html" }, "us.openai.gpt-6-astra": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1.1e-05, "input_cost_per_token_above_272k_tokens": 2.2e-05, "cache_creation_input_token_cost": 1.375e-05, @@ -58480,10 +58481,12 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-6-sol": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, "cache_creation_input_token_cost": 2.75e-06, @@ -58515,10 +58518,12 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-6-luna": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1.1e-07, "input_cost_per_token_above_272k_tokens": 2.2e-07, "cache_creation_input_token_cost": 1.375e-07, @@ -58550,10 +58555,12 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "global.openai.gpt-6-astra": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1e-05, "input_cost_per_token_above_272k_tokens": 2e-05, "cache_creation_input_token_cost": 1.25e-05, @@ -58585,6 +58592,7 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -58621,6 +58629,7 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-sol": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "cache_creation_input_token_cost": 2.5e-06, @@ -58652,6 +58661,7 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -58688,6 +58698,7 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-luna": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1e-07, "input_cost_per_token_above_272k_tokens": 2e-07, "cache_creation_input_token_cost": 1.25e-07, @@ -58719,6 +58730,7 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -79328,6 +79340,7 @@ "output_cost_per_token_above_272k_tokens": 1.5e-05, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-1-sol.html", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ], "supported_modalities": [ @@ -79337,6 +79350,7 @@ "supported_output_modalities": [ "text" ], + "supports_bedrock_runtime_chat_completions_response_format": true, "supports_function_calling": true, "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, @@ -79433,6 +79447,7 @@ "output_cost_per_token_above_272k_tokens": 1.65e-05, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-1-sol.html", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ], "supported_modalities": [ @@ -79442,6 +79457,7 @@ "supported_output_modalities": [ "text" ], + "supports_bedrock_runtime_chat_completions_response_format": true, "supports_function_calling": true, "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index e40ef13e25a..7539e8c8b43 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -1,4 +1,4 @@ -"""Opt-in Bedrock Runtime Chat Completions: ``bedrock/chat_completions/`` posts to /openai/v1/chat/completions.""" +"""Bedrock Runtime Chat Completions: the default for GPT 5.6 and newer, ``bedrock/chat_completions/`` for the rest.""" import json @@ -20,6 +20,7 @@ from litellm.llms.bedrock.common_utils import ( BedrockModelInfo, bedrock_request_needs_converse, bedrock_route_for_request, + bedrock_runtime_chat_completions_is_default, get_bedrock_chat_config, ) from litellm.llms.custom_httpx.http_handler import HTTPHandler @@ -63,9 +64,11 @@ def test_claude_stays_on_converse(local_cost_map): "us.xai.grok-4.6", "bedrock/openai.gpt-oss-20b-1:0", "openai.gpt-oss-120b-1:0", - "global.openai.gpt-5.6-sol", - "bedrock/us.openai.gpt-5.6-terra", + "global.openai.gpt-5.5", + "bedrock/us.openai.gpt-5.4", "bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", + "arn:aws:bedrock:us-east-1:123456789012:inference-profile/us.openai.gpt-6-astra", + "arn:aws:bedrock:us-west-2:123456789012:application-inference-profile/abc123xyz", ], ) def test_models_without_the_prefix_stay_on_converse(local_cost_map, model): @@ -86,6 +89,32 @@ def test_cost_map_row_listing_chat_completions_leaves_the_default_route_alone(mo assert BedrockModelInfo.get_bedrock_route("bedrock/chat_completions/openai.gpt-oss-20b-1:0", {}) == "chat_completions" +@pytest.mark.parametrize( + "model, supported_endpoints, expected_route", + [ + ("global.openai.gpt-5.5", ["/v1/chat/completions", "/v1/responses"], "converse"), + ("us.openai.gpt-5.6-sol", ["/v1/chat/completions", "/v1/responses"], "chat_completions"), + ("us.openai.gpt-5.6-sol", ["/v1/responses"], "converse"), + ("global.openai.gpt-6-sol", ["/v1/chat/completions", "/v1/responses"], "chat_completions"), + ("global.openai.gpt-6-sol", ["/v1/responses"], "converse"), + ("global.openai.gpt-6-sol", [], "converse"), + ("us.openai.gpt-6.1-sol", ["/v1/chat/completions"], "chat_completions"), + ("global.openai.gpt-10-sol", ["/v1/chat/completions"], "chat_completions"), + ("openai.gpt-oss-120b-1:0", ["/v1/chat/completions"], "converse"), + ("us.xai.grok-4.6", ["/v1/chat/completions"], "converse"), + ], +) +def test_default_route_needs_gpt_56_or_newer_and_a_row_listing_chat_completions( + monkeypatch, model, supported_endpoints, expected_route +): + entry = {"litellm_provider": "bedrock_converse", "supported_endpoints": supported_endpoints} + monkeypatch.setattr(litellm, "model_cost", {model: entry}) + assert bedrock_runtime_chat_completions_is_default(model) is (expected_route == "chat_completions") + assert BedrockModelInfo.get_bedrock_route(f"bedrock/{model}", {}) == expected_route + assert BedrockModelInfo.get_bedrock_route(f"bedrock/chat_completions/{model}", {}) == "chat_completions" + assert BedrockModelInfo.get_bedrock_route(f"bedrock/converse/{model}", {}) == "converse" + + @pytest.mark.parametrize("model", ["global.openai.gpt-5.6-sol", "openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"]) def test_chat_completions_prefix_prices_like_the_bare_model(local_cost_map, model): prefixed = litellm.get_model_info(model=f"bedrock/chat_completions/{model}") @@ -194,7 +223,7 @@ def _recording_client(**response_kwargs): [ ("bedrock/us.xai.grok-4.6", b"/model/us.xai.grok-4.6/converse"), ("bedrock/openai.gpt-oss-20b-1:0", b"/model/openai.gpt-oss-20b-1%3A0/converse"), - ("bedrock/global.openai.gpt-5.6-sol", b"/model/global.openai.gpt-5.6-sol/converse"), + ("bedrock/global.openai.gpt-5.5", b"/model/global.openai.gpt-5.5/converse"), ], ) def test_completion_without_the_prefix_posts_converse(local_cost_map, fake_aws_env, model, model_path): @@ -310,13 +339,40 @@ def test_openai_runtime_models_use_chat_completions_route(local_cost_map, model) assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig) +GPT_56_AND_NEWER_MODELS = ( + "global.openai.gpt-5.6-sol", + "bedrock/us.openai.gpt-5.6-terra", + "us.openai.gpt-5.6-luna", + "bedrock/global.openai.gpt-6-astra", + "us.openai.gpt-6-sol", + "global.openai.gpt-6-luna", + "bedrock/global.openai.gpt-6.1-sol", + "us.openai.gpt-6.1-sol", +) + + +@pytest.mark.parametrize("model", GPT_56_AND_NEWER_MODELS) +def test_gpt_56_and_newer_default_to_chat_completions(local_cost_map, model): + assert bedrock_runtime_chat_completions_is_default(model) is True + assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions" + assert BedrockModelInfo.get_bedrock_route(model, {}) == "chat_completions" + assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig) + + @pytest.mark.parametrize("model", ["us.amazon.nova-micro-v1:0", "us.anthropic.claude-haiku-4-5-20251001-v1:0"]) def test_nova_and_claude_stay_on_converse(local_cost_map, model): assert BedrockModelInfo.get_bedrock_route(model, {"tools": [GET_WEATHER_TOOL]}) == "converse" @pytest.mark.parametrize( - "model", ["chat_completions/openai.gpt-oss-20b-1:0", "bedrock/chat_completions/global.openai.gpt-5.6-sol"] + "model", + [ + "chat_completions/openai.gpt-oss-20b-1:0", + "bedrock/chat_completions/global.openai.gpt-5.6-sol", + "bedrock/us.openai.gpt-5.6-sol", + "global.openai.gpt-6-sol", + "us.openai.gpt-6.1-sol", + ], ) def test_guardrail_config_falls_back_to_converse(local_cost_map, model): guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"} @@ -363,6 +419,9 @@ def test_gpt56_tools_need_reasoning_none_on_chat_completions(local_cost_map, req BedrockModelInfo.get_bedrock_route("bedrock/chat_completions/us.openai.gpt-5.6-terra", request_params) == expected_route ) + assert BedrockModelInfo.get_bedrock_route("bedrock/us.openai.gpt-5.6-sol", request_params) == expected_route + assert BedrockModelInfo.get_bedrock_route("global.openai.gpt-6-sol", request_params) == expected_route + assert BedrockModelInfo.get_bedrock_route("bedrock/us.openai.gpt-6.1-sol", request_params) == expected_route @pytest.mark.parametrize("reasoning_effort", ["low", "high", None]) @@ -395,6 +454,8 @@ def test_thinking_block_goes_to_converse(local_cost_map): def test_explicit_converse_prefix_wins_for_openai_models(local_cost_map): assert BedrockModelInfo.get_bedrock_route("bedrock/converse/openai.gpt-oss-20b-1:0") == "converse" assert BedrockModelInfo.get_bedrock_route("converse/global.openai.gpt-5.6-sol", {}) == "converse" + assert BedrockModelInfo.get_bedrock_route("bedrock/converse/global.openai.gpt-6-sol", {}) == "converse" + assert isinstance(get_bedrock_chat_config("bedrock/converse/global.openai.gpt-6-sol"), litellm.AmazonConverseConfig) def test_map_openai_params_sends_max_tokens_as_max_completion_tokens(): @@ -769,6 +830,58 @@ def test_gpt56_tools_with_reasoning_none_stay_on_chat_completions(local_cost_map assert response.choices[0].message.tool_calls[0].function.name == "get_weather" +@pytest.mark.parametrize("model", ["global.openai.gpt-6-sol", "us.openai.gpt-5.6-sol", "us.openai.gpt-6.1-sol"]) +def test_gpt_56_and_newer_completion_without_the_prefix_posts_runtime_chat_completions( + local_cost_map, fake_aws_env, model +): + requests, client = _recording_client(json=_chat_completion_json("ok", model)) + response = litellm.completion( + model=f"bedrock/{model}", + messages=[{"role": "user", "content": "hello"}], + reasoning_effort="low", + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) + assert body["model"] == model + assert body["reasoning_effort"] == "low" + assert "inferenceConfig" not in body + assert response.choices[0].message.content == "ok" + assert response._hidden_params["response_cost"] > 0 + + +def test_gpt6_without_the_prefix_tools_with_reasoning_effort_go_to_converse(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + response = litellm.completion( + model="bedrock/global.openai.gpt-6-sol", + messages=[{"role": "user", "content": "hello"}], + tools=[GET_WEATHER_TOOL], + reasoning_effort="low", + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-6-sol/converse") + body = json.loads(requests[0].content) + assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "get_weather" + assert body["additionalModelRequestFields"]["reasoning"] == {"effort": "low"} + assert response.choices[0].message.content == "ok" + + +def test_gpt6_without_the_prefix_guardrail_config_goes_to_converse(local_cost_map, fake_aws_env): + guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"} + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/global.openai.gpt-6-sol", + messages=[{"role": "user", "content": "hello"}], + guardrailConfig=guardrail, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-6-sol/converse") + assert json.loads(requests[0].content)["guardrailConfig"] == guardrail + + @pytest.mark.parametrize( "converse_only_param", [ @@ -1017,6 +1130,8 @@ RESPONSE_FORMAT_ENFORCING_MODELS = [ "chat_completions/global.openai.gpt-5.6-sol", "chat_completions/us.xai.grok-4.6", "bedrock/chat_completions/us-gov.xai.grok-4.6", + "global.openai.gpt-6-sol", + "bedrock/us.openai.gpt-6.1-sol", ] diff --git a/tests/unit/llms/bedrock/test_cross_region_inference_profile_mapping.py b/tests/unit/llms/bedrock/test_cross_region_inference_profile_mapping.py index aa0827c5ae5..bcd1e9d6578 100644 --- a/tests/unit/llms/bedrock/test_cross_region_inference_profile_mapping.py +++ b/tests/unit/llms/bedrock/test_cross_region_inference_profile_mapping.py @@ -138,9 +138,10 @@ def _bedrock_response(model, usage): @pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id) -def test_bedrock_gpt_5_6_profiles_route_to_converse(profile, local_model_cost_map): - """GPT-5.6 is served by Converse on bedrock-runtime, never by Invoke.""" - assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "converse" +def test_bedrock_gpt_5_6_profiles_route_to_runtime_chat_completions(profile, local_model_cost_map): + """GPT-5.6 is served by bedrock-runtime's native Chat Completions by default and by Converse when pinned, never by Invoke.""" + assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "chat_completions" + assert BedrockModelInfo.get_bedrock_route(f"bedrock/converse/{profile.model_id}") == "converse" @pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id) From 22e35c6b216b2954c0e7961d030e2a4b60e139b5 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 30 Sep 2026 16:30:52 -0700 Subject: [PATCH 23/31] fix(bedrock): refuse temperature and top_p natively on GPT 5.6 and newer like Converse does AWS answers temperature and top_p with a 400 on the native Chat Completions endpoint for the GPT 5.6+ models, the same models whose Converse route already dropped both under drop_params via supports_sampling_params: false. The native config now honors that price-map flag, the gpt-6 and gpt-6.1 rows carry it, and the gpt-6 family joins gpt-5 in refusing frequency_penalty, presence_penalty, logprobs, and top_logprobs before the request reaches AWS. --- .../chat/chat_completions/transformation.py | 27 ++++++++++++++----- litellm/llms/bedrock/common_utils.py | 13 +++++++++ ...odel_prices_and_context_window_backup.json | 11 ++++++++ model_prices_and_context_window.json | 11 ++++++++ ...bedrock_chat_completions_transformation.py | 16 ++++++++--- 5 files changed, 68 insertions(+), 10 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 91a643fcd7b..34a756a54e2 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -32,7 +32,11 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import ( ) from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM -from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path +from litellm.llms.bedrock.common_utils import ( + BedrockError, + bedrock_model_supports_sampling_params, + split_bedrock_region_path, +) from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig from litellm.types.llms.openai import AllMessageValues @@ -46,26 +50,37 @@ if TYPE_CHECKING: REASONING_OPEN_TAG: Final = "" REASONING_CLOSE_TAG: Final = "" +GPT_CHAT_COMPLETIONS_REFUSED_PARAMS: Final = frozenset( + ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs") +) CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType( { - "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs")), + "openai.gpt-5": GPT_CHAT_COMPLETIONS_REFUSED_PARAMS, + "openai.gpt-6": GPT_CHAT_COMPLETIONS_REFUSED_PARAMS, "openai.gpt-oss": frozenset(("logit_bias",)), "xai.": frozenset(("frequency_penalty", "presence_penalty")), } ) +CHAT_COMPLETIONS_SAMPLING_PARAMS: Final = frozenset(("temperature", "top_p")) + + def chat_completions_params_refused_for(model: str) -> frozenset[str]: """The OpenAI params AWS's Chat Completions endpoint rejects for this model whatever else the request says. - Each family answers them with a 400 (GPT-5.6, gpt-oss) or a 503 (Grok), where Converse dropped the same - params under ``drop_params``, so the native config leaves them out of its supported list and the usual - drop-or-raise handling applies before the request reaches AWS. + Each family answers them with a 400 (GPT 5.6 and newer, gpt-oss) or a 503 (Grok), and a model whose price-map row + says ``supports_sampling_params: false`` answers ``temperature`` and ``top_p`` with a 400 too, where + Converse dropped the same params under ``drop_params``, so the native config leaves them out of its + supported list and the usual drop-or-raise handling applies before the request reaches AWS. """ model_id: Final = split_bedrock_region_path(model)[1] - return frozenset().union( + family_refused: Final = frozenset().union( *(refused for family, refused in CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY.items() if family in model_id) ) + if bedrock_model_supports_sampling_params(model): + return family_refused + return family_refused | CHAT_COMPLETIONS_SAMPLING_PARAMS CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY: Final = MappingProxyType({"xai.": frozenset(("none",))}) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 14ab71bd63a..bf40e9e93be 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -882,6 +882,19 @@ def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> boo return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_response_format") +def bedrock_model_supports_sampling_params(model: str) -> bool: + """Whether the model takes ``temperature`` and ``top_p``: false only when a price-map row says so. + + The GPT 5.6 and newer rows carry ``supports_sampling_params: false`` because AWS answers either param with + a 400 on Converse and on native Chat Completions alike, so both routes drop them under ``drop_params`` and + refuse them otherwise. + """ + return not any( + entry is not None and entry.get("supports_sampling_params") is False + for entry in _bedrock_price_map_entries(model) + ) + + BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( ( "guardrailConfig", diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 7c70f07329e..28690ed5afe 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -58479,6 +58479,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58516,6 +58517,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58553,6 +58555,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58590,6 +58593,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58626,6 +58630,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-sol": { @@ -58659,6 +58664,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58695,6 +58701,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-luna": { @@ -58728,6 +58735,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -79359,6 +79367,7 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true, + "supports_sampling_params": false, "supports_xhigh_reasoning_effort": true }, "openai.gpt-6.1-sol": { @@ -79391,6 +79400,7 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true, + "supports_sampling_params": false, "supports_xhigh_reasoning_effort": true }, "bedrock_mantle/openai.gpt-6.1-sol": { @@ -79466,6 +79476,7 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true, + "supports_sampling_params": false, "supports_xhigh_reasoning_effort": true }, "vertex_ai/gemini-3.8-flash-tts": { diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 7c70f07329e..28690ed5afe 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -58479,6 +58479,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58516,6 +58517,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58553,6 +58555,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58590,6 +58593,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58626,6 +58630,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-sol": { @@ -58659,6 +58664,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -58695,6 +58701,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-luna": { @@ -58728,6 +58735,7 @@ "supports_reasoning": true, "supports_xhigh_reasoning_effort": true, "supports_vision": true, + "supports_sampling_params": false, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ "/v1/chat/completions", @@ -79359,6 +79367,7 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true, + "supports_sampling_params": false, "supports_xhigh_reasoning_effort": true }, "openai.gpt-6.1-sol": { @@ -79391,6 +79400,7 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true, + "supports_sampling_params": false, "supports_xhigh_reasoning_effort": true }, "bedrock_mantle/openai.gpt-6.1-sol": { @@ -79466,6 +79476,7 @@ "supports_reasoning": true, "supports_tool_choice": true, "supports_vision": true, + "supports_sampling_params": false, "supports_xhigh_reasoning_effort": true }, "vertex_ai/gemini-3.8-flash-tts": { diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 7539e8c8b43..37d96c45d6d 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -463,7 +463,7 @@ def test_map_openai_params_sends_max_tokens_as_max_completion_tokens(): mapped = cfg.map_openai_params( non_default_params={"max_tokens": 64, "temperature": 0.1}, optional_params={}, - model="global.openai.gpt-5.6-sol", + model="us.xai.grok-4.6", drop_params=False, ) assert mapped == {"max_completion_tokens": 64, "temperature": 0.1} @@ -599,13 +599,18 @@ def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): [ ( "bedrock/global.openai.gpt-5.6-sol", - ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "n"), - ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions", "stop"), + ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "temperature", "top_p", "n"), + ("logit_bias", "reasoning_effort", "tools", "functions", "stop"), + ), + ( + "bedrock/us.openai.gpt-6.1-sol", + ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "temperature", "top_p", "n"), + ("logit_bias", "reasoning_effort", "tools", "functions", "stop"), ), ( "us.xai.grok-4.6", ("frequency_penalty", "presence_penalty", "n"), - ("stop", "logprobs", "top_p", "logit_bias", "reasoning_effort"), + ("stop", "logprobs", "temperature", "top_p", "logit_bias", "reasoning_effort"), ), ( "bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", @@ -625,6 +630,9 @@ def test_supported_params_leave_out_what_each_family_refuses(local_cost_map, mod [ ("bedrock/chat_completions/global.openai.gpt-5.6-sol", {"frequency_penalty": 0.5}), ("bedrock/chat_completions/global.openai.gpt-5.6-sol", {"logprobs": True, "top_logprobs": 2}), + ("bedrock/chat_completions/global.openai.gpt-5.6-sol", {"temperature": 0.2}), + ("bedrock/chat_completions/global.openai.gpt-6-sol", {"top_p": 0.9}), + ("bedrock/chat_completions/global.openai.gpt-6-sol", {"presence_penalty": 0.5}), ("bedrock/chat_completions/us.xai.grok-4.6", {"presence_penalty": 0.5}), ("bedrock/chat_completions/openai.gpt-oss-20b-1:0", {"logit_bias": {"1": 1}}), ], From 952adfc02519916e2a637c47af3467f9a2ccfcb2 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 30 Sep 2026 17:04:54 -0700 Subject: [PATCH 24/31] fix(bedrock): refuse GPT sampling and logprob params natively only while reasoning is on On bedrock-runtime's native chat completions endpoint, GPT-5.x and GPT-6.x accept temperature, top_p, frequency_penalty, presence_penalty, logprobs, and top_logprobs once reasoning_effort is "none", and refuse them with any other effort or when the effort is unset. The previous commit refused the sampling params unconditionally from the cost map's supports_sampling_params flag, which lost the reasoning-off case and never covered the penalties or logprobs. The refusal now keys on the model being a GPT id and reasoning being active, raises a 400 UnsupportedParamsError naming the params unless drop_params drops them, and lets everything through under "none". Grok and gpt-oss keep their unconditional family refusals. --- .../chat/chat_completions/transformation.py | 55 ++++++++++++------ litellm/llms/bedrock/common_utils.py | 14 +---- ...bedrock_chat_completions_transformation.py | 58 ++++++++++++++++--- 3 files changed, 88 insertions(+), 39 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 34a756a54e2..1477a46dcad 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -34,7 +34,7 @@ from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM from litellm.llms.bedrock.common_utils import ( BedrockError, - bedrock_model_supports_sampling_params, + bedrock_model_is_openai_gpt, split_bedrock_region_path, ) from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler @@ -50,37 +50,42 @@ if TYPE_CHECKING: REASONING_OPEN_TAG: Final = "" REASONING_CLOSE_TAG: Final = "" -GPT_CHAT_COMPLETIONS_REFUSED_PARAMS: Final = frozenset( - ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs") -) CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType( { - "openai.gpt-5": GPT_CHAT_COMPLETIONS_REFUSED_PARAMS, - "openai.gpt-6": GPT_CHAT_COMPLETIONS_REFUSED_PARAMS, "openai.gpt-oss": frozenset(("logit_bias",)), "xai.": frozenset(("frequency_penalty", "presence_penalty")), } ) - - -CHAT_COMPLETIONS_SAMPLING_PARAMS: Final = frozenset(("temperature", "top_p")) +GPT_CHAT_COMPLETIONS_PARAMS_REFUSED_WHILE_REASONING: Final = frozenset( + ("temperature", "top_p", "frequency_penalty", "presence_penalty", "logprobs", "top_logprobs") +) def chat_completions_params_refused_for(model: str) -> frozenset[str]: """The OpenAI params AWS's Chat Completions endpoint rejects for this model whatever else the request says. - Each family answers them with a 400 (GPT 5.6 and newer, gpt-oss) or a 503 (Grok), and a model whose price-map row - says ``supports_sampling_params: false`` answers ``temperature`` and ``top_p`` with a 400 too, where - Converse dropped the same params under ``drop_params``, so the native config leaves them out of its - supported list and the usual drop-or-raise handling applies before the request reaches AWS. + GPT-OSS answers ``logit_bias`` with a 400 and Grok answers the penalties with a 503, so the native config leaves + them out of its supported params and litellm refuses them, or drops them under ``drop_params``, before sending. """ model_id: Final = split_bedrock_region_path(model)[1] - family_refused: Final = frozenset().union( + return frozenset().union( *(refused for family, refused in CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY.items() if family in model_id) ) - if bedrock_model_supports_sampling_params(model): - return family_refused - return family_refused | CHAT_COMPLETIONS_SAMPLING_PARAMS + + +def chat_completions_params_refused_while_reasoning(model: str, params: Mapping[str, object]) -> frozenset[str]: + """The params of this request that AWS ties to ``reasoning_effort: "none"`` on the GPT-5.x and GPT-6.x families. + + AWS answers ``temperature``, ``top_p``, the penalties, and logprobs with a 400 while the model reasons, which + is every effort but ``"none"`` and the default when none is set, and accepts all of them under ``"none"``. + """ + if params.get("reasoning_effort") == "none" or not bedrock_model_is_openai_gpt(model): + return frozenset() + return GPT_CHAT_COMPLETIONS_PARAMS_REFUSED_WHILE_REASONING & frozenset(params) + + +def _without_params(params: Mapping[str, object], dropped: frozenset[str]) -> Mapping[str, object]: + return MappingProxyType({key: value for key, value in params.items() if key not in dropped}) CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY: Final = MappingProxyType({"xai.": frozenset(("none",))}) @@ -105,7 +110,7 @@ def chat_completions_reasoning_efforts_refused_for(model: str) -> frozenset[str] def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) -> Mapping[str, object]: if params.get("reasoning_effort") not in chat_completions_reasoning_efforts_refused_for(model): return params - return MappingProxyType({key: value for key, value in params.items() if key != "reasoning_effort"}) + return _without_params(params, frozenset(("reasoning_effort",))) def _held_close_tag_prefix(text: str) -> int: @@ -329,8 +334,20 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): drop_params=drop_params, replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens, ) + refused_while_reasoning: Final = chat_completions_params_refused_while_reasoning(model, non_default_params) + if refused_while_reasoning and not (litellm.drop_params or drop_params): + raise litellm.utils.UnsupportedParamsError( + message=( + f"{model} doesn't support {sorted(refused_while_reasoning)} while reasoning is active on " + "Bedrock's Chat Completions endpoint. Set reasoning_effort to 'none' to send them, or set " + "`litellm.drop_params = True` to drop them" + ), + status_code=400, + ) return dict( # mutable-ok: get_optional_params keeps filling this dict - without_refused_reasoning_effort(model, with_max_completion_tokens(mapped)) + without_refused_reasoning_effort( + model, with_max_completion_tokens(_without_params(mapped, refused_while_reasoning)) + ) ) def _inference_params( diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index bf40e9e93be..b45d4f40934 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -882,17 +882,9 @@ def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> boo return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_response_format") -def bedrock_model_supports_sampling_params(model: str) -> bool: - """Whether the model takes ``temperature`` and ``top_p``: false only when a price-map row says so. - - The GPT 5.6 and newer rows carry ``supports_sampling_params: false`` because AWS answers either param with - a 400 on Converse and on native Chat Completions alike, so both routes drop them under ``drop_params`` and - refuse them otherwise. - """ - return not any( - entry is not None and entry.get("supports_sampling_params") is False - for entry in _bedrock_price_map_entries(model) - ) +def bedrock_model_is_openai_gpt(model: str) -> bool: + """A GPT-5.x or GPT-6.x id, never GPT-OSS: the families whose sampling params AWS ties to reasoning being off.""" + return _openai_gpt_version(model) is not None BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 37d96c45d6d..5ea3c8c31bd 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -599,13 +599,13 @@ def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): [ ( "bedrock/global.openai.gpt-5.6-sol", - ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "temperature", "top_p", "n"), - ("logit_bias", "reasoning_effort", "tools", "functions", "stop"), + ("n",), + ("temperature", "top_p", "frequency_penalty", "logprobs", "logit_bias", "reasoning_effort", "stop"), ), ( "bedrock/us.openai.gpt-6.1-sol", - ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "temperature", "top_p", "n"), - ("logit_bias", "reasoning_effort", "tools", "functions", "stop"), + ("n",), + ("temperature", "top_p", "presence_penalty", "top_logprobs", "reasoning_effort", "tools", "functions"), ), ( "us.xai.grok-4.6", @@ -628,11 +628,6 @@ def test_supported_params_leave_out_what_each_family_refuses(local_cost_map, mod @pytest.mark.parametrize( "model, param", [ - ("bedrock/chat_completions/global.openai.gpt-5.6-sol", {"frequency_penalty": 0.5}), - ("bedrock/chat_completions/global.openai.gpt-5.6-sol", {"logprobs": True, "top_logprobs": 2}), - ("bedrock/chat_completions/global.openai.gpt-5.6-sol", {"temperature": 0.2}), - ("bedrock/chat_completions/global.openai.gpt-6-sol", {"top_p": 0.9}), - ("bedrock/chat_completions/global.openai.gpt-6-sol", {"presence_penalty": 0.5}), ("bedrock/chat_completions/us.xai.grok-4.6", {"presence_penalty": 0.5}), ("bedrock/chat_completions/openai.gpt-oss-20b-1:0", {"logit_bias": {"1": 1}}), ], @@ -650,6 +645,51 @@ def test_refused_params_are_dropped_or_refused_before_reaching_aws(local_cost_ma assert param.keys().isdisjoint(json.loads(requests[0].content)) +GPT_PARAMS_TIED_TO_REASONING_OFF = { + "temperature": 0.2, + "top_p": 0.9, + "frequency_penalty": 0.5, + "presence_penalty": 0.5, + "logprobs": True, + "top_logprobs": 2, +} + + +@pytest.mark.parametrize("model", ["bedrock/global.openai.gpt-5.6-sol", "bedrock/us.openai.gpt-6-sol"]) +@pytest.mark.parametrize("reasoning", [{}, {"reasoning_effort": "low"}], ids=["effort_unset", "effort_low"]) +@pytest.mark.parametrize("param", list(GPT_PARAMS_TIED_TO_REASONING_OFF)) +def test_gpt_sampling_params_are_refused_or_dropped_while_reasoning( + local_cost_map, fake_aws_env, model, reasoning, param +): + requests, client = _recording_client(json=_chat_completion_json("ok", model.removeprefix("bedrock/"))) + request = {"model": model, "messages": [{"role": "user", "content": "hello"}], "client": client, **reasoning} + with pytest.raises(litellm.UnsupportedParamsError, match=param): + litellm.completion(**request, **{param: GPT_PARAMS_TIED_TO_REASONING_OFF[param]}) + litellm.completion(**request, drop_params=True, **{param: GPT_PARAMS_TIED_TO_REASONING_OFF[param]}) + + body = json.loads(requests[0].content) + assert str(requests[0].url).endswith("/openai/v1/chat/completions") + assert param not in body + assert body.get("reasoning_effort") == reasoning.get("reasoning_effort") + + +@pytest.mark.parametrize("model", ["bedrock/global.openai.gpt-5.6-sol", "bedrock/us.openai.gpt-6-sol"]) +def test_gpt_sampling_params_reach_aws_with_reasoning_effort_none(local_cost_map, fake_aws_env, model): + requests, client = _recording_client(json=_chat_completion_json("ok", model.removeprefix("bedrock/"))) + litellm.completion( + model=model, + messages=[{"role": "user", "content": "hello"}], + reasoning_effort="none", + client=client, + **GPT_PARAMS_TIED_TO_REASONING_OFF, + ) + + body = json.loads(requests[0].content) + assert str(requests[0].url).endswith("/openai/v1/chat/completions") + assert body["reasoning_effort"] == "none" + assert {key: body[key] for key in GPT_PARAMS_TIED_TO_REASONING_OFF} == GPT_PARAMS_TIED_TO_REASONING_OFF + + def test_split_reasoning_tag_splits_leading_tag(): assert split_reasoning_tag("plan it\n\n\nHello") == ("plan it\n", "Hello") From 19176f675481b66f97a6cdb8e497afe99ee08988 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 30 Sep 2026 17:50:14 -0700 Subject: [PATCH 25/31] refactor(bedrock): keep the Converse route-prefix strip inside the bedrock llms module --- litellm/llms/bedrock/common_utils.py | 8 ++++++++ litellm/llms/bedrock/responses/transformation.py | 2 +- litellm/main.py | 8 ++++++-- .../unit/llms/bedrock/test_bedrock_common_utils.py | 14 ++++++++++++++ 4 files changed, 29 insertions(+), 3 deletions(-) diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index b45d4f40934..e5d950314c4 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -810,6 +810,14 @@ def strip_bedrock_routing_prefix(model: str) -> str: return model +BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX: Final = "chat_completions/" +BEDROCK_CONVERSE_ROUTE_PREFIX: Final = "converse/" + + +def without_bedrock_route_prefix(model: str) -> str: + return model.replace(BEDROCK_CONVERSE_ROUTE_PREFIX, "").replace(BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX, "") + + def split_bedrock_region_path(model: str) -> tuple[str | None, str]: """Split a ``/`` routing path into the region and the id AWS receives. diff --git a/litellm/llms/bedrock/responses/transformation.py b/litellm/llms/bedrock/responses/transformation.py index e6e8f853ef8..f989a96198b 100644 --- a/litellm/llms/bedrock/responses/transformation.py +++ b/litellm/llms/bedrock/responses/transformation.py @@ -50,6 +50,7 @@ from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.base_llm.responses.codex_compat import drop_unsupported_tools, normalize_codex_input_items from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM from litellm.llms.bedrock.common_utils import ( + BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX, BedrockError, bedrock_supports_openai_responses, ) @@ -71,7 +72,6 @@ BEDROCK_RUNTIME_SUPPORTED_RESPONSE_TOOL_TYPES: Final = frozenset( {"function", "mcp", "custom", "apply_patch", "namespace", "tool_search", "computer"} ) BEDROCK_RUNTIME_UNSUPPORTED_RESPONSE_PARAMS: Final = frozenset({"background"}) -BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX: Final = "chat_completions/" REMOTE_IMAGE_URL_SCHEMES: Final = ("http://", "https://") IMAGE_BLOCK_KEYS: Final = ("content", "output") IMAGE_BLOCK_TYPES: Final = frozenset({"input_image", "computer_screenshot"}) diff --git a/litellm/main.py b/litellm/main.py index 851052fdef1..3b50d36a0d9 100644 --- a/litellm/main.py +++ b/litellm/main.py @@ -116,7 +116,11 @@ from litellm.llms.base_llm import BaseConfig, BaseImageGenerationConfig from litellm.llms.base_llm.base_model_iterator import ( convert_model_response_to_streaming, ) -from litellm.llms.bedrock.common_utils import BedrockModelInfo, bedrock_route_for_request +from litellm.llms.bedrock.common_utils import ( + BedrockModelInfo, + bedrock_route_for_request, + without_bedrock_route_prefix, +) from litellm.llms.cohere.common_utils import CohereModelInfo from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler, http2_enabled from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config @@ -4234,7 +4238,7 @@ def _complete_bedrock(ctx: _CompletionDispatchContext) -> _CompletionDispatchRes provider_config=provider_config, ) elif bedrock_route == "converse": - model = model.replace("converse/", "").replace("chat_completions/", "") + model = without_bedrock_route_prefix(model) response = bedrock_converse_chat_completion.completion( model=model, messages=messages, diff --git a/tests/unit/llms/bedrock/test_bedrock_common_utils.py b/tests/unit/llms/bedrock/test_bedrock_common_utils.py index e5118f90e44..2e294a7760b 100644 --- a/tests/unit/llms/bedrock/test_bedrock_common_utils.py +++ b/tests/unit/llms/bedrock/test_bedrock_common_utils.py @@ -981,3 +981,17 @@ def test_unmapped_openai_family_model_routes_to_converse(): assert BedrockModelInfo.get_bedrock_route(unmapped) == "converse" imported: Final = "bedrock/openai/arn:aws:bedrock:us-east-1:123456789012:imported-model/abc123" assert BedrockModelInfo.get_bedrock_route(imported) == "openai" + + +@pytest.mark.parametrize( + ("model", "expected"), + [ + ("converse/us.anthropic.claude-haiku-4-5-20251001-v1:0", "us.anthropic.claude-haiku-4-5-20251001-v1:0"), + ("chat_completions/us.xai.grok-4.6", "us.xai.grok-4.6"), + ("global.openai.gpt-5.6-sol", "global.openai.gpt-5.6-sol"), + ], +) +def test_without_bedrock_route_prefix_hands_converse_the_bare_model_id(model, expected): + from litellm.llms.bedrock.common_utils import without_bedrock_route_prefix + + assert without_bedrock_route_prefix(model) == expected From a0cef91f0b355e659c3a11756f344d070f2324ae Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Wed, 30 Sep 2026 18:50:14 -0700 Subject: [PATCH 26/31] fix(bedrock): forward a non-string reasoning_effort on the native route instead of crashing A list or dict reasoning_effort hit a frozenset membership test in without_refused_reasoning_effort and raised TypeError, which the proxy surfaced as a 500 APIConnectionError with no upstream call. The value is now left alone unless it is a string Bedrock's native endpoint refuses, so AWS answers the malformed value with its own 400 like it does for an int --- .../bedrock/chat/chat_completions/transformation.py | 3 ++- .../test_bedrock_chat_completions_transformation.py | 13 +++++++++++++ 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index 1477a46dcad..ac1d2cf2498 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -108,7 +108,8 @@ def chat_completions_reasoning_efforts_refused_for(model: str) -> frozenset[str] def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) -> Mapping[str, object]: - if params.get("reasoning_effort") not in chat_completions_reasoning_efforts_refused_for(model): + effort: Final = params.get("reasoning_effort") + if not isinstance(effort, str) or effort not in chat_completions_reasoning_efforts_refused_for(model): return params return _without_params(params, frozenset(("reasoning_effort",))) diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 5ea3c8c31bd..db1efd49cb9 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -573,6 +573,19 @@ def test_map_openai_params_keeps_reasoning_effort_low_for_grok(): assert mapped["reasoning_effort"] == "low" +@pytest.mark.parametrize("model", ["us.xai.grok-4.6", "global.openai.gpt-5.6-sol"]) +@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5]) +def test_map_openai_params_forwards_a_malformed_reasoning_effort_for_aws_to_refuse(model, reasoning_effort): + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + mapped = cfg.map_openai_params( + non_default_params={"reasoning_effort": reasoning_effort, "max_tokens": 64}, + optional_params={}, + model=model, + drop_params=False, + ) + assert mapped["reasoning_effort"] == reasoning_effort + + def test_map_openai_params_keeps_reasoning_effort_none_for_gpt56(): cfg = AmazonBedrockRuntimeChatCompletionsConfig() mapped = cfg.map_openai_params( From dae29c138bb2c8ff69e3e3195ea9b9ca5bfb4b67 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 1 Oct 2026 14:53:25 -0700 Subject: [PATCH 27/31] fix(bedrock): route overlong GPT version digits to Converse and send native chat completions to the runtime endpoint A model id with more than 4300 version digits raised ValueError in the route check; the digits are now bounded so such ids fall back to Converse. The native chat completions URL now follows Converse's precedence: aws_bedrock_runtime_endpoint (or AWS_BEDROCK_RUNTIME_ENDPOINT) wins over api_base, so a deployment that sets both keeps sending to the same host --- .../chat/chat_completions/transformation.py | 4 +-- litellm/llms/bedrock/common_utils.py | 2 +- ...bedrock_chat_completions_transformation.py | 34 +++++++++++++++++++ 3 files changed, 37 insertions(+), 3 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index ac1d2cf2498..e16eeebd37a 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -277,12 +277,12 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): aws_region_name: Final = self._aws_signer._get_aws_region_name( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public region resolver optional_params=self._params_with_region_from_path(optional_params, model), model=model ) - endpoint_url, _ = self._aws_signer.get_runtime_endpoint( + _, proxy_endpoint_url = self._aws_signer.get_runtime_endpoint( api_base=api_base, aws_bedrock_runtime_endpoint=optional_params.get("aws_bedrock_runtime_endpoint"), aws_region_name=aws_region_name, ) - base: Final = endpoint_url.rstrip("/") + base: Final = proxy_endpoint_url.rstrip("/") if base.endswith("/openai/v1/chat/completions"): return base if base.endswith("/openai/v1"): diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index e5d950314c4..c2d2660326b 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -39,7 +39,7 @@ if TYPE_CHECKING: _ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs" _OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.") -_OPENAI_GPT_VERSION_RE: Final = re.compile(r"(^|[./])openai\.gpt-(\d+)(?:\.(\d+))?") +_OPENAI_GPT_VERSION_RE: Final = re.compile(r"(^|[./])openai\.gpt-(\d{1,3})(?!\d)(?:\.(\d{1,3})(?!\d))?") _BEDROCK_RUNTIME_CHAT_COMPLETIONS_DEFAULT_SINCE: Final = (5, 6) _BEDROCK_RUNTIME_CHAT_COMPLETIONS_ENDPOINT: Final = "/v1/chat/completions" BedrockRoute = Literal[ diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index db1efd49cb9..259c5d89add 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -149,6 +149,40 @@ def test_complete_url_appends_to_openai_v1_base(): assert url == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" +def test_complete_url_sends_to_the_runtime_endpoint_over_api_base_like_converse(monkeypatch): + monkeypatch.delenv("AWS_BEDROCK_RUNTIME_ENDPOINT", raising=False) + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + url = cfg.get_complete_url( + api_base="https://signing-host.example.com", + api_key=None, + model="us.openai.gpt-5.6-sol", + optional_params={"aws_region_name": "us-east-1", "aws_bedrock_runtime_endpoint": "https://egress.example.com/"}, + litellm_params={}, + ) + assert url == "https://egress.example.com/openai/v1/chat/completions" + + +def test_complete_url_sends_to_the_env_runtime_endpoint_over_api_base_like_converse(monkeypatch): + monkeypatch.setenv("AWS_BEDROCK_RUNTIME_ENDPOINT", "https://env-egress.example.com") + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + url = cfg.get_complete_url( + api_base="https://signing-host.example.com", + api_key=None, + model="us.openai.gpt-5.6-sol", + optional_params={"aws_region_name": "us-east-1"}, + litellm_params={}, + ) + assert url == "https://env-egress.example.com/openai/v1/chat/completions" + + +@pytest.mark.parametrize("digits", [4, 4301, 30000]) +@pytest.mark.parametrize("template", ["openai.gpt-{run}", "us.openai.gpt-5.{run}", "openai.gpt-{run}.{run}-sol"]) +def test_overlong_gpt_version_digits_route_to_converse_without_raising(local_cost_map, template, digits): + model = template.format(run="9" * digits) + assert bedrock_runtime_chat_completions_is_default(model) is False + assert bedrock_route_for_request(model, {}, None) == "converse" + + def test_project_id_is_not_sent_as_openai_project_header(): cfg = AmazonBedrockRuntimeChatCompletionsConfig() headers = cfg.validate_environment( From 578f26edba60acf2b53ca3ef04475fbccbecede9 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 1 Oct 2026 15:50:58 -0700 Subject: [PATCH 28/31] fix(bedrock): route model_id overrides to Converse and never send an empty bearer natively A deployment whose litellm_params carry model_id (an application inference profile or provisioned throughput ARN) went to the native Chat Completions route with the base model in the URL and model_id left in the body. It now takes Converse like the bedrock/arn:... model form, which encodes the override into the request URL A blank api_key on a SigV4 deployment became an Authorization header reading Bearer with nothing after it on the native route, since the OpenAI-like header builder writes any non-None key and the signer keeps a non-AWS4 Authorization header. validate_environment now resolves the key through bedrock_bearer_token, so a blank key is signed with SigV4 the way Converse signs it --- .../chat/chat_completions/transformation.py | 22 ++++++- litellm/llms/bedrock/common_utils.py | 5 +- ...bedrock_chat_completions_transformation.py | 64 ++++++++++++++++++- 3 files changed, 87 insertions(+), 4 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index e16eeebd37a..c47fa3944c1 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -31,7 +31,7 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import ( inline_remote_media, ) from litellm.llms.base_llm.chat.transformation import BaseLLMException -from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM +from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM, bedrock_bearer_token from litellm.llms.bedrock.common_utils import ( BedrockError, bedrock_model_is_openai_gpt, @@ -263,6 +263,26 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): ) -> BaseLLMException: return BedrockError(status_code=status_code, message=error_message, headers=headers) + def validate_environment( + self, + headers: dict, # mutable-ok: BaseConfig signature + model: str, + messages: list[AllMessageValues], + optional_params: dict, # mutable-ok: BaseConfig signature + litellm_params: dict, # mutable-ok: BaseConfig signature + api_key: str | None = None, + api_base: str | None = None, + ) -> dict: # mutable-ok: BaseConfig signature + return super().validate_environment( + headers=headers, + model=model, + messages=messages, + optional_params=optional_params, + litellm_params=litellm_params, + api_key=bedrock_bearer_token(api_key), + api_base=api_base, + ) + def get_complete_url( self, api_base: str | None, diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index c2d2660326b..b358bb395ab 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -906,6 +906,7 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( "additionalModelRequestFields", "top_k", "stop", + "model_id", ) ) @@ -931,7 +932,9 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking`` block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on - AWS's native OpenAI surface, ``stop`` stays on Converse where it fails loudly instead of silently + AWS's native OpenAI surface, a ``model_id`` override (an application inference profile or provisioned + throughput ARN) is only encoded into Converse's request URL and so stays on Converse like the + ``bedrock/arn:...`` model form, ``stop`` stays on Converse where it fails loudly instead of silently stopping hidden reasoning, operator-owned request metadata is only written onto the Converse body, function tools (``tools`` or legacy ``functions``) on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 259c5d89add..06fff49d500 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -25,6 +25,8 @@ from litellm.llms.bedrock.common_utils import ( ) from litellm.llms.custom_httpx.http_handler import HTTPHandler +APPLICATION_INFERENCE_PROFILE_ARN = "arn:aws:bedrock:us-west-2:123412341234:application-inference-profile/a1b2c3" + @pytest.fixture def local_cost_map(monkeypatch): @@ -425,8 +427,13 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model): ) @pytest.mark.parametrize( "request_params", - [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}, {"stop": ["END"]}], - ids=["additionalModelRequestFields", "top_k", "stop"], + [ + {"additionalModelRequestFields": {"reasoning_effort": "high"}}, + {"top_k": 40}, + {"stop": ["END"]}, + {"model_id": APPLICATION_INFERENCE_PROFILE_ARN}, + ], + ids=["additionalModelRequestFields", "top_k", "stop", "model_id"], ) def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params): assert bedrock_request_needs_converse(model, request_params) is True @@ -434,6 +441,59 @@ def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, assert BedrockModelInfo.get_bedrock_route(model, {key: None for key in request_params}) == "chat_completions" +@pytest.mark.parametrize( + "model", ["bedrock/us.openai.gpt-5.6-sol", "global.openai.gpt-6-sol", "bedrock/chat_completions/us.xai.grok-4.6"] +) +def test_model_id_override_is_served_by_converse_like_the_arn_model_form(local_cost_map, model): + assert bedrock_route_for_request(model, {"model_id": APPLICATION_INFERENCE_PROFILE_ARN}, None) == "converse" + assert bedrock_route_for_request(model, {"model_id": None}, None) == "chat_completions" + + +SIGV4_PARAMS = { + "aws_access_key_id": "AKIAIOSFODNN7EXAMPLE", + "aws_secret_access_key": "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY", + "aws_region_name": "us-east-1", +} + + +@pytest.mark.parametrize("api_key", ["", None], ids=["blank", "absent"]) +def test_blank_api_key_is_signed_with_sigv4_instead_of_an_empty_bearer(monkeypatch, api_key): + monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False) + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + url = "https://bedrock-runtime.us-east-1.amazonaws.com/openai/v1/chat/completions" + headers = cfg.validate_environment( + headers={}, + model="bedrock/us.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "hello"}], + optional_params=dict(SIGV4_PARAMS), + litellm_params={}, + api_key=api_key, + ) + assert "Authorization" not in headers + signed, _ = cfg.sign_request( + headers=headers, + optional_params=dict(SIGV4_PARAMS), + request_data={"model": "us.openai.gpt-5.6-sol", "messages": []}, + api_base=url, + api_key=api_key, + ) + assert signed["Authorization"].startswith("AWS4-HMAC-SHA256 Credential=AKIAIOSFODNN7EXAMPLE/"), signed + + +def test_bearer_api_key_is_sent_as_the_authorization_header(monkeypatch): + monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False) + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + headers = cfg.validate_environment( + headers={}, + model="bedrock/us.openai.gpt-5.6-sol", + messages=[{"role": "user", "content": "hello"}], + optional_params={}, + litellm_params={}, + api_key="bedrock-api-key", + ) + assert headers["Authorization"] == "Bearer bedrock-api-key" + + @pytest.mark.parametrize( "request_params, expected_route", [ From 4ee7ff4315e49ee77539fcf09e61fd6bb3f9d929 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 1 Oct 2026 16:41:03 -0700 Subject: [PATCH 29/31] test(bedrock): audit the native GPT chat completions route on the integration rig Adds the /audit cells for the runtime chat completions route: the scripted Bedrock runtime peer, the happy and fallback wire tests, the sad-path and regex worst-case tests, the chaos burst tests, the Messages adapter tests, and the Responses native-route tests. Tests only, no product diff. --- .../_support/bedrock_runtime_peer.py | 276 +++++++++ ...rock_messages_gpt_chat_completions_wire.py | 194 +++++++ .../test_bedrock_gpt_responses_native_wire.py | 117 ++++ ..._bedrock_runtime_chat_completions_chaos.py | 402 +++++++++++++ ...drock_runtime_chat_completions_sad_wire.py | 415 +++++++++++++ ...t_bedrock_runtime_chat_completions_wire.py | 549 ++++++++++++++++++ 6 files changed, 1953 insertions(+) create mode 100644 tests/integration/_support/bedrock_runtime_peer.py create mode 100644 tests/integration/messages_endpoint/providers/bedrock/test_bedrock_messages_gpt_chat_completions_wire.py create mode 100644 tests/integration/providers/test_bedrock_gpt_responses_native_wire.py create mode 100644 tests/integration/providers/test_bedrock_runtime_chat_completions_chaos.py create mode 100644 tests/integration/providers/test_bedrock_runtime_chat_completions_sad_wire.py create mode 100644 tests/integration/providers/test_bedrock_runtime_chat_completions_wire.py diff --git a/tests/integration/_support/bedrock_runtime_peer.py b/tests/integration/_support/bedrock_runtime_peer.py new file mode 100644 index 00000000000..3a547260590 --- /dev/null +++ b/tests/integration/_support/bedrock_runtime_peer.py @@ -0,0 +1,276 @@ +import json +import re +import threading +from collections.abc import Mapping +from multiprocessing.sharedctypes import Synchronized +from types import MappingProxyType +from typing import Final +from urllib.parse import unquote + +from integration._support.upstream import _aws_event_frame +from integration._support.wire import Reply, Request, wire_server +from pydantic import JsonValue, TypeAdapter + +MARKER: Final = re.compile(r"marker-([0-9a-f]{32})") +EVENT_STREAM: Final = "application/vnd.amazon.eventstream" +REASONING_EFFORTS: Final = frozenset(("none", "minimal", "low", "medium", "high", "xhigh")) +NATIVE_CHAT: Final = "/openai/v1/chat/completions" +NATIVE_RESPONSES: Final = "/openai/v1/responses" +PNG_1X1: Final = bytes.fromhex( + "89504e470d0a1a0a0000000d49484452000000010000000108060000001f15c489" + "0000000d49444154789c63f8cfc0f01f00050001ff89993d1d0000000049454e44ae426082" +) +USAGE: Final[Mapping[str, JsonValue]] = MappingProxyType( + { + "prompt_tokens": 9, + "completion_tokens": 5, + "total_tokens": 14, + "completion_tokens_details": {"reasoning_tokens": 3}, + } +) +_STATUS: Final = re.compile(r"status=(\d{3})") +_CONVERSE: Final = re.compile(r"^/model/(.+)/converse$") +_CONVERSE_STREAM: Final = re.compile(r"^/model/(.+)/converse-stream$") +_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) +_NO_MARKER: Final = "0" * 32 + + +def marker_of(request: Request) -> str: + found: Final = MARKER.search(request.body.decode(errors="replace")) + return _NO_MARKER if found is None else found.group(1) + + +def body_of(request: Request) -> Mapping[str, JsonValue]: + try: + return _JSON_OBJECT.validate_json(request.body) + except ValueError: + return {} + + +def target_of(request: Request) -> str: + return unquote(request.target) + + +def answer(marker: str) -> str: + return f"answer marker-{marker}" + + +def reasoning_answer(marker: str) -> str: + return f"why marker-{marker} {answer(marker)}" + + +def _headers(marker: str) -> Mapping[str, str]: + return MappingProxyType({"x-amzn-requestid": marker}) + + +def _json_reply(status: int, payload: Mapping[str, JsonValue], marker: str) -> Reply: + return Reply(status=status, body=json.dumps(payload).encode(), headers=_headers(marker)) + + +def _error(status: int, message: str, marker: str) -> Reply: + return _json_reply(status, {"message": message}, marker) + + +def _effort_of(target: str, body: Mapping[str, JsonValue]) -> JsonValue: + if not _CONVERSE.match(target) and not _CONVERSE_STREAM.match(target): + return body.get("reasoning_effort") + fields: Final = body.get("additionalModelRequestFields") + reasoning: Final = fields.get("reasoning") if isinstance(fields, Mapping) else None + return reasoning.get("effort") if isinstance(reasoning, Mapping) else None + + +def forwarded_effort(request: Request) -> JsonValue: + return _effort_of(target_of(request), body_of(request)) + + +def _sse(frames: tuple[Mapping[str, JsonValue], ...], pause: float) -> Reply: + return Reply( + content_type="text/event-stream", + chunks=(*(b"data: " + json.dumps(frame).encode() + b"\n\n" for frame in frames), b"data: [DONE]\n\n"), + pause_between_chunks=pause, + ) + + +def _with_headers(reply: Reply, marker: str) -> Reply: + return Reply( + status=reply.status, + body=reply.body, + content_type=reply.content_type, + chunks=reply.chunks, + abort_after=reply.abort_after, + gate_after_first=reply.gate_after_first, + pause_between_chunks=reply.pause_between_chunks, + headers=_headers(marker), + ) + + +def _content_deltas(model: str, marker: str) -> tuple[str, ...]: + if "gpt-oss" in model: + return ("why ", f"marker-{marker}", " answer ", f"marker-{marker}") + return ("answer ", f"marker-{marker}") + + +def _chat_text(model: str, marker: str) -> str: + return reasoning_answer(marker) if "gpt-oss" in model else answer(marker) + + +def _chat_reply(model: str, marker: str, stream: bool, pause: float) -> Reply: + identity: Final = f"chatcmpl-{marker}" + if not stream: + return _json_reply( + 200, + { + "id": identity, + "object": "chat.completion", + "created": 1, + "model": model, + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": _chat_text(model, marker)}, + "finish_reason": "stop", + } + ], + "usage": dict(USAGE), + }, + marker, + ) + deltas: Final = _content_deltas(model, marker) + frames: Final = tuple( + { + "id": identity, + "object": "chat.completion.chunk", + "created": 1, + "model": model, + "choices": [{"index": 0, "delta": {"role": "assistant", "content": delta}, "finish_reason": None}], + } + for delta in deltas + ) + finish: Final[Mapping[str, JsonValue]] = { + "id": identity, + "object": "chat.completion.chunk", + "created": 1, + "model": model, + "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}], + "usage": dict(USAGE), + } + return _with_headers(_sse((*frames, finish), pause), marker) + + +def _responses_reply(model: str, marker: str, stream: bool, pause: float) -> Reply: + identity: Final = f"resp_upstream_{marker}" + item_id: Final = f"msg_{marker}" + response: Final[Mapping[str, JsonValue]] = { + "id": identity, + "object": "response", + "created_at": 1, + "status": "completed", + "model": model, + "output": [ + { + "type": "message", + "id": item_id, + "status": "completed", + "role": "assistant", + "content": [{"type": "output_text", "text": answer(marker), "annotations": []}], + } + ], + "usage": {"input_tokens": 30, "output_tokens": 5, "total_tokens": 35}, + } + if not stream: + return _json_reply(200, response, marker) + events: Final[tuple[Mapping[str, JsonValue], ...]] = ( + { + "type": "response.created", + "sequence_number": 0, + "response": {**response, "status": "in_progress", "output": []}, + }, + { + "type": "response.output_text.delta", + "sequence_number": 1, + "item_id": item_id, + "output_index": 0, + "content_index": 0, + "delta": answer(marker), + }, + {"type": "response.completed", "sequence_number": 2, "response": response}, + ) + return Reply( + content_type="text/event-stream", + chunks=tuple(f"event: {event['type']}\ndata: {json.dumps(event)}\n\n".encode() for event in events), + pause_between_chunks=pause, + headers=_headers(marker), + ) + + +def _converse_reply(marker: str) -> Reply: + return _json_reply( + 200, + { + "output": {"message": {"role": "assistant", "content": [{"text": answer(marker)}]}}, + "stopReason": "end_turn", + "usage": {"inputTokens": 9, "outputTokens": 5, "totalTokens": 14}, + "metrics": {"latencyMs": 1}, + }, + marker, + ) + + +def _converse_stream_reply(marker: str, pause: float) -> Reply: + events: Final[tuple[tuple[str, Mapping[str, JsonValue]], ...]] = ( + ("messageStart", {"role": "assistant"}), + ("contentBlockDelta", {"delta": {"text": "answer "}, "contentBlockIndex": 0}), + ("contentBlockDelta", {"delta": {"text": f"marker-{marker}"}, "contentBlockIndex": 0}), + ("contentBlockStop", {"contentBlockIndex": 0}), + ("messageStop", {"stopReason": "end_turn"}), + ("metadata", {"usage": {"inputTokens": 9, "outputTokens": 5, "totalTokens": 14}, "metrics": {"latencyMs": 1}}), + ) + return Reply( + content_type=EVENT_STREAM, + chunks=tuple(_aws_event_frame(kind, payload, "sc", marker) for kind, payload in events), + pause_between_chunks=pause, + headers=_headers(marker), + ) + + +def respond(request: Request, *, pause: float = 0.0) -> Reply: + target: Final = target_of(request) + marker: Final = marker_of(request) + if request.method == "GET": + if target == "/image.png": + return Reply(body=PNG_1X1, content_type="image/png", headers=_headers(marker)) + return _error(404, f"no scripted object at {target}", marker) + scripted_status: Final = _STATUS.search(request.body.decode(errors="replace")) + if scripted_status is not None: + status: Final = int(scripted_status.group(1)) + return _error(status, f"scripted {status}", marker) + body: Final = body_of(request) + effort: Final = _effort_of(target, body) + if effort is not None and (not isinstance(effort, str) or effort not in REASONING_EFFORTS): + return _error(400, f"Invalid reasoning effort: {json.dumps(effort)}", marker) + model: Final = str(body.get("model", "")) + stream: Final = body.get("stream") is True + if request.method == "POST" and target == NATIVE_CHAT: + return _chat_reply(model, marker, stream, pause) + if request.method == "POST" and target == NATIVE_RESPONSES: + return _responses_reply(model, marker, stream, pause) + if request.method == "POST" and _CONVERSE.match(target): + return _converse_reply(marker) + if request.method == "POST" and _CONVERSE_STREAM.match(target): + return _converse_stream_reply(marker, pause) + return _error(404, f"unknown bedrock route {request.method} {target}", marker) + + +def serve_peer(port: int, received: Synchronized[int], answer_first: int) -> None: + held: Final = threading.Event() + + def respond_or_hold(request: Request) -> Reply: + with received.get_lock(): + received.value += 1 + ordinal: Final = received.value + if ordinal > answer_first: + held.wait() + return respond(request) + + with wire_server(respond_or_hold, port=port): + threading.Event().wait() diff --git a/tests/integration/messages_endpoint/providers/bedrock/test_bedrock_messages_gpt_chat_completions_wire.py b/tests/integration/messages_endpoint/providers/bedrock/test_bedrock_messages_gpt_chat_completions_wire.py new file mode 100644 index 00000000000..00945840808 --- /dev/null +++ b/tests/integration/messages_endpoint/providers/bedrock/test_bedrock_messages_gpt_chat_completions_wire.py @@ -0,0 +1,194 @@ +import json +import uuid +from collections.abc import Mapping +from typing import Final + +import anthropic +from integration._support.bedrock_runtime_peer import NATIVE_CHAT, answer, body_of, marker_of, respond, target_of +from integration._support.client import Gateway, Scenario, eventually +from integration._support.database import read_rows +from integration._support.wire import Request, Wire, wire_server +from pydantic import JsonValue + +BEDROCK_MODEL: Final = "us.openai.gpt-5.6-sol" +TOKEN: Final = "synthetic-bedrock-bearer" +NO_CACHE: Final[Mapping[str, JsonValue]] = {"cache": {"no-cache": True}} +ANTHROPIC_VERSION: Final[Mapping[str, str]] = {"anthropic-version": "2023-06-01"} + + +def _question(marker: str) -> str: + return f"Question marker-{marker}" + + +def _deployment(scenario: Scenario, wire: Wire) -> str: + return scenario.model( + model=f"bedrock/{BEDROCK_MODEL}", + api_key=TOKEN, + aws_region_name="us-east-1", + aws_bedrock_runtime_endpoint=wire.url, + ) + + +def _carrying(wire: Wire, marker: str) -> tuple[Request, ...]: + return tuple(request for request in wire.drain() if marker_of(request) == marker) + + +def _native_body(wire: Wire, marker: str) -> Mapping[str, JsonValue]: + received: Final = _carrying(wire, marker) + assert [(request.method, target_of(request)) for request in received] == [("POST", NATIVE_CHAT)] + assert received[0].headers["authorization"] == f"Bearer {TOKEN}", received[0].headers + return body_of(received[0]) + + +def _native_request(marker: str, max_tokens: int, effort: str) -> Mapping[str, JsonValue]: + return { + "model": BEDROCK_MODEL, + "messages": [{"role": "user", "content": _question(marker)}], + "max_completion_tokens": max_tokens, + "reasoning_effort": effort, + } + + +def _spend_rows(identity: str, expected: int) -> list[dict[str, JsonValue]]: + return eventually( + lambda: read_rows( + "SELECT request_id, call_type, status, model_group, prompt_tokens, completion_tokens, cache_hit" + ' FROM "LiteLLM_SpendLogs" WHERE starts_with(request_id, %s) ORDER BY "startTime"', + (identity,), + ), + lambda found: len(found) == expected, + seconds=70, + ) + + +def _success_row(identity: str, model: str, cache_hit: str = "None") -> dict[str, JsonValue]: + return { + "request_id": identity, + "call_type": "anthropic_messages", + "status": "success", + "model_group": model, + "prompt_tokens": 9, + "completion_tokens": 5, + "cache_hit": cache_hit, + } + + +def test_anthropic_sdk_thinking_budget_reaches_native_chat_completions_as_reasoning_effort(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + client: Final = anthropic.Anthropic(base_url=str(gateway.client.base_url), api_key=gateway.key, max_retries=0) + message: Final = client.messages.create( + model=model, + max_tokens=4096, + thinking={"type": "enabled", "budget_tokens": 2048}, + messages=[{"role": "user", "content": _question(marker)}], + extra_body=NO_CACHE, + ) + assert _native_body(wire, marker) == _native_request(marker, 4096, "medium") + assert message.id == f"chatcmpl-{marker}", message + assert [(block.type, getattr(block, "text", None)) for block in message.content] == [("text", answer(marker))] + assert (message.usage.input_tokens, message.usage.output_tokens) == (9, 5), message + assert _spend_rows(message.id, 1) == [_success_row(message.id, model)] + + +def test_anthropic_sdk_stream_with_thinking_budget_is_served_by_native_chat_completions(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + client: Final = anthropic.Anthropic(base_url=str(gateway.client.base_url), api_key=gateway.key, max_retries=0) + stream: Final = client.messages.create( + model=model, + max_tokens=4096, + thinking={"type": "enabled", "budget_tokens": 2048}, + messages=[{"role": "user", "content": _question(marker)}], + extra_body=NO_CACHE, + stream=True, + ) + events: Final = list(stream) + assert _native_body(wire, marker) == { + **_native_request(marker, 4096, "medium"), + "stream": True, + "stream_options": {"include_usage": True}, + } + assert events[0].type == "message_start" and events[-1].type == "message_stop", events + identity: Final = events[0].message.id + assert identity.startswith("msg_"), events + assert "".join( + event.delta.text + for event in events + if event.type == "content_block_delta" and event.delta.type == "text_delta" + ) == answer(marker) + assert _spend_rows(identity, 1) == [_success_row(identity, model, cache_hit="False")] + + +def test_raw_thinking_summary_reaches_native_chat_completions_as_the_plain_effort(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = gateway.request( + "POST", + "/v1/messages", + { + "model": model, + "max_tokens": 4096, + "thinking": {"type": "enabled", "budget_tokens": 2048, "summary": "detailed"}, + "messages": [{"role": "user", "content": _question(marker)}], + **NO_CACHE, + }, + headers=ANTHROPIC_VERSION, + ) + body: Final = _native_body(wire, marker) + assert body == _native_request(marker, 4096, "medium") + assert "summary" not in json.dumps(body), body + assert response.status_code == 200, response.text + assert response.json()["id"] == f"chatcmpl-{marker}", response.text + assert response.json()["content"] == [{"type": "text", "text": answer(marker)}], response.text + assert _spend_rows(f"chatcmpl-{marker}", 1) == [_success_row(f"chatcmpl-{marker}", model)] + + +async def test_async_anthropic_sdk_disabled_thinking_reaches_native_chat_completions_as_effort_none( + gateway: Gateway, +) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + client: Final = anthropic.AsyncAnthropic( + base_url=str(gateway.client.base_url), api_key=gateway.key, max_retries=0 + ) + message: Final = await client.messages.create( + model=model, + max_tokens=64, + thinking={"type": "disabled"}, + messages=[{"role": "user", "content": _question(marker)}], + extra_body=NO_CACHE, + ) + assert _native_body(wire, marker) == _native_request(marker, 64, "none") + assert message.id == f"chatcmpl-{marker}", message + assert [(block.type, getattr(block, "text", None)) for block in message.content] == [("text", answer(marker))] + assert _spend_rows(message.id, 1) == [_success_row(message.id, model)] + + +def test_identical_messages_requests_reach_the_peer_once_and_log_a_cache_hit_row(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + body: Final[dict[str, JsonValue]] = { + "model": model, + "max_tokens": 64, + "messages": [{"role": "user", "content": _question(marker)}], + } + first: Final = gateway.request("POST", "/v1/messages", body, headers=ANTHROPIC_VERSION) + assert first.status_code == 200, first.text + identity: Final = str(first.json()["id"]) + assert first.json()["content"] == [{"type": "text", "text": answer(marker)}], first.text + second: Final = gateway.request("POST", "/v1/messages", body, headers=ANTHROPIC_VERSION) + assert second.status_code == 200, second.text + assert second.json()["id"] == identity, (first.text, second.text) + assert second.json()["content"] == [{"type": "text", "text": answer(marker)}], second.text + received: Final = _carrying(wire, marker) + assert [(request.method, marker_of(request)) for request in received] == [("POST", marker)], received + rows: Final = _spend_rows(identity, 2) + assert rows[0] == _success_row(identity, model), rows + assert str(rows[1]["request_id"]).startswith(identity + "_cache_hit"), rows + assert {**rows[1], "request_id": identity, "cache_hit": "None"} == _success_row(identity, model), rows diff --git a/tests/integration/providers/test_bedrock_gpt_responses_native_wire.py b/tests/integration/providers/test_bedrock_gpt_responses_native_wire.py new file mode 100644 index 00000000000..6a365ac26ba --- /dev/null +++ b/tests/integration/providers/test_bedrock_gpt_responses_native_wire.py @@ -0,0 +1,117 @@ +import base64 +import uuid +from dataclasses import dataclass +from typing import Final + +import openai +from integration._support.bedrock_runtime_peer import NATIVE_RESPONSES, answer, respond, target_of +from integration._support.client import Gateway, Scenario, eventually +from integration._support.database import read_rows +from integration._support.wire import Request, Wire, wire_server +from openai.types.responses import ResponseCompletedEvent, ResponseTextDeltaEvent +from pydantic import JsonValue, TypeAdapter + +from litellm.proxy.common_utils.encrypt_decrypt_utils import decrypt_if_encrypted_with + +GPT: Final = "us.openai.gpt-5.6-sol" +TOKEN: Final = "synthetic-bedrock-bearer" +SALT: Final = "sk-integration-salt" +_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) + + +@dataclass(frozen=True, slots=True) +class _IssuedId: + issued: str + upstream: str + + +def _prompt(marker: str) -> str: + return f"synthetic responses request marker-{marker}" + + +def _deployment(scenario: Scenario, wire: Wire) -> str: + return scenario.model( + model=f"bedrock/{GPT}", + api_key=TOKEN, + aws_region_name="us-east-1", + aws_bedrock_runtime_endpoint=wire.url, + api_base=None, + ) + + +def _issued_id(client_id: str) -> _IssuedId: + decrypted: Final = decrypt_if_encrypted_with(client_id.removeprefix("resp_"), SALT) + assert decrypted is not None, client_id + issued: Final = decrypted.split(";")[0].split("response_id:")[-1] + decoded: Final = base64.b64decode(issued.removeprefix("resp_")).decode() + return _IssuedId(issued, decoded.split(";")[-1].removeprefix("response_id:")) + + +def _native_request(wire: Wire) -> Request: + received: Final = wire.drain() + assert [(request.method, target_of(request)) for request in received] == [("POST", NATIVE_RESPONSES)], received + assert received[0].headers["authorization"] == f"Bearer {TOKEN}", dict(received[0].headers) + return received[0] + + +def _body(request: Request) -> dict[str, JsonValue]: + return _JSON_OBJECT.validate_json(request.body) + + +def _spend_row(identity: str) -> dict[str, JsonValue]: + rows: Final = eventually( + lambda: read_rows( + 'SELECT model_group, status, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ), + lambda found: len(found) == 1, + seconds=70, + ) + return rows[0] + + +def _success_row(model: str) -> dict[str, JsonValue]: + return {"model_group": model, "status": "success", "prompt_tokens": 30, "completion_tokens": 5} + + +def test_openai_sdk_responses_request_is_served_by_the_native_responses_route(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + client: Final = openai.OpenAI(base_url=f"{gateway.client.base_url}/v1", api_key=gateway.key, max_retries=0) + raw: Final = client.responses.with_raw_response.create( + model=model, input=_prompt(marker), extra_body={"cache": {"no-cache": True}} + ) + response: Final = raw.parse() + assert response.output_text == answer(marker), raw.text + assert response.usage is not None and (response.usage.input_tokens, response.usage.output_tokens) == (30, 5) + assert _issued_id(response.id).upstream == f"resp_upstream_{marker}", response.id + request: Final = _native_request(wire) + assert _body(request) == {"model": GPT, "input": _prompt(marker)}, request.body + assert _spend_row(response.id) == _success_row(model) + + +async def test_async_openai_sdk_responses_stream_is_served_by_the_native_responses_route(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + client: Final = openai.AsyncOpenAI(base_url=f"{gateway.client.base_url}/v1", api_key=gateway.key, max_retries=0) + stream: Final = await client.responses.create( + model=model, input=_prompt(marker), stream=True, extra_body={"cache": {"no-cache": True}} + ) + events: Final = [event async for event in stream] + assert [event.type for event in events] == [ + "response.created", + "response.output_text.delta", + "response.completed", + ], events + deltas: Final = "".join(event.delta for event in events if isinstance(event, ResponseTextDeltaEvent)) + assert deltas == answer(marker), events + completed: Final = events[-1] + assert isinstance(completed, ResponseCompletedEvent), completed + assert completed.response.output_text == answer(marker), completed + issued: Final = _issued_id(completed.response.id) + assert issued.upstream == f"resp_upstream_{marker}", completed.response.id + request: Final = _native_request(wire) + assert _body(request) == {"model": GPT, "input": _prompt(marker), "stream": True}, request.body + assert _spend_row(issued.issued) == _success_row(model) diff --git a/tests/integration/providers/test_bedrock_runtime_chat_completions_chaos.py b/tests/integration/providers/test_bedrock_runtime_chat_completions_chaos.py new file mode 100644 index 00000000000..030a407bd8d --- /dev/null +++ b/tests/integration/providers/test_bedrock_runtime_chat_completions_chaos.py @@ -0,0 +1,402 @@ +import asyncio +import base64 +import binascii +import multiprocessing +import os +import re +import signal +import socket +import threading +import uuid +from collections.abc import Callable, Iterator, Mapping +from contextlib import contextmanager +from dataclasses import dataclass +from multiprocessing.process import BaseProcess +from multiprocessing.sharedctypes import Synchronized +from pathlib import Path +from queue import SimpleQueue +from types import MappingProxyType +from typing import Final, Literal +from urllib.parse import urlsplit, urlunsplit + +import httpx +import psutil +import pytest +import yaml +from integration._support.bedrock_runtime_peer import MARKER, marker_of, respond, serve_peer +from integration._support.client import Gateway, Scenario, eventually, object_value +from integration._support.database import read_rows +from integration._support.process import owned_proxy_process +from integration._support.wire import Reply, Request, Wire, wire_server +from pydantic import JsonValue, TypeAdapter + +BEDROCK_MODEL: Final = "us.openai.gpt-5.6-sol" +TOKEN: Final = "synthetic-bedrock-bearer" +_CONFIG_MODEL: Final = "bedrock-gpt-chat-completions-chaos" +_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) +_STARTED_WORKER: Final = re.compile(r"Started server process \[(\d+)\]") +_ENDPOINTS: Final[tuple["Endpoint", ...]] = ("chat", "messages", "responses") + +Endpoint = Literal["chat", "messages", "responses"] + + +@dataclass(frozen=True, slots=True) +class _Call: + endpoint: Endpoint + stream: bool + marker: str + + +@dataclass(frozen=True, slots=True) +class _Served: + call: _Call + status: int + text: str + call_id: str | None + + +@dataclass(frozen=True, slots=True) +class _ChildPeer: + process: BaseProcess + received: Synchronized[int] + url: str + + +def _path(endpoint: Endpoint) -> str: + match endpoint: + case "chat": + return "/v1/chat/completions" + case "messages": + return "/v1/messages" + case "responses": + return "/v1/responses" + + +def _terminal(endpoint: Endpoint) -> str: + match endpoint: + case "chat": + return "data: [DONE]" + case "messages": + return "event: message_stop" + case "responses": + return '"type":"response.completed"' + + +def _body(model: str, call: _Call) -> dict[str, JsonValue]: + question: Final = f"Question marker-{call.marker}" + common: Final[dict[str, JsonValue]] = {"model": model, "stream": call.stream, "cache": {"no-cache": True}} + match call.endpoint: + case "chat": + return {**common, "messages": [{"role": "user", "content": question}]} + case "messages": + return {**common, "max_tokens": 64, "messages": [{"role": "user", "content": question}]} + case "responses": + return {**common, "input": question} + + +def _deployment(scenario: Scenario, endpoint: str) -> str: + return scenario.model( + model=f"bedrock/{BEDROCK_MODEL}", + api_key=TOKEN, + api_base=None, + aws_region_name="us-east-1", + aws_bedrock_runtime_endpoint=endpoint, + ) + + +def _frames(text: str) -> tuple[dict[str, JsonValue], ...]: + return tuple( + _JSON_OBJECT.validate_json(line[6:]) + for line in text.splitlines() + if line.startswith("data: ") and line != "data: [DONE]" + ) + + +def _frame_id(frame: Mapping[str, JsonValue]) -> str | None: + if frame.get("type") == "message_start": + return str(object_value(frame["message"])["id"]) + response: Final = frame.get("response") + if isinstance(response, dict) and "id" in response: + return str(response["id"]) + identity: Final = frame.get("id") + return identity if isinstance(identity, str) else None + + +def _response_id(served: _Served) -> str: + if not served.call.stream: + return str(_JSON_OBJECT.validate_json(served.text)["id"]) + ids: Final = tuple(identity for identity in map(_frame_id, _frames(served.text)) if identity is not None) + assert ids, served.text + return ids[0] + + +def _assert_answered_with_its_own_marker(served: _Served) -> None: + assert served.status == 200, served.text + assert set(MARKER.findall(served.text)) == {served.call.marker}, served.text + if served.call.stream: + assert _terminal(served.call.endpoint) in served.text, served.text + + +def _spend_rows(model: str, expected: int) -> list[dict[str, JsonValue]]: + return eventually( + lambda: read_rows('SELECT request_id, status FROM "LiteLLM_SpendLogs" WHERE model_group=%s', (model,)), + lambda found: len(found) >= expected, + seconds=60, + ) + + +def _rows_by_status(rows: list[dict[str, JsonValue]], status: str) -> list[str]: + return sorted(str(row["request_id"]) for row in rows if row["status"] == status) + + +def _upstream_id_inside(row_id: str) -> str | None: + try: + payload: Final = base64.b64decode(row_id.removeprefix("resp_"), validate=True).decode() + except (binascii.Error, UnicodeDecodeError): + return None + return payload.rsplit("response_id:", 1)[1] if "response_id:" in payload else None + + +# TODO: a Bedrock non-stream /v1/responses spend row can carry the pre-encryption resp_ id instead of the +# ciphertext the caller received, because the spend row id is read from response_obj["id"] before the +# ResponsesIDSecurity hook rewrites it in place; such a row is matched by the upstream id inside that payload until +# that ordering is fixed on main +def _row_belongs_to(row_id: str, served: _Served) -> bool: + if row_id == _response_id(served): + return True + return served.call.endpoint == "responses" and _upstream_id_inside(row_id) == f"resp_upstream_{served.call.marker}" + + +def _assert_each_success_landed_once(rows: list[dict[str, JsonValue]], served: tuple[_Served, ...]) -> None: + success_ids: Final = _rows_by_status(rows, "success") + assert len(success_ids) == len(served), rows + for item in served: + owned: Final = [row_id for row_id in success_ids if _row_belongs_to(row_id, item)] + assert len(owned) == 1, (item.call, owned, success_ids) + + +async def _send(client: httpx.AsyncClient, key: str, model: str, call: _Call) -> _Served: + async with client.stream( + "POST", + _path(call.endpoint), + json=_body(model, call), + headers={"Authorization": f"Bearer {key}", "anthropic-version": "2023-06-01"}, + ) as response: + raw: Final = await response.aread() + return _Served( + call=call, status=response.status_code, text=raw.decode(), call_id=response.headers.get("x-litellm-call-id") + ) + + +async def _burst( + base_url: str, key: str, model: str, calls: tuple[_Call, ...], *, tolerate_transport_errors: bool = False +) -> tuple[_Served, ...]: + async with httpx.AsyncClient(base_url=base_url, timeout=60, trust_env=False) as client: + results: Final = await asyncio.gather( + *(_send(client, key, model, call) for call in calls), return_exceptions=tolerate_transport_errors + ) + for result in results: + assert not isinstance(result, BaseException) or isinstance(result, httpx.TransportError), repr(result) + return tuple(result for result in results if isinstance(result, _Served)) + + +def _calls(count: int, endpoints: tuple[Endpoint, ...], stream: Callable[[int], bool]) -> tuple[_Call, ...]: + return tuple( + _Call(endpoint=endpoints[index % len(endpoints)], stream=stream(index), marker=uuid.uuid4().hex) + for index in range(count) + ) + + +def _free_port() -> int: + with socket.socket() as reserve: + reserve.bind(("127.0.0.1", 0)) + return reserve.getsockname()[1] + + +def _accepts_connections(port: int) -> bool: + try: + with socket.create_connection(("127.0.0.1", port), timeout=0.2): + return True + except OSError: + return False + + +@contextmanager +def _child_peer(port: int, answer_first: int) -> Iterator[_ChildPeer]: + received: Final = multiprocessing.Value("i", 0) + process: Final = multiprocessing.get_context("spawn").Process( + target=serve_peer, args=(port, received, answer_first), daemon=True + ) + process.start() + try: + eventually(lambda: _accepts_connections(port), bool, seconds=30) + yield _ChildPeer(process=process, received=received, url=f"http://127.0.0.1:{port}") + finally: + process.kill() + process.join(timeout=10) + assert not process.is_alive(), "Owned peer survived cleanup" + + +async def test_burst_across_every_endpoint_lands_each_response_id_once(gateway: Gateway) -> None: + calls: Final = _calls(36, _ENDPOINTS, lambda index: index % 2 == 0) + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire.url) + served: Final = await _burst(str(gateway.client.base_url), gateway.key, model, calls) + assert len(served) == 36 + for item in served: + _assert_answered_with_its_own_marker(item) + ids: Final = sorted(_response_id(item) for item in served) + assert len(set(ids)) == 36, ids + assert sorted(marker_of(request) for request in wire.drain()) == sorted(call.marker for call in calls) + rows: Final = _spend_rows(model, 36) + _assert_each_success_landed_once(rows, served) + assert len(rows) == 36, rows + + +@pytest.mark.timeout(180) +async def test_peer_killed_mid_burst_fails_only_the_held_calls_and_a_restarted_peer_serves_again( + gateway: Gateway, +) -> None: + calls: Final = _calls(12, _ENDPOINTS, lambda index: index % 2 == 0) + recovery: Final = _calls(6, _ENDPOINTS, lambda index: index % 2 == 1) + port: Final = _free_port() + with gateway.scenario() as scenario: + model: Final = _deployment(scenario, f"http://127.0.0.1:{port}") + with _child_peer(port, answer_first=6) as peer: + burst: Final = asyncio.create_task(_burst(str(gateway.client.base_url), gateway.key, model, calls)) + await asyncio.to_thread(eventually, lambda: peer.received.value, lambda count: count == 12, 60) + peer.process.kill() + peer.process.join(timeout=10) + served: Final = await burst + succeeded: Final = tuple(item for item in served if item.status == 200) + failed: Final = tuple(item for item in served if item.status != 200) + assert (len(succeeded), len(failed)) == (6, 6), [(item.call.marker, item.status) for item in served] + for item in succeeded: + _assert_answered_with_its_own_marker(item) + assert {item.status for item in failed} == {503}, [ + (item.call.endpoint, item.call.stream, item.status, item.text) for item in failed + ] + for item in failed: + assert "ServiceUnavailableError: BedrockException - Server disconnected" in item.text, item.text + assert "marker-" not in item.text and item.call_id is not None, item.text + with _child_peer(port, answer_first=10**6) as revived: + recovered: Final = await _burst(str(gateway.client.base_url), gateway.key, model, recovery) + assert revived.received.value == 6, revived.received.value + for item in recovered: + _assert_answered_with_its_own_marker(item) + rows: Final = _spend_rows(model, 18) + _assert_each_success_landed_once(rows, (*succeeded, *recovered)) + assert _rows_by_status(rows, "failure") == sorted(str(item.call_id) for item in failed), rows + assert len(rows) == 18, rows + + +async def test_slow_peer_streams_are_forwarded_once_and_terminated(gateway: Gateway) -> None: + calls: Final = _calls(10, ("chat",), lambda _: True) + with wire_server(lambda request: respond(request, pause=0.3)) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire.url) + served: Final = await _burst(str(gateway.client.base_url), gateway.key, model, calls) + assert len(served) == 10 + for item in served: + _assert_answered_with_its_own_marker(item) + assert sorted(marker_of(request) for request in wire.drain()) == sorted(call.marker for call in calls) + ids: Final = sorted(_response_id(item) for item in served) + rows: Final = _spend_rows(model, 10) + assert _rows_by_status(rows, "success") == ids, rows + assert len(rows) == 10, rows + + +def _chaos_config(wire: Wire, tmp_path: Path) -> Path: + config: Final = yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + config["model_list"] = [ + { + "model_name": _CONFIG_MODEL, + "litellm_params": { + "model": f"bedrock/{BEDROCK_MODEL}", + "api_key": TOKEN, + "aws_region_name": "us-east-1", + "aws_bedrock_runtime_endpoint": wire.url, + }, + } + ] + path: Final = tmp_path / "bedrock-gpt-chat-completions-chaos.yaml" + path.write_text(yaml.safe_dump(config)) + return path + + +def _pooled_database_url() -> str: + parts: Final = urlsplit(os.environ["DATABASE_URL"]) + query: Final = "&".join(part for part in (parts.query, "connection_limit=5") if part) + return urlunsplit(parts._replace(query=query)) + + +def _open_upstream_connections(pid: int, upstream: str) -> int: + port: Final = urlsplit(upstream).port + return sum( + 1 + for connection in psutil.Process(pid).net_connections(kind="tcp") + if connection.status == psutil.CONN_ESTABLISHED and connection.raddr and connection.raddr.port == port + ) + + +def _landed_once(ids: tuple[str, ...]) -> list[dict[str, JsonValue]]: + return eventually( + lambda: read_rows( + 'SELECT request_id, status FROM "LiteLLM_SpendLogs" WHERE request_id = ANY(%s)', + (list(ids),), # pyright: ignore[reportArgumentType] # psycopg adapts the list to a text array + ), + lambda found: len(found) >= len(ids), + seconds=60, + ) + + +@pytest.mark.timeout(180) +async def test_worker_sigkill_mid_burst_leaves_the_sibling_serving(gateway: Gateway, tmp_path: Path) -> None: + calls: Final = _calls(20, ("chat",), lambda _: False) + release: Final = threading.Event() + held_markers: Final[SimpleQueue[str]] = SimpleQueue() + + def held(request: Request) -> Reply: + held_markers.put(marker_of(request)) + assert release.wait(timeout=60), "The burst was never released" + return respond(request) + + with wire_server(held) as wire: + path: Final = _chaos_config(wire, tmp_path) + overrides: Final = {"DATABASE_URL": _pooled_database_url()} + with owned_proxy_process(gateway, tmp_path, overrides, config=path, workers=2) as owned: + candidate: Final = owned.gateway + workers: Final = eventually( + lambda: tuple(int(pid) for pid in _STARTED_WORKER.findall(owned.log.read_text())), + lambda pids: len(pids) == 2, + seconds=30, + ) + burst: Final = asyncio.create_task( + _burst( + str(candidate.client.base_url), candidate.key, _CONFIG_MODEL, calls, tolerate_transport_errors=True + ) + ) + await asyncio.to_thread(eventually, held_markers.qsize, lambda size: size == 20, 60) + held_by: Final = MappingProxyType({pid: _open_upstream_connections(pid, wire.url) for pid in workers}) + assert sum(held_by.values()) == 20, held_by + victim_pid, survivor_pid = sorted(workers, key=held_by.__getitem__) + victim: Final = psutil.Process(victim_pid) + victim.suspend() + victim.send_signal(signal.SIGKILL) + release.set() + served: Final = await burst + assert held_by[survivor_pid] >= 10, held_by + assert len(served) == held_by[survivor_pid], (held_by, len(served)) + for item in served: + _assert_answered_with_its_own_marker(item) + follow_up: Final = _Call(endpoint="chat", stream=False, marker=uuid.uuid4().hex) + (answered,) = await _burst(str(candidate.client.base_url), candidate.key, _CONFIG_MODEL, (follow_up,)) + _assert_answered_with_its_own_marker(answered) + received: Final = wire.drain() + assert {request.method for request in received} == {"POST"}, received + assert sorted(marker_of(request) for request in received) == sorted( + call.marker for call in (*calls, follow_up) + ) + ids: Final = tuple(sorted(_response_id(item) for item in (*served, answered))) + rows: Final = _landed_once(ids) + assert _rows_by_status(rows, "success") == list(ids), rows + assert len(rows) == len(ids), rows diff --git a/tests/integration/providers/test_bedrock_runtime_chat_completions_sad_wire.py b/tests/integration/providers/test_bedrock_runtime_chat_completions_sad_wire.py new file mode 100644 index 00000000000..a96fe884c59 --- /dev/null +++ b/tests/integration/providers/test_bedrock_runtime_chat_completions_sad_wire.py @@ -0,0 +1,415 @@ +import json +import os +import time +import uuid +from collections.abc import Mapping +from concurrent.futures import ThreadPoolExecutor +from hashlib import sha256 +from pathlib import Path +from types import MappingProxyType +from typing import Final +from urllib.parse import urlsplit, urlunsplit + +import httpx +import pytest +import yaml +from integration._support.bedrock_runtime_peer import answer, forwarded_effort, marker_of, respond, target_of +from integration._support.client import Gateway, Scenario, eventually, object_value, string_value +from integration._support.database import read_rows +from integration._support.process import owned_proxy_process +from integration._support.wire import Request, Wire, wire_server +from pydantic import JsonValue, TypeAdapter + +GPT: Final = "us.openai.gpt-5.6-sol" +TOKEN: Final = "synthetic-bedrock-bearer" +BAD_KEY: Final = "sk-synthetic-bad-key" +NATIVE_TARGET: Final = "/openai/v1/chat/completions" +CONVERSE_TARGET: Final = f"/model/{GPT}/converse" +LONG_VERSION_GPT: Final = "openai.gpt-" + "1" * 30000 +PNG_DATA_URL: Final = ( + "data:image/png;base64," + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR4nGP4z8DwHwAFAAH/iZk9HQAAAABJRU5ErkJggg==" +) +GPT_DEPLOYMENT: Final[Mapping[str, JsonValue]] = MappingProxyType( + {"model": f"bedrock/{GPT}", "api_key": TOKEN, "aws_region_name": "us-east-1"} +) +_ALLOWLISTED_MODEL: Final = "bedrock-gpt-image-allowlist" +_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) + + +def _prompt(marker: str) -> str: + return f"synthetic sad request marker-{marker}" + + +def _messages(marker: str) -> list[dict[str, JsonValue]]: + return [{"role": "user", "content": _prompt(marker)}] + + +def _image_messages(marker: str, url: str) -> list[dict[str, JsonValue]]: + return [ + { + "role": "user", + "content": [{"type": "text", "text": _prompt(marker)}, {"type": "image_url", "image_url": {"url": url}}], + } + ] + + +def _deployment(scenario: Scenario, wire: Wire, **overrides: JsonValue) -> str: + return scenario.model(**{**GPT_DEPLOYMENT, "aws_bedrock_runtime_endpoint": wire.url, **overrides}) + + +def _chat(gateway: Gateway, model: str, marker: str, *, key: str | None = None, **params: JsonValue) -> httpx.Response: + return gateway.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": _messages(marker), "cache": {"no-cache": True}, **params}, + key=key, + ) + + +def _payload(response: httpx.Response) -> dict[str, JsonValue]: + assert response.status_code == 200, response.text + return _JSON_OBJECT.validate_json(response.content) + + +def _content(response: httpx.Response) -> JsonValue: + choices: Final = _payload(response)["choices"] + assert isinstance(choices, list), response.text + return object_value(object_value(choices[0])["message"])["content"] + + +def _error_message(response: httpx.Response) -> str: + return string_value(object_value(_JSON_OBJECT.validate_json(response.content)["error"])["message"]) + + +def _call_id(response: httpx.Response) -> str: + return response.headers["x-litellm-call-id"] + + +def _body(request: Request) -> dict[str, JsonValue]: + return _JSON_OBJECT.validate_json(request.body) + + +def _routes(received: tuple[Request, ...]) -> list[tuple[str, str]]: + return [(request.method, target_of(request)) for request in received] + + +def _only_request(wire: Wire, marker: str) -> Request: + received: Final = wire.drain() + assert len(received) == 1, _routes(received) + assert marker_of(received[0]) == marker, received[0].body + return received[0] + + +def _spend_rows(identity: str) -> list[dict[str, JsonValue]]: + return read_rows( + 'SELECT request_id, model_group, status, cache_hit, spend FROM "LiteLLM_SpendLogs" WHERE request_id=%s', + (identity,), + ) + + +def _spend_row(identity: str) -> dict[str, JsonValue]: + return eventually(lambda: _spend_rows(identity), lambda found: len(found) == 1, seconds=70)[0] + + +def _assert_row(identity: str, model: str, status: str) -> None: + row: Final = _spend_row(identity) + assert (row["model_group"], row["status"]) == (model, status), row + + +def _timed_liveliness(gateway: Gateway) -> tuple[int, float]: + started: Final = time.monotonic() + response: Final = gateway.request("GET", "/health/liveliness") + return response.status_code, time.monotonic() - started + + +def _pooled_database_url(url: str) -> str: + parts: Final = urlsplit(url) + query: Final = "&".join(part for part in (parts.query, "connection_limit=5") if part) + return urlunsplit(parts._replace(query=query)) + + +def _allowlist_config(wire: Wire, tmp_path: Path) -> Path: + config: Final = _JSON_OBJECT.validate_python( + yaml.safe_load(Path("tests/integration/proxy_config.yaml").read_text()) + ) + path: Final = tmp_path / "bedrock-gpt-image-allowlist.yaml" + path.write_text( + yaml.safe_dump( + { + **config, + "model_list": [ + { + "model_name": _ALLOWLISTED_MODEL, + "litellm_params": {**GPT_DEPLOYMENT, "aws_bedrock_runtime_endpoint": wire.url}, + } + ], + "general_settings": { + **object_value(config["general_settings"]), + "user_url_allowed_hosts": ["127.0.0.1"], + }, + } + ) + ) + return path + + +def test_remote_image_url_on_the_shared_proxy_is_rejected_before_any_fetch(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = gateway.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": _image_messages(marker, f"{wire.url}/image.png"), "cache": {"no-cache": True}}, + ) + assert response.status_code == 400, response.text + message: Final = _error_message(response) + assert "Unable to fetch image from URL" in message and "user_url_allowed_hosts" in message, response.text + _assert_row(_call_id(response), model, "failure") + assert _routes(wire.drain()) == [] + + +@pytest.mark.timeout(180) +def test_allowlisted_remote_image_is_inlined_for_the_native_route(gateway: Gateway, tmp_path: Path) -> None: + marker: Final = uuid.uuid4().hex + missing_marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire: + path: Final = _allowlist_config(wire, tmp_path) + overrides: Final = {"DATABASE_URL": _pooled_database_url(os.environ["DATABASE_URL"])} + with owned_proxy_process(gateway, tmp_path, overrides, config=path) as owned: + candidate: Final = owned.gateway + response: Final = candidate.request( + "POST", + "/v1/chat/completions", + { + "model": _ALLOWLISTED_MODEL, + "messages": _image_messages(marker, f"{wire.url}/image.png"), + "cache": {"no-cache": True}, + }, + ) + assert _content(response) == answer(marker), response.text + received: Final = wire.drain() + assert _routes(received) == [("GET", "/image.png"), ("POST", NATIVE_TARGET)], received + assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text + assert _body(received[1]) == { + "model": GPT, + "messages": _image_messages(marker, PNG_DATA_URL), + "stream": False, + }, received[1].body + _assert_row(f"chatcmpl-{marker}", _ALLOWLISTED_MODEL, "success") + missing: Final = candidate.request( + "POST", + "/v1/chat/completions", + { + "model": _ALLOWLISTED_MODEL, + "messages": _image_messages(missing_marker, f"{wire.url}/missing.png"), + "cache": {"no-cache": True}, + }, + ) + assert missing.status_code == 400, missing.text + assert "Unable to fetch image from URL. Status code: 404" in _error_message(missing), missing.text + _assert_row(_call_id(missing), _ALLOWLISTED_MODEL, "failure") + assert _routes(wire.drain()) == [("GET", "/missing.png")] + + +def test_response_cache_twin_serves_the_second_request_without_a_second_wire_call(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + body: Final[dict[str, JsonValue]] = {"model": model, "messages": _messages(marker)} + first: Final = gateway.request("POST", "/v1/chat/completions", body) + second: Final = gateway.request("POST", "/v1/chat/completions", body) + identity: Final = string_value(_payload(first)["id"]) + assert _content(first) == answer(marker), first.text + assert _payload(second)["id"] == identity, (first.text, second.text) + assert _content(second) == answer(marker), second.text + _only_request(wire, marker) + rows: Final = eventually( + lambda: read_rows( + 'SELECT request_id, cache_hit, spend FROM "LiteLLM_SpendLogs" WHERE starts_with(request_id, %s)' + " ORDER BY request_id", + (identity,), + ), + lambda found: len(found) == 2, + seconds=70, + ) + assert [(row["request_id"] == identity, row["cache_hit"]) for row in rows] == [(True, "None"), (False, "True")] + assert string_value(rows[1]["request_id"]).startswith(f"{identity}_cache_hit"), rows + assert rows[1]["spend"] == 0.0, rows + assert isinstance(rows[0]["spend"], float) and rows[0]["spend"] > 0.0, rows + + +def test_model_group_info_lists_the_native_supported_params(gateway: Gateway) -> None: + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + groups: Final = gateway.get("/model_group/info", {"model_group": model})["data"] + assert isinstance(groups, list) and len(groups) == 1, groups + group: Final = object_value(groups[0]) + assert group["model_group"] == model, group + params: Final = group["supported_openai_params"] + assert isinstance(params, list), group + assert {"reasoning_effort", "logprobs", "top_logprobs"} <= set(params) and "n" not in params, params + assert _routes(wire.drain()) == [] + + +def test_thirty_thousand_digit_version_is_classified_quickly_and_served_by_converse(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario, ThreadPoolExecutor(max_workers=1) as pool: + model: Final = _deployment(scenario, wire, model=f"bedrock/{LONG_VERSION_GPT}") + liveliness: Final = pool.submit(_timed_liveliness, gateway) + started: Final = time.monotonic() + response: Final = _chat(gateway, model, marker) + elapsed: Final = time.monotonic() - started + health_status, health_elapsed = liveliness.result() + assert _content(response) == answer(marker), response.text + assert elapsed < 10, elapsed + assert (health_status, health_elapsed < 2) == (200, True), (health_status, health_elapsed) + request: Final = _only_request(wire, marker) + assert (request.method, target_of(request)) == ("POST", f"/model/{LONG_VERSION_GPT}/converse"), request.target + _assert_row(string_value(_payload(response)["id"]), model, "success") + + +def test_bad_key_on_the_long_version_model_is_refused_before_any_route(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + control_marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire, model=f"bedrock/{LONG_VERSION_GPT}") + started: Final = time.monotonic() + refused: Final = _chat(gateway, model, marker, key=BAD_KEY) + elapsed: Final = time.monotonic() - started + assert refused.status_code == 401, refused.text + assert elapsed < 2, elapsed + assert "Authentication Error" in _error_message(refused), refused.text + refused_rows: Final = eventually( + lambda: read_rows( + "SELECT request_id, status, spend, metadata->'error_information'->>'error_code' AS error_code" + ' FROM "LiteLLM_SpendLogs" WHERE model_group=%s AND api_key=%s', + (model, sha256(BAD_KEY.encode()).hexdigest()), + ), + lambda found: len(found) == 1, + seconds=70, + ) + assert (refused_rows[0]["status"], refused_rows[0]["spend"], refused_rows[0]["error_code"]) == ( + "failure", + 0.0, + "401", + ), refused_rows + control: Final = _chat(gateway, model, control_marker) + control_id: Final = string_value(_payload(control)["id"]) + _assert_row(control_id, model, "success") + landed: Final = read_rows('SELECT request_id FROM "LiteLLM_SpendLogs" WHERE model_group=%s', (model,)) + assert {row["request_id"] for row in landed} == {control_id, refused_rows[0]["request_id"]}, landed + received: Final = wire.drain() + assert [marker_of(request) for request in received] == [control_marker], _routes(received) + + +@pytest.mark.parametrize( + "effort", + [ + pytest.param(7, id="int"), + pytest.param(["high"], id="list"), + pytest.param("", id="empty"), + pytest.param("x" * 5120, id="five_kb"), + ], +) +def test_invalid_reasoning_effort_reaches_the_peer_and_its_400_reaches_the_caller( + gateway: Gateway, effort: JsonValue +) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, reasoning_effort=effort) + assert response.status_code == 400, response.text + peer_error: Final = json.dumps({"message": f"Invalid reasoning effort: {json.dumps(effort)}"}) + assert f"BedrockException - {peer_error}" in _error_message(response), response.text + request: Final = _only_request(wire, marker) + assert forwarded_effort(request) == effort, request.body + _assert_row(_call_id(response), model, "failure") + + +def test_duplicated_reasoning_effort_key_lets_the_last_value_win(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + prefix: Final = json.dumps({"model": model, "messages": _messages(marker), "cache": {"no-cache": True}})[:-1] + response: Final = gateway.client.post( + "/v1/chat/completions", + content=f'{prefix}, "reasoning_effort": "low", "reasoning_effort": "high"}}'.encode(), + headers={"Authorization": f"Bearer {gateway.key}", "content-type": "application/json"}, + ) + assert _content(response) == answer(marker), response.text + request: Final = _only_request(wire, marker) + assert forwarded_effort(request) == "high", request.body + _assert_row(string_value(_payload(response)["id"]), model, "success") + + +def test_string_temperature_is_refused_before_any_wire_request(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, temperature="0.2") + assert response.status_code == 400, response.text + message: Final = _error_message(response) + assert message.startswith("litellm.UnsupportedParamsError") and "['temperature']" in message, response.text + _assert_row(_call_id(response), model, "failure") + assert _routes(wire.drain()) == [] + + +@pytest.mark.parametrize( + ("scripted", "expected"), + [pytest.param(401, 401, id="401"), pytest.param(429, 429, id="429"), pytest.param(500, 503, id="500")], +) +def test_peer_error_status_reaches_the_caller_and_unrelated_deployments_keep_serving( + gateway: Gateway, scripted: int, expected: int +) -> None: + marker: Final = uuid.uuid4().hex + control_marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + unrelated: Final = scenario.model() + response: Final = gateway.request( + "POST", + "/v1/chat/completions", + { + "model": model, + "messages": [{"role": "user", "content": f"status={scripted} marker-{marker}"}], + "cache": {"no-cache": True}, + }, + ) + assert response.status_code == expected, response.text + assert f'BedrockException - {{"message": "scripted {scripted}"}}' in _error_message(response), response.text + _only_request(wire, marker) + _assert_row(_call_id(response), model, "failure") + control: Final = _chat(gateway, unrelated, control_marker) + assert control.status_code == 200, control.text + _assert_row(string_value(_payload(control)["id"]), unrelated, "success") + assert _routes(wire.drain()) == [] + + +@pytest.mark.parametrize( + "params", [pytest.param({"reasoning_effort": None}, id="null"), pytest.param({}, id="missing")] +) +def test_absent_reasoning_effort_is_forwarded_as_absent_on_every_repeat( + gateway: Gateway, params: dict[str, JsonValue] +) -> None: + markers: Final = tuple(uuid.uuid4().hex for _ in range(3)) + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + responses: Final = tuple(_chat(gateway, model, marker, **params) for marker in markers) + assert [_content(response) for response in responses] == [answer(marker) for marker in markers] + ids: Final = tuple(string_value(_payload(response)["id"]) for response in responses) + assert len(set(ids)) == 3, ids + received: Final = wire.drain() + assert [marker_of(request) for request in received] == list(markers), _routes(received) + assert [forwarded_effort(request) for request in received] == [None, None, None], [_body(r) for r in received] + rows: Final = eventually( + lambda: read_rows( + 'SELECT request_id, status FROM "LiteLLM_SpendLogs" WHERE request_id IN (%s, %s, %s)', ids + ), + lambda found: len(found) == 3, + seconds=70, + ) + assert {(string_value(row["request_id"]), row["status"]) for row in rows} == { + (identity, "success") for identity in ids + }, rows diff --git a/tests/integration/providers/test_bedrock_runtime_chat_completions_wire.py b/tests/integration/providers/test_bedrock_runtime_chat_completions_wire.py new file mode 100644 index 00000000000..d44d9f154ec --- /dev/null +++ b/tests/integration/providers/test_bedrock_runtime_chat_completions_wire.py @@ -0,0 +1,549 @@ +import json +import uuid +from collections.abc import Mapping, Sequence +from types import MappingProxyType +from typing import Final +from urllib.parse import quote + +import httpx +import openai +import pytest +from integration._support.bedrock_runtime_peer import answer, respond, target_of +from integration._support.client import Gateway, Scenario, eventually +from integration._support.database import read_rows +from integration._support.sigv4 import signature +from integration._support.wire import Request, Wire, wire_server +from openai.types.chat import ChatCompletionChunk, ChatCompletionMessageParam +from openai.types.chat.chat_completion_chunk import ChoiceDelta +from pydantic import JsonValue, TypeAdapter + +GPT: Final = "us.openai.gpt-5.6-sol" +GLOBAL_GPT: Final = "global.openai.gpt-5.6-sol" +GPT_OSS: Final = "openai.gpt-oss-120b-1:0" +TOKEN: Final = "synthetic-bedrock-bearer" +ACCESS_KEY: Final = "AKIASYNTHETICKEY0001" +SECRET_KEY: Final = "synthetic-secret-key-for-testing" +PROFILE_ARN: Final = "arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/a1b2c3d4e5f6" +NATIVE_TARGET: Final = "/openai/v1/chat/completions" +CONVERSE_TARGET: Final = f"/model/{GPT}/converse" +GPT_DEPLOYMENT: Final[Mapping[str, JsonValue]] = MappingProxyType( + {"model": f"bedrock/{GPT}", "api_key": TOKEN, "aws_region_name": "us-east-1"} +) +GUARDRAIL: Final[Mapping[str, JsonValue]] = MappingProxyType( + {"guardrailIdentifier": "gr-synthetic", "guardrailVersion": "1"} +) +TOOL_PARAMETERS: Final[Mapping[str, JsonValue]] = MappingProxyType( + {"type": "object", "properties": {"id": {"type": "string"}}, "required": ["id"]} +) +TOOL: Final[Mapping[str, JsonValue]] = MappingProxyType( + { + "type": "function", + "function": { + "name": "lookup_invoice", + "description": "Look up an invoice", + "parameters": dict(TOOL_PARAMETERS), + }, + } +) +CONVERSE_TOOL: Final[Mapping[str, JsonValue]] = MappingProxyType( + { + "toolSpec": { + "inputSchema": {"json": dict(TOOL_PARAMETERS)}, + "name": "lookup_invoice", + "description": "Look up an invoice", + } + } +) +JSON_SCHEMA: Final[Mapping[str, JsonValue]] = MappingProxyType( + { + "type": "json_schema", + "json_schema": { + "name": "verdict", + "strict": True, + "schema": { + "type": "object", + "properties": {"ok": {"type": "boolean"}}, + "required": ["ok"], + "additionalProperties": False, + }, + }, + } +) +_JSON_OBJECT: Final = TypeAdapter(dict[str, JsonValue]) +_OBSERVATIONS: Final = TypeAdapter(list[dict[str, JsonValue]]) + + +def _prompt(marker: str) -> str: + return f"synthetic native request marker-{marker}" + + +def _messages(marker: str) -> list[JsonValue]: + return [{"role": "user", "content": _prompt(marker)}] + + +def _sdk_messages(marker: str) -> list[ChatCompletionMessageParam]: + return [{"role": "user", "content": _prompt(marker)}] + + +def _converse_messages(marker: str) -> list[JsonValue]: + return [{"role": "user", "content": [{"text": _prompt(marker)}]}] + + +def _native_body(model: str, marker: str, **params: JsonValue) -> dict[str, JsonValue]: + return {"model": model, "messages": _messages(marker), "stream": False, **params} + + +def _streamed_native_body(model: str, marker: str) -> dict[str, JsonValue]: + return _native_body(model, marker, stream=True, stream_options={"include_usage": True}) + + +def _deployment(scenario: Scenario, wire: Wire, **overrides: JsonValue) -> str: + return scenario.model(model_info=None, **{**GPT_DEPLOYMENT, "aws_bedrock_runtime_endpoint": wire.url, **overrides}) + + +def _openai_client(gateway: Gateway) -> openai.OpenAI: + return openai.OpenAI(base_url=str(gateway.client.base_url) + "/v1", api_key=gateway.key, max_retries=0) + + +def _async_openai_client(gateway: Gateway) -> openai.AsyncOpenAI: + return openai.AsyncOpenAI(base_url=str(gateway.client.base_url) + "/v1", api_key=gateway.key, max_retries=0) + + +def _chat(gateway: Gateway, model: str, marker: str, **params: JsonValue) -> httpx.Response: + return gateway.request( + "POST", + "/v1/chat/completions", + {"model": model, "messages": _messages(marker), "cache": {"no-cache": True}, **params}, + ) + + +def _payload(response: httpx.Response) -> dict[str, JsonValue]: + assert response.status_code == 200, response.text + return _JSON_OBJECT.validate_json(response.content) + + +def _only_request(wire: Wire) -> Request: + received: Final = wire.drain() + assert len(received) == 1, [(request.method, target_of(request)) for request in received] + return received[0] + + +def _body(request: Request) -> dict[str, JsonValue]: + return _JSON_OBJECT.validate_json(request.body) + + +def _native_request(wire: Wire) -> Request: + request: Final = _only_request(wire) + assert (request.method, target_of(request)) == ("POST", NATIVE_TARGET), request.target + assert request.headers["authorization"] == f"Bearer {TOKEN}", dict(request.headers) + return request + + +def _converse_request(wire: Wire, target: str = CONVERSE_TARGET) -> Request: + request: Final = _only_request(wire) + assert (request.method, target_of(request)) == ("POST", target), request.target + assert request.headers["authorization"] == f"Bearer {TOKEN}", dict(request.headers) + return request + + +def _spend_row(identity: str) -> dict[str, JsonValue]: + rows: Final = eventually( + lambda: read_rows( + 'SELECT model_group, status, prompt_tokens, completion_tokens, api_base FROM "LiteLLM_SpendLogs"' + " WHERE request_id=%s", + (identity,), + ), + lambda found: len(found) == 1, + seconds=70, + ) + return rows[0] + + +def _success_row(model: str, api_base: str) -> dict[str, JsonValue]: + return {"model_group": model, "status": "success", "prompt_tokens": 9, "completion_tokens": 5, "api_base": api_base} + + +def _delta_text(delta: ChoiceDelta, field: str) -> str: + value: Final = delta.model_dump().get(field) + return value if isinstance(value, str) else "" + + +def _chunk_text(chunk: ChatCompletionChunk, field: str) -> str: + return "".join(_delta_text(choice.delta, field) for choice in chunk.choices) + + +def _joined(chunks: Sequence[ChatCompletionChunk], field: str) -> str: + return "".join(_chunk_text(chunk, field) for chunk in chunks) + + +def _upstream_requests_mentioning(gateway: Gateway, marker: str) -> list[dict[str, JsonValue]]: + observed: Final = httpx.get(f"{gateway.upstream_url}/__observations", trust_env=False, timeout=15) + observed.raise_for_status() + requests: Final = _OBSERVATIONS.validate_python(_JSON_OBJECT.validate_json(observed.content)["requests"]) + return [request for request in requests if marker in json.dumps(request["body"])] + + +def _authorization_field(part: str) -> tuple[str, str]: + name, _, value = part.partition("=") + return name, value + + +def _assert_sigv4_signed(request: Request, path: str) -> None: + authorization: Final = request.headers["authorization"] + assert authorization.startswith("AWS4-HMAC-SHA256 "), dict(request.headers) + fields: Final = dict( + _authorization_field(part) for part in authorization.removeprefix("AWS4-HMAC-SHA256 ").split(", ") + ) + access_key, scope = fields["Credential"].split("/", 1) + assert access_key == ACCESS_KEY, authorization + assert scope == f"{request.headers['x-amz-date'][:8]}/us-east-1/bedrock/aws4_request", authorization + assert {"host", "x-amz-date"}.issubset(fields["SignedHeaders"].split(";")), authorization + expected: Final = signature("POST", path, request.headers, fields["SignedHeaders"], request.body, SECRET_KEY, scope) + assert fields["Signature"] == expected[1], authorization + + +def test_openai_sdk_reasoning_request_is_served_by_native_chat_completions(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + raw: Final = _openai_client(gateway).chat.completions.with_raw_response.create( + model=model, + messages=_sdk_messages(marker), + reasoning_effort="high", + max_tokens=16, + extra_body={"cache": {"no-cache": True}}, + ) + completion: Final = raw.parse() + assert completion.id == f"chatcmpl-{marker}", raw.text + assert completion.choices[0].message.content == answer(marker), raw.text + assert completion.usage is not None and completion.usage.model_dump(exclude_none=True) == { + "prompt_tokens": 9, + "completion_tokens": 5, + "total_tokens": 14, + "completion_tokens_details": {"reasoning_tokens": 3}, + }, raw.text + assert raw.headers["llm_provider-x-amzn-requestid"] == marker, dict(raw.headers) + request: Final = _native_request(wire) + assert _body(request) == _native_body(GPT, marker, max_completion_tokens=16, reasoning_effort="high") + assert _spend_row(completion.id) == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +async def test_async_openai_sdk_stream_keeps_the_upstream_id_and_usage(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + identity: Final = f"chatcmpl-{marker}" + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + stream: Final = await _async_openai_client(gateway).chat.completions.create( + model=model, + messages=_sdk_messages(marker), + stream=True, + stream_options={"include_usage": True}, + extra_body={"cache": {"no-cache": True}}, + ) + chunks: Final = [chunk async for chunk in stream] + assert {chunk.id for chunk in chunks} == {identity}, chunks + assert _joined(chunks, "content") == answer(marker), chunks + usage: Final = chunks[-1].usage + assert usage is not None and (usage.prompt_tokens, usage.completion_tokens) == (9, 5), chunks[-1] + assert usage.completion_tokens_details is not None and usage.completion_tokens_details.reasoning_tokens == 3 + assert all(chunk.usage is None for chunk in chunks[:-1]), chunks + assert _body(_native_request(wire)) == _streamed_native_body(GPT, marker) + assert _spend_row(identity) == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +def test_temperature_is_forwarded_natively_when_reasoning_is_off(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, temperature=0.2, reasoning_effort="none") + payload: Final = _payload(response) + assert payload["id"] == f"chatcmpl-{marker}", response.text + assert _body(_native_request(wire)) == _native_body(GPT, marker, temperature=0.2, reasoning_effort="none") + assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +def test_temperature_while_reasoning_is_refused_before_any_wire_request(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, temperature=0.2, reasoning_effort="high") + assert response.status_code == 400, response.text + assert "UnsupportedParamsError" in response.text and "'temperature'" in response.text, response.text + assert wire.drain() == (), response.text + row: Final = _spend_row(response.headers["x-litellm-call-id"]) + assert (row["status"], row["model_group"], row["prompt_tokens"]) == ("failure", model, 0), row + assert "while reasoning is active" in response.text, response.text + + +def test_drop_params_deployment_drops_temperature_while_reasoning(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire, drop_params=True) + response: Final = _chat(gateway, model, marker, temperature=0.2, reasoning_effort="high") + assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text + assert _body(_native_request(wire)) == _native_body(GPT, marker, reasoning_effort="high") + assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +def test_guardrail_config_keeps_converse(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, guardrailConfig=dict(GUARDRAIL)) + payload: Final = _payload(response) + assert payload["choices"] == [ + {"finish_reason": "stop", "index": 0, "message": {"content": answer(marker), "role": "assistant"}} + ], response.text + assert response.headers["llm_provider-x-amzn-requestid"] == marker, dict(response.headers) + body: Final = _body(_converse_request(wire)) + assert body["guardrailConfig"] == GUARDRAIL, body + assert body["messages"] == [ + {"role": "user", "content": [{"guardContent": {"text": {"text": _prompt(marker)}}}]} + ], body + assert _spend_row(str(payload["id"])) == _success_row(model, f"{wire.url}{CONVERSE_TARGET}") + + +def test_converse_prefix_pins_the_model_to_converse(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire, model=f"bedrock/converse/{GPT}") + response: Final = _chat(gateway, model, marker, reasoning_effort="high") + payload: Final = _payload(response) + assert payload["choices"] == [ + {"finish_reason": "stop", "index": 0, "message": {"content": answer(marker), "role": "assistant"}} + ], response.text + body: Final = _body(_converse_request(wire)) + assert body["messages"] == _converse_messages(marker), body + assert body["additionalModelRequestFields"] == {"reasoning": {"effort": "high"}}, body + assert _spend_row(str(payload["id"])) == _success_row(model, f"{wire.url}{CONVERSE_TARGET}") + + +def test_application_inference_profile_arn_keeps_converse(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire, model=f"bedrock/{PROFILE_ARN}") + response: Final = _chat(gateway, model, marker) + payload: Final = _payload(response) + assert payload["choices"] == [ + {"finish_reason": "stop", "index": 0, "message": {"content": answer(marker), "role": "assistant"}} + ], response.text + request: Final = _converse_request(wire, f"/model/{PROFILE_ARN}/converse") + assert request.target == f"/model/{quote(PROFILE_ARN, safe='')}/converse", request.target + assert _body(request)["messages"] == _converse_messages(marker), request.body + assert _spend_row(str(payload["id"])) == _success_row( + model, f"{wire.url}/model/{quote(PROFILE_ARN, safe='')}/converse" + ) + + +def test_model_id_application_inference_profile_keeps_converse_at_the_profile_url(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire, model_id=PROFILE_ARN) + response: Final = _chat(gateway, model, marker) + payload: Final = _payload(response) + assert payload["choices"] == [ + {"finish_reason": "stop", "index": 0, "message": {"content": answer(marker), "role": "assistant"}} + ], response.text + request: Final = _converse_request(wire, f"/model/{PROFILE_ARN}/converse") + assert request.target == f"/model/{quote(PROFILE_ARN, safe='')}/converse", request.target + body: Final = _body(request) + assert body["messages"] == _converse_messages(marker), request.body + assert "model_id" not in body and "model" not in body, request.body + assert _spend_row(str(payload["id"])) == _success_row( + model, f"{wire.url}/model/{quote(PROFILE_ARN, safe='')}/converse" + ) + + +def test_stop_sequences_keep_converse(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, stop=["END"]) + payload: Final = _payload(response) + assert payload["choices"] == [ + {"finish_reason": "stop", "index": 0, "message": {"content": answer(marker), "role": "assistant"}} + ], response.text + body: Final = _body(_converse_request(wire)) + assert body["messages"] == _converse_messages(marker), body + assert body["inferenceConfig"] == {"stopSequences": ["END"]}, body + assert _spend_row(str(payload["id"])) == _success_row(model, f"{wire.url}{CONVERSE_TARGET}") + + +def test_json_object_response_format_keeps_converse(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, response_format={"type": "json_object"}) + payload: Final = _payload(response) + assert payload["choices"] == [ + {"finish_reason": "stop", "index": 0, "message": {"content": answer(marker), "role": "assistant"}} + ], response.text + assert _body(_converse_request(wire))["messages"] == _converse_messages(marker), response.text + assert _spend_row(str(payload["id"])) == _success_row(model, f"{wire.url}{CONVERSE_TARGET}") + + +def test_json_schema_response_format_is_forwarded_natively(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, response_format=dict(JSON_SCHEMA)) + assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text + assert _body(_native_request(wire)) == _native_body(GPT, marker, response_format=dict(JSON_SCHEMA)) + assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +def test_tools_while_reasoning_keep_converse(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, tools=[dict(TOOL)], reasoning_effort="high") + payload: Final = _payload(response) + assert payload["choices"] == [ + {"finish_reason": "stop", "index": 0, "message": {"content": answer(marker), "role": "assistant"}} + ], response.text + body: Final = _body(_converse_request(wire)) + assert body["toolConfig"] == {"tools": [CONVERSE_TOOL]}, body + assert body["additionalModelRequestFields"] == {"reasoning": {"effort": "high"}}, body + assert _spend_row(str(payload["id"])) == _success_row(model, f"{wire.url}{CONVERSE_TARGET}") + + +def test_tools_with_reasoning_off_are_forwarded_natively(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, tools=[dict(TOOL)], reasoning_effort="none") + assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text + assert _body(_native_request(wire)) == _native_body(GPT, marker, tools=[dict(TOOL)], reasoning_effort="none") + assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +def test_empty_tools_list_while_reasoning_stays_native(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, tools=[], reasoning_effort="high") + assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text + assert _body(_native_request(wire)) == _native_body(GPT, marker, tools=[], reasoning_effort="high") + assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +def test_chat_completions_prefix_splits_gpt_oss_reasoning_tag(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire, model=f"bedrock/chat_completions/{GPT_OSS}") + raw: Final = _openai_client(gateway).chat.completions.with_raw_response.create( + model=model, messages=_sdk_messages(marker), extra_body={"cache": {"no-cache": True}} + ) + completion: Final = raw.parse() + assert completion.id == f"chatcmpl-{marker}", raw.text + message: Final = completion.choices[0].message + assert message.content == answer(marker), raw.text + assert (message.model_extra or {}).get("reasoning_content") == f"why marker-{marker}", raw.text + assert _body(_native_request(wire)) == _native_body(GPT_OSS, marker) + assert _spend_row(completion.id) == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +def test_chat_completions_prefix_splits_gpt_oss_reasoning_tag_across_stream_deltas(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + identity: Final = f"chatcmpl-{marker}" + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire, model=f"bedrock/chat_completions/{GPT_OSS}") + stream: Final = _openai_client(gateway).chat.completions.create( + model=model, + messages=_sdk_messages(marker), + stream=True, + stream_options={"include_usage": True}, + extra_body={"cache": {"no-cache": True}}, + ) + chunks: Final = list(stream) + assert {chunk.id for chunk in chunks} == {identity}, chunks + assert _joined(chunks, "reasoning_content") == f"why marker-{marker}", chunks + assert _joined(chunks, "content") == answer(marker), chunks + assert _body(_native_request(wire)) == _streamed_native_body(GPT_OSS, marker) + assert _spend_row(identity) == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +def test_region_path_model_is_served_natively_without_the_region(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"bedrock/us-west-2/{GLOBAL_GPT}", api_key=TOKEN, aws_bedrock_runtime_endpoint=wire.url + ) + response: Final = _chat(gateway, model, marker) + assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text + assert _body(_native_request(wire)) == _native_body(GLOBAL_GPT, marker) + assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +def test_sigv4_deployment_signs_the_native_request(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"bedrock/{GPT}", + api_key=None, + aws_access_key_id=ACCESS_KEY, + aws_secret_access_key=SECRET_KEY, + aws_region_name="us-east-1", + aws_bedrock_runtime_endpoint=wire.url, + ) + response: Final = _chat(gateway, model, marker) + assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text + request: Final = _only_request(wire) + assert (request.method, target_of(request)) == ("POST", NATIVE_TARGET), request.target + _assert_sigv4_signed(request, NATIVE_TARGET) + assert _body(request) == _native_body(GPT, marker) + assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +def test_blank_api_key_on_a_sigv4_deployment_is_signed_not_sent_as_an_empty_bearer(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model( + model=f"bedrock/{GPT}", + api_key="", + aws_access_key_id=ACCESS_KEY, + aws_secret_access_key=SECRET_KEY, + aws_region_name="us-east-1", + aws_bedrock_runtime_endpoint=wire.url, + ) + response: Final = _chat(gateway, model, marker) + assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text + request: Final = _only_request(wire) + assert (request.method, target_of(request)) == ("POST", NATIVE_TARGET), request.target + _assert_sigv4_signed(request, NATIVE_TARGET) + assert _body(request) == _native_body(GPT, marker) + assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +def test_runtime_endpoint_without_api_base_is_used_natively(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire, api_base=None) + response: Final = _chat(gateway, model, marker) + assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text + assert _body(_native_request(wire)) == _native_body(GPT, marker) + assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +def test_runtime_endpoint_wins_over_an_unrelated_api_base(gateway: Gateway) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker) + assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text + assert _body(_native_request(wire)) == _native_body(GPT, marker) + assert _upstream_requests_mentioning(gateway, marker) == [], response.text + assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}") + + +@pytest.mark.parametrize("suffix", ["/openai/v1", "/openai/v1/chat/completions"]) +def test_api_base_already_naming_the_native_path_is_not_doubled(gateway: Gateway, suffix: str) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = scenario.model(model_info=None, **{**GPT_DEPLOYMENT, "api_base": f"{wire.url}{suffix}"}) + response: Final = _chat(gateway, model, marker) + request: Final = _only_request(wire) + assert (request.method, request.target) == ("POST", NATIVE_TARGET), response.text + assert _payload(response)["id"] == f"chatcmpl-{marker}", response.text + assert _body(request) == _native_body(GPT, marker) + assert _spend_row(f"chatcmpl-{marker}") == _success_row(model, f"{wire.url}{NATIVE_TARGET}") From 2f27eb3f3606be726cf58fe1c3ffcd66c3216ffd Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 1 Oct 2026 17:02:10 -0700 Subject: [PATCH 30/31] test(bedrock): harden the runtime chat completions audit cells The chaos peer's shared counter and process now come from the same spawn context, since a fork-context Value handed to a spawn-context process raises on Linux. The peer-kill test waits for the first six answers to reach the client before killing the peer instead of counting accepted requests. The Responses wire tests look the spend row up under both the ciphertext id the caller received and the issued id behind it, matching the chaos file's rule for the pre-encryption row --- .../test_bedrock_gpt_responses_native_wire.py | 18 ++++++++---- ..._bedrock_runtime_chat_completions_chaos.py | 29 +++++++++++++------ 2 files changed, 32 insertions(+), 15 deletions(-) diff --git a/tests/integration/providers/test_bedrock_gpt_responses_native_wire.py b/tests/integration/providers/test_bedrock_gpt_responses_native_wire.py index 6a365ac26ba..3d6eb7fcaff 100644 --- a/tests/integration/providers/test_bedrock_gpt_responses_native_wire.py +++ b/tests/integration/providers/test_bedrock_gpt_responses_native_wire.py @@ -58,11 +58,16 @@ def _body(request: Request) -> dict[str, JsonValue]: return _JSON_OBJECT.validate_json(request.body) -def _spend_row(identity: str) -> dict[str, JsonValue]: +# TODO: a Bedrock non-stream /v1/responses spend row can carry the pre-encryption resp_ id instead of the +# ciphertext the caller received, because the spend row id is read from response_obj["id"] before the +# ResponsesIDSecurity hook rewrites it in place; the row is looked up under both ids until that ordering is fixed on +# main +def _spend_row(client_id: str, issued_id: str) -> dict[str, JsonValue]: rows: Final = eventually( lambda: read_rows( - 'SELECT model_group, status, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" WHERE request_id=%s', - (identity,), + 'SELECT model_group, status, prompt_tokens, completion_tokens FROM "LiteLLM_SpendLogs" ' + "WHERE request_id = ANY(%s)", + ([client_id, issued_id],), # pyright: ignore[reportArgumentType] # psycopg adapts the list to a text array ), lambda found: len(found) == 1, seconds=70, @@ -85,10 +90,11 @@ def test_openai_sdk_responses_request_is_served_by_the_native_responses_route(ga response: Final = raw.parse() assert response.output_text == answer(marker), raw.text assert response.usage is not None and (response.usage.input_tokens, response.usage.output_tokens) == (30, 5) - assert _issued_id(response.id).upstream == f"resp_upstream_{marker}", response.id + issued: Final = _issued_id(response.id) + assert issued.upstream == f"resp_upstream_{marker}", response.id request: Final = _native_request(wire) assert _body(request) == {"model": GPT, "input": _prompt(marker)}, request.body - assert _spend_row(response.id) == _success_row(model) + assert _spend_row(response.id, issued.issued) == _success_row(model) async def test_async_openai_sdk_responses_stream_is_served_by_the_native_responses_route(gateway: Gateway) -> None: @@ -114,4 +120,4 @@ async def test_async_openai_sdk_responses_stream_is_served_by_the_native_respons assert issued.upstream == f"resp_upstream_{marker}", completed.response.id request: Final = _native_request(wire) assert _body(request) == {"model": GPT, "input": _prompt(marker), "stream": True}, request.body - assert _spend_row(issued.issued) == _success_row(model) + assert _spend_row(completed.response.id, issued.issued) == _success_row(model) diff --git a/tests/integration/providers/test_bedrock_runtime_chat_completions_chaos.py b/tests/integration/providers/test_bedrock_runtime_chat_completions_chaos.py index 030a407bd8d..ef6e0ce60f2 100644 --- a/tests/integration/providers/test_bedrock_runtime_chat_completions_chaos.py +++ b/tests/integration/providers/test_bedrock_runtime_chat_completions_chaos.py @@ -1,6 +1,7 @@ import asyncio import base64 import binascii +import itertools import multiprocessing import os import re @@ -200,6 +201,19 @@ async def _burst( return tuple(result for result in results if isinstance(result, _Served)) +async def _burst_killing_the_peer_once_it_answered( + base_url: str, key: str, model: str, calls: tuple[_Call, ...], peer: _ChildPeer, answered: int +) -> tuple[_Served, ...]: + async with httpx.AsyncClient(base_url=base_url, timeout=60, trust_env=False) as client: + tasks: Final = tuple(asyncio.create_task(_send(client, key, model, call)) for call in calls) + await asyncio.to_thread(eventually, lambda: peer.received.value, lambda count: count == len(calls), 60) + first: Final = [await finished for finished in itertools.islice(asyncio.as_completed(tasks), answered)] + assert all(item.status == 200 for item in first), [(item.call.marker, item.status) for item in first] + peer.process.kill() + peer.process.join(timeout=10) + return tuple(await asyncio.gather(*tasks)) + + def _calls(count: int, endpoints: tuple[Endpoint, ...], stream: Callable[[int], bool]) -> tuple[_Call, ...]: return tuple( _Call(endpoint=endpoints[index % len(endpoints)], stream=stream(index), marker=uuid.uuid4().hex) @@ -223,10 +237,9 @@ def _accepts_connections(port: int) -> bool: @contextmanager def _child_peer(port: int, answer_first: int) -> Iterator[_ChildPeer]: - received: Final = multiprocessing.Value("i", 0) - process: Final = multiprocessing.get_context("spawn").Process( - target=serve_peer, args=(port, received, answer_first), daemon=True - ) + context: Final = multiprocessing.get_context("spawn") + received: Final = context.Value("i", 0) + process: Final = context.Process(target=serve_peer, args=(port, received, answer_first), daemon=True) process.start() try: eventually(lambda: _accepts_connections(port), bool, seconds=30) @@ -263,11 +276,9 @@ async def test_peer_killed_mid_burst_fails_only_the_held_calls_and_a_restarted_p with gateway.scenario() as scenario: model: Final = _deployment(scenario, f"http://127.0.0.1:{port}") with _child_peer(port, answer_first=6) as peer: - burst: Final = asyncio.create_task(_burst(str(gateway.client.base_url), gateway.key, model, calls)) - await asyncio.to_thread(eventually, lambda: peer.received.value, lambda count: count == 12, 60) - peer.process.kill() - peer.process.join(timeout=10) - served: Final = await burst + served: Final = await _burst_killing_the_peer_once_it_answered( + str(gateway.client.base_url), gateway.key, model, calls, peer, answered=6 + ) succeeded: Final = tuple(item for item in served if item.status == 200) failed: Final = tuple(item for item in served if item.status != 200) assert (len(succeeded), len(failed)) == (6, 6), [(item.call.marker, item.status) for item in served] From 6daf0b7136aa2587052a27f634fbca279d453840 Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Thu, 1 Oct 2026 19:00:09 -0700 Subject: [PATCH 31/31] fix(bedrock): refuse or drop a non-string reasoning_effort before the native chat completions call A reasoning_effort sent as an int, a list, or an object on a GPT 5.6+ deployment the native route serves now answers 400 from litellm before any wire request, naming the type and the drop_params way out, and is dropped under drop_params so AWS applies its default effort, the way Converse dropped it on main. The tip since a0cef91f0b forwarded it for AWS to refuse --- .../chat/chat_completions/transformation.py | 25 ++++++++- ...drock_runtime_chat_completions_sad_wire.py | 42 +++++++++++---- ...bedrock_chat_completions_transformation.py | 51 +++++++++++++++++-- 3 files changed, 102 insertions(+), 16 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index c47fa3944c1..4dea3e80cfe 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -114,6 +114,18 @@ def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) - return _without_params(params, frozenset(("reasoning_effort",))) +def non_string_reasoning_effort(params: Mapping[str, object]) -> frozenset[str]: + """``reasoning_effort`` when the request sends it as anything but a string (an int, a list, an object). + + AWS's Chat Completions endpoint answers such a value with a 400 where Converse silently dropped it, so the + native config refuses it before the call, or drops it under ``drop_params`` so AWS applies its default effort. + """ + effort: Final = params.get("reasoning_effort") + if effort is None or isinstance(effort, str): + return frozenset() + return frozenset(("reasoning_effort",)) + + def _held_close_tag_prefix(text: str) -> int: return next( ( @@ -355,7 +367,17 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): drop_params=drop_params, replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens, ) + malformed_effort: Final = non_string_reasoning_effort(non_default_params) refused_while_reasoning: Final = chat_completions_params_refused_while_reasoning(model, non_default_params) + if malformed_effort and not (litellm.drop_params or drop_params): + raise litellm.utils.UnsupportedParamsError( + message=( + f"{model} takes reasoning_effort as a string on Bedrock's Chat Completions endpoint, not " + f"{type(non_default_params['reasoning_effort']).__name__}. Send one of its named efforts, or " + "set `litellm.drop_params = True` to drop it" + ), + status_code=400, + ) if refused_while_reasoning and not (litellm.drop_params or drop_params): raise litellm.utils.UnsupportedParamsError( message=( @@ -367,7 +389,8 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): ) return dict( # mutable-ok: get_optional_params keeps filling this dict without_refused_reasoning_effort( - model, with_max_completion_tokens(_without_params(mapped, refused_while_reasoning)) + model, + with_max_completion_tokens(_without_params(mapped, refused_while_reasoning | malformed_effort)), ) ) diff --git a/tests/integration/providers/test_bedrock_runtime_chat_completions_sad_wire.py b/tests/integration/providers/test_bedrock_runtime_chat_completions_sad_wire.py index a96fe884c59..357d0ea4d0c 100644 --- a/tests/integration/providers/test_bedrock_runtime_chat_completions_sad_wire.py +++ b/tests/integration/providers/test_bedrock_runtime_chat_completions_sad_wire.py @@ -304,17 +304,9 @@ def test_bad_key_on_the_long_version_model_is_refused_before_any_route(gateway: assert [marker_of(request) for request in received] == [control_marker], _routes(received) -@pytest.mark.parametrize( - "effort", - [ - pytest.param(7, id="int"), - pytest.param(["high"], id="list"), - pytest.param("", id="empty"), - pytest.param("x" * 5120, id="five_kb"), - ], -) +@pytest.mark.parametrize("effort", [pytest.param("", id="empty"), pytest.param("x" * 5120, id="five_kb")]) def test_invalid_reasoning_effort_reaches_the_peer_and_its_400_reaches_the_caller( - gateway: Gateway, effort: JsonValue + gateway: Gateway, effort: str ) -> None: marker: Final = uuid.uuid4().hex with wire_server(respond) as wire, gateway.scenario() as scenario: @@ -328,6 +320,36 @@ def test_invalid_reasoning_effort_reaches_the_peer_and_its_400_reaches_the_calle _assert_row(_call_id(response), model, "failure") +NON_STRING_EFFORTS: Final = (pytest.param(7, id="int"), pytest.param(["high"], id="list")) + + +@pytest.mark.parametrize("effort", NON_STRING_EFFORTS) +def test_non_string_reasoning_effort_is_refused_before_any_wire_request(gateway: Gateway, effort: JsonValue) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire) + response: Final = _chat(gateway, model, marker, reasoning_effort=effort) + assert response.status_code == 400, response.text + message: Final = _error_message(response) + assert message.startswith("litellm.UnsupportedParamsError"), response.text + assert "reasoning_effort as a string" in message and "drop_params" in message, response.text + _assert_row(_call_id(response), model, "failure") + assert _routes(wire.drain()) == [] + + +@pytest.mark.parametrize("effort", NON_STRING_EFFORTS) +def test_drop_params_deployment_drops_a_non_string_reasoning_effort(gateway: Gateway, effort: JsonValue) -> None: + marker: Final = uuid.uuid4().hex + with wire_server(respond) as wire, gateway.scenario() as scenario: + model: Final = _deployment(scenario, wire, drop_params=True) + response: Final = _chat(gateway, model, marker, reasoning_effort=effort) + assert _content(response) == answer(marker), response.text + request: Final = _only_request(wire, marker) + assert target_of(request) == NATIVE_TARGET, request.body + assert "reasoning_effort" not in _body(request), request.body + _assert_row(string_value(_payload(response)["id"]), model, "success") + + def test_duplicated_reasoning_effort_key_lets_the_last_value_win(gateway: Gateway) -> None: marker: Final = uuid.uuid4().hex with wire_server(respond) as wire, gateway.scenario() as scenario: diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 06fff49d500..16ae1114402 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -668,16 +668,35 @@ def test_map_openai_params_keeps_reasoning_effort_low_for_grok(): @pytest.mark.parametrize("model", ["us.xai.grok-4.6", "global.openai.gpt-5.6-sol"]) -@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5]) -def test_map_openai_params_forwards_a_malformed_reasoning_effort_for_aws_to_refuse(model, reasoning_effort): +@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5], ids=["list", "object", "int"]) +def test_map_openai_params_refuses_a_non_string_reasoning_effort_without_drop_params(model, reasoning_effort): cfg = AmazonBedrockRuntimeChatCompletionsConfig() - mapped = cfg.map_openai_params( + with pytest.raises(litellm.UnsupportedParamsError, match="drop_params") as refused: + cfg.map_openai_params( + non_default_params={"reasoning_effort": reasoning_effort, "max_tokens": 64}, + optional_params={}, + model=model, + drop_params=False, + ) + assert refused.value.status_code == 400 + assert type(reasoning_effort).__name__ in str(refused.value) + + +@pytest.mark.parametrize("model", ["us.xai.grok-4.6", "global.openai.gpt-5.6-sol"]) +@pytest.mark.parametrize("reasoning_effort", [["low"], {"effort": "low"}, 5], ids=["list", "object", "int"]) +@pytest.mark.parametrize("drop_params_via", ["request", "litellm.drop_params"]) +def test_map_openai_params_drops_a_non_string_reasoning_effort_under_drop_params( + monkeypatch, model, reasoning_effort, drop_params_via +): + monkeypatch.setattr(litellm, "drop_params", drop_params_via == "litellm.drop_params") + mapped = AmazonBedrockRuntimeChatCompletionsConfig().map_openai_params( non_default_params={"reasoning_effort": reasoning_effort, "max_tokens": 64}, optional_params={}, model=model, - drop_params=False, + drop_params=drop_params_via == "request", ) - assert mapped["reasoning_effort"] == reasoning_effort + assert "reasoning_effort" not in mapped + assert mapped["max_completion_tokens"] == 64 def test_map_openai_params_keeps_reasoning_effort_none_for_gpt56(): @@ -752,6 +771,28 @@ def test_refused_params_are_dropped_or_refused_before_reaching_aws(local_cost_ma assert param.keys().isdisjoint(json.loads(requests[0].content)) +@pytest.mark.parametrize("reasoning_effort", [3, ["high"]], ids=["int", "list"]) +def test_non_string_reasoning_effort_is_refused_or_dropped_before_reaching_aws( + local_cost_map, fake_aws_env, reasoning_effort +): + requests, client = _recording_client(json=_chat_completion_json("ok", "global.openai.gpt-5.6-sol")) + request = { + "model": "bedrock/global.openai.gpt-5.6-sol", + "messages": [{"role": "user", "content": "hello"}], + "reasoning_effort": reasoning_effort, + "client": client, + } + with pytest.raises(litellm.UnsupportedParamsError, match="reasoning_effort") as refused: + litellm.completion(**request) + assert refused.value.status_code == 400 + assert requests == [] + + litellm.completion(**request, drop_params=True) + + assert str(requests[0].url).endswith("/openai/v1/chat/completions") + assert "reasoning_effort" not in json.loads(requests[0].content) + + GPT_PARAMS_TIED_TO_REASONING_OFF = { "temperature": 0.2, "top_p": 0.9,