diff --git a/litellm/litellm_core_utils/prompt_templates/image_handling.py b/litellm/litellm_core_utils/prompt_templates/image_handling.py index 57b4f545301..705a7b93381 100644 --- a/litellm/litellm_core_utils/prompt_templates/image_handling.py +++ b/litellm/litellm_core_utils/prompt_templates/image_handling.py @@ -4,8 +4,9 @@ Helper functions to handle images passed in messages import asyncio import base64 -from collections.abc import Callable, Mapping +from collections.abc import Callable, Iterable, Mapping from dataclasses import dataclass +from itertools import chain from types import MappingProxyType from typing import Final @@ -304,18 +305,19 @@ async def _fetch_data_urls(remote_urls: tuple[str, ...]) -> tuple[str, ...]: raise +def _remote_urls_to_inline( + messages: Iterable[AllMessageValues], should_inline: Callable[[RemoteMedia], bool] +) -> tuple[str, ...]: + parts: Final = chain.from_iterable(_content_parts(message) for message in messages) + remotes: Final = (remote for part in parts if (remote := _parse_remote_part(part)) is not None) + return tuple(dict.fromkeys(remote.url for remote in remotes if should_inline(_remote_media(remote)))) + + def inline_remote_media( messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues] should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url, ) -> list[AllMessageValues]: # mutable-ok: every transform_request takes list[AllMessageValues] - remote_urls: Final = tuple( - dict.fromkeys( - remote.url - for message in messages - for part in _content_parts(message) - if (remote := _parse_remote_part(part)) is not None and should_inline(_remote_media(remote)) - ) - ) + remote_urls: Final = _remote_urls_to_inline(messages, should_inline) if not remote_urls: return messages data_urls: Final = MappingProxyType({url: convert_url_to_base64(url) for url in remote_urls}) @@ -328,14 +330,7 @@ async def async_inline_remote_media( messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues] should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url, ) -> list[AllMessageValues]: # mutable-ok: every transform_request takes list[AllMessageValues] - remote_urls: Final = tuple( - dict.fromkeys( - remote.url - for message in messages - for part in _content_parts(message) - if (remote := _parse_remote_part(part)) is not None and should_inline(_remote_media(remote)) - ) - ) + remote_urls: Final = _remote_urls_to_inline(messages, should_inline) if not remote_urls: return messages data_urls: Final = await _fetch_data_urls(remote_urls) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index c424ca2a069..91a643fcd7b 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -3,12 +3,14 @@ Native OpenAI Chat Completions on Amazon Bedrock Runtime. AWS serves this surface at ``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions`` -for Grok 4.6, gpt-oss and the GPT-5.6 family. The ``chat_completions/`` route prefix -opts a model in, so chat completions stay chat completions instead of being rewritten -to Converse; without it these models stay on Converse. +for Grok 4.6, gpt-oss and GPT 5.6 and newer. GPT 5.6 and newer take it by default +(``bedrock_runtime_chat_completions_is_default`` in ``common_utils``), so their chat +completions stay chat completions instead of being rewritten to Converse; the +``chat_completions/`` route prefix opts any other model in, and ``converse/`` pins a +model to Converse. -Usage: model="bedrock/chat_completions/openai.gpt-oss-20b-1:0" or -model="bedrock/chat_completions/global.openai.gpt-5.6-sol". A request that needs a +Usage: model="bedrock/global.openai.gpt-6-sol" or +model="bedrock/chat_completions/openai.gpt-oss-20b-1:0". A request that needs a Converse-only feature (``bedrock_request_needs_converse`` in ``common_utils``) is still served by Converse. """ diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index b349f996192..14ab71bd63a 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -39,6 +39,9 @@ if TYPE_CHECKING: _ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs" _OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.") +_OPENAI_GPT_VERSION_RE: Final = re.compile(r"(^|[./])openai\.gpt-(\d+)(?:\.(\d+))?") +_BEDROCK_RUNTIME_CHAT_COMPLETIONS_DEFAULT_SINCE: Final = (5, 6) +_BEDROCK_RUNTIME_CHAT_COMPLETIONS_ENDPOINT: Final = "/v1/chat/completions" BedrockRoute = Literal[ "converse", "invoke", @@ -831,6 +834,34 @@ def _bedrock_price_map_flag(model: str, flag: str) -> bool: return any(entry is not None and entry.get(flag) is True for entry in _bedrock_price_map_entries(model)) +def _price_map_entry_lists_endpoint(entry: Mapping[str, object] | None, endpoint: str) -> bool: + endpoints: Final = None if entry is None else entry.get("supported_endpoints") + return isinstance(endpoints, (list, tuple)) and endpoint in endpoints + + +def _openai_gpt_version(model: str) -> tuple[int, int] | None: + match: Final = _OPENAI_GPT_VERSION_RE.search(model) + if match is None: + return None + return int(match.group(2)), int(match.group(3) or 0) + + +def bedrock_runtime_chat_completions_is_default(model: str) -> bool: + """Whether a model with no route prefix goes to bedrock-runtime's native Chat Completions by default. + + GPT 5.6 and newer (``openai.gpt-[.]`` at or above 5.6, which gpt-oss never matches) whose + price-map row lists ``/v1/chat/completions`` in ``supported_endpoints``. Older GPT rows, gpt-oss and Grok + stay on Converse unless the ``chat_completions/`` prefix opts them in. + """ + version: Final = _openai_gpt_version(model) + if version is None or version < _BEDROCK_RUNTIME_CHAT_COMPLETIONS_DEFAULT_SINCE: + return False + return any( + _price_map_entry_lists_endpoint(entry, _BEDROCK_RUNTIME_CHAT_COMPLETIONS_ENDPOINT) + for entry in _bedrock_price_map_entries(model) + ) + + def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool: """Whether AWS's native Chat Completions serves this model's function tools with any ``reasoning_effort``. @@ -879,7 +910,10 @@ def _response_format_needs_converse(model: str, response_format: object) -> bool def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool: - """Whether a request on the opt-in ``chat_completions/`` route must still be served by Converse. + """Whether a request on the native Chat Completions route must still be served by Converse. + + The route is the default for GPT 5.6 and newer (``bedrock_runtime_chat_completions_is_default``) and the + ``chat_completions/`` prefix's opt-in for the rest; this decides the fallback for both alike. Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking`` block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse @@ -909,6 +943,14 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje ) +def _chat_completions_unless_converse_needed( + model: str, request_params: Mapping[str, object] | None +) -> Literal["converse", "chat_completions"]: + if request_params is not None and bedrock_request_needs_converse(model, request_params): + return "converse" + return "chat_completions" + + def bedrock_route_for_request( model: str, request_params: Mapping[str, object], additional_drop_params: Sequence[str] | None ) -> BedrockRoute: @@ -1285,9 +1327,11 @@ class BedrockModelInfo(BaseLLMModelInfo): """ Get the bedrock route for the given model. - ``chat_completions/`` opts a model into bedrock-runtime's native OpenAI Chat Completions; - ``request_params`` (the caller's chat params) sends such a request to Converse when it - needs a feature only Converse serves. Without the prefix, OpenAI-family models stay on Converse. + GPT 5.6 and newer go to bedrock-runtime's native OpenAI Chat Completions by default + (``bedrock_runtime_chat_completions_is_default``) and ``chat_completions/`` opts any other model in; + ``request_params`` (the caller's chat params) sends such a request to Converse when it needs a + feature only Converse serves, and ``converse/`` pins a model to Converse. Every other OpenAI-family + model stays on Converse without the prefix. """ route_mappings: dict[ str, @@ -1324,9 +1368,7 @@ class BedrockModelInfo(BaseLLMModelInfo): return route_type if BedrockModelInfo._model_has_route_prefix(model, "chat_completions/"): - if request_params is not None and bedrock_request_needs_converse(model, request_params): - return "converse" - return "chat_completions" + return _chat_completions_unless_converse_needed(model, request_params) # Check for nova spec prefixes (nova/ and nova-2/) _model_after_bedrock: Final = model.replace("bedrock/", "", 1) @@ -1336,6 +1378,9 @@ class BedrockModelInfo(BaseLLMModelInfo): if is_bedrock_application_inference_profile_arn(model): return "converse" + if bedrock_runtime_chat_completions_is_default(model): + return _chat_completions_unless_converse_needed(model, request_params) + base_model: Final = BedrockModelInfo.get_base_model(model) alt_model: Final = BedrockModelInfo.get_non_litellm_routing_model_name(model=model) if base_model in litellm.bedrock_converse_models or alt_model in litellm.bedrock_converse_models: diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index d847b9712ef..7c70f07329e 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -58449,6 +58449,7 @@ "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html" }, "us.openai.gpt-6-astra": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1.1e-05, "input_cost_per_token_above_272k_tokens": 2.2e-05, "cache_creation_input_token_cost": 1.375e-05, @@ -58480,10 +58481,12 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-6-sol": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, "cache_creation_input_token_cost": 2.75e-06, @@ -58515,10 +58518,12 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-6-luna": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1.1e-07, "input_cost_per_token_above_272k_tokens": 2.2e-07, "cache_creation_input_token_cost": 1.375e-07, @@ -58550,10 +58555,12 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "global.openai.gpt-6-astra": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1e-05, "input_cost_per_token_above_272k_tokens": 2e-05, "cache_creation_input_token_cost": 1.25e-05, @@ -58585,6 +58592,7 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -58621,6 +58629,7 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-sol": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "cache_creation_input_token_cost": 2.5e-06, @@ -58652,6 +58661,7 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -58688,6 +58698,7 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-luna": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1e-07, "input_cost_per_token_above_272k_tokens": 2e-07, "cache_creation_input_token_cost": 1.25e-07, @@ -58719,6 +58730,7 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -79328,6 +79340,7 @@ "output_cost_per_token_above_272k_tokens": 1.5e-05, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-1-sol.html", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ], "supported_modalities": [ @@ -79337,6 +79350,7 @@ "supported_output_modalities": [ "text" ], + "supports_bedrock_runtime_chat_completions_response_format": true, "supports_function_calling": true, "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, @@ -79433,6 +79447,7 @@ "output_cost_per_token_above_272k_tokens": 1.65e-05, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-1-sol.html", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ], "supported_modalities": [ @@ -79442,6 +79457,7 @@ "supported_output_modalities": [ "text" ], + "supports_bedrock_runtime_chat_completions_response_format": true, "supports_function_calling": true, "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index d847b9712ef..7c70f07329e 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -58449,6 +58449,7 @@ "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html" }, "us.openai.gpt-6-astra": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1.1e-05, "input_cost_per_token_above_272k_tokens": 2.2e-05, "cache_creation_input_token_cost": 1.375e-05, @@ -58480,10 +58481,12 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-6-sol": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2.2e-06, "input_cost_per_token_above_272k_tokens": 4.4e-06, "cache_creation_input_token_cost": 2.75e-06, @@ -58515,10 +58518,12 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "us.openai.gpt-6-luna": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1.1e-07, "input_cost_per_token_above_272k_tokens": 2.2e-07, "cache_creation_input_token_cost": 1.375e-07, @@ -58550,10 +58555,12 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, "global.openai.gpt-6-astra": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1e-05, "input_cost_per_token_above_272k_tokens": 2e-05, "cache_creation_input_token_cost": 1.25e-05, @@ -58585,6 +58592,7 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -58621,6 +58629,7 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-sol": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 2e-06, "input_cost_per_token_above_272k_tokens": 4e-06, "cache_creation_input_token_cost": 2.5e-06, @@ -58652,6 +58661,7 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -58688,6 +58698,7 @@ "source": "https://aws.amazon.com/bedrock/pricing/" }, "global.openai.gpt-6-luna": { + "supports_bedrock_runtime_chat_completions_response_format": true, "input_cost_per_token": 1e-07, "input_cost_per_token_above_272k_tokens": 2e-07, "cache_creation_input_token_cost": 1.25e-07, @@ -58719,6 +58730,7 @@ "supports_vision": true, "source": "https://aws.amazon.com/bedrock/pricing/", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ] }, @@ -79328,6 +79340,7 @@ "output_cost_per_token_above_272k_tokens": 1.5e-05, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-1-sol.html", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ], "supported_modalities": [ @@ -79337,6 +79350,7 @@ "supported_output_modalities": [ "text" ], + "supports_bedrock_runtime_chat_completions_response_format": true, "supports_function_calling": true, "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, @@ -79433,6 +79447,7 @@ "output_cost_per_token_above_272k_tokens": 1.65e-05, "source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-1-sol.html", "supported_endpoints": [ + "/v1/chat/completions", "/v1/responses" ], "supported_modalities": [ @@ -79442,6 +79457,7 @@ "supported_output_modalities": [ "text" ], + "supports_bedrock_runtime_chat_completions_response_format": true, "supports_function_calling": true, "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": false, diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index e40ef13e25a..7539e8c8b43 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -1,4 +1,4 @@ -"""Opt-in Bedrock Runtime Chat Completions: ``bedrock/chat_completions/`` posts to /openai/v1/chat/completions.""" +"""Bedrock Runtime Chat Completions: the default for GPT 5.6 and newer, ``bedrock/chat_completions/`` for the rest.""" import json @@ -20,6 +20,7 @@ from litellm.llms.bedrock.common_utils import ( BedrockModelInfo, bedrock_request_needs_converse, bedrock_route_for_request, + bedrock_runtime_chat_completions_is_default, get_bedrock_chat_config, ) from litellm.llms.custom_httpx.http_handler import HTTPHandler @@ -63,9 +64,11 @@ def test_claude_stays_on_converse(local_cost_map): "us.xai.grok-4.6", "bedrock/openai.gpt-oss-20b-1:0", "openai.gpt-oss-120b-1:0", - "global.openai.gpt-5.6-sol", - "bedrock/us.openai.gpt-5.6-terra", + "global.openai.gpt-5.5", + "bedrock/us.openai.gpt-5.4", "bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0", + "arn:aws:bedrock:us-east-1:123456789012:inference-profile/us.openai.gpt-6-astra", + "arn:aws:bedrock:us-west-2:123456789012:application-inference-profile/abc123xyz", ], ) def test_models_without_the_prefix_stay_on_converse(local_cost_map, model): @@ -86,6 +89,32 @@ def test_cost_map_row_listing_chat_completions_leaves_the_default_route_alone(mo assert BedrockModelInfo.get_bedrock_route("bedrock/chat_completions/openai.gpt-oss-20b-1:0", {}) == "chat_completions" +@pytest.mark.parametrize( + "model, supported_endpoints, expected_route", + [ + ("global.openai.gpt-5.5", ["/v1/chat/completions", "/v1/responses"], "converse"), + ("us.openai.gpt-5.6-sol", ["/v1/chat/completions", "/v1/responses"], "chat_completions"), + ("us.openai.gpt-5.6-sol", ["/v1/responses"], "converse"), + ("global.openai.gpt-6-sol", ["/v1/chat/completions", "/v1/responses"], "chat_completions"), + ("global.openai.gpt-6-sol", ["/v1/responses"], "converse"), + ("global.openai.gpt-6-sol", [], "converse"), + ("us.openai.gpt-6.1-sol", ["/v1/chat/completions"], "chat_completions"), + ("global.openai.gpt-10-sol", ["/v1/chat/completions"], "chat_completions"), + ("openai.gpt-oss-120b-1:0", ["/v1/chat/completions"], "converse"), + ("us.xai.grok-4.6", ["/v1/chat/completions"], "converse"), + ], +) +def test_default_route_needs_gpt_56_or_newer_and_a_row_listing_chat_completions( + monkeypatch, model, supported_endpoints, expected_route +): + entry = {"litellm_provider": "bedrock_converse", "supported_endpoints": supported_endpoints} + monkeypatch.setattr(litellm, "model_cost", {model: entry}) + assert bedrock_runtime_chat_completions_is_default(model) is (expected_route == "chat_completions") + assert BedrockModelInfo.get_bedrock_route(f"bedrock/{model}", {}) == expected_route + assert BedrockModelInfo.get_bedrock_route(f"bedrock/chat_completions/{model}", {}) == "chat_completions" + assert BedrockModelInfo.get_bedrock_route(f"bedrock/converse/{model}", {}) == "converse" + + @pytest.mark.parametrize("model", ["global.openai.gpt-5.6-sol", "openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"]) def test_chat_completions_prefix_prices_like_the_bare_model(local_cost_map, model): prefixed = litellm.get_model_info(model=f"bedrock/chat_completions/{model}") @@ -194,7 +223,7 @@ def _recording_client(**response_kwargs): [ ("bedrock/us.xai.grok-4.6", b"/model/us.xai.grok-4.6/converse"), ("bedrock/openai.gpt-oss-20b-1:0", b"/model/openai.gpt-oss-20b-1%3A0/converse"), - ("bedrock/global.openai.gpt-5.6-sol", b"/model/global.openai.gpt-5.6-sol/converse"), + ("bedrock/global.openai.gpt-5.5", b"/model/global.openai.gpt-5.5/converse"), ], ) def test_completion_without_the_prefix_posts_converse(local_cost_map, fake_aws_env, model, model_path): @@ -310,13 +339,40 @@ def test_openai_runtime_models_use_chat_completions_route(local_cost_map, model) assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig) +GPT_56_AND_NEWER_MODELS = ( + "global.openai.gpt-5.6-sol", + "bedrock/us.openai.gpt-5.6-terra", + "us.openai.gpt-5.6-luna", + "bedrock/global.openai.gpt-6-astra", + "us.openai.gpt-6-sol", + "global.openai.gpt-6-luna", + "bedrock/global.openai.gpt-6.1-sol", + "us.openai.gpt-6.1-sol", +) + + +@pytest.mark.parametrize("model", GPT_56_AND_NEWER_MODELS) +def test_gpt_56_and_newer_default_to_chat_completions(local_cost_map, model): + assert bedrock_runtime_chat_completions_is_default(model) is True + assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions" + assert BedrockModelInfo.get_bedrock_route(model, {}) == "chat_completions" + assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig) + + @pytest.mark.parametrize("model", ["us.amazon.nova-micro-v1:0", "us.anthropic.claude-haiku-4-5-20251001-v1:0"]) def test_nova_and_claude_stay_on_converse(local_cost_map, model): assert BedrockModelInfo.get_bedrock_route(model, {"tools": [GET_WEATHER_TOOL]}) == "converse" @pytest.mark.parametrize( - "model", ["chat_completions/openai.gpt-oss-20b-1:0", "bedrock/chat_completions/global.openai.gpt-5.6-sol"] + "model", + [ + "chat_completions/openai.gpt-oss-20b-1:0", + "bedrock/chat_completions/global.openai.gpt-5.6-sol", + "bedrock/us.openai.gpt-5.6-sol", + "global.openai.gpt-6-sol", + "us.openai.gpt-6.1-sol", + ], ) def test_guardrail_config_falls_back_to_converse(local_cost_map, model): guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"} @@ -363,6 +419,9 @@ def test_gpt56_tools_need_reasoning_none_on_chat_completions(local_cost_map, req BedrockModelInfo.get_bedrock_route("bedrock/chat_completions/us.openai.gpt-5.6-terra", request_params) == expected_route ) + assert BedrockModelInfo.get_bedrock_route("bedrock/us.openai.gpt-5.6-sol", request_params) == expected_route + assert BedrockModelInfo.get_bedrock_route("global.openai.gpt-6-sol", request_params) == expected_route + assert BedrockModelInfo.get_bedrock_route("bedrock/us.openai.gpt-6.1-sol", request_params) == expected_route @pytest.mark.parametrize("reasoning_effort", ["low", "high", None]) @@ -395,6 +454,8 @@ def test_thinking_block_goes_to_converse(local_cost_map): def test_explicit_converse_prefix_wins_for_openai_models(local_cost_map): assert BedrockModelInfo.get_bedrock_route("bedrock/converse/openai.gpt-oss-20b-1:0") == "converse" assert BedrockModelInfo.get_bedrock_route("converse/global.openai.gpt-5.6-sol", {}) == "converse" + assert BedrockModelInfo.get_bedrock_route("bedrock/converse/global.openai.gpt-6-sol", {}) == "converse" + assert isinstance(get_bedrock_chat_config("bedrock/converse/global.openai.gpt-6-sol"), litellm.AmazonConverseConfig) def test_map_openai_params_sends_max_tokens_as_max_completion_tokens(): @@ -769,6 +830,58 @@ def test_gpt56_tools_with_reasoning_none_stay_on_chat_completions(local_cost_map assert response.choices[0].message.tool_calls[0].function.name == "get_weather" +@pytest.mark.parametrize("model", ["global.openai.gpt-6-sol", "us.openai.gpt-5.6-sol", "us.openai.gpt-6.1-sol"]) +def test_gpt_56_and_newer_completion_without_the_prefix_posts_runtime_chat_completions( + local_cost_map, fake_aws_env, model +): + requests, client = _recording_client(json=_chat_completion_json("ok", model)) + response = litellm.completion( + model=f"bedrock/{model}", + messages=[{"role": "user", "content": "hello"}], + reasoning_effort="low", + client=client, + ) + + assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions" + body = json.loads(requests[0].content) + assert body["model"] == model + assert body["reasoning_effort"] == "low" + assert "inferenceConfig" not in body + assert response.choices[0].message.content == "ok" + assert response._hidden_params["response_cost"] > 0 + + +def test_gpt6_without_the_prefix_tools_with_reasoning_effort_go_to_converse(local_cost_map, fake_aws_env): + requests, client = _recording_client(json=CONVERSE_JSON) + response = litellm.completion( + model="bedrock/global.openai.gpt-6-sol", + messages=[{"role": "user", "content": "hello"}], + tools=[GET_WEATHER_TOOL], + reasoning_effort="low", + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-6-sol/converse") + body = json.loads(requests[0].content) + assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "get_weather" + assert body["additionalModelRequestFields"]["reasoning"] == {"effort": "low"} + assert response.choices[0].message.content == "ok" + + +def test_gpt6_without_the_prefix_guardrail_config_goes_to_converse(local_cost_map, fake_aws_env): + guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"} + requests, client = _recording_client(json=CONVERSE_JSON) + litellm.completion( + model="bedrock/global.openai.gpt-6-sol", + messages=[{"role": "user", "content": "hello"}], + guardrailConfig=guardrail, + client=client, + ) + + assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-6-sol/converse") + assert json.loads(requests[0].content)["guardrailConfig"] == guardrail + + @pytest.mark.parametrize( "converse_only_param", [ @@ -1017,6 +1130,8 @@ RESPONSE_FORMAT_ENFORCING_MODELS = [ "chat_completions/global.openai.gpt-5.6-sol", "chat_completions/us.xai.grok-4.6", "bedrock/chat_completions/us-gov.xai.grok-4.6", + "global.openai.gpt-6-sol", + "bedrock/us.openai.gpt-6.1-sol", ] diff --git a/tests/unit/llms/bedrock/test_cross_region_inference_profile_mapping.py b/tests/unit/llms/bedrock/test_cross_region_inference_profile_mapping.py index aa0827c5ae5..bcd1e9d6578 100644 --- a/tests/unit/llms/bedrock/test_cross_region_inference_profile_mapping.py +++ b/tests/unit/llms/bedrock/test_cross_region_inference_profile_mapping.py @@ -138,9 +138,10 @@ def _bedrock_response(model, usage): @pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id) -def test_bedrock_gpt_5_6_profiles_route_to_converse(profile, local_model_cost_map): - """GPT-5.6 is served by Converse on bedrock-runtime, never by Invoke.""" - assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "converse" +def test_bedrock_gpt_5_6_profiles_route_to_runtime_chat_completions(profile, local_model_cost_map): + """GPT-5.6 is served by bedrock-runtime's native Chat Completions by default and by Converse when pinned, never by Invoke.""" + assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "chat_completions" + assert BedrockModelInfo.get_bedrock_route(f"bedrock/converse/{profile.model_id}") == "converse" @pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id)