From 093a9d4ddfbb010977ad7a50a0c8c3f7745dbcaf Mon Sep 17 00:00:00 2001 From: mateo-berri <277851410+mateo-berri@users.noreply.github.com> Date: Sat, 26 Sep 2026 18:37:36 +0000 Subject: [PATCH] fix(bedrock): inline http image urls and keep stop on converse for native chat completions Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../chat/chat_completions/transformation.py | 57 +++++++++++++- litellm/llms/bedrock/common_utils.py | 4 +- ...bedrock_chat_completions_transformation.py | 75 +++++++++++++++++-- 3 files changed, 127 insertions(+), 9 deletions(-) diff --git a/litellm/llms/bedrock/chat/chat_completions/transformation.py b/litellm/llms/bedrock/chat/chat_completions/transformation.py index be2eb8c7713..123531cf227 100644 --- a/litellm/llms/bedrock/chat/chat_completions/transformation.py +++ b/litellm/llms/bedrock/chat/chat_completions/transformation.py @@ -22,6 +22,11 @@ import httpx from typing_extensions import assert_never import litellm +from litellm.litellm_core_utils.prompt_templates.image_handling import ( + async_inline_remote_media, + convert_url_to_base64, + inline_remote_image_urls, +) from litellm.llms.base_llm.chat.transformation import BaseLLMException from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path @@ -40,7 +45,7 @@ REASONING_CLOSE_TAG: Final = "" CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType( { - "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs")), + "openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs")), "openai.gpt-oss": frozenset(("logit_bias",)), "xai.": frozenset(("frequency_penalty", "presence_penalty")), } @@ -167,6 +172,48 @@ def split_reasoning_tag(content: str) -> tuple[str | None, str]: return reasoning or None, body +def _remote_http_url(candidate: object) -> str | None: + return candidate if isinstance(candidate, str) and candidate.startswith(("http://", "https://")) else None + + +def _inlined_image_url_part(part: object) -> object: + fields: Final = part if isinstance(part, Mapping) else None + if fields is None or fields.get("type") != "image_url": + return part + image_url: Final = fields.get("image_url") + image_url_fields: Final = image_url if isinstance(image_url, Mapping) else None + url: Final = _remote_http_url(image_url_fields.get("url") if image_url_fields is not None else image_url) + if url is None: + return part + data_url: Final = convert_url_to_base64(url) + inlined: Final = {**image_url_fields, "url": data_url} if image_url_fields is not None else data_url + return {**fields, "image_url": inlined} # mutable-ok: json-serialized message part + + +def _inlined_image_url_message(message: AllMessageValues) -> AllMessageValues: + content: Final = message.get("content") + if not isinstance(content, list): + return message + inlined_message: Final = { # mutable-ok: json-serialized message + **message, + "content": [_inlined_image_url_part(part) for part in content], + } + return inlined_message # pyright: ignore[reportReturnType] # the same message with remote image parts inlined + + +def _with_inlined_remote_image_urls( + messages: list[AllMessageValues], +) -> list[AllMessageValues]: # mutable-ok: transform_request takes a list + """Inline every remote ``image_url`` so AWS never sees the ``http(s)://`` URLs it rejects. + + AWS's native surface only takes inline ``data:`` URLs and S3 URLs where Converse downloaded + remote images itself, so the bytes are fetched and inlined here exactly like Converse did. + """ + return [ # mutable-ok: transform_request takes a list + _inlined_image_url_message(message) for message in messages + ] + + class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler): """OpenAI chunk parsing plus the ```` split, tracked per choice index.""" @@ -222,6 +269,10 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): def custom_llm_provider(self) -> str | None: return "bedrock" + @property + def uses_async_transform_request(self) -> bool: + return True + def get_error_class( self, error_message: str, @@ -325,7 +376,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): ) -> dict: # mutable-ok: BaseConfig signature return super().transform_request( model=split_bedrock_region_path(model)[1], - messages=messages, + messages=_with_inlined_remote_image_urls(messages), optional_params=self._inference_params(optional_params), litellm_params=litellm_params, headers=headers, @@ -341,7 +392,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig): ) -> dict: # mutable-ok: BaseConfig signature return await super().async_transform_request( model=split_bedrock_region_path(model)[1], - messages=messages, + messages=await async_inline_remote_media(messages, should_inline=inline_remote_image_urls), optional_params=self._inference_params(optional_params), litellm_params=litellm_params, headers=headers, diff --git a/litellm/llms/bedrock/common_utils.py b/litellm/llms/bedrock/common_utils.py index 098be7082d3..6bd4a8d634c 100644 --- a/litellm/llms/bedrock/common_utils.py +++ b/litellm/llms/bedrock/common_utils.py @@ -899,6 +899,7 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset( "thinking", "additionalModelRequestFields", "top_k", + "stop", ) ) @@ -921,7 +922,8 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking`` block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on - AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body, + AWS's native OpenAI surface, ``stop`` stays on Converse where it fails loudly instead of silently + stopping hidden reasoning, operator-owned request metadata is only written onto the Converse body, function tools (``tools`` or legacy ``functions``) on a model without ``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless ``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as diff --git a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py index 17c0f8d3c1d..4bfa7bf0fc4 100644 --- a/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py +++ b/tests/unit/llms/bedrock/chat/chat_completions/test_bedrock_chat_completions_transformation.py @@ -258,11 +258,13 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model): assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": None}) == "chat_completions" -@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"]) +@pytest.mark.parametrize( + "model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6", "global.openai.gpt-5.6-sol"] +) @pytest.mark.parametrize( "request_params", - [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}], - ids=["additionalModelRequestFields", "top_k"], + [{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}, {"stop": ["END"]}], + ids=["additionalModelRequestFields", "top_k", "stop"], ) def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params): assert bedrock_request_needs_converse(model, request_params) is True @@ -331,6 +333,69 @@ def test_map_openai_params_sends_max_tokens_as_max_completion_tokens(): assert mapped == {"max_completion_tokens": 64, "temperature": 0.1} +HTTPS_IMAGE_URL = "https://example.com/cat.png" +IMAGE_MESSAGES = [ + { + "role": "user", + "content": [ + {"type": "text", "text": "what is this"}, + {"type": "image_url", "image_url": HTTPS_IMAGE_URL}, + {"type": "image_url", "image_url": {"url": HTTPS_IMAGE_URL, "detail": "high"}}, + {"type": "image_url", "image_url": {"url": "data:image/png;base64,AAA"}}, + {"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}}, + ], + } +] + + +def _assert_remote_images_inlined(content): + assert content[0] == {"type": "text", "text": "what is this"} + assert content[1]["image_url"]["url"] == f"data:image/png;base64,{HTTPS_IMAGE_URL}" + assert content[2] == { + "type": "image_url", + "image_url": {"url": f"data:image/png;base64,{HTTPS_IMAGE_URL}", "detail": "high"}, + } + assert content[3]["image_url"]["url"] == "data:image/png;base64,AAA" + assert content[4]["image_url"]["url"] == "s3://bucket/key.png" + + +def test_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch): + import litellm.llms.bedrock.chat.chat_completions.transformation as native_cc + + monkeypatch.setattr( + native_cc, "convert_url_to_base64", lambda url: f"data:image/png;base64,{url}" + ) + body = AmazonBedrockRuntimeChatCompletionsConfig().transform_request( + model="us.xai.grok-4.6", + messages=IMAGE_MESSAGES, + optional_params={}, + litellm_params={}, + headers={}, + ) + + _assert_remote_images_inlined(body["messages"][0]["content"]) + + +async def test_async_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch): + import litellm.litellm_core_utils.prompt_templates.image_handling as image_handling + + async def fake_convert(url): + return f"data:image/png;base64,{url}" + + monkeypatch.setattr(image_handling, "async_convert_url_to_base64", fake_convert) + cfg = AmazonBedrockRuntimeChatCompletionsConfig() + assert cfg.uses_async_transform_request is True + body = await cfg.async_transform_request( + model="us.xai.grok-4.6", + messages=IMAGE_MESSAGES, + optional_params={}, + litellm_params={}, + headers={}, + ) + + _assert_remote_images_inlined(body["messages"][0]["content"]) + + def test_map_openai_params_keeps_explicit_max_completion_tokens(): cfg = AmazonBedrockRuntimeChatCompletionsConfig() mapped = cfg.map_openai_params( @@ -398,8 +463,8 @@ def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map): [ ( "bedrock/global.openai.gpt-5.6-sol", - ("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs", "n"), - ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions"), + ("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "n"), + ("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions", "stop"), ), ( "us.xai.grok-4.6",