fix(bedrock): inline http image urls and keep stop on converse for native chat completions

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
mateo-berri 2026-09-26 18:37:36 +00:00 • committed by mateo
parent e653217228
commit 093a9d4ddf
3 changed files with 127 additions and 9 deletions

View file

@ -22,6 +22,11 @@ import httpx
from typing_extensions import assert_never
import litellm
from litellm.litellm_core_utils.prompt_templates.image_handling import (
async_inline_remote_media,
convert_url_to_base64,
inline_remote_image_urls,
)
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path
@ -40,7 +45,7 @@ REASONING_CLOSE_TAG: Final = "</reasoning>"
CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType(
{
"openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs")),
"openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs")),
"openai.gpt-oss": frozenset(("logit_bias",)),
"xai.": frozenset(("frequency_penalty", "presence_penalty")),
}
@ -167,6 +172,48 @@ def split_reasoning_tag(content: str) -> tuple[str | None, str]:
return reasoning or None, body
def _remote_http_url(candidate: object) -> str | None:
return candidate if isinstance(candidate, str) and candidate.startswith(("http://", "https://")) else None
def _inlined_image_url_part(part: object) -> object:
fields: Final = part if isinstance(part, Mapping) else None
if fields is None or fields.get("type") != "image_url":
return part
image_url: Final = fields.get("image_url")
image_url_fields: Final = image_url if isinstance(image_url, Mapping) else None
url: Final = _remote_http_url(image_url_fields.get("url") if image_url_fields is not None else image_url)
if url is None:
return part
data_url: Final = convert_url_to_base64(url)
inlined: Final = {**image_url_fields, "url": data_url} if image_url_fields is not None else data_url
return {**fields, "image_url": inlined} # mutable-ok: json-serialized message part
def _inlined_image_url_message(message: AllMessageValues) -> AllMessageValues:
content: Final = message.get("content")
if not isinstance(content, list):
return message
inlined_message: Final = { # mutable-ok: json-serialized message
**message,
"content": [_inlined_image_url_part(part) for part in content],
}
return inlined_message # pyright: ignore[reportReturnType] # the same message with remote image parts inlined
def _with_inlined_remote_image_urls(
messages: list[AllMessageValues],
) -> list[AllMessageValues]: # mutable-ok: transform_request takes a list
"""Inline every remote ``image_url`` so AWS never sees the ``http(s)://`` URLs it rejects.
AWS's native surface only takes inline ``data:`` URLs and S3 URLs where Converse downloaded
remote images itself, so the bytes are fetched and inlined here exactly like Converse did.
"""
return [ # mutable-ok: transform_request takes a list
_inlined_image_url_message(message) for message in messages
]
class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler):
"""OpenAI chunk parsing plus the ``<reasoning>`` split, tracked per choice index."""
@ -222,6 +269,10 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
def custom_llm_provider(self) -> str | None:
return "bedrock"
@property
def uses_async_transform_request(self) -> bool:
return True
def get_error_class(
self,
error_message: str,
@ -325,7 +376,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
) -> dict: # mutable-ok: BaseConfig signature
return super().transform_request(
model=split_bedrock_region_path(model)[1],
messages=messages,
messages=_with_inlined_remote_image_urls(messages),
optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
headers=headers,
@ -341,7 +392,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
) -> dict: # mutable-ok: BaseConfig signature
return await super().async_transform_request(
model=split_bedrock_region_path(model)[1],
messages=messages,
messages=await async_inline_remote_media(messages, should_inline=inline_remote_image_urls),
optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
headers=headers,

View file

@ -899,6 +899,7 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
"thinking",
"additionalModelRequestFields",
"top_k",
"stop",
)
)
@ -921,7 +922,8 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje
Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking``
block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse
forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on
AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body,
AWS's native OpenAI surface, ``stop`` stays on Converse where it fails loudly instead of silently
stopping hidden reasoning, operator-owned request metadata is only written onto the Converse body,
function tools (``tools`` or legacy ``functions``) on a model without
``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless
``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as

View file

@ -258,11 +258,13 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model):
assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": None}) == "chat_completions"
@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"])
@pytest.mark.parametrize(
"model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6", "global.openai.gpt-5.6-sol"]
)
@pytest.mark.parametrize(
"request_params",
[{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}],
ids=["additionalModelRequestFields", "top_k"],
[{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}, {"stop": ["END"]}],
ids=["additionalModelRequestFields", "top_k", "stop"],
)
def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params):
assert bedrock_request_needs_converse(model, request_params) is True
@ -331,6 +333,69 @@ def test_map_openai_params_sends_max_tokens_as_max_completion_tokens():
assert mapped == {"max_completion_tokens": 64, "temperature": 0.1}
HTTPS_IMAGE_URL = "https://example.com/cat.png"
IMAGE_MESSAGES = [
{
"role": "user",
"content": [
{"type": "text", "text": "what is this"},
{"type": "image_url", "image_url": HTTPS_IMAGE_URL},
{"type": "image_url", "image_url": {"url": HTTPS_IMAGE_URL, "detail": "high"}},
{"type": "image_url", "image_url": {"url": "data:image/png;base64,AAA"}},
{"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}},
],
}
]
def _assert_remote_images_inlined(content):
assert content[0] == {"type": "text", "text": "what is this"}
assert content[1]["image_url"]["url"] == f"data:image/png;base64,{HTTPS_IMAGE_URL}"
assert content[2] == {
"type": "image_url",
"image_url": {"url": f"data:image/png;base64,{HTTPS_IMAGE_URL}", "detail": "high"},
}
assert content[3]["image_url"]["url"] == "data:image/png;base64,AAA"
assert content[4]["image_url"]["url"] == "s3://bucket/key.png"
def test_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch):
import litellm.llms.bedrock.chat.chat_completions.transformation as native_cc
monkeypatch.setattr(
native_cc, "convert_url_to_base64", lambda url: f"data:image/png;base64,{url}"
)
body = AmazonBedrockRuntimeChatCompletionsConfig().transform_request(
model="us.xai.grok-4.6",
messages=IMAGE_MESSAGES,
optional_params={},
litellm_params={},
headers={},
)
_assert_remote_images_inlined(body["messages"][0]["content"])
async def test_async_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch):
import litellm.litellm_core_utils.prompt_templates.image_handling as image_handling
async def fake_convert(url):
return f"data:image/png;base64,{url}"
monkeypatch.setattr(image_handling, "async_convert_url_to_base64", fake_convert)
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
assert cfg.uses_async_transform_request is True
body = await cfg.async_transform_request(
model="us.xai.grok-4.6",
messages=IMAGE_MESSAGES,
optional_params={},
litellm_params={},
headers={},
)
_assert_remote_images_inlined(body["messages"][0]["content"])
def test_map_openai_params_keeps_explicit_max_completion_tokens():
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
mapped = cfg.map_openai_params(
@ -398,8 +463,8 @@ def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map):
[
(
"bedrock/global.openai.gpt-5.6-sol",
("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs", "n"),
("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions"),
("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "n"),
("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions", "stop"),
),
(
"us.xai.grok-4.6",