mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-03 02:22:24 +00:00
fix(bedrock): inline http image urls and keep stop on converse for native chat completions
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
e653217228
commit
093a9d4ddf
3 changed files with 127 additions and 9 deletions
|
|
@ -22,6 +22,11 @@ import httpx
|
|||
from typing_extensions import assert_never
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.prompt_templates.image_handling import (
|
||||
async_inline_remote_media,
|
||||
convert_url_to_base64,
|
||||
inline_remote_image_urls,
|
||||
)
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
|
||||
from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path
|
||||
|
|
@ -40,7 +45,7 @@ REASONING_CLOSE_TAG: Final = "</reasoning>"
|
|||
|
||||
CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType(
|
||||
{
|
||||
"openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs")),
|
||||
"openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs")),
|
||||
"openai.gpt-oss": frozenset(("logit_bias",)),
|
||||
"xai.": frozenset(("frequency_penalty", "presence_penalty")),
|
||||
}
|
||||
|
|
@ -167,6 +172,48 @@ def split_reasoning_tag(content: str) -> tuple[str | None, str]:
|
|||
return reasoning or None, body
|
||||
|
||||
|
||||
def _remote_http_url(candidate: object) -> str | None:
|
||||
return candidate if isinstance(candidate, str) and candidate.startswith(("http://", "https://")) else None
|
||||
|
||||
|
||||
def _inlined_image_url_part(part: object) -> object:
|
||||
fields: Final = part if isinstance(part, Mapping) else None
|
||||
if fields is None or fields.get("type") != "image_url":
|
||||
return part
|
||||
image_url: Final = fields.get("image_url")
|
||||
image_url_fields: Final = image_url if isinstance(image_url, Mapping) else None
|
||||
url: Final = _remote_http_url(image_url_fields.get("url") if image_url_fields is not None else image_url)
|
||||
if url is None:
|
||||
return part
|
||||
data_url: Final = convert_url_to_base64(url)
|
||||
inlined: Final = {**image_url_fields, "url": data_url} if image_url_fields is not None else data_url
|
||||
return {**fields, "image_url": inlined} # mutable-ok: json-serialized message part
|
||||
|
||||
|
||||
def _inlined_image_url_message(message: AllMessageValues) -> AllMessageValues:
|
||||
content: Final = message.get("content")
|
||||
if not isinstance(content, list):
|
||||
return message
|
||||
inlined_message: Final = { # mutable-ok: json-serialized message
|
||||
**message,
|
||||
"content": [_inlined_image_url_part(part) for part in content],
|
||||
}
|
||||
return inlined_message # pyright: ignore[reportReturnType] # the same message with remote image parts inlined
|
||||
|
||||
|
||||
def _with_inlined_remote_image_urls(
|
||||
messages: list[AllMessageValues],
|
||||
) -> list[AllMessageValues]: # mutable-ok: transform_request takes a list
|
||||
"""Inline every remote ``image_url`` so AWS never sees the ``http(s)://`` URLs it rejects.
|
||||
|
||||
AWS's native surface only takes inline ``data:`` URLs and S3 URLs where Converse downloaded
|
||||
remote images itself, so the bytes are fetched and inlined here exactly like Converse did.
|
||||
"""
|
||||
return [ # mutable-ok: transform_request takes a list
|
||||
_inlined_image_url_message(message) for message in messages
|
||||
]
|
||||
|
||||
|
||||
class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler):
|
||||
"""OpenAI chunk parsing plus the ``<reasoning>`` split, tracked per choice index."""
|
||||
|
||||
|
|
@ -222,6 +269,10 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
|
|||
def custom_llm_provider(self) -> str | None:
|
||||
return "bedrock"
|
||||
|
||||
@property
|
||||
def uses_async_transform_request(self) -> bool:
|
||||
return True
|
||||
|
||||
def get_error_class(
|
||||
self,
|
||||
error_message: str,
|
||||
|
|
@ -325,7 +376,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
|
|||
) -> dict: # mutable-ok: BaseConfig signature
|
||||
return super().transform_request(
|
||||
model=split_bedrock_region_path(model)[1],
|
||||
messages=messages,
|
||||
messages=_with_inlined_remote_image_urls(messages),
|
||||
optional_params=self._inference_params(optional_params),
|
||||
litellm_params=litellm_params,
|
||||
headers=headers,
|
||||
|
|
@ -341,7 +392,7 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
|
|||
) -> dict: # mutable-ok: BaseConfig signature
|
||||
return await super().async_transform_request(
|
||||
model=split_bedrock_region_path(model)[1],
|
||||
messages=messages,
|
||||
messages=await async_inline_remote_media(messages, should_inline=inline_remote_image_urls),
|
||||
optional_params=self._inference_params(optional_params),
|
||||
litellm_params=litellm_params,
|
||||
headers=headers,
|
||||
|
|
|
|||
|
|
@ -899,6 +899,7 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
|
|||
"thinking",
|
||||
"additionalModelRequestFields",
|
||||
"top_k",
|
||||
"stop",
|
||||
)
|
||||
)
|
||||
|
||||
|
|
@ -921,7 +922,8 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje
|
|||
Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking``
|
||||
block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse
|
||||
forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on
|
||||
AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body,
|
||||
AWS's native OpenAI surface, ``stop`` stays on Converse where it fails loudly instead of silently
|
||||
stopping hidden reasoning, operator-owned request metadata is only written onto the Converse body,
|
||||
function tools (``tools`` or legacy ``functions``) on a model without
|
||||
``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless
|
||||
``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as
|
||||
|
|
|
|||
|
|
@ -258,11 +258,13 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model):
|
|||
assert BedrockModelInfo.get_bedrock_route(model, {"guardrailConfig": None}) == "chat_completions"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"])
|
||||
@pytest.mark.parametrize(
|
||||
"model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6", "global.openai.gpt-5.6-sol"]
|
||||
)
|
||||
@pytest.mark.parametrize(
|
||||
"request_params",
|
||||
[{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}],
|
||||
ids=["additionalModelRequestFields", "top_k"],
|
||||
[{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}, {"stop": ["END"]}],
|
||||
ids=["additionalModelRequestFields", "top_k", "stop"],
|
||||
)
|
||||
def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params):
|
||||
assert bedrock_request_needs_converse(model, request_params) is True
|
||||
|
|
@ -331,6 +333,69 @@ def test_map_openai_params_sends_max_tokens_as_max_completion_tokens():
|
|||
assert mapped == {"max_completion_tokens": 64, "temperature": 0.1}
|
||||
|
||||
|
||||
HTTPS_IMAGE_URL = "https://example.com/cat.png"
|
||||
IMAGE_MESSAGES = [
|
||||
{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "what is this"},
|
||||
{"type": "image_url", "image_url": HTTPS_IMAGE_URL},
|
||||
{"type": "image_url", "image_url": {"url": HTTPS_IMAGE_URL, "detail": "high"}},
|
||||
{"type": "image_url", "image_url": {"url": "data:image/png;base64,AAA"}},
|
||||
{"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}},
|
||||
],
|
||||
}
|
||||
]
|
||||
|
||||
|
||||
def _assert_remote_images_inlined(content):
|
||||
assert content[0] == {"type": "text", "text": "what is this"}
|
||||
assert content[1]["image_url"]["url"] == f"data:image/png;base64,{HTTPS_IMAGE_URL}"
|
||||
assert content[2] == {
|
||||
"type": "image_url",
|
||||
"image_url": {"url": f"data:image/png;base64,{HTTPS_IMAGE_URL}", "detail": "high"},
|
||||
}
|
||||
assert content[3]["image_url"]["url"] == "data:image/png;base64,AAA"
|
||||
assert content[4]["image_url"]["url"] == "s3://bucket/key.png"
|
||||
|
||||
|
||||
def test_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch):
|
||||
import litellm.llms.bedrock.chat.chat_completions.transformation as native_cc
|
||||
|
||||
monkeypatch.setattr(
|
||||
native_cc, "convert_url_to_base64", lambda url: f"data:image/png;base64,{url}"
|
||||
)
|
||||
body = AmazonBedrockRuntimeChatCompletionsConfig().transform_request(
|
||||
model="us.xai.grok-4.6",
|
||||
messages=IMAGE_MESSAGES,
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
_assert_remote_images_inlined(body["messages"][0]["content"])
|
||||
|
||||
|
||||
async def test_async_transform_request_inlines_remote_image_urls(local_cost_map, monkeypatch):
|
||||
import litellm.litellm_core_utils.prompt_templates.image_handling as image_handling
|
||||
|
||||
async def fake_convert(url):
|
||||
return f"data:image/png;base64,{url}"
|
||||
|
||||
monkeypatch.setattr(image_handling, "async_convert_url_to_base64", fake_convert)
|
||||
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
|
||||
assert cfg.uses_async_transform_request is True
|
||||
body = await cfg.async_transform_request(
|
||||
model="us.xai.grok-4.6",
|
||||
messages=IMAGE_MESSAGES,
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
_assert_remote_images_inlined(body["messages"][0]["content"])
|
||||
|
||||
|
||||
def test_map_openai_params_keeps_explicit_max_completion_tokens():
|
||||
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
|
||||
mapped = cfg.map_openai_params(
|
||||
|
|
@ -398,8 +463,8 @@ def test_supported_params_include_reasoning_effort_for_gpt56(local_cost_map):
|
|||
[
|
||||
(
|
||||
"bedrock/global.openai.gpt-5.6-sol",
|
||||
("frequency_penalty", "presence_penalty", "stop", "logprobs", "top_logprobs", "n"),
|
||||
("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions"),
|
||||
("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs", "n"),
|
||||
("temperature", "top_p", "logit_bias", "reasoning_effort", "tools", "functions", "stop"),
|
||||
),
|
||||
(
|
||||
"us.xai.grok-4.6",
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue