This commit is contained in:
Mateo Wang 2026-09-28 19:40:30 +00:00 • committed by GitHub
commit c915eca0cd
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
18 changed files with 1984 additions and 38 deletions

View file

@ -441,6 +441,10 @@ pub struct ModelInfo {
#[serde(skip_serializing_if = "Option::is_none")]
pub supports_audio_output: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub supports_bedrock_runtime_chat_completions_response_format: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub supports_computer_use: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub supports_embedding_image_input: Option<bool>,

View file

@ -1779,6 +1779,9 @@ if TYPE_CHECKING:
from .llms.bedrock.chat.invoke_transformations.amazon_openai_transformation import (
AmazonBedrockOpenAIConfig as AmazonBedrockOpenAIConfig,
)
from .llms.bedrock.chat.chat_completions.transformation import (
AmazonBedrockRuntimeChatCompletionsConfig as AmazonBedrockRuntimeChatCompletionsConfig,
)
from .llms.bedrock.image_generation.amazon_stability1_transformation import (
AmazonStabilityConfig as AmazonStabilityConfig,
)

View file

@ -206,6 +206,7 @@ LLM_CONFIG_NAMES: Final = (
"AmazonTwelveLabsPegasusConfig",
"AmazonInvokeConfig",
"AmazonBedrockOpenAIConfig",
"AmazonBedrockRuntimeChatCompletionsConfig",
"AmazonStabilityConfig",
"AmazonStability3Config",
"AmazonNovaCanvasConfig",
@ -868,6 +869,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
".llms.bedrock.chat.invoke_transformations.amazon_openai_transformation",
"AmazonBedrockOpenAIConfig",
),
"AmazonBedrockRuntimeChatCompletionsConfig": (
".llms.bedrock.chat.chat_completions.transformation",
"AmazonBedrockRuntimeChatCompletionsConfig",
),
"AmazonStabilityConfig": (
".llms.bedrock.image_generation.amazon_stability1_transformation",
"AmazonStabilityConfig",

View file

@ -15,6 +15,7 @@ import litellm
from litellm import verbose_logger
from litellm.caching.caching import InMemoryCache
from litellm.constants import MAX_IMAGE_URL_DOWNLOAD_SIZE_MB
from litellm.litellm_core_utils.prompt_templates.common_utils import infer_content_type_from_url_and_content
from litellm.litellm_core_utils.url_utils import SSRFError, async_safe_get, safe_get
from litellm.types.llms.openai import AllMessageValues
@ -55,23 +56,16 @@ def _process_image_response(response: Response, url: str) -> str:
base64_image: Final = base64.b64encode(image_bytes).decode("utf-8")
image_type: Final = response.headers.get("Content-Type")
if image_type is None:
img_type = url.split(".")[-1].lower()
_img_type: Final = {
"jpg": "image/jpeg",
"jpeg": "image/jpeg",
"png": "image/png",
"gif": "image/gif",
"webp": "image/webp",
}.get(img_type)
if _img_type is None:
raise Exception(
f"Error: Unsupported image format. Format={_img_type}. Supported types = ['image/jpeg', 'image/png', 'image/gif', 'image/webp']"
)
img_type = _img_type
else:
img_type = image_type
try:
img_type: Final = infer_content_type_from_url_and_content(
url=url,
content=bytes(image_bytes),
current_content_type=response.headers.get("Content-Type"),
)
except ValueError as e:
raise litellm.ImageFetchError(
f"Error: Unable to determine image content type from the server's headers, the URL, or the image bytes. url={url}"
) from e
result: Final = f"data:{img_type};base64,{base64_image}"
in_memory_cache.set_cache(url, result)
@ -310,6 +304,26 @@ async def _fetch_data_urls(remote_urls: tuple[str, ...]) -> tuple[str, ...]:
raise
def inline_remote_media(
messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues]
should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url,
) -> list[AllMessageValues]: # mutable-ok: every transform_request takes list[AllMessageValues]
remote_urls: Final = tuple(
dict.fromkeys(
remote.url
for message in messages
for part in _content_parts(message)
if (remote := _parse_remote_part(part)) is not None and should_inline(_remote_media(remote))
)
)
if not remote_urls:
return messages
data_urls: Final = MappingProxyType({url: convert_url_to_base64(url) for url in remote_urls})
return [ # mutable-ok: transform_request takes a list
_inline_message(message, data_urls, should_inline) for message in messages
]
async def async_inline_remote_media(
messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues]
should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url,

View file

@ -0,0 +1,420 @@
"""
Native OpenAI Chat Completions on Amazon Bedrock Runtime.
AWS serves this surface at
``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions``
for the models whose price-map ``supported_endpoints`` lists ``/v1/chat/completions``
(Grok 4.6, gpt-oss, the GPT-5.6 family): chat completions stay chat completions
instead of being rewritten to Converse.
Usage: model="us.xai.grok-4.6", model="bedrock/openai.gpt-oss-20b-1:0" or
model="bedrock/global.openai.gpt-5.6-sol". Explicit ``bedrock/converse/...``
still uses Converse, and so does a request that needs a Converse-only feature
(``bedrock_request_needs_converse`` in ``common_utils``).
"""
from collections.abc import AsyncIterator, Iterator, Mapping
from dataclasses import dataclass, replace
from types import MappingProxyType
from typing import TYPE_CHECKING, Final, Literal
import httpx
from typing_extensions import assert_never
import litellm
from litellm.litellm_core_utils.prompt_templates.image_handling import (
async_inline_remote_media,
inline_remote_image_urls,
inline_remote_media,
)
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path
from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler
from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig
from litellm.types.llms.openai import AllMessageValues
from litellm.types.utils import Choices, ModelResponse, ModelResponseStream
if TYPE_CHECKING:
import tiktoken
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
REASONING_OPEN_TAG: Final = "<reasoning>"
REASONING_CLOSE_TAG: Final = "</reasoning>"
CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType(
{
"openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs")),
"openai.gpt-oss": frozenset(("logit_bias",)),
"xai.": frozenset(("frequency_penalty", "presence_penalty")),
}
)
def chat_completions_params_refused_for(model: str) -> frozenset[str]:
"""The OpenAI params AWS's Chat Completions endpoint rejects for this model whatever else the request says.
Each family answers them with a 400 (GPT-5.6, gpt-oss) or a 503 (Grok), where Converse dropped the same
params under ``drop_params``, so the native config leaves them out of its supported list and the usual
drop-or-raise handling applies before the request reaches AWS.
"""
model_id: Final = split_bedrock_region_path(model)[1]
return frozenset().union(
*(refused for family, refused in CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY.items() if family in model_id)
)
CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY: Final = MappingProxyType({"xai.": frozenset(("none",))})
def chat_completions_reasoning_efforts_refused_for(model: str) -> frozenset[str]:
"""The ``reasoning_effort`` values AWS's Chat Completions endpoint rejects for this model.
Grok answers ``"none"`` with a 400 (it takes low, medium, high, and xhigh) where Converse dropped every
``reasoning_effort`` for it, so the native config drops the value and AWS applies its default effort as before.
"""
model_id: Final = split_bedrock_region_path(model)[1]
return frozenset().union(
*(
refused
for family, refused in CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY.items()
if family in model_id
)
)
def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) -> Mapping[str, object]:
if params.get("reasoning_effort") not in chat_completions_reasoning_efforts_refused_for(model):
return params
return MappingProxyType({key: value for key, value in params.items() if key != "reasoning_effort"})
def _held_close_tag_prefix(text: str) -> int:
return next(
(
size
for size in range(min(len(text), len(REASONING_CLOSE_TAG) - 1), 0, -1)
if REASONING_CLOSE_TAG.startswith(text[-size:])
),
0,
)
@dataclass(frozen=True, slots=True)
class ReasoningTagSplitter:
"""
The same split for a stream of content deltas, where a tag can arrive across chunks.
``feed`` returns the next state plus the reasoning and content text the delta contributes;
``flush`` releases what the stream ended on before a tag resolved.
"""
phase: Literal["start", "reasoning", "after_close", "content"] = "start"
pending: str = ""
def feed(self, text: str) -> tuple["ReasoningTagSplitter", str, str]:
match self.phase:
case "content":
return self, "", text
case "after_close":
content: Final = text.lstrip()
return (replace(self, phase="content") if content else self), "", content
case "start":
return self._feed_start(self.pending + text)
case "reasoning":
return self._feed_reasoning(self.pending + text)
case _:
assert_never(self.phase)
def _feed_start(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]:
if buffered.startswith(REASONING_OPEN_TAG):
return replace(self, phase="reasoning", pending="")._feed_reasoning(buffered[len(REASONING_OPEN_TAG) :])
if REASONING_OPEN_TAG.startswith(buffered):
return replace(self, pending=buffered), "", ""
return replace(self, phase="content", pending=""), "", buffered
def _feed_reasoning(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]:
close_at: Final = buffered.find(REASONING_CLOSE_TAG)
if close_at >= 0:
after_close: Final = replace(self, phase="after_close", pending="")
next_state, _, content = after_close.feed(buffered[close_at + len(REASONING_CLOSE_TAG) :])
return next_state, buffered[:close_at], content
held: Final = _held_close_tag_prefix(buffered)
return replace(self, pending=buffered[len(buffered) - held :]), buffered[: len(buffered) - held], ""
def flush(self) -> tuple["ReasoningTagSplitter", str, str]:
drained: Final = replace(self, phase="content", pending="")
if self.phase == "reasoning":
return drained, self.pending, ""
return drained, "", self.pending
def _split_streamed_content(
splitter: ReasoningTagSplitter, content: str | None, finished: bool
) -> tuple[ReasoningTagSplitter, str, str]:
fed_state, fed_reasoning, fed_content = splitter.feed(content or "")
if not finished:
return fed_state, fed_reasoning, fed_content
drained, flushed_reasoning, flushed_content = fed_state.flush()
return drained, fed_reasoning + flushed_reasoning, fed_content + flushed_content
def split_reasoning_tag(content: str) -> tuple[str | None, str]:
"""
Split gpt-oss's inline ``<reasoning>...</reasoning>`` prefix out of a complete message.
Runs the streaming splitter over the whole message, so a streamed and a non-streamed
response to the same completion split identically. Returns ``(None, content)`` when the
message does not start with the tag.
"""
_, reasoning, body = _split_streamed_content(ReasoningTagSplitter(), content, finished=True)
return reasoning or None, body
class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler):
"""OpenAI chunk parsing plus the ``<reasoning>`` split, tracked per choice index."""
def __init__(
self,
streaming_response: Iterator[str] | AsyncIterator[str] | ModelResponse,
sync_stream: bool,
json_mode: bool | None = False,
) -> None:
super().__init__(streaming_response=streaming_response, sync_stream=sync_stream, json_mode=json_mode)
self._splitters: Mapping[int, ReasoningTagSplitter] = MappingProxyType({})
def chunk_parser(self, chunk: dict) -> ModelResponseStream: # mutable-ok: BaseModelResponseIterator signature
parsed: Final = super().chunk_parser(chunk)
for choice in parsed.choices:
next_state, reasoning, content = _split_streamed_content(
self._splitters.get(choice.index, ReasoningTagSplitter()),
choice.delta.content,
choice.finish_reason is not None,
)
self._splitters = MappingProxyType({**self._splitters, choice.index: next_state})
if reasoning:
choice.delta.reasoning_content = f"{getattr(choice.delta, 'reasoning_content', None) or ''}{reasoning}"
if content or choice.delta.content is not None:
choice.delta.content = content
return parsed
def with_max_completion_tokens(params: Mapping[str, object]) -> Mapping[str, object]:
"""
Send the caller's ``max_tokens`` as ``max_completion_tokens``.
Every model on this surface accepts ``max_completion_tokens`` and the GPT-5.6 family
rejects ``max_tokens``; an explicit ``max_completion_tokens`` wins when both are set.
"""
if "max_tokens" not in params:
return params
return MappingProxyType(
{
key: value
for key, value in (("max_completion_tokens", params["max_tokens"]), *params.items())
if key != "max_tokens"
}
)
class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
def __init__(self, aws_signer: BaseAWSLLM | None = None) -> None:
super().__init__()
self._aws_signer: Final = aws_signer or BaseAWSLLM()
@property
def custom_llm_provider(self) -> str | None:
return "bedrock"
@property
def uses_async_transform_request(self) -> bool:
return True
def get_error_class(
self,
error_message: str,
status_code: int,
headers: dict[str, object] | httpx.Headers, # mutable-ok: BaseConfig signature
) -> BaseLLMException:
return BedrockError(status_code=status_code, message=error_message, headers=headers)
def get_complete_url(
self,
api_base: str | None,
api_key: str | None,
model: str,
optional_params: dict, # mutable-ok: BaseConfig signature
litellm_params: dict, # mutable-ok: BaseConfig signature
stream: bool | None = None,
) -> str:
if api_base is not None and "chat/completions" in api_base:
return api_base.rstrip("/")
aws_region_name: Final = self._aws_signer._get_aws_region_name( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public region resolver
optional_params=self._params_with_region_from_path(optional_params, model), model=model
)
endpoint_url, _ = self._aws_signer.get_runtime_endpoint(
api_base=api_base,
aws_bedrock_runtime_endpoint=optional_params.get("aws_bedrock_runtime_endpoint"),
aws_region_name=aws_region_name,
)
base: Final = endpoint_url.rstrip("/")
if base.endswith("/openai/v1/chat/completions"):
return base
if base.endswith("/openai/v1"):
return f"{base}/chat/completions"
return f"{base}/openai/v1/chat/completions"
def _params_with_region_from_path(
self, optional_params: dict, model: str | None
) -> dict: # mutable-ok: BaseAWSLLM's region resolver and signer take a plain dict
region_from_path, _ = split_bedrock_region_path(model or "")
if region_from_path is None or optional_params.get("aws_region_name") is not None:
return optional_params
return {**optional_params, "aws_region_name": region_from_path} # mutable-ok: BaseAWSLLM takes a plain dict
def sign_request(
self,
headers: dict, # mutable-ok: BaseConfig signature
optional_params: dict, # mutable-ok: BaseConfig signature
request_data: dict, # mutable-ok: BaseConfig signature
api_base: str,
api_key: str | None = None,
model: str | None = None,
stream: bool | None = None,
fake_stream: bool | None = None,
) -> tuple[dict, bytes | None]: # mutable-ok: BaseConfig signature
return self._aws_signer._sign_request( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public signer
service_name="bedrock",
headers=headers,
optional_params=self._params_with_region_from_path(optional_params, model),
request_data=request_data,
api_base=api_base,
api_key=api_key,
model=model,
stream=stream,
fake_stream=fake_stream,
)
def map_openai_params(
self,
non_default_params: dict, # mutable-ok: BaseConfig signature
optional_params: dict, # mutable-ok: BaseConfig signature
model: str,
drop_params: bool,
replace_max_completion_tokens_with_max_tokens: bool = False,
) -> dict: # mutable-ok: BaseConfig signature
mapped: Final = super().map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model=model,
drop_params=drop_params,
replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens,
)
return dict( # mutable-ok: get_optional_params keeps filling this dict
without_refused_reasoning_effort(model, with_max_completion_tokens(mapped))
)
def _inference_params(
self, optional_params: Mapping[str, object]
) -> dict[str, object]: # mutable-ok: BaseConfig signature of transform_request
return { # mutable-ok: OpenAILikeChatConfig.transform_request takes a plain dict
key: value
for key, value in optional_params.items()
if key not in self._aws_signer.aws_authentication_params
}
def transform_request(
self,
model: str,
messages: list[AllMessageValues], # mutable-ok: BaseConfig signature
optional_params: dict, # mutable-ok: BaseConfig signature
litellm_params: dict, # mutable-ok: BaseConfig signature
headers: dict, # mutable-ok: BaseConfig signature
) -> dict: # mutable-ok: BaseConfig signature
return super().transform_request(
model=split_bedrock_region_path(model)[1],
messages=inline_remote_media(messages, should_inline=inline_remote_image_urls),
optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
headers=headers,
)
async def async_transform_request(
self,
model: str,
messages: list[AllMessageValues], # mutable-ok: BaseConfig signature
optional_params: dict, # mutable-ok: BaseConfig signature
litellm_params: dict, # mutable-ok: BaseConfig signature
headers: dict, # mutable-ok: BaseConfig signature
) -> dict: # mutable-ok: BaseConfig signature
return await super().async_transform_request(
model=split_bedrock_region_path(model)[1],
messages=await async_inline_remote_media(messages, should_inline=inline_remote_image_urls),
optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
headers=headers,
)
def transform_response(
self,
model: str,
raw_response: httpx.Response,
model_response: ModelResponse,
logging_obj: "LiteLLMLoggingObj",
request_data: dict, # mutable-ok: BaseConfig signature
messages: list[AllMessageValues], # mutable-ok: BaseConfig signature
optional_params: dict, # mutable-ok: BaseConfig signature
litellm_params: dict, # mutable-ok: BaseConfig signature
encoding: "tiktoken.Encoding | None",
api_key: str | None = None,
json_mode: bool | None = None,
) -> ModelResponse:
response: Final = super().transform_response(
model=model,
raw_response=raw_response,
model_response=model_response,
logging_obj=logging_obj,
request_data=request_data,
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
encoding=encoding,
api_key=api_key,
json_mode=json_mode,
)
for choice in response.choices:
if not isinstance(choice, Choices) or not isinstance(choice.message.content, str):
continue
reasoning, content = split_reasoning_tag(choice.message.content)
if reasoning is not None:
choice.message.reasoning_content = (
f"{getattr(choice.message, 'reasoning_content', None) or ''}{reasoning}"
)
choice.message.content = content
return response
def get_supported_openai_params(self, model: str) -> list: # mutable-ok: BaseConfig signature
refused: Final = frozenset(("n", *chat_completions_params_refused_for(model)))
base_params: Final = tuple(
param for param in super().get_supported_openai_params(model) if param not in refused
)
reasoning_param: Final = (
("reasoning_effort",)
if "reasoning_effort" not in base_params
and litellm.supports_reasoning(model=model, custom_llm_provider=self.custom_llm_provider)
else ()
)
return [*base_params, *reasoning_param] # mutable-ok: BaseConfig signature returns a list
def get_model_response_iterator(
self,
streaming_response: Iterator[str] | AsyncIterator[str] | ModelResponse,
sync_stream: bool,
json_mode: bool | None = False,
) -> BedrockRuntimeChatCompletionsStreamingHandler:
return BedrockRuntimeChatCompletionsStreamingHandler(
streaming_response=streaming_response,
sync_stream=sync_stream,
json_mode=json_mode,
)

View file

@ -10,6 +10,7 @@ import json
import os
import re
from collections.abc import Mapping, Sequence
from types import MappingProxyType
from typing import TYPE_CHECKING, Any, Final, Literal, TypedDict
if TYPE_CHECKING:
@ -28,6 +29,7 @@ from litellm.llms.base_llm.anthropic_messages.transformation import (
)
from litellm.llms.base_llm.base_utils import BaseLLMModelInfo, BaseTokenCounter
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.bedrock.request_metadata import bedrock_request_metadata_is_owned
from litellm.secret_managers.main import get_secret, get_secret_str
from litellm.types.llms.bedrock import AWS_AUTH_PARAM_KEYS, AwsAuthParams
@ -37,6 +39,18 @@ if TYPE_CHECKING:
_ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs"
_OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.")
BedrockRoute = Literal[
"converse",
"invoke",
"claude_platform",
"converse_like",
"agent",
"agentcore",
"async_invoke",
"openai",
"mantle",
"chat_completions",
]
def error_response_text(response: httpx.Response) -> str:
@ -793,6 +807,151 @@ def strip_bedrock_routing_prefix(model: str) -> str:
return model
def split_bedrock_region_path(model: str) -> tuple[str | None, str]:
"""Split a ``<region>/<model-id>`` routing path into the region and the id AWS receives.
``bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0`` -> ``("us-gov-west-1", "openai.gpt-oss-20b-1:0")``;
a model without a region path comes back as ``(None, <routing-prefix-stripped id>)``.
"""
stripped: Final = strip_bedrock_routing_prefix(model)
region, separator, model_id = stripped.partition("/")
if separator and region in _get_all_bedrock_regions():
return region, model_id
return None, stripped
BEDROCK_RUNTIME_PRICE_MAP_PROVIDERS: Final = frozenset(("bedrock", "bedrock_converse"))
def _bedrock_price_map_entries(model: str) -> tuple[Mapping[str, object] | None, ...]:
return tuple(
litellm.model_cost.get(key)
for key in (model, strip_bedrock_routing_prefix(model), split_bedrock_region_path(model)[1])
)
def _bedrock_price_map_flag(model: str, flag: str) -> bool:
return any(entry is not None and entry.get(flag) is True for entry in _bedrock_price_map_entries(model))
def _bedrock_runtime_row_lists_chat_completions(entry: Mapping[str, object]) -> bool:
endpoints: Final = entry.get("supported_endpoints")
return (
entry.get("litellm_provider") in BEDROCK_RUNTIME_PRICE_MAP_PROVIDERS
and isinstance(endpoints, (list, tuple))
and "/v1/chat/completions" in endpoints
)
def uses_bedrock_runtime_chat_completions(model: str) -> bool:
"""Whether this Bedrock model should use runtime native Chat Completions.
Data-driven from ``/v1/chat/completions`` in the price-map row's ``supported_endpoints``,
the same per-model signal ``bedrock_supports_openai_responses`` reads for ``/v1/responses``,
so onboarding a model is a JSON change. Explicit ``converse/`` still wins in
``get_bedrock_route`` because prefix routes are checked first, and a request
that needs a Converse-only feature (``bedrock_request_needs_converse``) is
served by Converse even on a listed model. Only a bedrock-runtime row counts: a
``bedrock_mantle`` row lists the endpoints of the Mantle host, not this one.
"""
return any(
entry is not None and _bedrock_runtime_row_lists_chat_completions(entry)
for entry in _bedrock_price_map_entries(model)
)
def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool:
"""Whether AWS's native Chat Completions serves this model's function tools with any ``reasoning_effort``.
Data-driven from the price-map ``supports_bedrock_runtime_chat_completions_tools_with_reasoning``
flag (gpt-oss, Grok). Without it AWS only takes tools with ``reasoning_effort="none"``
(the GPT-5.6 family), and Converse serves tools with any effort, so those requests fall back to it.
"""
return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_tools_with_reasoning")
def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> bool:
"""Whether AWS's native Chat Completions enforces a ``response_format`` schema for this model.
Data-driven from the price-map ``supports_bedrock_runtime_chat_completions_response_format`` flag
(GPT-5.6, Grok). Without it AWS accepts the field and answers with unconstrained text (gpt-oss), so
Converse, which emulates the schema through a forced ``json_tool_call`` tool, serves those requests.
"""
return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_response_format")
BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
(
"guardrailConfig",
"performanceConfig",
"serviceTier",
"requestMetadata",
"outputConfig",
"thinking",
"additionalModelRequestFields",
"top_k",
"stop",
)
)
def _response_format_needs_converse(model: str, response_format: object) -> bool:
if response_format is None:
return False
if not isinstance(response_format, Mapping):
return not bedrock_runtime_chat_completions_enforces_response_format(model)
response_format_type: Final = response_format.get("type")
if response_format_type == "text":
return False
is_json_schema: Final = response_format_type == "json_schema" and "json_schema" in response_format
return not (is_json_schema and bedrock_runtime_chat_completions_enforces_response_format(model))
def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool:
"""Whether a request on a runtime-Chat-Completions model must still be served by Converse.
Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking``
block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse
forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on
AWS's native OpenAI surface, ``stop`` stays on Converse where it fails loudly instead of silently
stopping hidden reasoning, operator-owned request metadata is only written onto the Converse body,
function tools (``tools`` or legacy ``functions``) on a model without
``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless
``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as
``{"type": "json_schema", "json_schema": ...}`` (a pydantic model is converted to that) on a model with
``supports_bedrock_runtime_chat_completions_response_format``: a schema on any other model is only
honored by Converse, and every ``json_object`` form (``response_schema`` included) keeps Converse's
handling everywhere, since AWS's native surface rejects that type with a 400 unless the prompt
mentions json.
"""
if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS):
return True
if bedrock_request_metadata_is_owned():
return True
if _response_format_needs_converse(model, request_params.get("response_format")):
return True
if not (request_params.get("tools") or request_params.get("functions")):
return False
return (
not bedrock_runtime_chat_completions_serves_tools_with_reasoning(model)
and request_params.get("reasoning_effort") != "none"
)
def bedrock_route_for_request(
model: str, request_params: Mapping[str, object], additional_drop_params: Sequence[str] | None
) -> BedrockRoute:
"""The route for one request, decided from the caller's raw params before any provider mapping.
Param mapping and dispatch both call this with the same inputs, so a request that falls back to
Converse is mapped with the Converse config and sent to Converse, never one without the other.
"""
dropped: Final = frozenset(additional_drop_params or ())
return BedrockModelInfo.get_bedrock_route(
model, MappingProxyType({key: value for key, value in request_params.items() if key not in dropped})
)
def strip_bedrock_throughput_suffix(model: str) -> str:
"""Strip throughput tier suffixes and context window suffixes from Bedrock model names."""
import re
@ -1150,19 +1309,13 @@ class BedrockModelInfo(BaseLLMModelInfo):
@staticmethod
def get_bedrock_route(
model: str,
) -> Literal[
"converse",
"invoke",
"claude_platform",
"converse_like",
"agent",
"agentcore",
"async_invoke",
"openai",
"mantle",
]:
request_params: Mapping[str, object] | None = None,
) -> BedrockRoute:
"""
Get the bedrock route for the given model.
``request_params`` (the caller's chat params) lets a runtime Chat Completions
model fall back to Converse for the requests only Converse can serve.
"""
route_mappings: dict[
str,
@ -1176,6 +1329,7 @@ class BedrockModelInfo(BaseLLMModelInfo):
"async_invoke",
"openai",
"mantle",
"chat_completions",
],
] = {
"invoke/": "invoke",
@ -1205,6 +1359,11 @@ class BedrockModelInfo(BaseLLMModelInfo):
if is_bedrock_application_inference_profile_arn(model):
return "converse"
if uses_bedrock_runtime_chat_completions(model) and not (
request_params is not None and bedrock_request_needs_converse(model, request_params)
):
return "chat_completions"
base_model: Final = BedrockModelInfo.get_base_model(model)
alt_model: Final = BedrockModelInfo.get_non_litellm_routing_model_name(model=model)
if base_model in litellm.bedrock_converse_models or alt_model in litellm.bedrock_converse_models:
@ -1383,6 +1542,8 @@ def get_bedrock_chat_config(model: str):
return litellm.AmazonConverseConfig()
elif bedrock_route == "openai":
return litellm.AmazonBedrockOpenAIConfig()
elif bedrock_route == "chat_completions":
return litellm.AmazonBedrockRuntimeChatCompletionsConfig()
elif bedrock_route == "agent":
from litellm.llms.bedrock.chat.invoke_agent.transformation import (
AmazonInvokeAgentConfig,

View file

@ -116,7 +116,7 @@ from litellm.llms.base_llm import BaseConfig, BaseImageGenerationConfig
from litellm.llms.base_llm.base_model_iterator import (
convert_model_response_to_streaming,
)
from litellm.llms.bedrock.common_utils import BedrockModelInfo
from litellm.llms.bedrock.common_utils import BedrockModelInfo, bedrock_route_for_request
from litellm.llms.cohere.common_utils import CohereModelInfo
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler, http2_enabled
from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config
@ -4205,7 +4205,9 @@ def _complete_bedrock(ctx: _CompletionDispatchContext) -> _CompletionDispatchRes
if "aws_region_name" not in optional_params or optional_params["aws_region_name"] is None:
optional_params["aws_region_name"] = aws_bedrock_client.meta.region_name
bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model)
bedrock_route: Final = bedrock_route_for_request(
model, ctx.request_params, ctx.kwargs.get("additional_drop_params")
)
if bedrock_route == "claude_platform":
provider_config = ProviderConfigManager.get_provider_chat_config(
model=model,
@ -5836,6 +5838,7 @@ def completion(
optional_params=optional_params,
organization=organization,
provider_config=provider_config,
request_params=MappingProxyType({**optional_param_args, **non_default_params}),
shared_session=shared_session,
stream=stream,
temperature=temperature,

View file

@ -41410,6 +41410,10 @@
"output_cost_per_token": 0.0
},
"openai.gpt-oss-120b-1:0": {
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 1.5e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@ -41424,6 +41428,10 @@
"supports_tool_choice": true
},
"openai.gpt-oss-20b-1:0": {
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@ -47162,6 +47170,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -47175,6 +47187,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -47184,6 +47200,11 @@
"input_cost_per_token": 2.64e-06,
"output_cost_per_token": 7.92e-06,
"cache_read_input_token_cost": 6.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@ -57748,6 +57769,7 @@
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-luna.html"
},
"us.openai.gpt-5.6-sol": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4.4e-06,
"input_cost_per_token_above_272k_tokens": 8.8e-06,
"cache_creation_input_token_cost": 5.5e-06,
@ -57778,10 +57800,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"global.openai.gpt-5.6-sol": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4e-06,
"input_cost_per_token_above_272k_tokens": 8e-06,
"cache_creation_input_token_cost": 5e-06,
@ -57812,10 +57836,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"us.openai.gpt-5.6-terra": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_above_272k_tokens": 4.4e-06,
"cache_creation_input_token_cost": 2.75e-06,
@ -57846,10 +57872,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"global.openai.gpt-5.6-terra": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-06,
"input_cost_per_token_above_272k_tokens": 4e-06,
"cache_creation_input_token_cost": 2.5e-06,
@ -57880,10 +57908,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"us.openai.gpt-5.6-luna": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-07,
"input_cost_per_token_above_272k_tokens": 4.4e-07,
"cache_creation_input_token_cost": 2.75e-07,
@ -57914,6 +57944,7 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
@ -58042,6 +58073,7 @@
]
},
"global.openai.gpt-5.6-luna": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_272k_tokens": 4e-07,
"cache_creation_input_token_cost": 2.5e-07,
@ -58072,6 +58104,7 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
@ -58675,6 +58708,11 @@
"input_cost_per_token": 2.2e-06,
"output_cost_per_token": 6.6e-06,
"cache_read_input_token_cost": 5.5e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@ -58691,6 +58729,11 @@
"input_cost_per_token": 2e-06,
"output_cost_per_token": 6e-06,
"cache_read_input_token_cost": 5e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@ -64617,6 +64660,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -64630,6 +64677,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -64871,6 +64922,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -64884,6 +64939,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,

View file

@ -1,6 +1,6 @@
from __future__ import annotations
from collections.abc import Callable, Coroutine, Iterable
from collections.abc import Callable, Coroutine, Iterable, Mapping
from dataclasses import dataclass
from typing import TYPE_CHECKING, Any, Literal, Union
@ -229,6 +229,7 @@ class _CompletionDispatchContext:
optional_params: dict
organization: str | None
provider_config: BaseConfig | None
request_params: Mapping[str, object]
shared_session: ClientSession | None
stream: bool | None
temperature: float | None

View file

@ -408,7 +408,7 @@ if TYPE_CHECKING:
BaseVectorStoreFilesConfig,
)
from litellm.llms.base_llm.videos.transformation import BaseVideoConfig
from litellm.llms.bedrock.common_utils import BedrockModelInfo
from litellm.llms.bedrock.common_utils import BedrockModelInfo, BedrockRoute
from litellm.llms.bedrock.embed.amazon_nova_transformation import (
AmazonNovaEmbeddingConfig,
)
@ -3453,6 +3453,14 @@ def _should_drop_param(k, additional_drop_params) -> bool:
return False
def _bedrock_route_for_request(
model: str, passed_params: Mapping[str, object], additional_drop_params: list | None
) -> BedrockRoute:
from litellm.llms.bedrock.common_utils import bedrock_route_for_request
return bedrock_route_for_request(model, passed_params, additional_drop_params)
def _get_non_default_params(passed_params: dict, default_params: dict, additional_drop_params: list | None) -> dict:
non_default_params: Final = {}
for k, v in passed_params.items():
@ -4493,9 +4501,17 @@ def get_optional_params(
message=f"{custom_llm_provider} does not support parameters: {list(unsupported_params.keys())}, for model={model}. To drop these, set `litellm.drop_params=True` or for proxy:\n\n`litellm_settings:\n drop_params: true`\n. \n If you want to use these params dynamically send allowed_openai_params={list(unsupported_params.keys())} in your request.",
)
bedrock_route: Final = (
_bedrock_route_for_request(model, passed_params, additional_drop_params)
if custom_llm_provider == "bedrock"
else None
)
get_supported_openai_params: Final[_SupportedOpenAIParamsGetter] = litellm_utils.get_supported_openai_params
supported_params = get_supported_openai_params(
model=model, custom_llm_provider=custom_llm_provider, base_model=base_model
supported_params = (
litellm.AmazonConverseConfig().get_supported_openai_params(model=model)
if bedrock_route == "converse"
and isinstance(provider_config, litellm.AmazonBedrockRuntimeChatCompletionsConfig)
else get_supported_openai_params(model=model, custom_llm_provider=custom_llm_provider, base_model=base_model)
)
if supported_params is None:
supported_params = get_supported_openai_params(model=model, custom_llm_provider="openai")
@ -4665,7 +4681,6 @@ def get_optional_params(
)
elif custom_llm_provider == "bedrock":
bedrock_model_info: Final[type[BedrockModelInfo]] = litellm_utils.BedrockModelInfo
bedrock_route: Final = bedrock_model_info.get_bedrock_route(model)
bedrock_base_model: Final = bedrock_model_info.get_base_model(model)
if bedrock_route == "converse" or bedrock_route == "converse_like":
optional_params = litellm.AmazonConverseConfig().map_openai_params(

View file

@ -41410,6 +41410,10 @@
"output_cost_per_token": 0.0
},
"openai.gpt-oss-120b-1:0": {
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 1.5e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@ -41424,6 +41428,10 @@
"supports_tool_choice": true
},
"openai.gpt-oss-20b-1:0": {
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@ -47162,6 +47170,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -47175,6 +47187,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -47184,6 +47200,11 @@
"input_cost_per_token": 2.64e-06,
"output_cost_per_token": 7.92e-06,
"cache_read_input_token_cost": 6.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@ -57748,6 +57769,7 @@
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-luna.html"
},
"us.openai.gpt-5.6-sol": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4.4e-06,
"input_cost_per_token_above_272k_tokens": 8.8e-06,
"cache_creation_input_token_cost": 5.5e-06,
@ -57778,10 +57800,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"global.openai.gpt-5.6-sol": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4e-06,
"input_cost_per_token_above_272k_tokens": 8e-06,
"cache_creation_input_token_cost": 5e-06,
@ -57812,10 +57836,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"us.openai.gpt-5.6-terra": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_above_272k_tokens": 4.4e-06,
"cache_creation_input_token_cost": 2.75e-06,
@ -57846,10 +57872,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"global.openai.gpt-5.6-terra": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-06,
"input_cost_per_token_above_272k_tokens": 4e-06,
"cache_creation_input_token_cost": 2.5e-06,
@ -57880,10 +57908,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"us.openai.gpt-5.6-luna": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-07,
"input_cost_per_token_above_272k_tokens": 4.4e-07,
"cache_creation_input_token_cost": 2.75e-07,
@ -57914,6 +57944,7 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
@ -58042,6 +58073,7 @@
]
},
"global.openai.gpt-5.6-luna": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_272k_tokens": 4e-07,
"cache_creation_input_token_cost": 2.5e-07,
@ -58072,6 +58104,7 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
@ -58675,6 +58708,11 @@
"input_cost_per_token": 2.2e-06,
"output_cost_per_token": 6.6e-06,
"cache_read_input_token_cost": 5.5e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@ -58691,6 +58729,11 @@
"input_cost_per_token": 2e-06,
"output_cost_per_token": 6e-06,
"cache_read_input_token_cost": 5e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@ -64617,6 +64660,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -64630,6 +64677,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -64871,6 +64922,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -64884,6 +64939,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,

View file

@ -950,6 +950,12 @@
"supports_audio_output": {
"type": "boolean"
},
"supports_bedrock_runtime_chat_completions_response_format": {
"type": "boolean"
},
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": {
"type": "boolean"
},
"supports_computer_use": {
"type": "boolean"
},

View file

@ -1,4 +1,5 @@
import asyncio
import base64
import copy
import time
import uuid
@ -16,6 +17,7 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import (
async_convert_url_to_base64,
async_inline_remote_media,
convert_url_to_base64,
inline_remote_media,
)
from litellm.litellm_core_utils.url_utils import SSRFError
@ -258,6 +260,54 @@ async def test_async_data_url_is_returned_unchanged_without_fetch(monkeypatch):
assert await async_convert_url_to_base64(data_url) == data_url
REAL_PNG_BYTES = base64.b64decode(
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="
)
def _stub_image_client(content, content_type):
class _Client:
def get(self, url, follow_redirects=True):
headers = {} if content_type is None else {"Content-Type": content_type}
return Response(200, content=content, headers=headers, request=Request("GET", url))
return _Client()
def test_convert_url_to_base64_infers_the_type_when_the_server_sends_octet_stream(monkeypatch):
monkeypatch.setattr(
litellm, "module_level_client", _stub_image_client(REAL_PNG_BYTES, "application/octet-stream")
)
result = convert_url_to_base64(f"http://img.example/{uuid.uuid4()}")
assert result.startswith("data:image/png;base64,")
def test_convert_url_to_base64_keeps_a_real_content_type(monkeypatch):
monkeypatch.setattr(
litellm, "module_level_client", _stub_image_client(REAL_PNG_BYTES, "image/jpeg")
)
result = convert_url_to_base64(f"http://img.example/{uuid.uuid4()}.png")
assert result.startswith("data:image/jpeg;base64,")
def test_convert_url_to_base64_raises_when_no_content_type_is_determinable(monkeypatch):
monkeypatch.setattr(
litellm,
"module_level_client",
_stub_image_client(b"\x00\x01\x02\x03not-an-image", "application/octet-stream"),
)
url = f"http://img.example/{uuid.uuid4()}"
with pytest.raises(litellm.ImageFetchError) as excinfo:
convert_url_to_base64(url)
assert url in str(excinfo.value)
def test_image_size_limit_disabled(monkeypatch):
"""
Test that setting MAX_IMAGE_URL_DOWNLOAD_SIZE_MB to 0 disables all image URL downloads.
@ -320,6 +370,50 @@ async def test_async_inline_remote_media_inlines_every_remote_part_shape(async_o
assert messages == snapshot
def test_inline_remote_media_inlines_every_remote_part_shape(monkeypatch):
image_url = f"http://img.example/{uuid.uuid4()}.png"
pdf_url = f"http://docs.example/{uuid.uuid4()}.pdf"
fetched = []
def fake_convert(url):
fetched.append(url)
return f"data:image/png;base64,{url}"
monkeypatch.setattr(image_handling, "convert_url_to_base64", fake_convert)
messages = [
{"role": "system", "content": "be terse"},
{
"role": "user",
"content": [
{"type": "text", "text": "what is this?"},
{"type": "image_url", "image_url": {"url": image_url, "detail": "low"}},
{"type": "image_url", "image_url": image_url},
{"type": "image_url", "image_url": {"url": "data:image/png;base64,iVBORw0KGgo="}},
{"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}},
{"type": "file", "file": {"file_id": pdf_url}},
{"type": "document", "source": {"type": "url", "url": pdf_url}, "title": "the doc"},
],
},
]
snapshot = copy.deepcopy(messages)
inlined = inline_remote_media(messages, should_inline=image_handling.inline_remote_image_urls)
data_url = f"data:image/png;base64,{image_url}"
assert inlined[0] == {"role": "system", "content": "be terse"}
assert inlined[1]["content"] == [
{"type": "text", "text": "what is this?"},
{"type": "image_url", "image_url": {"url": data_url, "detail": "low"}},
{"type": "image_url", "image_url": data_url},
{"type": "image_url", "image_url": {"url": "data:image/png;base64,iVBORw0KGgo="}},
{"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}},
{"type": "file", "file": {"file_id": pdf_url}},
{"type": "document", "source": {"type": "url", "url": pdf_url}, "title": "the doc"},
]
assert fetched == [image_url]
assert messages == snapshot
async def test_async_inline_remote_media_inlines_only_the_parts_the_predicate_accepts(async_only_image_fetch):
files_api_prefix = "https://generativelanguage.googleapis.com/v1beta/files/"
files_api_pdf = f"{files_api_prefix}{uuid.uuid4().hex}"

View file

@ -138,9 +138,12 @@ def _bedrock_response(model, usage):
@pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id)
def test_bedrock_gpt_5_6_profiles_route_to_converse(profile, local_model_cost_map):
"""GPT-5.6 is served by Converse on bedrock-runtime, never by Invoke."""
assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "converse"
def test_bedrock_gpt_5_6_profiles_never_route_to_invoke(profile, local_model_cost_map):
"""GPT-5.6 is served by bedrock-runtime's native Chat Completions, and by Converse when
the request carries function tools without reasoning_effort "none", never by Invoke."""
assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "chat_completions"
tools_with_reasoning = {"tools": [{"type": "function", "function": {"name": "f"}}], "reasoning_effort": "low"}
assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}", tools_with_reasoning) == "converse"
@pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id)

View file

@ -942,6 +942,8 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"supports_video_input": {"type": "boolean"},
"supports_vision": {"type": "boolean"},
"supports_web_search": {"type": "boolean"},
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": {"type": "boolean"},
"supports_bedrock_runtime_chat_completions_response_format": {"type": "boolean"},
"supports_url_context": {"type": "boolean"},
"supports_multimodal": {"type": "boolean"},
"uses_embed_content": {"type": "boolean"},

View file

@ -181,6 +181,7 @@ def _build_dispatch_context() -> _CompletionDispatchContext:
optional_params={},
organization=None,
provider_config=None,
request_params={},
shared_session=None,
stream=None,
temperature=None,