This commit is contained in:
Mateo Wang 2026-09-29 17:35:07 +00:00 • committed by GitHub
commit af58f90e75
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
19 changed files with 2049 additions and 40 deletions

View file

@ -441,6 +441,10 @@ pub struct ModelInfo {
#[serde(skip_serializing_if = "Option::is_none")]
pub supports_audio_output: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub supports_bedrock_runtime_chat_completions_response_format: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub supports_bedrock_runtime_chat_completions_tools_with_reasoning: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub supports_computer_use: Option<bool>,
#[serde(skip_serializing_if = "Option::is_none")]
pub supports_embedding_image_input: Option<bool>,

View file

@ -1779,6 +1779,9 @@ if TYPE_CHECKING:
from .llms.bedrock.chat.invoke_transformations.amazon_openai_transformation import (
AmazonBedrockOpenAIConfig as AmazonBedrockOpenAIConfig,
)
from .llms.bedrock.chat.chat_completions.transformation import (
AmazonBedrockRuntimeChatCompletionsConfig as AmazonBedrockRuntimeChatCompletionsConfig,
)
from .llms.bedrock.image_generation.amazon_stability1_transformation import (
AmazonStabilityConfig as AmazonStabilityConfig,
)

View file

@ -206,6 +206,7 @@ LLM_CONFIG_NAMES: Final = (
"AmazonTwelveLabsPegasusConfig",
"AmazonInvokeConfig",
"AmazonBedrockOpenAIConfig",
"AmazonBedrockRuntimeChatCompletionsConfig",
"AmazonStabilityConfig",
"AmazonStability3Config",
"AmazonNovaCanvasConfig",
@ -868,6 +869,10 @@ _LLM_CONFIGS_IMPORT_MAP: Final = {
".llms.bedrock.chat.invoke_transformations.amazon_openai_transformation",
"AmazonBedrockOpenAIConfig",
),
"AmazonBedrockRuntimeChatCompletionsConfig": (
".llms.bedrock.chat.chat_completions.transformation",
"AmazonBedrockRuntimeChatCompletionsConfig",
),
"AmazonStabilityConfig": (
".llms.bedrock.image_generation.amazon_stability1_transformation",
"AmazonStabilityConfig",

View file

@ -15,6 +15,7 @@ import litellm
from litellm import verbose_logger
from litellm.caching.caching import InMemoryCache
from litellm.constants import MAX_IMAGE_URL_DOWNLOAD_SIZE_MB
from litellm.litellm_core_utils.prompt_templates.common_utils import infer_content_type_from_url_and_content
from litellm.litellm_core_utils.url_utils import SSRFError, async_safe_get, safe_get
from litellm.types.llms.openai import AllMessageValues
@ -55,23 +56,16 @@ def _process_image_response(response: Response, url: str) -> str:
base64_image: Final = base64.b64encode(image_bytes).decode("utf-8")
image_type: Final = response.headers.get("Content-Type")
if image_type is None:
img_type = url.split(".")[-1].lower()
_img_type: Final = {
"jpg": "image/jpeg",
"jpeg": "image/jpeg",
"png": "image/png",
"gif": "image/gif",
"webp": "image/webp",
}.get(img_type)
if _img_type is None:
raise Exception(
f"Error: Unsupported image format. Format={_img_type}. Supported types = ['image/jpeg', 'image/png', 'image/gif', 'image/webp']"
)
img_type = _img_type
else:
img_type = image_type
try:
img_type: Final = infer_content_type_from_url_and_content(
url=url,
content=bytes(image_bytes),
current_content_type=response.headers.get("Content-Type"),
)
except ValueError as e:
raise litellm.ImageFetchError(
f"Error: Unable to determine image content type from the server's headers, the URL, or the image bytes. url={url}"
) from e
result: Final = f"data:{img_type};base64,{base64_image}"
in_memory_cache.set_cache(url, result)
@ -310,6 +304,26 @@ async def _fetch_data_urls(remote_urls: tuple[str, ...]) -> tuple[str, ...]:
raise
def inline_remote_media(
messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues]
should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url,
) -> list[AllMessageValues]: # mutable-ok: every transform_request takes list[AllMessageValues]
remote_urls: Final = tuple(
dict.fromkeys(
remote.url
for message in messages
for part in _content_parts(message)
if (remote := _parse_remote_part(part)) is not None and should_inline(_remote_media(remote))
)
)
if not remote_urls:
return messages
data_urls: Final = MappingProxyType({url: convert_url_to_base64(url) for url in remote_urls})
return [ # mutable-ok: transform_request takes a list
_inline_message(message, data_urls, should_inline) for message in messages
]
async def async_inline_remote_media(
messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues]
should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url,

View file

@ -0,0 +1,422 @@
"""
Native OpenAI Chat Completions on Amazon Bedrock Runtime.
AWS serves this surface at
``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions``
for Grok 4.6, gpt-oss and the GPT-5.6 family. The ``chat_completions/`` route prefix
opts a model in, so chat completions stay chat completions instead of being rewritten
to Converse; without it these models stay on Converse.
Usage: model="bedrock/chat_completions/openai.gpt-oss-20b-1:0" or
model="bedrock/chat_completions/global.openai.gpt-5.6-sol". A request that needs a
Converse-only feature (``bedrock_request_needs_converse`` in ``common_utils``) is
still served by Converse.
"""
from collections.abc import AsyncIterator, Iterator, Mapping
from dataclasses import dataclass, replace
from types import MappingProxyType
from typing import TYPE_CHECKING, Final, Literal
import httpx
from typing_extensions import assert_never
import litellm
from litellm.litellm_core_utils.core_helpers import set_provider_response_headers_in_hidden_params
from litellm.litellm_core_utils.prompt_templates.image_handling import (
async_inline_remote_media,
inline_remote_image_urls,
inline_remote_media,
)
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
from litellm.llms.bedrock.common_utils import BedrockError, split_bedrock_region_path
from litellm.llms.openai.chat.gpt_transformation import OpenAIChatCompletionStreamingHandler
from litellm.llms.openai_like.chat.transformation import OpenAILikeChatConfig
from litellm.types.llms.openai import AllMessageValues
from litellm.types.utils import Choices, ModelResponse, ModelResponseStream
if TYPE_CHECKING:
import tiktoken
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
REASONING_OPEN_TAG: Final = "<reasoning>"
REASONING_CLOSE_TAG: Final = "</reasoning>"
CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY: Final = MappingProxyType(
{
"openai.gpt-5": frozenset(("frequency_penalty", "presence_penalty", "logprobs", "top_logprobs")),
"openai.gpt-oss": frozenset(("logit_bias",)),
"xai.": frozenset(("frequency_penalty", "presence_penalty")),
}
)
def chat_completions_params_refused_for(model: str) -> frozenset[str]:
"""The OpenAI params AWS's Chat Completions endpoint rejects for this model whatever else the request says.
Each family answers them with a 400 (GPT-5.6, gpt-oss) or a 503 (Grok), where Converse dropped the same
params under ``drop_params``, so the native config leaves them out of its supported list and the usual
drop-or-raise handling applies before the request reaches AWS.
"""
model_id: Final = split_bedrock_region_path(model)[1]
return frozenset().union(
*(refused for family, refused in CHAT_COMPLETIONS_REFUSED_PARAMS_BY_FAMILY.items() if family in model_id)
)
CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY: Final = MappingProxyType({"xai.": frozenset(("none",))})
def chat_completions_reasoning_efforts_refused_for(model: str) -> frozenset[str]:
"""The ``reasoning_effort`` values AWS's Chat Completions endpoint rejects for this model.
Grok answers ``"none"`` with a 400 (it takes low, medium, high, and xhigh) where Converse dropped every
``reasoning_effort`` for it, so the native config drops the value and AWS applies its default effort as before.
"""
model_id: Final = split_bedrock_region_path(model)[1]
return frozenset().union(
*(
refused
for family, refused in CHAT_COMPLETIONS_REFUSED_REASONING_EFFORTS_BY_FAMILY.items()
if family in model_id
)
)
def without_refused_reasoning_effort(model: str, params: Mapping[str, object]) -> Mapping[str, object]:
if params.get("reasoning_effort") not in chat_completions_reasoning_efforts_refused_for(model):
return params
return MappingProxyType({key: value for key, value in params.items() if key != "reasoning_effort"})
def _held_close_tag_prefix(text: str) -> int:
return next(
(
size
for size in range(min(len(text), len(REASONING_CLOSE_TAG) - 1), 0, -1)
if REASONING_CLOSE_TAG.startswith(text[-size:])
),
0,
)
@dataclass(frozen=True, slots=True)
class ReasoningTagSplitter:
"""
The same split for a stream of content deltas, where a tag can arrive across chunks.
``feed`` returns the next state plus the reasoning and content text the delta contributes;
``flush`` releases what the stream ended on before a tag resolved.
"""
phase: Literal["start", "reasoning", "after_close", "content"] = "start"
pending: str = ""
def feed(self, text: str) -> tuple["ReasoningTagSplitter", str, str]:
match self.phase:
case "content":
return self, "", text
case "after_close":
content: Final = text.lstrip()
return (replace(self, phase="content") if content else self), "", content
case "start":
return self._feed_start(self.pending + text)
case "reasoning":
return self._feed_reasoning(self.pending + text)
case _:
assert_never(self.phase)
def _feed_start(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]:
if buffered.startswith(REASONING_OPEN_TAG):
return replace(self, phase="reasoning", pending="")._feed_reasoning(buffered[len(REASONING_OPEN_TAG) :])
if REASONING_OPEN_TAG.startswith(buffered):
return replace(self, pending=buffered), "", ""
return replace(self, phase="content", pending=""), "", buffered
def _feed_reasoning(self, buffered: str) -> tuple["ReasoningTagSplitter", str, str]:
close_at: Final = buffered.find(REASONING_CLOSE_TAG)
if close_at >= 0:
after_close: Final = replace(self, phase="after_close", pending="")
next_state, _, content = after_close.feed(buffered[close_at + len(REASONING_CLOSE_TAG) :])
return next_state, buffered[:close_at], content
held: Final = _held_close_tag_prefix(buffered)
return replace(self, pending=buffered[len(buffered) - held :]), buffered[: len(buffered) - held], ""
def flush(self) -> tuple["ReasoningTagSplitter", str, str]:
drained: Final = replace(self, phase="content", pending="")
if self.phase == "reasoning":
return drained, self.pending, ""
return drained, "", self.pending
def _split_streamed_content(
splitter: ReasoningTagSplitter, content: str | None, finished: bool
) -> tuple[ReasoningTagSplitter, str, str]:
fed_state, fed_reasoning, fed_content = splitter.feed(content or "")
if not finished:
return fed_state, fed_reasoning, fed_content
drained, flushed_reasoning, flushed_content = fed_state.flush()
return drained, fed_reasoning + flushed_reasoning, fed_content + flushed_content
def split_reasoning_tag(content: str) -> tuple[str | None, str]:
"""
Split gpt-oss's inline ``<reasoning>...</reasoning>`` prefix out of a complete message.
Runs the streaming splitter over the whole message, so a streamed and a non-streamed
response to the same completion split identically. Returns ``(None, content)`` when the
message does not start with the tag.
"""
_, reasoning, body = _split_streamed_content(ReasoningTagSplitter(), content, finished=True)
return reasoning or None, body
class BedrockRuntimeChatCompletionsStreamingHandler(OpenAIChatCompletionStreamingHandler):
"""OpenAI chunk parsing plus the ``<reasoning>`` split, tracked per choice index."""
def __init__(
self,
streaming_response: Iterator[str] | AsyncIterator[str] | ModelResponse,
sync_stream: bool,
json_mode: bool | None = False,
) -> None:
super().__init__(streaming_response=streaming_response, sync_stream=sync_stream, json_mode=json_mode)
self._splitters: Mapping[int, ReasoningTagSplitter] = MappingProxyType({})
def chunk_parser(self, chunk: dict) -> ModelResponseStream: # mutable-ok: BaseModelResponseIterator signature
parsed: Final = super().chunk_parser(chunk)
for choice in parsed.choices:
next_state, reasoning, content = _split_streamed_content(
self._splitters.get(choice.index, ReasoningTagSplitter()),
choice.delta.content,
choice.finish_reason is not None,
)
self._splitters = MappingProxyType({**self._splitters, choice.index: next_state})
if reasoning:
choice.delta.reasoning_content = f"{getattr(choice.delta, 'reasoning_content', None) or ''}{reasoning}"
if content or choice.delta.content is not None:
choice.delta.content = content
return parsed
def with_max_completion_tokens(params: Mapping[str, object]) -> Mapping[str, object]:
"""
Send the caller's ``max_tokens`` as ``max_completion_tokens``.
Every model on this surface accepts ``max_completion_tokens`` and the GPT-5.6 family
rejects ``max_tokens``; an explicit ``max_completion_tokens`` wins when both are set.
"""
if "max_tokens" not in params:
return params
return MappingProxyType(
{
key: value
for key, value in (("max_completion_tokens", params["max_tokens"]), *params.items())
if key != "max_tokens"
}
)
class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
def __init__(self, aws_signer: BaseAWSLLM | None = None) -> None:
super().__init__()
self._aws_signer: Final = aws_signer or BaseAWSLLM()
@property
def custom_llm_provider(self) -> str | None:
return "bedrock"
@property
def uses_async_transform_request(self) -> bool:
return True
def get_error_class(
self,
error_message: str,
status_code: int,
headers: dict[str, object] | httpx.Headers, # mutable-ok: BaseConfig signature
) -> BaseLLMException:
return BedrockError(status_code=status_code, message=error_message, headers=headers)
def get_complete_url(
self,
api_base: str | None,
api_key: str | None,
model: str,
optional_params: dict, # mutable-ok: BaseConfig signature
litellm_params: dict, # mutable-ok: BaseConfig signature
stream: bool | None = None,
) -> str:
if api_base is not None and "chat/completions" in api_base:
return api_base.rstrip("/")
aws_region_name: Final = self._aws_signer._get_aws_region_name( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public region resolver
optional_params=self._params_with_region_from_path(optional_params, model), model=model
)
endpoint_url, _ = self._aws_signer.get_runtime_endpoint(
api_base=api_base,
aws_bedrock_runtime_endpoint=optional_params.get("aws_bedrock_runtime_endpoint"),
aws_region_name=aws_region_name,
)
base: Final = endpoint_url.rstrip("/")
if base.endswith("/openai/v1/chat/completions"):
return base
if base.endswith("/openai/v1"):
return f"{base}/chat/completions"
return f"{base}/openai/v1/chat/completions"
def _params_with_region_from_path(
self, optional_params: dict, model: str | None
) -> dict: # mutable-ok: BaseAWSLLM's region resolver and signer take a plain dict
region_from_path, _ = split_bedrock_region_path(model or "")
if region_from_path is None or optional_params.get("aws_region_name") is not None:
return optional_params
return {**optional_params, "aws_region_name": region_from_path} # mutable-ok: BaseAWSLLM takes a plain dict
def sign_request(
self,
headers: dict, # mutable-ok: BaseConfig signature
optional_params: dict, # mutable-ok: BaseConfig signature
request_data: dict, # mutable-ok: BaseConfig signature
api_base: str,
api_key: str | None = None,
model: str | None = None,
stream: bool | None = None,
fake_stream: bool | None = None,
) -> tuple[dict, bytes | None]: # mutable-ok: BaseConfig signature
return self._aws_signer._sign_request( # pyright: ignore[reportPrivateUsage] # BaseAWSLLM has no public signer
service_name="bedrock",
headers=headers,
optional_params=self._params_with_region_from_path(optional_params, model),
request_data=request_data,
api_base=api_base,
api_key=api_key,
model=model,
stream=stream,
fake_stream=fake_stream,
)
def map_openai_params(
self,
non_default_params: dict, # mutable-ok: BaseConfig signature
optional_params: dict, # mutable-ok: BaseConfig signature
model: str,
drop_params: bool,
replace_max_completion_tokens_with_max_tokens: bool = False,
) -> dict: # mutable-ok: BaseConfig signature
mapped: Final = super().map_openai_params(
non_default_params=non_default_params,
optional_params=optional_params,
model=model,
drop_params=drop_params,
replace_max_completion_tokens_with_max_tokens=replace_max_completion_tokens_with_max_tokens,
)
return dict( # mutable-ok: get_optional_params keeps filling this dict
without_refused_reasoning_effort(model, with_max_completion_tokens(mapped))
)
def _inference_params(
self, optional_params: Mapping[str, object]
) -> dict[str, object]: # mutable-ok: BaseConfig signature of transform_request
return { # mutable-ok: OpenAILikeChatConfig.transform_request takes a plain dict
key: value
for key, value in optional_params.items()
if key not in self._aws_signer.aws_authentication_params
}
def transform_request(
self,
model: str,
messages: list[AllMessageValues], # mutable-ok: BaseConfig signature
optional_params: dict, # mutable-ok: BaseConfig signature
litellm_params: dict, # mutable-ok: BaseConfig signature
headers: dict, # mutable-ok: BaseConfig signature
) -> dict: # mutable-ok: BaseConfig signature
return super().transform_request(
model=split_bedrock_region_path(model)[1],
messages=inline_remote_media(messages, should_inline=inline_remote_image_urls),
optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
headers=headers,
)
async def async_transform_request(
self,
model: str,
messages: list[AllMessageValues], # mutable-ok: BaseConfig signature
optional_params: dict, # mutable-ok: BaseConfig signature
litellm_params: dict, # mutable-ok: BaseConfig signature
headers: dict, # mutable-ok: BaseConfig signature
) -> dict: # mutable-ok: BaseConfig signature
return await super().async_transform_request(
model=split_bedrock_region_path(model)[1],
messages=await async_inline_remote_media(messages, should_inline=inline_remote_image_urls),
optional_params=self._inference_params(optional_params),
litellm_params=litellm_params,
headers=headers,
)
def transform_response(
self,
model: str,
raw_response: httpx.Response,
model_response: ModelResponse,
logging_obj: "LiteLLMLoggingObj",
request_data: dict, # mutable-ok: BaseConfig signature
messages: list[AllMessageValues], # mutable-ok: BaseConfig signature
optional_params: dict, # mutable-ok: BaseConfig signature
litellm_params: dict, # mutable-ok: BaseConfig signature
encoding: "tiktoken.Encoding | None",
api_key: str | None = None,
json_mode: bool | None = None,
) -> ModelResponse:
response: Final = super().transform_response(
model=model,
raw_response=raw_response,
model_response=model_response,
logging_obj=logging_obj,
request_data=request_data,
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
encoding=encoding,
api_key=api_key,
json_mode=json_mode,
)
set_provider_response_headers_in_hidden_params(response, raw_response.headers)
for choice in response.choices:
if not isinstance(choice, Choices) or not isinstance(choice.message.content, str):
continue
reasoning, content = split_reasoning_tag(choice.message.content)
if reasoning is not None:
choice.message.reasoning_content = (
f"{getattr(choice.message, 'reasoning_content', None) or ''}{reasoning}"
)
choice.message.content = content
return response
def get_supported_openai_params(self, model: str) -> list: # mutable-ok: BaseConfig signature
refused: Final = frozenset(("n", *chat_completions_params_refused_for(model)))
base_params: Final = tuple(
param for param in super().get_supported_openai_params(model) if param not in refused
)
reasoning_param: Final = (
("reasoning_effort",)
if "reasoning_effort" not in base_params
and litellm.supports_reasoning(model=model, custom_llm_provider=self.custom_llm_provider)
else ()
)
return [*base_params, *reasoning_param] # mutable-ok: BaseConfig signature returns a list
def get_model_response_iterator(
self,
streaming_response: Iterator[str] | AsyncIterator[str] | ModelResponse,
sync_stream: bool,
json_mode: bool | None = False,
) -> BedrockRuntimeChatCompletionsStreamingHandler:
return BedrockRuntimeChatCompletionsStreamingHandler(
streaming_response=streaming_response,
sync_stream=sync_stream,
json_mode=json_mode,
)

View file

@ -10,6 +10,7 @@ import json
import os
import re
from collections.abc import Mapping, Sequence
from types import MappingProxyType
from typing import TYPE_CHECKING, Any, Final, Literal, TypedDict
if TYPE_CHECKING:
@ -28,6 +29,7 @@ from litellm.llms.base_llm.anthropic_messages.transformation import (
)
from litellm.llms.base_llm.base_utils import BaseLLMModelInfo, BaseTokenCounter
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.bedrock.request_metadata import bedrock_request_metadata_is_owned
from litellm.secret_managers.main import get_secret, get_secret_str
from litellm.types.llms.bedrock import AWS_AUTH_PARAM_KEYS, AwsAuthParams
@ -37,6 +39,18 @@ if TYPE_CHECKING:
_ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs"
_OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.")
BedrockRoute = Literal[
"converse",
"invoke",
"claude_platform",
"converse_like",
"agent",
"agentcore",
"async_invoke",
"openai",
"mantle",
"chat_completions",
]
def error_response_text(response: httpx.Response) -> str:
@ -787,12 +801,128 @@ def is_bedrock_application_inference_profile_arn(model: str) -> bool:
def strip_bedrock_routing_prefix(model: str) -> str:
"""Strip LiteLLM routing prefixes from model name."""
for prefix in ["bedrock/", "converse/", "invoke/", "openai/", "mantle/", "nova-2/", "nova/"]:
for prefix in ["bedrock/", "chat_completions/", "converse/", "invoke/", "openai/", "mantle/", "nova-2/", "nova/"]:
if model.startswith(prefix):
model = model.split("/", 1)[1]
return model
def split_bedrock_region_path(model: str) -> tuple[str | None, str]:
"""Split a ``<region>/<model-id>`` routing path into the region and the id AWS receives.
``bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0`` -> ``("us-gov-west-1", "openai.gpt-oss-20b-1:0")``;
a model without a region path comes back as ``(None, <routing-prefix-stripped id>)``.
"""
stripped: Final = strip_bedrock_routing_prefix(model)
region, separator, model_id = stripped.partition("/")
if separator and region in _get_all_bedrock_regions():
return region, model_id
return None, stripped
def _bedrock_price_map_entries(model: str) -> tuple[Mapping[str, object] | None, ...]:
return tuple(
litellm.model_cost.get(key)
for key in (model, strip_bedrock_routing_prefix(model), split_bedrock_region_path(model)[1])
)
def _bedrock_price_map_flag(model: str, flag: str) -> bool:
return any(entry is not None and entry.get(flag) is True for entry in _bedrock_price_map_entries(model))
def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool:
"""Whether AWS's native Chat Completions serves this model's function tools with any ``reasoning_effort``.
Data-driven from the price-map ``supports_bedrock_runtime_chat_completions_tools_with_reasoning``
flag (gpt-oss, Grok). Without it AWS only takes tools with ``reasoning_effort="none"``
(the GPT-5.6 family), and Converse serves tools with any effort, so those requests fall back to it.
"""
return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_tools_with_reasoning")
def bedrock_runtime_chat_completions_enforces_response_format(model: str) -> bool:
"""Whether AWS's native Chat Completions enforces a ``response_format`` schema for this model.
Data-driven from the price-map ``supports_bedrock_runtime_chat_completions_response_format`` flag
(GPT-5.6, Grok). Without it AWS accepts the field and answers with unconstrained text (gpt-oss), so
Converse, which emulates the schema through a forced ``json_tool_call`` tool, serves those requests.
"""
return _bedrock_price_map_flag(model, "supports_bedrock_runtime_chat_completions_response_format")
BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
(
"guardrailConfig",
"performanceConfig",
"serviceTier",
"requestMetadata",
"outputConfig",
"thinking",
"additionalModelRequestFields",
"top_k",
"stop",
)
)
def _response_format_needs_converse(model: str, response_format: object) -> bool:
if response_format is None:
return False
if not isinstance(response_format, Mapping):
return not bedrock_runtime_chat_completions_enforces_response_format(model)
response_format_type: Final = response_format.get("type")
if response_format_type == "text":
return False
is_json_schema: Final = response_format_type == "json_schema" and "json_schema" in response_format
return not (is_json_schema and bedrock_runtime_chat_completions_enforces_response_format(model))
def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool:
"""Whether a request on the opt-in ``chat_completions/`` route must still be served by Converse.
Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking``
block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse
forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on
AWS's native OpenAI surface, ``stop`` stays on Converse where it fails loudly instead of silently
stopping hidden reasoning, operator-owned request metadata is only written onto the Converse body,
function tools (``tools`` or legacy ``functions``) on a model without
``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless
``reasoning_effort`` is exactly ``"none"``, and a ``response_format`` goes native only as
``{"type": "json_schema", "json_schema": ...}`` (a pydantic model is converted to that) on a model with
``supports_bedrock_runtime_chat_completions_response_format``: a schema on any other model is only
honored by Converse, and every ``json_object`` form (``response_schema`` included) keeps Converse's
handling everywhere, since AWS's native surface rejects that type with a 400 unless the prompt
mentions json.
"""
if any(request_params.get(key) is not None for key in BEDROCK_CONVERSE_ONLY_REQUEST_KEYS):
return True
if bedrock_request_metadata_is_owned():
return True
if _response_format_needs_converse(model, request_params.get("response_format")):
return True
if not (request_params.get("tools") or request_params.get("functions")):
return False
return (
not bedrock_runtime_chat_completions_serves_tools_with_reasoning(model)
and request_params.get("reasoning_effort") != "none"
)
def bedrock_route_for_request(
model: str, request_params: Mapping[str, object], additional_drop_params: Sequence[str] | None
) -> BedrockRoute:
"""The route for one request, decided from the caller's raw params before any provider mapping.
Param mapping and dispatch both call this with the same inputs, so a request that falls back to
Converse is mapped with the Converse config and sent to Converse, never one without the other.
"""
dropped: Final = frozenset(additional_drop_params or ())
return BedrockModelInfo.get_bedrock_route(
model, MappingProxyType({key: value for key, value in request_params.items() if key not in dropped})
)
def strip_bedrock_throughput_suffix(model: str) -> str:
"""Strip throughput tier suffixes and context window suffixes from Bedrock model names."""
import re
@ -1150,19 +1280,14 @@ class BedrockModelInfo(BaseLLMModelInfo):
@staticmethod
def get_bedrock_route(
model: str,
) -> Literal[
"converse",
"invoke",
"claude_platform",
"converse_like",
"agent",
"agentcore",
"async_invoke",
"openai",
"mantle",
]:
request_params: Mapping[str, object] | None = None,
) -> BedrockRoute:
"""
Get the bedrock route for the given model.
``chat_completions/`` opts a model into bedrock-runtime's native OpenAI Chat Completions;
``request_params`` (the caller's chat params) sends such a request to Converse when it
needs a feature only Converse serves. Without the prefix, OpenAI-family models stay on Converse.
"""
route_mappings: dict[
str,
@ -1176,6 +1301,7 @@ class BedrockModelInfo(BaseLLMModelInfo):
"async_invoke",
"openai",
"mantle",
"chat_completions",
],
] = {
"invoke/": "invoke",
@ -1197,6 +1323,11 @@ class BedrockModelInfo(BaseLLMModelInfo):
if BedrockModelInfo._model_has_route_prefix(model, prefix):
return route_type
if BedrockModelInfo._model_has_route_prefix(model, "chat_completions/"):
if request_params is not None and bedrock_request_needs_converse(model, request_params):
return "converse"
return "chat_completions"
# Check for nova spec prefixes (nova/ and nova-2/)
_model_after_bedrock: Final = model.replace("bedrock/", "", 1)
if _model_after_bedrock.startswith("nova-2/") or _model_after_bedrock.startswith("nova/"):
@ -1383,6 +1514,8 @@ def get_bedrock_chat_config(model: str):
return litellm.AmazonConverseConfig()
elif bedrock_route == "openai":
return litellm.AmazonBedrockOpenAIConfig()
elif bedrock_route == "chat_completions":
return litellm.AmazonBedrockRuntimeChatCompletionsConfig()
elif bedrock_route == "agent":
from litellm.llms.bedrock.chat.invoke_agent.transformation import (
AmazonInvokeAgentConfig,

View file

@ -71,11 +71,16 @@ BEDROCK_RUNTIME_SUPPORTED_RESPONSE_TOOL_TYPES: Final = frozenset(
{"function", "mcp", "custom", "apply_patch", "namespace", "tool_search", "computer"}
)
BEDROCK_RUNTIME_UNSUPPORTED_RESPONSE_PARAMS: Final = frozenset({"background"})
BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX: Final = "chat_completions/"
REMOTE_IMAGE_URL_SCHEMES: Final = ("http://", "https://")
IMAGE_BLOCK_KEYS: Final = ("content", "output")
IMAGE_BLOCK_TYPES: Final = frozenset({"input_image", "computer_screenshot"})
def _without_chat_completions_route(model: str) -> str:
return model.removeprefix(BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX)
def resolve_bedrock_bearer_token(api_key: str | None) -> str | None:
return api_key or get_secret_str("AWS_BEARER_TOKEN_BEDROCK")
@ -168,9 +173,13 @@ class BedrockOpenAIResponsesConfig(BaseAWSLLM, OpenAIResponsesAPIConfig):
The capability decision lives here rather than in the shared dispatch so that
onboarding a model, or changing how the signal is read, stays inside the
Bedrock adapter. ``None`` leaves the caller's existing behaviour untouched --
chat-only Bedrock models keep the Chat Completions bridge.
chat-only Bedrock models keep the Chat Completions bridge. The ``chat_completions/``
opt-in only moves Chat Completions calls off Converse, so a Responses call on such a
deployment still takes this surface instead of being bridged.
"""
if not bedrock_supports_openai_responses(model, litellm.model_cost):
if not model or not bedrock_supports_openai_responses(
_without_chat_completions_route(model), litellm.model_cost
):
return None
return cls()
@ -330,7 +339,7 @@ class BedrockOpenAIResponsesConfig(BaseAWSLLM, OpenAIResponsesAPIConfig):
rewritten_types,
)
return super().transform_responses_api_request(
model=model,
model=_without_chat_completions_route(model),
input=normalized_input,
response_api_optional_request_params=response_api_optional_request_params,
litellm_params=litellm_params,

View file

@ -116,7 +116,7 @@ from litellm.llms.base_llm import BaseConfig, BaseImageGenerationConfig
from litellm.llms.base_llm.base_model_iterator import (
convert_model_response_to_streaming,
)
from litellm.llms.bedrock.common_utils import BedrockModelInfo
from litellm.llms.bedrock.common_utils import BedrockModelInfo, bedrock_route_for_request
from litellm.llms.cohere.common_utils import CohereModelInfo
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler, http2_enabled
from litellm.llms.openai.chat.gpt_5_transformation import OpenAIGPT5Config
@ -4205,7 +4205,9 @@ def _complete_bedrock(ctx: _CompletionDispatchContext) -> _CompletionDispatchRes
if "aws_region_name" not in optional_params or optional_params["aws_region_name"] is None:
optional_params["aws_region_name"] = aws_bedrock_client.meta.region_name
bedrock_route: Final = BedrockModelInfo.get_bedrock_route(model)
bedrock_route: Final = bedrock_route_for_request(
model, ctx.request_params, ctx.kwargs.get("additional_drop_params")
)
if bedrock_route == "claude_platform":
provider_config = ProviderConfigManager.get_provider_chat_config(
model=model,
@ -4232,7 +4234,7 @@ def _complete_bedrock(ctx: _CompletionDispatchContext) -> _CompletionDispatchRes
provider_config=provider_config,
)
elif bedrock_route == "converse":
model = model.replace("converse/", "")
model = model.replace("converse/", "").replace("chat_completions/", "")
response = bedrock_converse_chat_completion.completion(
model=model,
messages=messages,
@ -5836,6 +5838,7 @@ def completion(
optional_params=optional_params,
organization=organization,
provider_config=provider_config,
request_params=MappingProxyType({**optional_param_args, **non_default_params}),
shared_session=shared_session,
stream=stream,
temperature=temperature,

View file

@ -41417,6 +41417,10 @@
"output_cost_per_token": 0.0
},
"openai.gpt-oss-120b-1:0": {
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 1.5e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@ -41431,6 +41435,10 @@
"supports_tool_choice": true
},
"openai.gpt-oss-20b-1:0": {
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@ -47169,6 +47177,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -47182,6 +47194,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -47191,6 +47207,11 @@
"input_cost_per_token": 2.64e-06,
"output_cost_per_token": 7.92e-06,
"cache_read_input_token_cost": 6.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@ -57759,6 +57780,7 @@
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-luna.html"
},
"us.openai.gpt-5.6-sol": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4.4e-06,
"input_cost_per_token_above_272k_tokens": 8.8e-06,
"cache_creation_input_token_cost": 5.5e-06,
@ -57789,10 +57811,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"global.openai.gpt-5.6-sol": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4e-06,
"input_cost_per_token_above_272k_tokens": 8e-06,
"cache_creation_input_token_cost": 5e-06,
@ -57823,10 +57847,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"us.openai.gpt-5.6-terra": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_above_272k_tokens": 4.4e-06,
"cache_creation_input_token_cost": 2.75e-06,
@ -57857,10 +57883,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"global.openai.gpt-5.6-terra": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-06,
"input_cost_per_token_above_272k_tokens": 4e-06,
"cache_creation_input_token_cost": 2.5e-06,
@ -57891,10 +57919,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"us.openai.gpt-5.6-luna": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-07,
"input_cost_per_token_above_272k_tokens": 4.4e-07,
"cache_creation_input_token_cost": 2.75e-07,
@ -57925,6 +57955,7 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
@ -58053,6 +58084,7 @@
]
},
"global.openai.gpt-5.6-luna": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_272k_tokens": 4e-07,
"cache_creation_input_token_cost": 2.5e-07,
@ -58083,6 +58115,7 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
@ -58767,6 +58800,11 @@
"input_cost_per_token": 2.2e-06,
"output_cost_per_token": 6.6e-06,
"cache_read_input_token_cost": 5.5e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@ -58783,6 +58821,11 @@
"input_cost_per_token": 2e-06,
"output_cost_per_token": 6e-06,
"cache_read_input_token_cost": 5e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@ -64736,6 +64779,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -64749,6 +64796,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -64990,6 +65041,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -65003,6 +65058,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,

View file

@ -1,6 +1,6 @@
from __future__ import annotations
from collections.abc import Callable, Coroutine, Iterable
from collections.abc import Callable, Coroutine, Iterable, Mapping
from dataclasses import dataclass
from typing import TYPE_CHECKING, Any, Literal, Union
@ -229,6 +229,7 @@ class _CompletionDispatchContext:
optional_params: dict
organization: str | None
provider_config: BaseConfig | None
request_params: Mapping[str, object]
shared_session: ClientSession | None
stream: bool | None
temperature: float | None

View file

@ -408,7 +408,7 @@ if TYPE_CHECKING:
BaseVectorStoreFilesConfig,
)
from litellm.llms.base_llm.videos.transformation import BaseVideoConfig
from litellm.llms.bedrock.common_utils import BedrockModelInfo
from litellm.llms.bedrock.common_utils import BedrockModelInfo, BedrockRoute
from litellm.llms.bedrock.embed.amazon_nova_transformation import (
AmazonNovaEmbeddingConfig,
)
@ -3453,6 +3453,14 @@ def _should_drop_param(k, additional_drop_params) -> bool:
return False
def _bedrock_route_for_request(
model: str, passed_params: Mapping[str, object], additional_drop_params: list | None
) -> BedrockRoute:
from litellm.llms.bedrock.common_utils import bedrock_route_for_request
return bedrock_route_for_request(model, passed_params, additional_drop_params)
def _get_non_default_params(passed_params: dict, default_params: dict, additional_drop_params: list | None) -> dict:
non_default_params: Final = {}
for k, v in passed_params.items():
@ -4494,9 +4502,17 @@ def get_optional_params(
message=f"{custom_llm_provider} does not support parameters: {list(unsupported_params.keys())}, for model={model}. To drop these, set `litellm.drop_params=True` or for proxy:\n\n`litellm_settings:\n drop_params: true`\n. \n If you want to use these params dynamically send allowed_openai_params={list(unsupported_params.keys())} in your request.",
)
bedrock_route: Final = (
_bedrock_route_for_request(model, passed_params, additional_drop_params)
if custom_llm_provider == "bedrock"
else None
)
get_supported_openai_params: Final[_SupportedOpenAIParamsGetter] = litellm_utils.get_supported_openai_params
supported_params = get_supported_openai_params(
model=model, custom_llm_provider=custom_llm_provider, base_model=base_model
supported_params = (
litellm.AmazonConverseConfig().get_supported_openai_params(model=model)
if bedrock_route == "converse"
and isinstance(provider_config, litellm.AmazonBedrockRuntimeChatCompletionsConfig)
else get_supported_openai_params(model=model, custom_llm_provider=custom_llm_provider, base_model=base_model)
)
if supported_params is None:
supported_params = get_supported_openai_params(model=model, custom_llm_provider="openai")
@ -4666,7 +4682,6 @@ def get_optional_params(
)
elif custom_llm_provider == "bedrock":
bedrock_model_info: Final[type[BedrockModelInfo]] = litellm_utils.BedrockModelInfo
bedrock_route: Final = bedrock_model_info.get_bedrock_route(model)
bedrock_base_model: Final = bedrock_model_info.get_base_model(model)
if bedrock_route == "converse" or bedrock_route == "converse_like":
optional_params = litellm.AmazonConverseConfig().map_openai_params(

View file

@ -41417,6 +41417,10 @@
"output_cost_per_token": 0.0
},
"openai.gpt-oss-120b-1:0": {
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 1.5e-07,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@ -41431,6 +41435,10 @@
"supports_tool_choice": true
},
"openai.gpt-oss-20b-1:0": {
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"input_cost_per_token": 7e-08,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 128000,
@ -47169,6 +47177,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -47182,6 +47194,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -47191,6 +47207,11 @@
"input_cost_per_token": 2.64e-06,
"output_cost_per_token": 7.92e-06,
"cache_read_input_token_cost": 6.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@ -57759,6 +57780,7 @@
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-56-luna.html"
},
"us.openai.gpt-5.6-sol": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4.4e-06,
"input_cost_per_token_above_272k_tokens": 8.8e-06,
"cache_creation_input_token_cost": 5.5e-06,
@ -57789,10 +57811,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"global.openai.gpt-5.6-sol": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 4e-06,
"input_cost_per_token_above_272k_tokens": 8e-06,
"cache_creation_input_token_cost": 5e-06,
@ -57823,10 +57847,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"us.openai.gpt-5.6-terra": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-06,
"input_cost_per_token_above_272k_tokens": 4.4e-06,
"cache_creation_input_token_cost": 2.75e-06,
@ -57857,10 +57883,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"global.openai.gpt-5.6-terra": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-06,
"input_cost_per_token_above_272k_tokens": 4e-06,
"cache_creation_input_token_cost": 2.5e-06,
@ -57891,10 +57919,12 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
"us.openai.gpt-5.6-luna": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2.2e-07,
"input_cost_per_token_above_272k_tokens": 4.4e-07,
"cache_creation_input_token_cost": 2.75e-07,
@ -57925,6 +57955,7 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
@ -58053,6 +58084,7 @@
]
},
"global.openai.gpt-5.6-luna": {
"supports_bedrock_runtime_chat_completions_response_format": true,
"input_cost_per_token": 2e-07,
"input_cost_per_token_above_272k_tokens": 4e-07,
"cache_creation_input_token_cost": 2.5e-07,
@ -58083,6 +58115,7 @@
"supports_vision": true,
"supports_sampling_params": false,
"supported_endpoints": [
"/v1/chat/completions",
"/v1/responses"
]
},
@ -58767,6 +58800,11 @@
"input_cost_per_token": 2.2e-06,
"output_cost_per_token": 6.6e-06,
"cache_read_input_token_cost": 5.5e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@ -58783,6 +58821,11 @@
"input_cost_per_token": 2e-06,
"output_cost_per_token": 6e-06,
"cache_read_input_token_cost": 5e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_bedrock_runtime_chat_completions_response_format": true,
"litellm_provider": "bedrock_converse",
"max_input_tokens": 500000,
"max_output_tokens": 500000,
@ -64736,6 +64779,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -64749,6 +64796,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -64990,6 +65041,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 3.6e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -65003,6 +65058,10 @@
"max_tokens": 128000,
"mode": "chat",
"output_cost_per_token": 7.2e-07,
"supported_endpoints": [
"/v1/chat/completions"
],
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": true,
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,

View file

@ -950,6 +950,12 @@
"supports_audio_output": {
"type": "boolean"
},
"supports_bedrock_runtime_chat_completions_response_format": {
"type": "boolean"
},
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": {
"type": "boolean"
},
"supports_computer_use": {
"type": "boolean"
},

View file

@ -1,4 +1,5 @@
import asyncio
import base64
import copy
import time
import uuid
@ -16,6 +17,7 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import (
async_convert_url_to_base64,
async_inline_remote_media,
convert_url_to_base64,
inline_remote_media,
)
from litellm.litellm_core_utils.url_utils import SSRFError
@ -258,6 +260,54 @@ async def test_async_data_url_is_returned_unchanged_without_fetch(monkeypatch):
assert await async_convert_url_to_base64(data_url) == data_url
REAL_PNG_BYTES = base64.b64decode(
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNkYPhfDwAChwGA60e6kgAAAABJRU5ErkJggg=="
)
def _stub_image_client(content, content_type):
class _Client:
def get(self, url, follow_redirects=True):
headers = {} if content_type is None else {"Content-Type": content_type}
return Response(200, content=content, headers=headers, request=Request("GET", url))
return _Client()
def test_convert_url_to_base64_infers_the_type_when_the_server_sends_octet_stream(monkeypatch):
monkeypatch.setattr(
litellm, "module_level_client", _stub_image_client(REAL_PNG_BYTES, "application/octet-stream")
)
result = convert_url_to_base64(f"http://img.example/{uuid.uuid4()}")
assert result.startswith("data:image/png;base64,")
def test_convert_url_to_base64_keeps_a_real_content_type(monkeypatch):
monkeypatch.setattr(
litellm, "module_level_client", _stub_image_client(REAL_PNG_BYTES, "image/jpeg")
)
result = convert_url_to_base64(f"http://img.example/{uuid.uuid4()}.png")
assert result.startswith("data:image/jpeg;base64,")
def test_convert_url_to_base64_raises_when_no_content_type_is_determinable(monkeypatch):
monkeypatch.setattr(
litellm,
"module_level_client",
_stub_image_client(b"\x00\x01\x02\x03not-an-image", "application/octet-stream"),
)
url = f"http://img.example/{uuid.uuid4()}"
with pytest.raises(litellm.ImageFetchError) as excinfo:
convert_url_to_base64(url)
assert url in str(excinfo.value)
def test_image_size_limit_disabled(monkeypatch):
"""
Test that setting MAX_IMAGE_URL_DOWNLOAD_SIZE_MB to 0 disables all image URL downloads.
@ -320,6 +370,50 @@ async def test_async_inline_remote_media_inlines_every_remote_part_shape(async_o
assert messages == snapshot
def test_inline_remote_media_inlines_every_remote_part_shape(monkeypatch):
image_url = f"http://img.example/{uuid.uuid4()}.png"
pdf_url = f"http://docs.example/{uuid.uuid4()}.pdf"
fetched = []
def fake_convert(url):
fetched.append(url)
return f"data:image/png;base64,{url}"
monkeypatch.setattr(image_handling, "convert_url_to_base64", fake_convert)
messages = [
{"role": "system", "content": "be terse"},
{
"role": "user",
"content": [
{"type": "text", "text": "what is this?"},
{"type": "image_url", "image_url": {"url": image_url, "detail": "low"}},
{"type": "image_url", "image_url": image_url},
{"type": "image_url", "image_url": {"url": "data:image/png;base64,iVBORw0KGgo="}},
{"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}},
{"type": "file", "file": {"file_id": pdf_url}},
{"type": "document", "source": {"type": "url", "url": pdf_url}, "title": "the doc"},
],
},
]
snapshot = copy.deepcopy(messages)
inlined = inline_remote_media(messages, should_inline=image_handling.inline_remote_image_urls)
data_url = f"data:image/png;base64,{image_url}"
assert inlined[0] == {"role": "system", "content": "be terse"}
assert inlined[1]["content"] == [
{"type": "text", "text": "what is this?"},
{"type": "image_url", "image_url": {"url": data_url, "detail": "low"}},
{"type": "image_url", "image_url": data_url},
{"type": "image_url", "image_url": {"url": "data:image/png;base64,iVBORw0KGgo="}},
{"type": "image_url", "image_url": {"url": "s3://bucket/key.png"}},
{"type": "file", "file": {"file_id": pdf_url}},
{"type": "document", "source": {"type": "url", "url": pdf_url}, "title": "the doc"},
]
assert fetched == [image_url]
assert messages == snapshot
async def test_async_inline_remote_media_inlines_only_the_parts_the_predicate_accepts(async_only_image_fetch):
files_api_prefix = "https://generativelanguage.googleapis.com/v1beta/files/"
files_api_pdf = f"{files_api_prefix}{uuid.uuid4().hex}"

View file

@ -162,6 +162,27 @@ class TestForModelGate:
):
assert BedrockOpenAIResponsesConfig.for_model(None) is None
def test_chat_completions_route_keeps_the_native_responses_surface(self):
with patch.object( # test-quality-ok: the gate reads the global cost map by design; no injection point exists
litellm, "model_cost", {MODEL: {"supported_endpoints": ["/v1/responses"]}}
):
cfg = BedrockOpenAIResponsesConfig.for_model(f"chat_completions/{MODEL}")
assert isinstance(cfg, BedrockOpenAIResponsesConfig)
body = cfg.transform_responses_api_request(
model=f"chat_completions/{MODEL}",
input="hi",
response_api_optional_request_params={},
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert body["model"] == MODEL
def test_converse_route_keeps_the_chat_completions_bridge(self):
with patch.object( # test-quality-ok: the gate reads the global cost map by design; no injection point exists
litellm, "model_cost", {MODEL: {"supported_endpoints": ["/v1/responses"]}}
):
assert BedrockOpenAIResponsesConfig.for_model(f"converse/{MODEL}") is None
class TestProviderResolution:
"""model_cost is patched explicitly: it is populated at import time from a GitHub

View file

@ -942,6 +942,8 @@ def test_aaamodel_prices_and_context_window_json_is_valid():
"supports_video_input": {"type": "boolean"},
"supports_vision": {"type": "boolean"},
"supports_web_search": {"type": "boolean"},
"supports_bedrock_runtime_chat_completions_tools_with_reasoning": {"type": "boolean"},
"supports_bedrock_runtime_chat_completions_response_format": {"type": "boolean"},
"supports_url_context": {"type": "boolean"},
"supports_multimodal": {"type": "boolean"},
"uses_embed_content": {"type": "boolean"},

View file

@ -181,6 +181,7 @@ def _build_dispatch_context() -> _CompletionDispatchContext:
optional_params={},
organization=None,
provider_config=None,
request_params={},
shared_session=None,
stream=None,
temperature=None,