mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
feat(bedrock): serve gpt-5.6 and newer on runtime chat completions by default
Unprefixed bedrock/<gpt-5.6+> models whose cost-map row lists /v1/chat/completions now route to the native OpenAI-compatible endpoint; converse/ pins Converse and chat_completions/ still opts gpt-oss and Grok in. Guardrails, application inference profile ARNs, and tools with reasoning keep falling back to Converse per request. Hoist the remote-media url comprehension into a single-clause helper.
This commit is contained in:
parent
615eff5eb8
commit
3e3b9cdfc7
7 changed files with 227 additions and 37 deletions
|
|
@ -4,8 +4,9 @@ Helper functions to handle images passed in messages
|
|||
|
||||
import asyncio
|
||||
import base64
|
||||
from collections.abc import Callable, Mapping
|
||||
from collections.abc import Callable, Iterable, Mapping
|
||||
from dataclasses import dataclass
|
||||
from itertools import chain
|
||||
from types import MappingProxyType
|
||||
from typing import Final
|
||||
|
||||
|
|
@ -304,18 +305,19 @@ async def _fetch_data_urls(remote_urls: tuple[str, ...]) -> tuple[str, ...]:
|
|||
raise
|
||||
|
||||
|
||||
def _remote_urls_to_inline(
|
||||
messages: Iterable[AllMessageValues], should_inline: Callable[[RemoteMedia], bool]
|
||||
) -> tuple[str, ...]:
|
||||
parts: Final = chain.from_iterable(_content_parts(message) for message in messages)
|
||||
remotes: Final = (remote for part in parts if (remote := _parse_remote_part(part)) is not None)
|
||||
return tuple(dict.fromkeys(remote.url for remote in remotes if should_inline(_remote_media(remote))))
|
||||
|
||||
|
||||
def inline_remote_media(
|
||||
messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues]
|
||||
should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url,
|
||||
) -> list[AllMessageValues]: # mutable-ok: every transform_request takes list[AllMessageValues]
|
||||
remote_urls: Final = tuple(
|
||||
dict.fromkeys(
|
||||
remote.url
|
||||
for message in messages
|
||||
for part in _content_parts(message)
|
||||
if (remote := _parse_remote_part(part)) is not None and should_inline(_remote_media(remote))
|
||||
)
|
||||
)
|
||||
remote_urls: Final = _remote_urls_to_inline(messages, should_inline)
|
||||
if not remote_urls:
|
||||
return messages
|
||||
data_urls: Final = MappingProxyType({url: convert_url_to_base64(url) for url in remote_urls})
|
||||
|
|
@ -328,14 +330,7 @@ async def async_inline_remote_media(
|
|||
messages: list[AllMessageValues], # mutable-ok: every transform_request takes list[AllMessageValues]
|
||||
should_inline: Callable[[RemoteMedia], bool] = inline_every_remote_url,
|
||||
) -> list[AllMessageValues]: # mutable-ok: every transform_request takes list[AllMessageValues]
|
||||
remote_urls: Final = tuple(
|
||||
dict.fromkeys(
|
||||
remote.url
|
||||
for message in messages
|
||||
for part in _content_parts(message)
|
||||
if (remote := _parse_remote_part(part)) is not None and should_inline(_remote_media(remote))
|
||||
)
|
||||
)
|
||||
remote_urls: Final = _remote_urls_to_inline(messages, should_inline)
|
||||
if not remote_urls:
|
||||
return messages
|
||||
data_urls: Final = await _fetch_data_urls(remote_urls)
|
||||
|
|
|
|||
|
|
@ -3,12 +3,14 @@ Native OpenAI Chat Completions on Amazon Bedrock Runtime.
|
|||
|
||||
AWS serves this surface at
|
||||
``https://bedrock-runtime.{region}.amazonaws.com/openai/v1/chat/completions``
|
||||
for Grok 4.6, gpt-oss and the GPT-5.6 family. The ``chat_completions/`` route prefix
|
||||
opts a model in, so chat completions stay chat completions instead of being rewritten
|
||||
to Converse; without it these models stay on Converse.
|
||||
for Grok 4.6, gpt-oss and GPT 5.6 and newer. GPT 5.6 and newer take it by default
|
||||
(``bedrock_runtime_chat_completions_is_default`` in ``common_utils``), so their chat
|
||||
completions stay chat completions instead of being rewritten to Converse; the
|
||||
``chat_completions/`` route prefix opts any other model in, and ``converse/`` pins a
|
||||
model to Converse.
|
||||
|
||||
Usage: model="bedrock/chat_completions/openai.gpt-oss-20b-1:0" or
|
||||
model="bedrock/chat_completions/global.openai.gpt-5.6-sol". A request that needs a
|
||||
Usage: model="bedrock/global.openai.gpt-6-sol" or
|
||||
model="bedrock/chat_completions/openai.gpt-oss-20b-1:0". A request that needs a
|
||||
Converse-only feature (``bedrock_request_needs_converse`` in ``common_utils``) is
|
||||
still served by Converse.
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -39,6 +39,9 @@ if TYPE_CHECKING:
|
|||
|
||||
_ERROR_REQUEST_URL: Final = "https://docs.litellm.ai/docs"
|
||||
_OPENAI_FAMILY_MODEL_RE: Final = re.compile(r"(^|[./])openai\.")
|
||||
_OPENAI_GPT_VERSION_RE: Final = re.compile(r"(^|[./])openai\.gpt-(\d+)(?:\.(\d+))?")
|
||||
_BEDROCK_RUNTIME_CHAT_COMPLETIONS_DEFAULT_SINCE: Final = (5, 6)
|
||||
_BEDROCK_RUNTIME_CHAT_COMPLETIONS_ENDPOINT: Final = "/v1/chat/completions"
|
||||
BedrockRoute = Literal[
|
||||
"converse",
|
||||
"invoke",
|
||||
|
|
@ -831,6 +834,34 @@ def _bedrock_price_map_flag(model: str, flag: str) -> bool:
|
|||
return any(entry is not None and entry.get(flag) is True for entry in _bedrock_price_map_entries(model))
|
||||
|
||||
|
||||
def _price_map_entry_lists_endpoint(entry: Mapping[str, object] | None, endpoint: str) -> bool:
|
||||
endpoints: Final = None if entry is None else entry.get("supported_endpoints")
|
||||
return isinstance(endpoints, (list, tuple)) and endpoint in endpoints
|
||||
|
||||
|
||||
def _openai_gpt_version(model: str) -> tuple[int, int] | None:
|
||||
match: Final = _OPENAI_GPT_VERSION_RE.search(model)
|
||||
if match is None:
|
||||
return None
|
||||
return int(match.group(2)), int(match.group(3) or 0)
|
||||
|
||||
|
||||
def bedrock_runtime_chat_completions_is_default(model: str) -> bool:
|
||||
"""Whether a model with no route prefix goes to bedrock-runtime's native Chat Completions by default.
|
||||
|
||||
GPT 5.6 and newer (``openai.gpt-<major>[.<minor>]`` at or above 5.6, which gpt-oss never matches) whose
|
||||
price-map row lists ``/v1/chat/completions`` in ``supported_endpoints``. Older GPT rows, gpt-oss and Grok
|
||||
stay on Converse unless the ``chat_completions/`` prefix opts them in.
|
||||
"""
|
||||
version: Final = _openai_gpt_version(model)
|
||||
if version is None or version < _BEDROCK_RUNTIME_CHAT_COMPLETIONS_DEFAULT_SINCE:
|
||||
return False
|
||||
return any(
|
||||
_price_map_entry_lists_endpoint(entry, _BEDROCK_RUNTIME_CHAT_COMPLETIONS_ENDPOINT)
|
||||
for entry in _bedrock_price_map_entries(model)
|
||||
)
|
||||
|
||||
|
||||
def bedrock_runtime_chat_completions_serves_tools_with_reasoning(model: str) -> bool:
|
||||
"""Whether AWS's native Chat Completions serves this model's function tools with any ``reasoning_effort``.
|
||||
|
||||
|
|
@ -879,7 +910,10 @@ def _response_format_needs_converse(model: str, response_format: object) -> bool
|
|||
|
||||
|
||||
def bedrock_request_needs_converse(model: str, request_params: Mapping[str, object]) -> bool:
|
||||
"""Whether a request on the opt-in ``chat_completions/`` route must still be served by Converse.
|
||||
"""Whether a request on the native Chat Completions route must still be served by Converse.
|
||||
|
||||
The route is the default for GPT 5.6 and newer (``bedrock_runtime_chat_completions_is_default``) and the
|
||||
``chat_completions/`` prefix's opt-in for the rest; this decides the fallback for both alike.
|
||||
|
||||
Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking``
|
||||
block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse
|
||||
|
|
@ -909,6 +943,14 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje
|
|||
)
|
||||
|
||||
|
||||
def _chat_completions_unless_converse_needed(
|
||||
model: str, request_params: Mapping[str, object] | None
|
||||
) -> Literal["converse", "chat_completions"]:
|
||||
if request_params is not None and bedrock_request_needs_converse(model, request_params):
|
||||
return "converse"
|
||||
return "chat_completions"
|
||||
|
||||
|
||||
def bedrock_route_for_request(
|
||||
model: str, request_params: Mapping[str, object], additional_drop_params: Sequence[str] | None
|
||||
) -> BedrockRoute:
|
||||
|
|
@ -1285,9 +1327,11 @@ class BedrockModelInfo(BaseLLMModelInfo):
|
|||
"""
|
||||
Get the bedrock route for the given model.
|
||||
|
||||
``chat_completions/`` opts a model into bedrock-runtime's native OpenAI Chat Completions;
|
||||
``request_params`` (the caller's chat params) sends such a request to Converse when it
|
||||
needs a feature only Converse serves. Without the prefix, OpenAI-family models stay on Converse.
|
||||
GPT 5.6 and newer go to bedrock-runtime's native OpenAI Chat Completions by default
|
||||
(``bedrock_runtime_chat_completions_is_default``) and ``chat_completions/`` opts any other model in;
|
||||
``request_params`` (the caller's chat params) sends such a request to Converse when it needs a
|
||||
feature only Converse serves, and ``converse/`` pins a model to Converse. Every other OpenAI-family
|
||||
model stays on Converse without the prefix.
|
||||
"""
|
||||
route_mappings: dict[
|
||||
str,
|
||||
|
|
@ -1324,9 +1368,7 @@ class BedrockModelInfo(BaseLLMModelInfo):
|
|||
return route_type
|
||||
|
||||
if BedrockModelInfo._model_has_route_prefix(model, "chat_completions/"):
|
||||
if request_params is not None and bedrock_request_needs_converse(model, request_params):
|
||||
return "converse"
|
||||
return "chat_completions"
|
||||
return _chat_completions_unless_converse_needed(model, request_params)
|
||||
|
||||
# Check for nova spec prefixes (nova/ and nova-2/)
|
||||
_model_after_bedrock: Final = model.replace("bedrock/", "", 1)
|
||||
|
|
@ -1336,6 +1378,9 @@ class BedrockModelInfo(BaseLLMModelInfo):
|
|||
if is_bedrock_application_inference_profile_arn(model):
|
||||
return "converse"
|
||||
|
||||
if bedrock_runtime_chat_completions_is_default(model):
|
||||
return _chat_completions_unless_converse_needed(model, request_params)
|
||||
|
||||
base_model: Final = BedrockModelInfo.get_base_model(model)
|
||||
alt_model: Final = BedrockModelInfo.get_non_litellm_routing_model_name(model=model)
|
||||
if base_model in litellm.bedrock_converse_models or alt_model in litellm.bedrock_converse_models:
|
||||
|
|
|
|||
|
|
@ -58449,6 +58449,7 @@
|
|||
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html"
|
||||
},
|
||||
"us.openai.gpt-6-astra": {
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"input_cost_per_token": 1.1e-05,
|
||||
"input_cost_per_token_above_272k_tokens": 2.2e-05,
|
||||
"cache_creation_input_token_cost": 1.375e-05,
|
||||
|
|
@ -58480,10 +58481,12 @@
|
|||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
]
|
||||
},
|
||||
"us.openai.gpt-6-sol": {
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"input_cost_per_token": 2.2e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4.4e-06,
|
||||
"cache_creation_input_token_cost": 2.75e-06,
|
||||
|
|
@ -58515,10 +58518,12 @@
|
|||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
]
|
||||
},
|
||||
"us.openai.gpt-6-luna": {
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"input_cost_per_token": 1.1e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 2.2e-07,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
|
|
@ -58550,10 +58555,12 @@
|
|||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
]
|
||||
},
|
||||
"global.openai.gpt-6-astra": {
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"input_cost_per_token": 1e-05,
|
||||
"input_cost_per_token_above_272k_tokens": 2e-05,
|
||||
"cache_creation_input_token_cost": 1.25e-05,
|
||||
|
|
@ -58585,6 +58592,7 @@
|
|||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
]
|
||||
},
|
||||
|
|
@ -58621,6 +58629,7 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.openai.gpt-6-sol": {
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"input_cost_per_token": 2e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4e-06,
|
||||
"cache_creation_input_token_cost": 2.5e-06,
|
||||
|
|
@ -58652,6 +58661,7 @@
|
|||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
]
|
||||
},
|
||||
|
|
@ -58688,6 +58698,7 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.openai.gpt-6-luna": {
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 2e-07,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
|
|
@ -58719,6 +58730,7 @@
|
|||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
]
|
||||
},
|
||||
|
|
@ -79328,6 +79340,7 @@
|
|||
"output_cost_per_token_above_272k_tokens": 1.5e-05,
|
||||
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-1-sol.html",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
|
|
@ -79337,6 +79350,7 @@
|
|||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": false,
|
||||
|
|
@ -79433,6 +79447,7 @@
|
|||
"output_cost_per_token_above_272k_tokens": 1.65e-05,
|
||||
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-1-sol.html",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
|
|
@ -79442,6 +79457,7 @@
|
|||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": false,
|
||||
|
|
|
|||
|
|
@ -58449,6 +58449,7 @@
|
|||
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-cards-openai.html"
|
||||
},
|
||||
"us.openai.gpt-6-astra": {
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"input_cost_per_token": 1.1e-05,
|
||||
"input_cost_per_token_above_272k_tokens": 2.2e-05,
|
||||
"cache_creation_input_token_cost": 1.375e-05,
|
||||
|
|
@ -58480,10 +58481,12 @@
|
|||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
]
|
||||
},
|
||||
"us.openai.gpt-6-sol": {
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"input_cost_per_token": 2.2e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4.4e-06,
|
||||
"cache_creation_input_token_cost": 2.75e-06,
|
||||
|
|
@ -58515,10 +58518,12 @@
|
|||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
]
|
||||
},
|
||||
"us.openai.gpt-6-luna": {
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"input_cost_per_token": 1.1e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 2.2e-07,
|
||||
"cache_creation_input_token_cost": 1.375e-07,
|
||||
|
|
@ -58550,10 +58555,12 @@
|
|||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
]
|
||||
},
|
||||
"global.openai.gpt-6-astra": {
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"input_cost_per_token": 1e-05,
|
||||
"input_cost_per_token_above_272k_tokens": 2e-05,
|
||||
"cache_creation_input_token_cost": 1.25e-05,
|
||||
|
|
@ -58585,6 +58592,7 @@
|
|||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
]
|
||||
},
|
||||
|
|
@ -58621,6 +58629,7 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.openai.gpt-6-sol": {
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"input_cost_per_token": 2e-06,
|
||||
"input_cost_per_token_above_272k_tokens": 4e-06,
|
||||
"cache_creation_input_token_cost": 2.5e-06,
|
||||
|
|
@ -58652,6 +58661,7 @@
|
|||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
]
|
||||
},
|
||||
|
|
@ -58688,6 +58698,7 @@
|
|||
"source": "https://aws.amazon.com/bedrock/pricing/"
|
||||
},
|
||||
"global.openai.gpt-6-luna": {
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"input_cost_per_token": 1e-07,
|
||||
"input_cost_per_token_above_272k_tokens": 2e-07,
|
||||
"cache_creation_input_token_cost": 1.25e-07,
|
||||
|
|
@ -58719,6 +58730,7 @@
|
|||
"supports_vision": true,
|
||||
"source": "https://aws.amazon.com/bedrock/pricing/",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
]
|
||||
},
|
||||
|
|
@ -79328,6 +79340,7 @@
|
|||
"output_cost_per_token_above_272k_tokens": 1.5e-05,
|
||||
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-1-sol.html",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
|
|
@ -79337,6 +79350,7 @@
|
|||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": false,
|
||||
|
|
@ -79433,6 +79447,7 @@
|
|||
"output_cost_per_token_above_272k_tokens": 1.65e-05,
|
||||
"source": "https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-openai-gpt-6-1-sol.html",
|
||||
"supported_endpoints": [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses"
|
||||
],
|
||||
"supported_modalities": [
|
||||
|
|
@ -79442,6 +79457,7 @@
|
|||
"supported_output_modalities": [
|
||||
"text"
|
||||
],
|
||||
"supports_bedrock_runtime_chat_completions_response_format": true,
|
||||
"supports_function_calling": true,
|
||||
"supports_max_reasoning_effort": true,
|
||||
"supports_minimal_reasoning_effort": false,
|
||||
|
|
|
|||
|
|
@ -1,4 +1,4 @@
|
|||
"""Opt-in Bedrock Runtime Chat Completions: ``bedrock/chat_completions/<model>`` posts to /openai/v1/chat/completions."""
|
||||
"""Bedrock Runtime Chat Completions: the default for GPT 5.6 and newer, ``bedrock/chat_completions/<model>`` for the rest."""
|
||||
|
||||
import json
|
||||
|
||||
|
|
@ -20,6 +20,7 @@ from litellm.llms.bedrock.common_utils import (
|
|||
BedrockModelInfo,
|
||||
bedrock_request_needs_converse,
|
||||
bedrock_route_for_request,
|
||||
bedrock_runtime_chat_completions_is_default,
|
||||
get_bedrock_chat_config,
|
||||
)
|
||||
from litellm.llms.custom_httpx.http_handler import HTTPHandler
|
||||
|
|
@ -63,9 +64,11 @@ def test_claude_stays_on_converse(local_cost_map):
|
|||
"us.xai.grok-4.6",
|
||||
"bedrock/openai.gpt-oss-20b-1:0",
|
||||
"openai.gpt-oss-120b-1:0",
|
||||
"global.openai.gpt-5.6-sol",
|
||||
"bedrock/us.openai.gpt-5.6-terra",
|
||||
"global.openai.gpt-5.5",
|
||||
"bedrock/us.openai.gpt-5.4",
|
||||
"bedrock/us-gov-west-1/openai.gpt-oss-20b-1:0",
|
||||
"arn:aws:bedrock:us-east-1:123456789012:inference-profile/us.openai.gpt-6-astra",
|
||||
"arn:aws:bedrock:us-west-2:123456789012:application-inference-profile/abc123xyz",
|
||||
],
|
||||
)
|
||||
def test_models_without_the_prefix_stay_on_converse(local_cost_map, model):
|
||||
|
|
@ -86,6 +89,32 @@ def test_cost_map_row_listing_chat_completions_leaves_the_default_route_alone(mo
|
|||
assert BedrockModelInfo.get_bedrock_route("bedrock/chat_completions/openai.gpt-oss-20b-1:0", {}) == "chat_completions"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model, supported_endpoints, expected_route",
|
||||
[
|
||||
("global.openai.gpt-5.5", ["/v1/chat/completions", "/v1/responses"], "converse"),
|
||||
("us.openai.gpt-5.6-sol", ["/v1/chat/completions", "/v1/responses"], "chat_completions"),
|
||||
("us.openai.gpt-5.6-sol", ["/v1/responses"], "converse"),
|
||||
("global.openai.gpt-6-sol", ["/v1/chat/completions", "/v1/responses"], "chat_completions"),
|
||||
("global.openai.gpt-6-sol", ["/v1/responses"], "converse"),
|
||||
("global.openai.gpt-6-sol", [], "converse"),
|
||||
("us.openai.gpt-6.1-sol", ["/v1/chat/completions"], "chat_completions"),
|
||||
("global.openai.gpt-10-sol", ["/v1/chat/completions"], "chat_completions"),
|
||||
("openai.gpt-oss-120b-1:0", ["/v1/chat/completions"], "converse"),
|
||||
("us.xai.grok-4.6", ["/v1/chat/completions"], "converse"),
|
||||
],
|
||||
)
|
||||
def test_default_route_needs_gpt_56_or_newer_and_a_row_listing_chat_completions(
|
||||
monkeypatch, model, supported_endpoints, expected_route
|
||||
):
|
||||
entry = {"litellm_provider": "bedrock_converse", "supported_endpoints": supported_endpoints}
|
||||
monkeypatch.setattr(litellm, "model_cost", {model: entry})
|
||||
assert bedrock_runtime_chat_completions_is_default(model) is (expected_route == "chat_completions")
|
||||
assert BedrockModelInfo.get_bedrock_route(f"bedrock/{model}", {}) == expected_route
|
||||
assert BedrockModelInfo.get_bedrock_route(f"bedrock/chat_completions/{model}", {}) == "chat_completions"
|
||||
assert BedrockModelInfo.get_bedrock_route(f"bedrock/converse/{model}", {}) == "converse"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["global.openai.gpt-5.6-sol", "openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"])
|
||||
def test_chat_completions_prefix_prices_like_the_bare_model(local_cost_map, model):
|
||||
prefixed = litellm.get_model_info(model=f"bedrock/chat_completions/{model}")
|
||||
|
|
@ -194,7 +223,7 @@ def _recording_client(**response_kwargs):
|
|||
[
|
||||
("bedrock/us.xai.grok-4.6", b"/model/us.xai.grok-4.6/converse"),
|
||||
("bedrock/openai.gpt-oss-20b-1:0", b"/model/openai.gpt-oss-20b-1%3A0/converse"),
|
||||
("bedrock/global.openai.gpt-5.6-sol", b"/model/global.openai.gpt-5.6-sol/converse"),
|
||||
("bedrock/global.openai.gpt-5.5", b"/model/global.openai.gpt-5.5/converse"),
|
||||
],
|
||||
)
|
||||
def test_completion_without_the_prefix_posts_converse(local_cost_map, fake_aws_env, model, model_path):
|
||||
|
|
@ -310,13 +339,40 @@ def test_openai_runtime_models_use_chat_completions_route(local_cost_map, model)
|
|||
assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig)
|
||||
|
||||
|
||||
GPT_56_AND_NEWER_MODELS = (
|
||||
"global.openai.gpt-5.6-sol",
|
||||
"bedrock/us.openai.gpt-5.6-terra",
|
||||
"us.openai.gpt-5.6-luna",
|
||||
"bedrock/global.openai.gpt-6-astra",
|
||||
"us.openai.gpt-6-sol",
|
||||
"global.openai.gpt-6-luna",
|
||||
"bedrock/global.openai.gpt-6.1-sol",
|
||||
"us.openai.gpt-6.1-sol",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", GPT_56_AND_NEWER_MODELS)
|
||||
def test_gpt_56_and_newer_default_to_chat_completions(local_cost_map, model):
|
||||
assert bedrock_runtime_chat_completions_is_default(model) is True
|
||||
assert BedrockModelInfo.get_bedrock_route(model) == "chat_completions"
|
||||
assert BedrockModelInfo.get_bedrock_route(model, {}) == "chat_completions"
|
||||
assert isinstance(get_bedrock_chat_config(model), AmazonBedrockRuntimeChatCompletionsConfig)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["us.amazon.nova-micro-v1:0", "us.anthropic.claude-haiku-4-5-20251001-v1:0"])
|
||||
def test_nova_and_claude_stay_on_converse(local_cost_map, model):
|
||||
assert BedrockModelInfo.get_bedrock_route(model, {"tools": [GET_WEATHER_TOOL]}) == "converse"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model", ["chat_completions/openai.gpt-oss-20b-1:0", "bedrock/chat_completions/global.openai.gpt-5.6-sol"]
|
||||
"model",
|
||||
[
|
||||
"chat_completions/openai.gpt-oss-20b-1:0",
|
||||
"bedrock/chat_completions/global.openai.gpt-5.6-sol",
|
||||
"bedrock/us.openai.gpt-5.6-sol",
|
||||
"global.openai.gpt-6-sol",
|
||||
"us.openai.gpt-6.1-sol",
|
||||
],
|
||||
)
|
||||
def test_guardrail_config_falls_back_to_converse(local_cost_map, model):
|
||||
guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"}
|
||||
|
|
@ -363,6 +419,9 @@ def test_gpt56_tools_need_reasoning_none_on_chat_completions(local_cost_map, req
|
|||
BedrockModelInfo.get_bedrock_route("bedrock/chat_completions/us.openai.gpt-5.6-terra", request_params)
|
||||
== expected_route
|
||||
)
|
||||
assert BedrockModelInfo.get_bedrock_route("bedrock/us.openai.gpt-5.6-sol", request_params) == expected_route
|
||||
assert BedrockModelInfo.get_bedrock_route("global.openai.gpt-6-sol", request_params) == expected_route
|
||||
assert BedrockModelInfo.get_bedrock_route("bedrock/us.openai.gpt-6.1-sol", request_params) == expected_route
|
||||
|
||||
|
||||
@pytest.mark.parametrize("reasoning_effort", ["low", "high", None])
|
||||
|
|
@ -395,6 +454,8 @@ def test_thinking_block_goes_to_converse(local_cost_map):
|
|||
def test_explicit_converse_prefix_wins_for_openai_models(local_cost_map):
|
||||
assert BedrockModelInfo.get_bedrock_route("bedrock/converse/openai.gpt-oss-20b-1:0") == "converse"
|
||||
assert BedrockModelInfo.get_bedrock_route("converse/global.openai.gpt-5.6-sol", {}) == "converse"
|
||||
assert BedrockModelInfo.get_bedrock_route("bedrock/converse/global.openai.gpt-6-sol", {}) == "converse"
|
||||
assert isinstance(get_bedrock_chat_config("bedrock/converse/global.openai.gpt-6-sol"), litellm.AmazonConverseConfig)
|
||||
|
||||
|
||||
def test_map_openai_params_sends_max_tokens_as_max_completion_tokens():
|
||||
|
|
@ -769,6 +830,58 @@ def test_gpt56_tools_with_reasoning_none_stay_on_chat_completions(local_cost_map
|
|||
assert response.choices[0].message.tool_calls[0].function.name == "get_weather"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["global.openai.gpt-6-sol", "us.openai.gpt-5.6-sol", "us.openai.gpt-6.1-sol"])
|
||||
def test_gpt_56_and_newer_completion_without_the_prefix_posts_runtime_chat_completions(
|
||||
local_cost_map, fake_aws_env, model
|
||||
):
|
||||
requests, client = _recording_client(json=_chat_completion_json("ok", model))
|
||||
response = litellm.completion(
|
||||
model=f"bedrock/{model}",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
reasoning_effort="low",
|
||||
client=client,
|
||||
)
|
||||
|
||||
assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
|
||||
body = json.loads(requests[0].content)
|
||||
assert body["model"] == model
|
||||
assert body["reasoning_effort"] == "low"
|
||||
assert "inferenceConfig" not in body
|
||||
assert response.choices[0].message.content == "ok"
|
||||
assert response._hidden_params["response_cost"] > 0
|
||||
|
||||
|
||||
def test_gpt6_without_the_prefix_tools_with_reasoning_effort_go_to_converse(local_cost_map, fake_aws_env):
|
||||
requests, client = _recording_client(json=CONVERSE_JSON)
|
||||
response = litellm.completion(
|
||||
model="bedrock/global.openai.gpt-6-sol",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
tools=[GET_WEATHER_TOOL],
|
||||
reasoning_effort="low",
|
||||
client=client,
|
||||
)
|
||||
|
||||
assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-6-sol/converse")
|
||||
body = json.loads(requests[0].content)
|
||||
assert body["toolConfig"]["tools"][0]["toolSpec"]["name"] == "get_weather"
|
||||
assert body["additionalModelRequestFields"]["reasoning"] == {"effort": "low"}
|
||||
assert response.choices[0].message.content == "ok"
|
||||
|
||||
|
||||
def test_gpt6_without_the_prefix_guardrail_config_goes_to_converse(local_cost_map, fake_aws_env):
|
||||
guardrail = {"guardrailIdentifier": "gr-1", "guardrailVersion": "1"}
|
||||
requests, client = _recording_client(json=CONVERSE_JSON)
|
||||
litellm.completion(
|
||||
model="bedrock/global.openai.gpt-6-sol",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
guardrailConfig=guardrail,
|
||||
client=client,
|
||||
)
|
||||
|
||||
assert requests[0].url.raw_path.endswith(b"/model/global.openai.gpt-6-sol/converse")
|
||||
assert json.loads(requests[0].content)["guardrailConfig"] == guardrail
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"converse_only_param",
|
||||
[
|
||||
|
|
@ -1017,6 +1130,8 @@ RESPONSE_FORMAT_ENFORCING_MODELS = [
|
|||
"chat_completions/global.openai.gpt-5.6-sol",
|
||||
"chat_completions/us.xai.grok-4.6",
|
||||
"bedrock/chat_completions/us-gov.xai.grok-4.6",
|
||||
"global.openai.gpt-6-sol",
|
||||
"bedrock/us.openai.gpt-6.1-sol",
|
||||
]
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -138,9 +138,10 @@ def _bedrock_response(model, usage):
|
|||
|
||||
|
||||
@pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id)
|
||||
def test_bedrock_gpt_5_6_profiles_route_to_converse(profile, local_model_cost_map):
|
||||
"""GPT-5.6 is served by Converse on bedrock-runtime, never by Invoke."""
|
||||
assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "converse"
|
||||
def test_bedrock_gpt_5_6_profiles_route_to_runtime_chat_completions(profile, local_model_cost_map):
|
||||
"""GPT-5.6 is served by bedrock-runtime's native Chat Completions by default and by Converse when pinned, never by Invoke."""
|
||||
assert BedrockModelInfo.get_bedrock_route(f"bedrock/{profile.model_id}") == "chat_completions"
|
||||
assert BedrockModelInfo.get_bedrock_route(f"bedrock/converse/{profile.model_id}") == "converse"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("profile", GPT_5_6_PROFILES, ids=lambda p: p.model_id)
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue