fix: make Bedrock Mantle Responses routing data-driven per model (#30700)

* Make Bedrock Mantle Responses routing data-driven per model

Route Bedrock Mantle models to the native Responses API based on each
model's price-map capability signal instead of a hardcoded model-name
heuristic, and derive the OpenAI-compatible base path segment per model.

Responses dispatch now selects the native config when the model advertises
responses support (/v1/responses in supported_endpoints, or mode=responses),
both overridable via register_model and proxy model_info. This enables
native Responses for gpt-oss-120b/20b and the gemma-4 family while keeping
chat-only models (gpt-oss safeguard, nvidia, mistral, ...) on the existing
chat-completions emulation. Capability is per-model, so gpt-oss-120b routes
natively while gpt-oss-safeguard-120b does not despite sharing the gpt-oss
substring.

The wire path is a separate concern, driven by the existing
use_openai_responses_path flag rather than a model-name match: gpt-5.x and
gemma-4-* on /openai/v1, everything else (incl. gpt-oss) on /v1. The chat
config now derives its base from the same flag, fixing gemma-4
chat-completions requests that previously went to /v1 instead of /openai/v1.

Cost maps: add supported_endpoints to the gpt-oss entries (responses for the
non-safeguard variants, chat-only for safeguard) and supported_endpoints +
use_openai_responses_path to all three gemma-4 entries.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

* Address review: move capability helper into bedrock_mantle package

Move the Responses capability check out of utils.py into
litellm/llms/bedrock_mantle/common_utils.py as mantle_supports_responses,
alongside its companion wire-path helper mantle_base_segment. Both are now
pure functions of (model, model_cost): the price-map mode/supported_endpoints
read replaces the get_model_info call, so the rules are unit-testable without
patching global state and the Bedrock Mantle package is self-contained.

Use str | None instead of Optional[str] on the new signatures to satisfy the
ruff UP045 strict-rule gate. Add direct unit tests for both helpers.

Fix test_register_model_restore_undoes_existing_key_overwrite: gpt-oss-120b
now legitimately supports Responses, so it can no longer be the
"None after restore" vehicle; use the chat-only safeguard variant, which
isolates the register/restore effect from the model's own capability.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
Co-authored-by: Sameer Kankute <sameer@berri.ai>
This commit is contained in:
Matthew Lapointe 2026-06-18 08:16:09 -04:00 • committed by GitHub
parent 3edfc0c405
commit ada3eee1f3
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
8 changed files with 340 additions and 163 deletions

View file

@ -641,7 +641,7 @@ def _get_openai_compatible_provider_info(
api_base,
dynamic_api_key,
) = litellm.BedrockMantleChatConfig()._get_openai_compatible_provider_info(
api_base, api_key, litellm_params=litellm_params
api_base, api_key, litellm_params=litellm_params, model=model
)
elif custom_llm_provider == "nvidia_nim":
# nvidia_nim is openai compatible, we just need to set this to custom_openai and have the api_base be https://api.endpoints.anyscale.com/v1

View file

@ -23,6 +23,7 @@ from litellm.secret_managers.main import get_secret_str
from litellm.types.llms.openai import AllMessageValues
from litellm.types.router import GenericLiteLLMParams
from ..common_utils import mantle_base_segment
from ...openai_like.chat.transformation import OpenAILikeChatConfig
@ -48,6 +49,7 @@ class BedrockMantleChatConfig(BedrockMantleAuthMixin, OpenAILikeChatConfig):
api_base: Optional[str],
api_key: Optional[str],
litellm_params: Optional[GenericLiteLLMParams] = None,
model: str | None = None,
) -> Tuple[Optional[str], Optional[str]]:
region = (
(litellm_params.aws_region_name if litellm_params else None)
@ -57,10 +59,13 @@ class BedrockMantleChatConfig(BedrockMantleAuthMixin, OpenAILikeChatConfig):
or BEDROCK_MANTLE_DEFAULT_REGION
)
BaseAWSLLM._validate_aws_region_name(region)
# The base path segment is data-driven per model (use_openai_responses_path
# flag): gemma-4-* and gpt-5.x are served on /openai/v1, everything else on
# /v1. An explicit api_base still wins over the derived default.
api_base = (
api_base
or get_secret_str("BEDROCK_MANTLE_API_BASE")
or f"https://bedrock-mantle.{region}.api.aws/v1"
or f"https://bedrock-mantle.{region}.api.aws/{mantle_base_segment(model, litellm.model_cost)}"
)
dynamic_api_key = self._resolve_bearer_token(api_key)
return api_base, dynamic_api_key

View file

@ -1,115 +1,38 @@
"""
Shared auth and region resolution for the Amazon Bedrock Mantle backends.
"""Shared helpers for the Amazon Bedrock Mantle OpenAI-compatible provider.
Mantle authenticates with a Bearer token when one is available
(litellm_params.api_key, BEDROCK_MANTLE_API_KEY, or the standard
AWS_BEARER_TOKEN_BEDROCK); otherwise it falls back to AWS SigV4 (service
"bedrock") over the standard credential chain (IAM role / access key / profile /
web identity). The Chat Completions and Responses backends share this behaviour
through BedrockMantleAuthMixin so the two paths can never drift apart.
Both helpers are pure functions of (model, model_cost) so the routing rules can be
unit-tested without patching global state.
"""
import re
from typing import Tuple
from botocore.exceptions import (
CredentialRetrievalError,
NoCredentialsError,
PartialCredentialsError,
ProfileNotFound,
)
def mantle_supports_responses(model: str | None, model_cost: dict) -> bool:
"""Whether a Bedrock Mantle model can serve the native Responses API.
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
from litellm.secret_managers.main import get_secret_str
BEDROCK_MANTLE_DEFAULT_REGION = "us-east-1"
# Standard Mantle host: https://bedrock-mantle.<region>.api.aws (group 1 = region).
MANTLE_HOST_RE = re.compile(
r"^https?://bedrock-mantle\.([^/.]+)\.api\.aws", re.IGNORECASE
)
Purely data-driven from the model's price-map capability signal -- either
/v1/responses in supported_endpoints, or mode=responses -- both overridable
via register_model and proxy model_info, so onboarding a model is a JSON
change, never a code change. There is deliberately NO model-name match here:
capability is per-model, not per-family (openai.gpt-oss-120b supports
Responses while openai.gpt-oss-safeguard-120b does not, despite sharing the
gpt-oss substring), so a substring gate would be wrong. A model absent from
model_cost simply has no signal and returns False (chat-completions emulation).
"""
entry = model_cost.get(f"bedrock_mantle/{model}", {})
if "/v1/responses" in (entry.get("supported_endpoints") or []):
return True
return entry.get("mode") == "responses"
class BedrockMantleAuthMixin:
_aws_signer: BaseAWSLLM
def mantle_base_segment(model: str | None, model_cost: dict) -> str:
"""Return the base path segment for a Bedrock Mantle model's OpenAI surface.
@staticmethod
def _resolve_bearer_token(api_key: str | None) -> str | None:
return (
api_key
or get_secret_str("BEDROCK_MANTLE_API_KEY")
or get_secret_str("AWS_BEARER_TOKEN_BEDROCK")
)
@staticmethod
def _resolve_region(params: dict) -> str:
region = params.get("aws_region_name")
if region:
BaseAWSLLM._validate_aws_region_name(region)
return region
base = params.get("api_base") or get_secret_str("BEDROCK_MANTLE_API_BASE")
if base:
match = MANTLE_HOST_RE.match(base.rstrip("/"))
if match:
return match.group(1)
return (
get_secret_str("BEDROCK_MANTLE_REGION")
or get_secret_str("AWS_REGION_NAME")
or get_secret_str("AWS_REGION")
or BEDROCK_MANTLE_DEFAULT_REGION
)
def sign_request(
self,
headers: dict,
optional_params: dict,
request_data: dict,
api_base: str,
api_key: str | None = None,
model: str | None = None,
stream: bool | None = None,
fake_stream: bool | None = None,
) -> Tuple[dict, bytes | None]:
bearer = self._resolve_bearer_token(api_key)
if not bearer:
# SigV4 path. Pin the credential-scope region to the region of the actual
# signing URL so the SigV4 scope and the URL host can never disagree, even
# when a stale api_base and aws_region_name point at different regions.
# Fall back to _resolve_region only for custom proxy hosts that do not
# match the standard Mantle URL pattern. Also drop any caller Authorization
# so _sign_request's restore-original-Authorization step cannot override
# the SigV4 header.
host_match = MANTLE_HOST_RE.match(api_base.rstrip("/"))
optional_params = {
**optional_params,
"aws_region_name": (
host_match.group(1)
if host_match
else self._resolve_region({**optional_params, "api_base": api_base})
),
}
headers = {k: v for k, v in headers.items() if k.lower() != "authorization"}
try:
return self._aws_signer._sign_request(
service_name="bedrock",
headers=headers,
optional_params=optional_params,
request_data=request_data,
api_base=api_base,
api_key=bearer,
model=model,
stream=stream,
fake_stream=fake_stream,
)
except (
NoCredentialsError,
PartialCredentialsError,
ProfileNotFound,
CredentialRetrievalError,
) as e:
raise ValueError(
"Bedrock Mantle auth failed: no Bearer token and no usable AWS "
"credentials. Set BEDROCK_MANTLE_API_KEY (or AWS_BEARER_TOKEN_BEDROCK) "
"or pass api_key for Bearer auth, or provide AWS credentials "
"(IAM role / access key / profile / web identity) for SigV4."
) from e
Data-driven from the model's price-map use_openai_responses_path flag
(overridable via register_model / proxy model_info). Per the AWS model cards,
gpt-5.x and the google gemma-4-* family carry that flag and are served on the
/openai/v1 base (.../openai/v1/responses and .../openai/v1/chat/completions);
every other model including gpt-oss uses the standard /v1 base. The segment is
the base for the model's whole OpenAI-compatible surface, so both the chat and
responses configs derive from it -- there is no separate model-name rule.
"""
entry = model_cost.get(f"bedrock_mantle/{model}", {})
return "openai/v1" if entry.get("use_openai_responses_path") is True else "v1"

View file

@ -42383,6 +42383,7 @@
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_reasoning": true,
@ -42397,6 +42398,7 @@
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_reasoning": true,
@ -42411,6 +42413,7 @@
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
"supported_endpoints": ["/v1/chat/completions"],
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -42424,6 +42427,7 @@
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
"supported_endpoints": ["/v1/chat/completions"],
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -42477,6 +42481,8 @@
"max_output_tokens": 256000,
"max_tokens": 256000,
"mode": "chat",
"use_openai_responses_path": true,
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
"supports_function_calling": true,
"supports_parallel_function_calling": false,
"supports_reasoning": true,
@ -42491,6 +42497,8 @@
"max_output_tokens": 256000,
"max_tokens": 256000,
"mode": "chat",
"use_openai_responses_path": true,
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
"supports_function_calling": true,
"supports_parallel_function_calling": false,
"supports_reasoning": true,
@ -42505,6 +42513,8 @@
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"use_openai_responses_path": true,
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
"supports_function_calling": true,
"supports_parallel_function_calling": false,
"supports_reasoning": true,

View file

@ -9070,34 +9070,26 @@ class ProviderConfigManager:
elif litellm.LlmProviders.HOSTED_VLLM == provider:
return litellm.HostedVLLMResponsesAPIConfig()
elif litellm.LlmProviders.BEDROCK_MANTLE == provider:
# Mantle serves Responses on two upstream paths. A model takes the
# /openai/v1/responses path when its price-map entry declares
# use_openai_responses_path (data-driven, so a non-gpt-named frontier
# model can be onboarded by JSON alone), or, as a fallback needing no
# price-map entry, when its name matches the openai.gpt- frontier
# convention (minus gpt-oss) -- this keeps a future gpt-6 routing
# correctly before its entry loads. Any other model declared
# mode=responses takes the standard /v1/responses path. Everything
# else returns None and keeps the chat-completions emulation (see
# responses/main.py "config is None").
if not model:
return None
model_lower = model.lower()
entry = litellm.model_cost.get(f"bedrock_mantle/{model}", {})
on_openai_path = entry.get("use_openai_responses_path") is True
name_is_frontier = (
"openai.gpt-" in model_lower and "gpt-oss" not in model_lower
# Both decisions are data-driven from the model's price-map entry, with
# no model-name logic. Capability (can it serve Responses?) comes from
# mantle_supports_responses (supported_endpoints / mode);
# chat-only models (gpt-oss safeguard, nvidia, ...) return None and keep
# the chat-completions emulation (responses/main.py "config is None").
# The wire path comes from mantle_base_segment, which reads the
# use_openai_responses_path flag: gpt-5.x and gemma-4-* on
# /openai/v1/responses, everything else (incl. gpt-oss) on
# /v1/responses.
from litellm.llms.bedrock_mantle.common_utils import (
mantle_base_segment,
mantle_supports_responses,
)
if not model or not mantle_supports_responses(model, litellm.model_cost):
return None
return litellm.BedrockMantleResponsesAPIConfig(
use_openai_path=mantle_base_segment(model, litellm.model_cost)
== "openai/v1"
)
if on_openai_path or name_is_frontier:
return litellm.BedrockMantleResponsesAPIConfig(use_openai_path=True)
try:
if get_model_info(model, "bedrock_mantle").get("mode") == "responses":
return litellm.BedrockMantleResponsesAPIConfig(
use_openai_path=False
)
except Exception:
pass
return None
return None
@staticmethod

View file

@ -42585,6 +42585,7 @@
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_reasoning": true,
@ -42599,6 +42600,7 @@
"max_output_tokens": 32768,
"max_tokens": 32768,
"mode": "chat",
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
"supports_function_calling": true,
"supports_parallel_function_calling": true,
"supports_reasoning": true,
@ -42613,6 +42615,7 @@
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
"supported_endpoints": ["/v1/chat/completions"],
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -42626,6 +42629,7 @@
"max_output_tokens": 65536,
"max_tokens": 65536,
"mode": "chat",
"supported_endpoints": ["/v1/chat/completions"],
"supports_function_calling": true,
"supports_reasoning": true,
"supports_response_schema": true,
@ -42679,6 +42683,8 @@
"max_output_tokens": 256000,
"max_tokens": 256000,
"mode": "chat",
"use_openai_responses_path": true,
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
"supports_function_calling": true,
"supports_parallel_function_calling": false,
"supports_reasoning": true,
@ -42693,6 +42699,8 @@
"max_output_tokens": 256000,
"max_tokens": 256000,
"mode": "chat",
"use_openai_responses_path": true,
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
"supports_function_calling": true,
"supports_parallel_function_calling": false,
"supports_reasoning": true,
@ -42707,6 +42715,8 @@
"max_output_tokens": 128000,
"max_tokens": 128000,
"mode": "chat",
"use_openai_responses_path": true,
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"],
"supports_function_calling": true,
"supports_parallel_function_calling": false,
"supports_reasoning": true,

View file

@ -174,7 +174,9 @@ class TestBedrockMantleResponsesURL:
class TestBedrockMantleGetLlmProviderRegion:
def test_get_llm_provider_uses_supplemental_litellm_params(self, monkeypatch):
def test_get_llm_provider_uses_supplemental_litellm_params(
self, monkeypatch, local_cost_map
):
monkeypatch.delenv("BEDROCK_MANTLE_REGION", raising=False)
monkeypatch.delenv("BEDROCK_MANTLE_API_BASE", raising=False)
monkeypatch.delenv("AWS_REGION", raising=False)
@ -187,9 +189,13 @@ class TestBedrockMantleGetLlmProviderRegion:
litellm_params=GenericLiteLLMParams(aws_region_name="us-east-2"),
)
assert provider == "bedrock_mantle"
assert api_base == "https://bedrock-mantle.us-east-2.api.aws/v1"
# gpt-5.x carries use_openai_responses_path, so its whole surface (incl.
# the resolved chat base) is on the /openai/v1 base per the AWS card.
assert api_base == "https://bedrock-mantle.us-east-2.api.aws/openai/v1"
def test_get_llm_provider_uses_aws_region_from_litellm_params(self, monkeypatch):
def test_get_llm_provider_uses_aws_region_from_litellm_params(
self, monkeypatch, local_cost_map
):
monkeypatch.delenv("BEDROCK_MANTLE_REGION", raising=False)
monkeypatch.delenv("BEDROCK_MANTLE_API_BASE", raising=False)
monkeypatch.delenv("AWS_REGION", raising=False)
@ -205,7 +211,7 @@ class TestBedrockMantleGetLlmProviderRegion:
litellm_params=params,
)
assert provider == "bedrock_mantle"
assert api_base == "https://bedrock-mantle.us-east-2.api.aws/v1"
assert api_base == "https://bedrock-mantle.us-east-2.api.aws/openai/v1"
class TestBedrockMantleResponsesAuth:
@ -368,7 +374,10 @@ class TestBedrockMantleResponsesTools:
class TestBedrockMantleResponsesRegistry:
def test_registry_returns_config_for_gpt_5_5(self):
def test_registry_returns_config_for_gpt_5_5(self, local_cost_map):
# gpt-5.x advertises /v1/responses in supported_endpoints (capability)
# and use_openai_responses_path (wire path), so it gets the native config
# on the /openai/v1/responses path. local_cost_map loads the entry.
from litellm.utils import ProviderConfigManager
cfg = ProviderConfigManager.get_provider_responses_api_config(
@ -378,7 +387,7 @@ class TestBedrockMantleResponsesRegistry:
assert isinstance(cfg, BedrockMantleResponsesAPIConfig)
assert cfg.use_openai_path is True
def test_registry_returns_config_for_gpt_5_4_enum(self):
def test_registry_returns_config_for_gpt_5_4_enum(self, local_cost_map):
from litellm.utils import ProviderConfigManager
cfg = ProviderConfigManager.get_provider_responses_api_config(
@ -388,39 +397,76 @@ class TestBedrockMantleResponsesRegistry:
assert isinstance(cfg, BedrockMantleResponsesAPIConfig)
assert cfg.use_openai_path is True
def test_registry_returns_none_for_gpt_oss(self):
# Regression guard: gpt-oss must NOT get the native Responses config; it
# keeps the chat-completions emulation path (responses/main.py ~line 1109).
def test_registry_returns_native_config_for_gpt_oss(self, local_cost_map):
# Core regression: gpt-oss-120b supports the native Responses API (AWS
# model card), so it must get a BedrockMantleResponsesAPIConfig on the
# STANDARD /v1/responses path -- NOT fall through to None / chat-completions
# emulation. Driven by /v1/responses in its price-map supported_endpoints.
# Fails on the old gate, which had no responses entry for gpt-oss.
from litellm.utils import ProviderConfigManager
cfg = ProviderConfigManager.get_provider_responses_api_config(
provider="bedrock_mantle",
model="openai.gpt-oss-120b",
)
assert cfg is None
assert isinstance(cfg, BedrockMantleResponsesAPIConfig)
assert cfg.use_openai_path is False
def test_registry_returns_none_for_gpt_oss_safeguard(self):
def test_registry_returns_native_config_for_gpt_oss_20b(self, local_cost_map):
from litellm.utils import ProviderConfigManager
cfg = ProviderConfigManager.get_provider_responses_api_config(
provider="bedrock_mantle",
model="openai.gpt-oss-safeguard-20b",
model="openai.gpt-oss-20b",
)
assert cfg is None
assert isinstance(cfg, BedrockMantleResponsesAPIConfig)
assert cfg.use_openai_path is False
def test_registry_returns_config_for_future_frontier_model(self):
# Forward-compatibility: an unseen OpenAI gpt frontier model (e.g. gpt-6),
# not yet in the price map, must get the openai-path Responses config with
# no code or JSON change. The name-convention fallback (openai.gpt- minus
# gpt-oss) catches it before any price-map entry exists.
def test_registry_returns_none_for_gpt_oss_safeguard(self, local_cost_map):
# Key discriminator: gpt-oss-safeguard shares the "gpt-oss" substring with
# gpt-oss-120b but does NOT support Responses (AWS card), so it must return
# None. Proves the gate is per-model (supported_endpoints) and not a naive
# gpt-oss substring match. local_cost_map loads the chat-only entry.
from litellm.utils import ProviderConfigManager
for model in ("openai.gpt-oss-safeguard-120b", "openai.gpt-oss-safeguard-20b"):
cfg = ProviderConfigManager.get_provider_responses_api_config(
provider="bedrock_mantle",
model=model,
)
assert cfg is None, model
@pytest.mark.parametrize(
"model",
["google.gemma-4-31b", "google.gemma-4-26b-a4b", "google.gemma-4-e2b"],
)
def test_registry_returns_native_config_for_gemma_4(self, local_cost_map, model):
# All three gemma-4 models support Responses (AWS cards) on the /openai/v1
# base, so each must get the native config with the openai path.
from litellm.utils import ProviderConfigManager
cfg = ProviderConfigManager.get_provider_responses_api_config(
provider="bedrock_mantle",
model=model,
)
assert isinstance(cfg, BedrockMantleResponsesAPIConfig)
assert cfg.use_openai_path is True
def test_unmapped_frontier_model_falls_through_to_none(self, restore_model_cost):
# The gate is data-driven, not name-based: an unseen model not yet in the
# price map (e.g. a future gpt-6) has no capability signal, so it falls
# through to None (chat-completions emulation) rather than being routed
# natively by a model-name guess. Onboarding it is a JSON / register_model
# change, never a code change (see the register_model tests below).
from litellm.utils import ProviderConfigManager
litellm.model_cost.pop("bedrock_mantle/openai.gpt-6", None)
litellm.get_model_info.cache_clear()
cfg = ProviderConfigManager.get_provider_responses_api_config(
provider="bedrock_mantle",
model="openai.gpt-6",
)
assert isinstance(cfg, BedrockMantleResponsesAPIConfig)
assert cfg.use_openai_path is True
assert cfg is None
def test_price_map_flag_routes_non_gpt_name_to_openai_path(
self, restore_model_cost
@ -542,8 +588,8 @@ class TestBedrockMantleResponsesRegistry:
assert cfg.use_openai_path is False
def test_unmapped_model_degrades_to_none_without_crashing(self, restore_model_cost):
# A non-frontier model that is not in model_cost makes get_model_info
# raise; the gate must swallow it and return None rather than crash.
# A model absent from model_cost has no capability signal, so the gate
# returns None (chat-completions emulation) rather than crashing.
from litellm.utils import ProviderConfigManager
litellm.model_cost.pop("bedrock_mantle/somelab.unmapped-model", None)
@ -560,6 +606,9 @@ class TestBedrockMantleResponsesRegistry:
# place, so the snapshot must be a deepcopy: a shallow dict() copy would
# share that nested dict and leave mode=responses after restore, making
# the final assertion fail. The in-place clear+update mirrors the fixture.
# gpt-oss-safeguard is the right vehicle here: it is chat-only, so without
# the registered mode=responses it resolves to None, isolating the effect
# of the register/restore from the model's own (lack of) capability.
from litellm.utils import ProviderConfigManager, register_model
snapshot = copy.deepcopy(litellm.model_cost)
@ -567,14 +616,14 @@ class TestBedrockMantleResponsesRegistry:
try:
register_model(
{
"bedrock_mantle/openai.gpt-oss-120b": {
"bedrock_mantle/openai.gpt-oss-safeguard-120b": {
"litellm_provider": "bedrock_mantle",
"mode": "responses",
}
}
)
during = ProviderConfigManager.get_provider_responses_api_config(
provider="bedrock_mantle", model="openai.gpt-oss-120b"
provider="bedrock_mantle", model="openai.gpt-oss-safeguard-120b"
)
assert isinstance(during, BedrockMantleResponsesAPIConfig)
finally:
@ -582,11 +631,151 @@ class TestBedrockMantleResponsesRegistry:
litellm.model_cost.update(snapshot)
litellm.get_model_info.cache_clear()
after = ProviderConfigManager.get_provider_responses_api_config(
provider="bedrock_mantle", model="openai.gpt-oss-120b"
provider="bedrock_mantle", model="openai.gpt-oss-safeguard-120b"
)
assert after is None
class TestMantleBaseSegment:
"""The wire-path helper is data-driven from the price-map
use_openai_responses_path flag (NOT a model-name match): flagged models are on
the /openai/v1 base, everything else on /v1. An unmapped model defaults to /v1.
"""
@pytest.mark.parametrize(
"model,model_cost,expected",
[
(
"openai.gpt-5.5",
{"bedrock_mantle/openai.gpt-5.5": {"use_openai_responses_path": True}},
"openai/v1",
),
(
"google.gemma-4-31b",
{
"bedrock_mantle/google.gemma-4-31b": {
"use_openai_responses_path": True
}
},
"openai/v1",
),
(
"openai.gpt-oss-120b",
{"bedrock_mantle/openai.gpt-oss-120b": {}},
"v1",
),
("openai.gpt-oss-120b", {}, "v1"),
(None, {}, "v1"),
],
)
def test_base_segment(self, model, model_cost, expected):
from litellm.llms.bedrock_mantle.common_utils import mantle_base_segment
assert mantle_base_segment(model, model_cost) == expected
class TestMantleSupportsResponses:
"""The capability helper is data-driven (supported_endpoints / mode), with no
model-name match: per-model, so gpt-oss-120b is supported but the safeguard
variant is not despite the shared substring."""
@pytest.mark.parametrize(
"model,model_cost,expected",
[
# supported_endpoints lists responses -> supported
(
"openai.gpt-oss-120b",
{
"bedrock_mantle/openai.gpt-oss-120b": {
"supported_endpoints": ["/v1/chat/completions", "/v1/responses"]
}
},
True,
),
# chat-only supported_endpoints -> not supported (the discriminator)
(
"openai.gpt-oss-safeguard-120b",
{
"bedrock_mantle/openai.gpt-oss-safeguard-120b": {
"supported_endpoints": ["/v1/chat/completions"]
}
},
False,
),
# mode=responses (no supported_endpoints) -> supported
(
"somelab.future-model",
{"bedrock_mantle/somelab.future-model": {"mode": "responses"}},
True,
),
# mode=chat, no responses endpoint -> not supported
(
"google.gemma-3-27b-it",
{"bedrock_mantle/google.gemma-3-27b-it": {"mode": "chat"}},
False,
),
# absent from model_cost -> no signal -> not supported
("somelab.unmapped", {}, False),
(None, {}, False),
],
)
def test_supports_responses(self, model, model_cost, expected):
from litellm.llms.bedrock_mantle.common_utils import mantle_supports_responses
assert mantle_supports_responses(model, model_cost) is expected
class TestBedrockMantlePerModelResponsesURL:
"""End-to-end: the registry-selected config must build the correct wire URL
per model. gpt-oss on /v1/responses, gpt-5.x and gemma-4 on
/openai/v1/responses."""
def _url_for(self, model, region="us-east-2"):
from litellm.utils import ProviderConfigManager
cfg = ProviderConfigManager.get_provider_responses_api_config(
provider="bedrock_mantle",
model=model,
)
assert isinstance(cfg, BedrockMantleResponsesAPIConfig)
return cfg.get_complete_url(
api_base=None, litellm_params={"aws_region_name": region}
)
def test_gpt_oss_uses_standard_responses_path(self, local_cost_map):
url = self._url_for("openai.gpt-oss-120b")
assert url == "https://bedrock-mantle.us-east-2.api.aws/v1/responses"
assert "/openai/v1/responses" not in url
def test_gpt_5_5_uses_openai_responses_path(self, local_cost_map):
url = self._url_for("openai.gpt-5.5")
assert url == "https://bedrock-mantle.us-east-2.api.aws/openai/v1/responses"
@pytest.mark.parametrize(
"model",
["google.gemma-4-31b", "google.gemma-4-26b-a4b", "google.gemma-4-e2b"],
)
def test_gemma_4_uses_openai_responses_path(self, local_cost_map, model):
url = self._url_for(model)
assert url == "https://bedrock-mantle.us-east-2.api.aws/openai/v1/responses"
class TestBedrockMantleEndpointHonoring:
def test_plain_chat_call_to_gpt_oss_is_not_bridged(self, local_cost_map):
# Adding native Responses support to gpt-oss must NOT reroute its plain
# chat-completions traffic. responses_api_bridge_check keys off mode, and
# gpt-oss stays mode=chat, so a completion() call is not flipped to the
# Responses API. Guards the dual-capability contract.
from litellm.main import responses_api_bridge_check
model_info, resolved_model = responses_api_bridge_check(
model="openai.gpt-oss-120b",
custom_llm_provider="bedrock_mantle",
)
assert model_info.get("mode") != "responses"
assert resolved_model == "openai.gpt-oss-120b"
@pytest.fixture
def restore_model_cost():
"""Snapshot litellm.model_cost so register_model edits don't leak across tests.

View file

@ -131,7 +131,9 @@ class TestBedrockMantleConfig:
),
)
def test_get_llm_provider_uses_aws_region_name_for_responses(self, monkeypatch):
def test_get_llm_provider_uses_aws_region_name_for_responses(
self, monkeypatch, local_cost_map
):
from litellm.types.router import GenericLiteLLMParams
monkeypatch.delenv("BEDROCK_MANTLE_REGION", raising=False)
@ -143,7 +145,9 @@ class TestBedrockMantleConfig:
litellm_params=GenericLiteLLMParams(aws_region_name="us-east-2"),
)
assert provider == "bedrock_mantle"
assert api_base == "https://bedrock-mantle.us-east-2.api.aws/v1"
# gpt-5.x carries use_openai_responses_path, so it is served on the
# /openai/v1 base per the AWS model card.
assert api_base == "https://bedrock-mantle.us-east-2.api.aws/openai/v1"
def test_default_api_base_fallback_to_us_east_1(self, monkeypatch):
monkeypatch.delenv("BEDROCK_MANTLE_REGION", raising=False)
@ -159,6 +163,50 @@ class TestBedrockMantleConfig:
api_base, _ = cfg._get_openai_compatible_provider_info(custom_base, None)
assert api_base == custom_base
def test_chat_base_for_gpt_oss_uses_v1(self, monkeypatch):
# gpt-oss carries no use_openai_responses_path flag, so it stays on the
# standard /v1 base; no regression for existing chat usage now that the
# segment is data-driven.
monkeypatch.setenv("BEDROCK_MANTLE_REGION", "us-east-2")
monkeypatch.delenv("BEDROCK_MANTLE_API_BASE", raising=False)
cfg = BedrockMantleChatConfig()
api_base, _ = cfg._get_openai_compatible_provider_info(
None, None, model="openai.gpt-oss-120b"
)
assert api_base == "https://bedrock-mantle.us-east-2.api.aws/v1"
@pytest.mark.parametrize(
"model_id",
["google.gemma-4-31b", "google.gemma-4-26b-a4b", "google.gemma-4-e2b"],
)
def test_chat_base_for_gemma_4_uses_openai_v1(
self, monkeypatch, local_cost_map, model_id
):
# The chat-config bug the Gemma 4 cards exposed: gemma-4-* is served on the
# /openai/v1 base, not the hardcoded /v1. Driven by the price-map
# use_openai_responses_path flag (loaded by local_cost_map). Fails before
# the data-driven segment lands.
monkeypatch.setenv("BEDROCK_MANTLE_REGION", "us-east-2")
monkeypatch.delenv("BEDROCK_MANTLE_API_BASE", raising=False)
cfg = BedrockMantleChatConfig()
api_base, _ = cfg._get_openai_compatible_provider_info(
None, None, model=model_id
)
assert api_base == "https://bedrock-mantle.us-east-2.api.aws/openai/v1"
def test_chat_base_explicit_api_base_wins_over_derived(
self, monkeypatch, local_cost_map
):
# An explicit api_base must not be overridden by the data-driven default,
# even for a model whose default differs (gemma-4 -> openai/v1).
monkeypatch.delenv("BEDROCK_MANTLE_API_BASE", raising=False)
custom_base = "https://bedrock-mantle.us-west-2.api.aws/v1"
cfg = BedrockMantleChatConfig()
api_base, _ = cfg._get_openai_compatible_provider_info(
custom_base, None, model="google.gemma-4-31b"
)
assert api_base == custom_base
def test_api_key_from_env(self, monkeypatch):
monkeypatch.setenv("BEDROCK_MANTLE_API_KEY", "test-key-123")
cfg = BedrockMantleChatConfig()