fix(bedrock): keep chat_completions/-prefixed deployments on the native Responses surface

This commit is contained in:
mateo-berri 2026-09-28 21:46:35 -07:00
parent 3a390e6206
commit 84ae4596b5
2 changed files with 33 additions and 3 deletions

View file

@ -71,11 +71,16 @@ BEDROCK_RUNTIME_SUPPORTED_RESPONSE_TOOL_TYPES: Final = frozenset(
{"function", "mcp", "custom", "apply_patch", "namespace", "tool_search", "computer"}
)
BEDROCK_RUNTIME_UNSUPPORTED_RESPONSE_PARAMS: Final = frozenset({"background"})
BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX: Final = "chat_completions/"
REMOTE_IMAGE_URL_SCHEMES: Final = ("http://", "https://")
IMAGE_BLOCK_KEYS: Final = ("content", "output")
IMAGE_BLOCK_TYPES: Final = frozenset({"input_image", "computer_screenshot"})
def _without_chat_completions_route(model: str) -> str:
return model.removeprefix(BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX)
def resolve_bedrock_bearer_token(api_key: str | None) -> str | None:
return api_key or get_secret_str("AWS_BEARER_TOKEN_BEDROCK")
@ -168,9 +173,13 @@ class BedrockOpenAIResponsesConfig(BaseAWSLLM, OpenAIResponsesAPIConfig):
The capability decision lives here rather than in the shared dispatch so that
onboarding a model, or changing how the signal is read, stays inside the
Bedrock adapter. ``None`` leaves the caller's existing behaviour untouched --
chat-only Bedrock models keep the Chat Completions bridge.
chat-only Bedrock models keep the Chat Completions bridge. The ``chat_completions/``
opt-in only moves Chat Completions calls off Converse, so a Responses call on such a
deployment still takes this surface instead of being bridged.
"""
if not bedrock_supports_openai_responses(model, litellm.model_cost):
if not model or not bedrock_supports_openai_responses(
_without_chat_completions_route(model), litellm.model_cost
):
return None
return cls()
@ -330,7 +339,7 @@ class BedrockOpenAIResponsesConfig(BaseAWSLLM, OpenAIResponsesAPIConfig):
rewritten_types,
)
return super().transform_responses_api_request(
model=model,
model=_without_chat_completions_route(model),
input=normalized_input,
response_api_optional_request_params=response_api_optional_request_params,
litellm_params=litellm_params,

View file

@ -162,6 +162,27 @@ class TestForModelGate:
):
assert BedrockOpenAIResponsesConfig.for_model(None) is None
def test_chat_completions_route_keeps_the_native_responses_surface(self):
with patch.object( # test-quality-ok: the gate reads the global cost map by design; no injection point exists
litellm, "model_cost", {MODEL: {"supported_endpoints": ["/v1/responses"]}}
):
cfg = BedrockOpenAIResponsesConfig.for_model(f"chat_completions/{MODEL}")
assert isinstance(cfg, BedrockOpenAIResponsesConfig)
body = cfg.transform_responses_api_request(
model=f"chat_completions/{MODEL}",
input="hi",
response_api_optional_request_params={},
litellm_params=GenericLiteLLMParams(),
headers={},
)
assert body["model"] == MODEL
def test_converse_route_keeps_the_chat_completions_bridge(self):
with patch.object( # test-quality-ok: the gate reads the global cost map by design; no injection point exists
litellm, "model_cost", {MODEL: {"supported_endpoints": ["/v1/responses"]}}
):
assert BedrockOpenAIResponsesConfig.for_model(f"converse/{MODEL}") is None
class TestProviderResolution:
"""model_cost is patched explicitly: it is populated at import time from a GitHub