mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
fix(bedrock): keep chat_completions/-prefixed deployments on the native Responses surface
This commit is contained in:
parent
3a390e6206
commit
84ae4596b5
2 changed files with 33 additions and 3 deletions
|
|
@ -71,11 +71,16 @@ BEDROCK_RUNTIME_SUPPORTED_RESPONSE_TOOL_TYPES: Final = frozenset(
|
|||
{"function", "mcp", "custom", "apply_patch", "namespace", "tool_search", "computer"}
|
||||
)
|
||||
BEDROCK_RUNTIME_UNSUPPORTED_RESPONSE_PARAMS: Final = frozenset({"background"})
|
||||
BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX: Final = "chat_completions/"
|
||||
REMOTE_IMAGE_URL_SCHEMES: Final = ("http://", "https://")
|
||||
IMAGE_BLOCK_KEYS: Final = ("content", "output")
|
||||
IMAGE_BLOCK_TYPES: Final = frozenset({"input_image", "computer_screenshot"})
|
||||
|
||||
|
||||
def _without_chat_completions_route(model: str) -> str:
|
||||
return model.removeprefix(BEDROCK_CHAT_COMPLETIONS_ROUTE_PREFIX)
|
||||
|
||||
|
||||
def resolve_bedrock_bearer_token(api_key: str | None) -> str | None:
|
||||
return api_key or get_secret_str("AWS_BEARER_TOKEN_BEDROCK")
|
||||
|
||||
|
|
@ -168,9 +173,13 @@ class BedrockOpenAIResponsesConfig(BaseAWSLLM, OpenAIResponsesAPIConfig):
|
|||
The capability decision lives here rather than in the shared dispatch so that
|
||||
onboarding a model, or changing how the signal is read, stays inside the
|
||||
Bedrock adapter. ``None`` leaves the caller's existing behaviour untouched --
|
||||
chat-only Bedrock models keep the Chat Completions bridge.
|
||||
chat-only Bedrock models keep the Chat Completions bridge. The ``chat_completions/``
|
||||
opt-in only moves Chat Completions calls off Converse, so a Responses call on such a
|
||||
deployment still takes this surface instead of being bridged.
|
||||
"""
|
||||
if not bedrock_supports_openai_responses(model, litellm.model_cost):
|
||||
if not model or not bedrock_supports_openai_responses(
|
||||
_without_chat_completions_route(model), litellm.model_cost
|
||||
):
|
||||
return None
|
||||
return cls()
|
||||
|
||||
|
|
@ -330,7 +339,7 @@ class BedrockOpenAIResponsesConfig(BaseAWSLLM, OpenAIResponsesAPIConfig):
|
|||
rewritten_types,
|
||||
)
|
||||
return super().transform_responses_api_request(
|
||||
model=model,
|
||||
model=_without_chat_completions_route(model),
|
||||
input=normalized_input,
|
||||
response_api_optional_request_params=response_api_optional_request_params,
|
||||
litellm_params=litellm_params,
|
||||
|
|
|
|||
|
|
@ -162,6 +162,27 @@ class TestForModelGate:
|
|||
):
|
||||
assert BedrockOpenAIResponsesConfig.for_model(None) is None
|
||||
|
||||
def test_chat_completions_route_keeps_the_native_responses_surface(self):
|
||||
with patch.object( # test-quality-ok: the gate reads the global cost map by design; no injection point exists
|
||||
litellm, "model_cost", {MODEL: {"supported_endpoints": ["/v1/responses"]}}
|
||||
):
|
||||
cfg = BedrockOpenAIResponsesConfig.for_model(f"chat_completions/{MODEL}")
|
||||
assert isinstance(cfg, BedrockOpenAIResponsesConfig)
|
||||
body = cfg.transform_responses_api_request(
|
||||
model=f"chat_completions/{MODEL}",
|
||||
input="hi",
|
||||
response_api_optional_request_params={},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
assert body["model"] == MODEL
|
||||
|
||||
def test_converse_route_keeps_the_chat_completions_bridge(self):
|
||||
with patch.object( # test-quality-ok: the gate reads the global cost map by design; no injection point exists
|
||||
litellm, "model_cost", {MODEL: {"supported_endpoints": ["/v1/responses"]}}
|
||||
):
|
||||
assert BedrockOpenAIResponsesConfig.for_model(f"converse/{MODEL}") is None
|
||||
|
||||
|
||||
class TestProviderResolution:
|
||||
"""model_cost is patched explicitly: it is populated at import time from a GitHub
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue