fix(bedrock): route model_id overrides to Converse and never send an empty bearer natively

A deployment whose litellm_params carry model_id (an application inference profile or provisioned throughput ARN) went to the native Chat Completions route with the base model in the URL and model_id left in the body. It now takes Converse like the bedrock/arn:... model form, which encodes the override into the request URL

A blank api_key on a SigV4 deployment became an Authorization header reading Bearer with nothing after it on the native route, since the OpenAI-like header builder writes any non-None key and the signer keeps a non-AWS4 Authorization header. validate_environment now resolves the key through bedrock_bearer_token, so a blank key is signed with SigV4 the way Converse signs it
This commit is contained in:
mateo-berri 2026-10-01 15:50:58 -07:00
parent dae29c138b
commit 578f26edba
3 changed files with 87 additions and 4 deletions

View file

@ -31,7 +31,7 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import (
inline_remote_media,
)
from litellm.llms.base_llm.chat.transformation import BaseLLMException
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM, bedrock_bearer_token
from litellm.llms.bedrock.common_utils import (
BedrockError,
bedrock_model_is_openai_gpt,
@ -263,6 +263,26 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
) -> BaseLLMException:
return BedrockError(status_code=status_code, message=error_message, headers=headers)
def validate_environment(
self,
headers: dict, # mutable-ok: BaseConfig signature
model: str,
messages: list[AllMessageValues],
optional_params: dict, # mutable-ok: BaseConfig signature
litellm_params: dict, # mutable-ok: BaseConfig signature
api_key: str | None = None,
api_base: str | None = None,
) -> dict: # mutable-ok: BaseConfig signature
return super().validate_environment(
headers=headers,
model=model,
messages=messages,
optional_params=optional_params,
litellm_params=litellm_params,
api_key=bedrock_bearer_token(api_key),
api_base=api_base,
)
def get_complete_url(
self,
api_base: str | None,

View file

@ -906,6 +906,7 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
"additionalModelRequestFields",
"top_k",
"stop",
"model_id",
)
)
@ -931,7 +932,9 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje
Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking``
block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse
forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on
AWS's native OpenAI surface, ``stop`` stays on Converse where it fails loudly instead of silently
AWS's native OpenAI surface, a ``model_id`` override (an application inference profile or provisioned
throughput ARN) is only encoded into Converse's request URL and so stays on Converse like the
``bedrock/arn:...`` model form, ``stop`` stays on Converse where it fails loudly instead of silently
stopping hidden reasoning, operator-owned request metadata is only written onto the Converse body,
function tools (``tools`` or legacy ``functions``) on a model without
``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless

View file

@ -25,6 +25,8 @@ from litellm.llms.bedrock.common_utils import (
)
from litellm.llms.custom_httpx.http_handler import HTTPHandler
APPLICATION_INFERENCE_PROFILE_ARN = "arn:aws:bedrock:us-west-2:123412341234:application-inference-profile/a1b2c3"
@pytest.fixture
def local_cost_map(monkeypatch):
@ -425,8 +427,13 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model):
)
@pytest.mark.parametrize(
"request_params",
[{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}, {"stop": ["END"]}],
ids=["additionalModelRequestFields", "top_k", "stop"],
[
{"additionalModelRequestFields": {"reasoning_effort": "high"}},
{"top_k": 40},
{"stop": ["END"]},
{"model_id": APPLICATION_INFERENCE_PROFILE_ARN},
],
ids=["additionalModelRequestFields", "top_k", "stop", "model_id"],
)
def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params):
assert bedrock_request_needs_converse(model, request_params) is True
@ -434,6 +441,59 @@ def test_converse_extension_params_fall_back_to_converse(local_cost_map, model,
assert BedrockModelInfo.get_bedrock_route(model, {key: None for key in request_params}) == "chat_completions"
@pytest.mark.parametrize(
"model", ["bedrock/us.openai.gpt-5.6-sol", "global.openai.gpt-6-sol", "bedrock/chat_completions/us.xai.grok-4.6"]
)
def test_model_id_override_is_served_by_converse_like_the_arn_model_form(local_cost_map, model):
assert bedrock_route_for_request(model, {"model_id": APPLICATION_INFERENCE_PROFILE_ARN}, None) == "converse"
assert bedrock_route_for_request(model, {"model_id": None}, None) == "chat_completions"
SIGV4_PARAMS = {
"aws_access_key_id": "AKIAIOSFODNN7EXAMPLE",
"aws_secret_access_key": "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY",
"aws_region_name": "us-east-1",
}
@pytest.mark.parametrize("api_key", ["", None], ids=["blank", "absent"])
def test_blank_api_key_is_signed_with_sigv4_instead_of_an_empty_bearer(monkeypatch, api_key):
monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False)
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
url = "https://bedrock-runtime.us-east-1.amazonaws.com/openai/v1/chat/completions"
headers = cfg.validate_environment(
headers={},
model="bedrock/us.openai.gpt-5.6-sol",
messages=[{"role": "user", "content": "hello"}],
optional_params=dict(SIGV4_PARAMS),
litellm_params={},
api_key=api_key,
)
assert "Authorization" not in headers
signed, _ = cfg.sign_request(
headers=headers,
optional_params=dict(SIGV4_PARAMS),
request_data={"model": "us.openai.gpt-5.6-sol", "messages": []},
api_base=url,
api_key=api_key,
)
assert signed["Authorization"].startswith("AWS4-HMAC-SHA256 Credential=AKIAIOSFODNN7EXAMPLE/"), signed
def test_bearer_api_key_is_sent_as_the_authorization_header(monkeypatch):
monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False)
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
headers = cfg.validate_environment(
headers={},
model="bedrock/us.openai.gpt-5.6-sol",
messages=[{"role": "user", "content": "hello"}],
optional_params={},
litellm_params={},
api_key="bedrock-api-key",
)
assert headers["Authorization"] == "Bearer bedrock-api-key"
@pytest.mark.parametrize(
"request_params, expected_route",
[