mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
fix(bedrock): route model_id overrides to Converse and never send an empty bearer natively
A deployment whose litellm_params carry model_id (an application inference profile or provisioned throughput ARN) went to the native Chat Completions route with the base model in the URL and model_id left in the body. It now takes Converse like the bedrock/arn:... model form, which encodes the override into the request URL A blank api_key on a SigV4 deployment became an Authorization header reading Bearer with nothing after it on the native route, since the OpenAI-like header builder writes any non-None key and the signer keeps a non-AWS4 Authorization header. validate_environment now resolves the key through bedrock_bearer_token, so a blank key is signed with SigV4 the way Converse signs it
This commit is contained in:
parent
dae29c138b
commit
578f26edba
3 changed files with 87 additions and 4 deletions
|
|
@ -31,7 +31,7 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import (
|
|||
inline_remote_media,
|
||||
)
|
||||
from litellm.llms.base_llm.chat.transformation import BaseLLMException
|
||||
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM
|
||||
from litellm.llms.bedrock.base_aws_llm import BaseAWSLLM, bedrock_bearer_token
|
||||
from litellm.llms.bedrock.common_utils import (
|
||||
BedrockError,
|
||||
bedrock_model_is_openai_gpt,
|
||||
|
|
@ -263,6 +263,26 @@ class AmazonBedrockRuntimeChatCompletionsConfig(OpenAILikeChatConfig):
|
|||
) -> BaseLLMException:
|
||||
return BedrockError(status_code=status_code, message=error_message, headers=headers)
|
||||
|
||||
def validate_environment(
|
||||
self,
|
||||
headers: dict, # mutable-ok: BaseConfig signature
|
||||
model: str,
|
||||
messages: list[AllMessageValues],
|
||||
optional_params: dict, # mutable-ok: BaseConfig signature
|
||||
litellm_params: dict, # mutable-ok: BaseConfig signature
|
||||
api_key: str | None = None,
|
||||
api_base: str | None = None,
|
||||
) -> dict: # mutable-ok: BaseConfig signature
|
||||
return super().validate_environment(
|
||||
headers=headers,
|
||||
model=model,
|
||||
messages=messages,
|
||||
optional_params=optional_params,
|
||||
litellm_params=litellm_params,
|
||||
api_key=bedrock_bearer_token(api_key),
|
||||
api_base=api_base,
|
||||
)
|
||||
|
||||
def get_complete_url(
|
||||
self,
|
||||
api_base: str | None,
|
||||
|
|
|
|||
|
|
@ -906,6 +906,7 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
|
|||
"additionalModelRequestFields",
|
||||
"top_k",
|
||||
"stop",
|
||||
"model_id",
|
||||
)
|
||||
)
|
||||
|
||||
|
|
@ -931,7 +932,9 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje
|
|||
Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking``
|
||||
block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse
|
||||
forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on
|
||||
AWS's native OpenAI surface, ``stop`` stays on Converse where it fails loudly instead of silently
|
||||
AWS's native OpenAI surface, a ``model_id`` override (an application inference profile or provisioned
|
||||
throughput ARN) is only encoded into Converse's request URL and so stays on Converse like the
|
||||
``bedrock/arn:...`` model form, ``stop`` stays on Converse where it fails loudly instead of silently
|
||||
stopping hidden reasoning, operator-owned request metadata is only written onto the Converse body,
|
||||
function tools (``tools`` or legacy ``functions``) on a model without
|
||||
``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless
|
||||
|
|
|
|||
|
|
@ -25,6 +25,8 @@ from litellm.llms.bedrock.common_utils import (
|
|||
)
|
||||
from litellm.llms.custom_httpx.http_handler import HTTPHandler
|
||||
|
||||
APPLICATION_INFERENCE_PROFILE_ARN = "arn:aws:bedrock:us-west-2:123412341234:application-inference-profile/a1b2c3"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def local_cost_map(monkeypatch):
|
||||
|
|
@ -425,8 +427,13 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model):
|
|||
)
|
||||
@pytest.mark.parametrize(
|
||||
"request_params",
|
||||
[{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}, {"stop": ["END"]}],
|
||||
ids=["additionalModelRequestFields", "top_k", "stop"],
|
||||
[
|
||||
{"additionalModelRequestFields": {"reasoning_effort": "high"}},
|
||||
{"top_k": 40},
|
||||
{"stop": ["END"]},
|
||||
{"model_id": APPLICATION_INFERENCE_PROFILE_ARN},
|
||||
],
|
||||
ids=["additionalModelRequestFields", "top_k", "stop", "model_id"],
|
||||
)
|
||||
def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params):
|
||||
assert bedrock_request_needs_converse(model, request_params) is True
|
||||
|
|
@ -434,6 +441,59 @@ def test_converse_extension_params_fall_back_to_converse(local_cost_map, model,
|
|||
assert BedrockModelInfo.get_bedrock_route(model, {key: None for key in request_params}) == "chat_completions"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model", ["bedrock/us.openai.gpt-5.6-sol", "global.openai.gpt-6-sol", "bedrock/chat_completions/us.xai.grok-4.6"]
|
||||
)
|
||||
def test_model_id_override_is_served_by_converse_like_the_arn_model_form(local_cost_map, model):
|
||||
assert bedrock_route_for_request(model, {"model_id": APPLICATION_INFERENCE_PROFILE_ARN}, None) == "converse"
|
||||
assert bedrock_route_for_request(model, {"model_id": None}, None) == "chat_completions"
|
||||
|
||||
|
||||
SIGV4_PARAMS = {
|
||||
"aws_access_key_id": "AKIAIOSFODNN7EXAMPLE",
|
||||
"aws_secret_access_key": "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY",
|
||||
"aws_region_name": "us-east-1",
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize("api_key", ["", None], ids=["blank", "absent"])
|
||||
def test_blank_api_key_is_signed_with_sigv4_instead_of_an_empty_bearer(monkeypatch, api_key):
|
||||
monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False)
|
||||
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
|
||||
url = "https://bedrock-runtime.us-east-1.amazonaws.com/openai/v1/chat/completions"
|
||||
headers = cfg.validate_environment(
|
||||
headers={},
|
||||
model="bedrock/us.openai.gpt-5.6-sol",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
optional_params=dict(SIGV4_PARAMS),
|
||||
litellm_params={},
|
||||
api_key=api_key,
|
||||
)
|
||||
assert "Authorization" not in headers
|
||||
signed, _ = cfg.sign_request(
|
||||
headers=headers,
|
||||
optional_params=dict(SIGV4_PARAMS),
|
||||
request_data={"model": "us.openai.gpt-5.6-sol", "messages": []},
|
||||
api_base=url,
|
||||
api_key=api_key,
|
||||
)
|
||||
assert signed["Authorization"].startswith("AWS4-HMAC-SHA256 Credential=AKIAIOSFODNN7EXAMPLE/"), signed
|
||||
|
||||
|
||||
def test_bearer_api_key_is_sent_as_the_authorization_header(monkeypatch):
|
||||
monkeypatch.delenv("AWS_BEARER_TOKEN_BEDROCK", raising=False)
|
||||
cfg = AmazonBedrockRuntimeChatCompletionsConfig()
|
||||
headers = cfg.validate_environment(
|
||||
headers={},
|
||||
model="bedrock/us.openai.gpt-5.6-sol",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
api_key="bedrock-api-key",
|
||||
)
|
||||
assert headers["Authorization"] == "Bearer bedrock-api-key"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"request_params, expected_route",
|
||||
[
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue