mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-01 02:02:20 +00:00
fix(bedrock): keep native chat completions when only top_k is set
Converse only forwards top_k for Anthropic and Nova, so treating it as a converse-only key sent gpt-oss, GPT-5.6, and Grok off the native route and still dropped the value
This commit is contained in:
parent
e653217228
commit
da1c77be2c
2 changed files with 24 additions and 9 deletions
|
|
@ -898,7 +898,6 @@ BEDROCK_CONVERSE_ONLY_REQUEST_KEYS: Final = frozenset(
|
|||
"outputConfig",
|
||||
"thinking",
|
||||
"additionalModelRequestFields",
|
||||
"top_k",
|
||||
)
|
||||
)
|
||||
|
||||
|
|
@ -919,8 +918,8 @@ def bedrock_request_needs_converse(model: str, request_params: Mapping[str, obje
|
|||
"""Whether a request on a runtime-Chat-Completions model must still be served by Converse.
|
||||
|
||||
Converse-shaped body keys (``BEDROCK_CONVERSE_ONLY_REQUEST_KEYS``, the Anthropic-style ``thinking``
|
||||
block and the ``additionalModelRequestFields`` / ``top_k`` extension params included, which only Converse
|
||||
forwards as ``additionalModelRequestFields`` and ``inferenceConfig``) have no field on
|
||||
block and ``additionalModelRequestFields`` included, which only Converse
|
||||
forwards as ``additionalModelRequestFields``) have no field on
|
||||
AWS's native OpenAI surface, operator-owned request metadata is only written onto the Converse body,
|
||||
function tools (``tools`` or legacy ``functions``) on a model without
|
||||
``supports_bedrock_runtime_chat_completions_tools_with_reasoning`` are rejected there unless
|
||||
|
|
|
|||
|
|
@ -259,17 +259,33 @@ def test_guardrail_config_falls_back_to_converse(local_cost_map, model):
|
|||
|
||||
|
||||
@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6"])
|
||||
@pytest.mark.parametrize(
|
||||
"request_params",
|
||||
[{"additionalModelRequestFields": {"reasoning_effort": "high"}}, {"top_k": 40}],
|
||||
ids=["additionalModelRequestFields", "top_k"],
|
||||
)
|
||||
def test_converse_extension_params_fall_back_to_converse(local_cost_map, model, request_params):
|
||||
def test_additional_model_request_fields_fall_back_to_converse(local_cost_map, model):
|
||||
request_params = {"additionalModelRequestFields": {"reasoning_effort": "high"}}
|
||||
assert bedrock_request_needs_converse(model, request_params) is True
|
||||
assert BedrockModelInfo.get_bedrock_route(model, request_params) == "converse"
|
||||
assert BedrockModelInfo.get_bedrock_route(model, {key: None for key in request_params}) == "chat_completions"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["openai.gpt-oss-20b-1:0", "us.xai.grok-4.6", "global.openai.gpt-5.6-sol"])
|
||||
def test_top_k_stays_on_chat_completions(local_cost_map, fake_aws_env, model):
|
||||
request_params = {"top_k": 40}
|
||||
assert bedrock_request_needs_converse(model, request_params) is False
|
||||
assert BedrockModelInfo.get_bedrock_route(model, request_params) == "chat_completions"
|
||||
|
||||
requests, client = _recording_client(json=_chat_completion_json("ok", model))
|
||||
litellm.completion(
|
||||
model=f"bedrock/{model}",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
top_k=40,
|
||||
client=client,
|
||||
)
|
||||
|
||||
assert str(requests[0].url) == "https://bedrock-runtime.us-west-2.amazonaws.com/openai/v1/chat/completions"
|
||||
body = json.loads(requests[0].content)
|
||||
assert body["top_k"] == 40
|
||||
assert "inferenceConfig" not in body
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"request_params, expected_route",
|
||||
[
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue