From 80868704986978660b4a99072939f2149354c364 Mon Sep 17 00:00:00 2001 From: Arun Mittal Date: Sun, 7 Jun 2026 20:54:24 -0400 Subject: [PATCH] fix(bedrock-invoke): use native output_format for Claude models that support it, gate via registry flag MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit PR #19877 forced every response_format request on Bedrock invoke through synthetic tool-injection by spoofing the model name to claude-3-sonnet-20240229. AWS has since shipped native output_format support for most Claude 4.5+ models (opus-4-6, sonnet-4-5/4-6, haiku-4-5). For those, tool-injection wastes ~3.4x input tokens vs. native and is fragile under reasoning_effort. Side-effect of the spoof: when response_format is combined with reasoning_effort, _is_adaptive_thinking_model is called with the spoofed model name, fails, and emits legacy thinking.enabled+budget_tokens — which AWS opus-4-7/4-8 reject with: "thinking.type.enabled is not supported for this model. Use thinking.type.adaptive and output_config.effort to control thinking behavior." Fix: pre-hoc strip-and-inject pattern, gated on the existing supports_native_structured_output flag on the Bedrock entry in model_prices_and_context_window.json (single source of truth — same flag the Converse transformer reads at converse_transformation.py:1014). When the flag is true (opus-4-6, sonnet-4-5/4-6, haiku-4-5): let the parent emit native output_format; auto-add the structured-outputs-2025-11-13 beta header. When the flag is false (opus-4-7, opus-4-8 today): pop response_format from non_default_params BEFORE delegating to AnthropicConfig.map_openai_params so the parent doesn't emit output_format. The parent then sees the real model name — adaptive thinking, effort mapping, and beta-header derivation all work correctly. After the parent returns, manually inject the json_tool_call tool using the parent's own public helpers. Companion changes: - Flip supports_native_structured_output from true to false on opus-4-7 and opus-4-8 Bedrock entries in model_prices_and_context_window.json and litellm/model_prices_and_context_window_backup.json (22 entries total across both files). AWS rejects native output_format on both Invoke and Converse for these two models; this flip fixes both surfaces simultaneously since they read the same flag. - Flip bedrock.structured-outputs-2025-11-13 from null to "structured-outputs-2025-11-13" in anthropic_beta_headers_config.json so filter_and_transform_beta_headers no longer drops the gating header. When AWS ships native output_format support on opus-4-7/4-8 Bedrock, flip the registry flag back to true (one-line revert per entry). Tests: - test_response_format_emits_native_output_format_for_opus_4_6_invoke - test_response_format_uses_tool_workaround_for_opus_4_7_invoke - test_response_format_plus_reasoning_preserves_adaptive_thinking_on_opus_4_7 - test_response_format_plus_reasoning_uses_native_with_adaptive_on_opus_4_6 - Updated test_bedrock_chat_invoke_checks_output_config_support_with_bedrock_provider to assert_any_call (two _supports_factory calls now happen) Refs: AWS doc confirming opus-4-7 requires adaptive thinking — https://docs.aws.amazon.com/bedrock/latest/userguide/model-card-anthropic-claude-opus-4-7.html --- litellm/anthropic_beta_headers_config.json | 2 +- .../anthropic_claude3_transformation.py | 101 ++++++-- ...odel_prices_and_context_window_backup.json | 22 +- model_prices_and_context_window.json | 22 +- ...ations_anthropic_claude3_transformation.py | 217 +++++++++++++++++- .../chat/test_converse_transformation.py | 14 +- 6 files changed, 336 insertions(+), 42 deletions(-) diff --git a/litellm/anthropic_beta_headers_config.json b/litellm/anthropic_beta_headers_config.json index d02afe37569..115dd1a2d85 100644 --- a/litellm/anthropic_beta_headers_config.json +++ b/litellm/anthropic_beta_headers_config.json @@ -115,7 +115,7 @@ "structured-output-2024-03-01": null, "prompt-caching-scope-2026-01-05": null, "skills-2025-10-02": null, - "structured-outputs-2025-11-13": null, + "structured-outputs-2025-11-13": "structured-outputs-2025-11-13", "text_editor_20241022": null, "text_editor_20250124": null, "token-efficient-tools-2025-02-19": null, diff --git a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py index a13336b6c88..f84f0683c53 100644 --- a/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py +++ b/litellm/llms/bedrock/chat/invoke_transformations/anthropic_claude3_transformation.py @@ -11,7 +11,10 @@ from litellm.litellm_core_utils.prompt_templates.image_handling import ( async_convert_url_to_base64, convert_url_to_base64, ) -from litellm.llms.anthropic.chat.transformation import AnthropicConfig +from litellm.llms.anthropic.chat.transformation import ( + AnthropicConfig, + RESPONSE_FORMAT_TOOL_NAME, +) from litellm.llms.bedrock.chat.invoke_transformations.base_invoke_transformation import ( AmazonInvokeConfig, ) @@ -63,6 +66,24 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig): def get_supported_openai_params(self, model: str) -> List[str]: return AnthropicConfig.get_supported_openai_params(self, model) + @staticmethod + def _bedrock_invoke_supports_native_output_format(model: str) -> bool: + """Whether AWS Bedrock invoke accepts native ``output_format`` for + ``model`` today. + + Reads ``supports_native_structured_output`` from + ``model_prices_and_context_window.json`` via the same + ``_supports_factory`` path the Converse transformer uses. Set the + flag to ``false`` on a Bedrock model entry to force the + synthetic-tool-injection workaround; ``true`` (the default for + Claude 4.5+) lets the native field flow to the wire body. + """ + return _supports_factory( + model=model, + custom_llm_provider="bedrock", + key="supports_native_structured_output", + ) + def map_openai_params( self, non_default_params: dict, @@ -70,14 +91,6 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig): model: str, drop_params: bool, ) -> dict: - # Force tool-based structured outputs for Bedrock Invoke - # (similar to VertexAI fix in #19201) - # Bedrock Invoke doesn't support output_format parameter - original_model = model - if "response_format" in non_default_params: - # Use a model name that forces tool-based approach - model = "claude-3-sonnet-20240229" - # Clamp ``reasoning_effort`` to the Bedrock effort ceiling before the # parent mapping converts it to ``output_config.effort`` and the # downstream effort gate runs. Mirrors the converse path's @@ -86,9 +99,37 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig): # requests degrade ``xhigh`` -> ``max`` rather than 400-ing on # models like Opus 4.6 that don't natively advertise xhigh. self._clamp_adaptive_reasoning_effort_for_bedrock( - model=original_model, params=non_default_params + model=model, params=non_default_params ) + # For models that don't yet support native ``output_format`` on + # Bedrock invoke (per the ``supports_native_structured_output`` flag + # on the model's entry in ``model_prices_and_context_window.json``), + # pop ``response_format`` BEFORE delegating to the parent so the + # parent doesn't emit ``output_format``. We inject the synthetic + # ``json_tool_call`` tool ourselves after the parent returns, using + # the same public helpers the parent's tool-injection branch uses. + # + # Letting the parent see the real model name preserves correct + # ``_is_adaptive_thinking_model`` / effort-mapping behavior. The + # previous workaround spoofed ``model = "claude-3-sonnet-20240229"`` + # to force the tool-injection branch; that side-effect poisoned the + # adaptive-thinking lookup when ``response_format`` was combined + # with ``reasoning_effort``, producing legacy + # ``thinking.enabled+budget`` bodies that AWS opus-4-7/4-8 reject + # with: + # "thinking.type.enabled is not supported for this model. + # Use thinking.type.adaptive and output_config.effort ..." + rf = None + if ( + "response_format" in non_default_params + and not self._bedrock_invoke_supports_native_output_format(model) + ): + rf = non_default_params["response_format"] + non_default_params = { + k: v for k, v in non_default_params.items() if k != "response_format" + } + optional_params = AnthropicConfig.map_openai_params( self, non_default_params, @@ -97,8 +138,24 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig): drop_params, ) - # Restore original model name - model = original_model + if rf is not None: + is_thinking = self.is_thinking_enabled(non_default_params) + tool = self.map_response_format_to_anthropic_tool( + rf, optional_params, is_thinking + ) + if tool is not None: + optional_params = self._add_tools_to_optional_params( + optional_params=optional_params, tools=[tool] + ) + if not is_thinking: + # AWS rejects ``tool_choice`` forcing alongside thinking, + # so only set it when thinking is off. Mirrors + # ``AnthropicConfig.map_openai_params`` line ~1493. + optional_params["tool_choice"] = { + "name": RESPONSE_FORMAT_TOOL_NAME, + "type": "tool", + } + optional_params["json_mode"] = True return optional_params @@ -212,7 +269,17 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig): anthropic_request.pop("model", None) anthropic_request.pop("stream", None) - output_format = anthropic_request.pop("output_format", None) + # Only strip ``output_format`` for models that don't natively support + # it on Bedrock invoke (per ``supports_native_structured_output`` in + # ``model_prices_and_context_window.json``). For everything else + # (opus-4-6, sonnet-4-5/4-6, haiku-4-5, ...) AWS accepts the field + # natively — leave it in the wire body and let the + # structured-outputs-2025-11-13 beta header gate it (added in + # ``_compute_bedrock_invoke_beta_headers``). + if not self._bedrock_invoke_supports_native_output_format(model): + output_format = anthropic_request.pop("output_format", None) + else: + output_format = None output_config_format = pop_bedrock_invoke_output_config_format( anthropic_request ) @@ -276,6 +343,14 @@ class AmazonAnthropicClaudeConfig(AmazonInvokeConfig, AnthropicConfig): ) beta_set.update(auto_betas) + # When the wire body carries native ``output_format``, AWS Bedrock + # invoke requires the gating beta header. The header itself must + # also be enabled for ``provider="bedrock"`` in + # ``litellm/anthropic_beta_headers_config.json`` to survive + # ``filter_and_transform_beta_headers``. + if optional_params.get("output_format") is not None: + beta_set.add("structured-outputs-2025-11-13") + if tool_search_used and not ( programmatic_tool_calling_used or input_examples_used ): diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 397f96fdb1e..8b874ab285f 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -1159,7 +1159,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1205,7 +1205,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1236,7 +1236,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1267,7 +1267,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1298,7 +1298,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1330,7 +1330,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1362,7 +1362,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1394,7 +1394,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1426,7 +1426,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1458,7 +1458,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1489,7 +1489,7 @@ "supports_vision": true, "supports_xhigh_reasoning_effort": true, "tool_use_system_prompt_tokens": 346, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": true }, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index b2836a096b7..373c0bbfae4 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -1159,7 +1159,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1205,7 +1205,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1236,7 +1236,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1267,7 +1267,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1298,7 +1298,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1330,7 +1330,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1362,7 +1362,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1394,7 +1394,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1426,7 +1426,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1458,7 +1458,7 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_output_config": true, "bedrock_output_config_effort_ceiling": "xhigh" @@ -1489,7 +1489,7 @@ "supports_vision": true, "supports_xhigh_reasoning_effort": true, "tool_use_system_prompt_tokens": 346, - "supports_native_structured_output": true, + "supports_native_structured_output": false, "supports_max_reasoning_effort": true, "supports_minimal_reasoning_effort": true }, diff --git a/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py b/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py index 4c4c0e17a38..b49536eafee 100644 --- a/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/invoke_transformations/test_bedrock_chat_invoke_transformations_anthropic_claude3_transformation.py @@ -1,4 +1,3 @@ -import asyncio import json import os import sys @@ -502,11 +501,19 @@ def test_bedrock_chat_invoke_checks_output_config_support_with_bedrock_provider( headers={}, ) - mock_supports_factory.assert_called_once_with( + # Two distinct flag lookups now happen (both scoped to bedrock provider): + # - supports_native_structured_output (decides whether to strip output_format) + # - supports_output_config (decides whether to strip output_config) + mock_supports_factory.assert_any_call( model="us.anthropic.claude-opus-4-7", custom_llm_provider="bedrock", key="supports_output_config", ) + mock_supports_factory.assert_any_call( + model="us.anthropic.claude-opus-4-7", + custom_llm_provider="bedrock", + key="supports_native_structured_output", + ) assert result["output_config"] == {"effort": "high"} @@ -545,3 +552,209 @@ def test_output_format_removed_from_bedrock_invoke_request(): assert ( "output_format" not in result ), f"output_format should be removed for Bedrock Invoke, got keys: {result.keys()}" + + +# ----------------------------------------------------------------------- +# Native ``output_format`` vs tool-injection — Bedrock invoke per-model +# ----------------------------------------------------------------------- +# AWS Bedrock invoke accepts native ``output_format`` for Claude 4.5+ except +# opus-4-7 and opus-4-8 (verified Jun 2026). For those two, LiteLLM falls +# back to the synthetic ``json_tool_call`` tool. For every other Claude 4.5+ +# model the native field flows to AWS, and the gating beta header +# ``structured-outputs-2025-11-13`` is auto-added. + +_REVENUE_SCHEMA = { + "type": "object", + "additionalProperties": False, + "properties": { + "revenue_b": {"type": "number"}, + "yoy_growth_pct": {"type": "number"}, + }, + "required": ["revenue_b", "yoy_growth_pct"], +} + +_OPENAI_RESPONSE_FORMAT_JSON_SCHEMA = { + "type": "json_schema", + "json_schema": {"name": "RevenueData", "schema": _REVENUE_SCHEMA}, +} + + +def _build_invoke_request( + model: str, + non_default_params: dict, + *, + bedrock_supports_native_output_format: bool, +) -> dict: + """Run map_openai_params + transform_request with the in-module + ``_supports_factory`` patched so the native-``output_format`` capability + lookup is deterministic regardless of remote model_cost state. + + ``bedrock_supports_native_output_format`` is what + ``_bedrock_invoke_supports_native_output_format(model)`` should return — + i.e. the value of ``supports_native_structured_output`` for ``model``'s + Bedrock entry in ``model_prices_and_context_window.json``. + """ + + def fake_supports_factory(model, custom_llm_provider, key): + if key == "supports_native_structured_output": + return bedrock_supports_native_output_format + # supports_output_config is True for opus-4-X in the cost map; preserve + # that so the downstream output_config strip behaves as in production. + if key == "supports_output_config": + return True + return False + + config = AmazonAnthropicClaudeConfig() + messages = [{"role": "user", "content": "test"}] + + with patch( + "litellm.llms.bedrock.chat.invoke_transformations.anthropic_claude3_transformation._supports_factory", + side_effect=fake_supports_factory, + ): + optional_params = config.map_openai_params( + non_default_params=non_default_params, + optional_params={}, + model=model, + drop_params=False, + ) + return config.transform_request( + model=model, + messages=messages, + optional_params=optional_params, + litellm_params={}, + headers={}, + ) + + +def test_response_format_emits_native_output_format_for_opus_4_6_invoke(): + """Models AWS accepts native ``output_format`` for must receive it on the wire + (no synthetic tool-injection) plus the gating beta header.""" + result = _build_invoke_request( + model="anthropic.claude-opus-4-6-v1", + non_default_params={"response_format": _OPENAI_RESPONSE_FORMAT_JSON_SCHEMA}, + bedrock_supports_native_output_format=True, + ) + + assert "output_format" in result, ( + f"opus-4-6 invoke should keep native output_format on the wire; " + f"got keys: {sorted(result.keys())}" + ) + assert result["output_format"].get("type") == "json_schema" + + assert "tools" not in result, ( + f"opus-4-6 invoke with native output_format must NOT inject json_tool_call; " + f"got tools={result.get('tools')!r}" + ) + assert "tool_choice" not in result + + beta = result.get("anthropic_beta", []) + assert "structured-outputs-2025-11-13" in beta, ( + f"structured-outputs-2025-11-13 beta header must be present when " + f"output_format is sent; got anthropic_beta={beta!r}" + ) + + +def test_response_format_uses_tool_workaround_for_opus_4_7_invoke(): + """opus-4-7's Bedrock entry has ``supports_native_structured_output: false`` + — AWS rejects native ``output_format`` here, so LiteLLM must fall back to + the synthetic ``json_tool_call`` tool.""" + result = _build_invoke_request( + model="anthropic.claude-opus-4-7", + non_default_params={"response_format": _OPENAI_RESPONSE_FORMAT_JSON_SCHEMA}, + bedrock_supports_native_output_format=False, + ) + + assert "output_format" not in result, ( + f"opus-4-7 invoke must NOT send native output_format (AWS rejects); " + f"got output_format={result.get('output_format')!r}" + ) + + tools = result.get("tools") or [] + assert any( + t.get("name") == "json_tool_call" for t in tools + ), f"opus-4-7 invoke must inject json_tool_call tool; got tools={tools!r}" + assert ( + result.get("tool_choice", {}).get("name") == "json_tool_call" + ), "opus-4-7 invoke must force tool_choice to json_tool_call (no thinking)" + + beta = result.get("anthropic_beta", []) + assert ( + "structured-outputs-2025-11-13" not in beta + ), "structured-outputs beta must NOT be set when output_format is absent" + + +def test_response_format_plus_reasoning_preserves_adaptive_thinking_on_opus_4_7(): + """Regression test for the model-spoof side-effect: when ``response_format`` is + combined with ``reasoning_effort``, the previous workaround spoofed the model + name and corrupted ``_is_adaptive_thinking_model`` to return False, producing + legacy ``thinking.enabled+budget_tokens`` that AWS opus-4-7 rejects. After the + fix, the real model name flows through and adaptive thinking is preserved.""" + result = _build_invoke_request( + model="anthropic.claude-opus-4-7", + non_default_params={ + "response_format": _OPENAI_RESPONSE_FORMAT_JSON_SCHEMA, + "reasoning_effort": "medium", + }, + bedrock_supports_native_output_format=False, + ) + + # Tool injection still happens (AWS rejects native output_format on 4-7) + tools = result.get("tools") or [] + assert any(t.get("name") == "json_tool_call" for t in tools), ( + f"opus-4-7 invoke must still inject json_tool_call when reasoning is on; " + f"got tools={tools!r}" + ) + + # No tool_choice — AWS rejects forcing alongside thinking + assert ( + "tool_choice" not in result + ), f"AWS rejects tool_choice forcing when thinking is on; got tool_choice={result.get('tool_choice')!r}" + + # Modern adaptive-thinking interface (NOT legacy enabled+budget) + thinking = result.get("thinking") or {} + assert ( + thinking.get("type") == "adaptive" + ), f"opus-4-7 requires thinking.type=adaptive; got thinking={thinking!r}" + assert ( + "budget_tokens" not in thinking + ), f"opus-4-7 rejects budget_tokens; got thinking={thinking!r}" + + output_config = result.get("output_config") or {} + assert output_config.get("effort") == "medium", ( + f"opus-4-7 requires output_config.effort alongside adaptive thinking; " + f"got output_config={output_config!r}" + ) + + +def test_response_format_plus_reasoning_uses_native_with_adaptive_on_opus_4_6(): + """Combined case for a model AWS accepts native output_format on — both + ``output_format`` (structured output) and ``thinking.adaptive + + output_config.effort`` (reasoning) coexist in the wire body, no tools.""" + result = _build_invoke_request( + model="anthropic.claude-opus-4-6-v1", + non_default_params={ + "response_format": _OPENAI_RESPONSE_FORMAT_JSON_SCHEMA, + "reasoning_effort": "medium", + }, + bedrock_supports_native_output_format=True, + ) + + assert "output_format" in result + assert result["output_format"].get("type") == "json_schema" + + thinking = result.get("thinking") or {} + assert ( + thinking.get("type") == "adaptive" + ), f"opus-4-6 should use adaptive thinking; got thinking={thinking!r}" + assert "budget_tokens" not in thinking + + output_config = result.get("output_config") or {} + assert output_config.get("effort") == "medium" + + assert "tools" not in result, ( + f"opus-4-6 with native output_format must not inject tools; " + f"got tools={result.get('tools')!r}" + ) + + beta = result.get("anthropic_beta", []) + assert "structured-outputs-2025-11-13" in beta diff --git a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py index a6aa35ee6d1..9e07d8137e0 100644 --- a/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py +++ b/tests/test_litellm/llms/bedrock/chat/test_converse_transformation.py @@ -371,7 +371,13 @@ def test_output_config_effort_forwarded_into_additional_request_fields(model): def test_output_config_format_translated_to_native_output_config_converse(): - """``output_config.format`` becomes Bedrock ``outputConfig`` and is not forwarded raw.""" + """``output_config.format`` becomes Bedrock ``outputConfig`` and is not forwarded raw. + + Uses ``us.anthropic.claude-opus-4-6-v1`` because it has + ``supports_native_structured_output: true`` in the model registry. opus-4-7 + no longer does — AWS doesn't accept native structured output on Bedrock + invoke or converse for that model. + """ config = AmazonConverseConfig() schema = { "type": "object", @@ -379,13 +385,13 @@ def test_output_config_format_translated_to_native_output_config_converse(): } result = config._transform_request( - model="bedrock/converse/us.anthropic.claude-opus-4-7", + model="bedrock/converse/us.anthropic.claude-opus-4-6-v1", messages=[{"role": "user", "content": "hi"}], optional_params={ "maxTokens": 256, "thinking": {"type": "adaptive"}, "output_config": { - "effort": "xhigh", + "effort": "high", "format": {"type": "json_schema", "schema": schema}, }, }, @@ -394,7 +400,7 @@ def test_output_config_format_translated_to_native_output_config_converse(): ) additional = result.get("additionalModelRequestFields", {}) - assert additional.get("output_config") == {"effort": "xhigh"} + assert additional.get("output_config") == {"effort": "high"} assert "format" not in additional["output_config"] assert result["outputConfig"]["textFormat"]["type"] == "json_schema" parsed_schema = json.loads(