From b8a8083308c8768607e624a874fa9bc91e0377d3 Mon Sep 17 00:00:00 2001 From: Kent <72616338+kingdoooo@users.noreply.github.com> Date: Wed, 17 Jun 2026 20:18:11 +0800 Subject: [PATCH] fix(bedrock): handle role:"system" inside the messages array on /v1/messages (#29698) (#30443) * feat(anthropic): hoist leading in-array system to top-level (helper) * test(anthropic): cover _system_content_to_blocks edge cases; deepcopy cache_control * test(anthropic): mid-conversation system normalization cases * feat: add supports_mid_conversation_system flag to Claude Opus 4.8 Add supports_mid_conversation_system: true to all 9 claude-opus-4-8 cost-map entries (Anthropic-native, Bedrock, Vertex, Azure AI) in both the root cost map and the bundled package backup, since the runtime helper and tests read the backup in local/offline mode. Pin the mid-system passthrough regression test to the local cost map via the existing local_model_cost_map fixture so it reads the branch-local flag rather than the network-fetched main copy. * fix(bedrock): normalize in-array system in /v1/messages handler (#29698) Wire normalize_system_messages_for_anthropic into anthropic_messages_handler so all Bedrock /v1/messages paths (Invoke / Mantle / ClaudePlatform / Converse-bridge) hoist leading in-array system entries (and demote mid-conversation ones on models lacking supports_mid_conversation_system) into the top-level system field. The normalized messages/system are written back into the local_vars snapshot the base_llm branch reads from, otherwise the Invoke/Mantle fix would silently no-op. Also fix the helper to resolve supports_mid_conversation_system through the prefix-aware AnthropicModelInfo._supports_model_capability resolver. The raw _supports_factory could not see the flag once get_llm_provider left the invoke/ prefix on the model id, which would have wrongly demoted mid-conversation system on a Bedrock invoke opus-4-8 path. * fix(bedrock): resolve mid-conversation-system flag through mantle/invoke/converse route prefixes; drop unused param * fix(types): widen system param to Union[str, List] for hoisted system blocks * refactor(bedrock): drop dead local_vars messages writeback * fix(bedrock/converse): translate in-array system in anthropic->openai adapter (#29698) * fix(bedrock/converse): preserve cache_control on in-array system; test drop-empty * fix(bedrock/converse): rename colliding local to satisfy mypy; test handler system-merge branches * fix(types): register supports_mid_conversation_system in model-info schema The cost-map JSON-schema validation test (test_aaamodel_prices_and_context_window_json_is_valid) rejects unknown properties, so adding supports_mid_conversation_system to the opus-4-8 cost-map entries failed CI with 'Additional properties are not allowed'. Register the flag in the INTENDED_SCHEMA allow-list and in the ProviderSpecificModelInfo TypedDict so it is a typed, first-class capability flag alongside its peers (supports_output_config, etc.). --------- Co-authored-by: Sameer Kankute --- litellm/llms/anthropic/common_utils.py | 71 ++++++- .../adapters/transformation.py | 30 +++ .../messages/handler.py | 21 +- ...odel_prices_and_context_window_backup.json | 27 ++- litellm/types/utils.py | 1 + model_prices_and_context_window.json | 27 ++- ...al_pass_through_adapters_transformation.py | 84 ++++++++ ...erimental_pass_through_messages_handler.py | 142 ++++++++++++++ .../anthropic/test_anthropic_common_utils.py | 179 ++++++++++++++++++ tests/test_litellm/test_utils.py | 1 + 10 files changed, 562 insertions(+), 21 deletions(-) diff --git a/litellm/llms/anthropic/common_utils.py b/litellm/llms/anthropic/common_utils.py index 329789616b3..9e2e6e11337 100644 --- a/litellm/llms/anthropic/common_utils.py +++ b/litellm/llms/anthropic/common_utils.py @@ -4,7 +4,7 @@ This file contains common utils for anthropic calls. import copy from datetime import datetime, timezone -from typing import Any, Dict, List, Optional, Union +from typing import Any, Dict, List, Optional, Tuple, Union import httpx @@ -340,7 +340,11 @@ class AnthropicModelInfo(BaseLLMModelInfo): for prefix in ( "bedrock/converse/", "bedrock/invoke/", + "bedrock/mantle/", "bedrock/", + "converse/", + "invoke/", + "mantle/", "vertex_ai/", ): if model.startswith(prefix): @@ -991,6 +995,71 @@ def _is_empty_text_block(block: Any) -> bool: return not isinstance(text, str) or not text.strip() +def _system_content_to_blocks(content: Any) -> List[Dict[str, Any]]: + """Normalize a system message's content into a list of Anthropic text blocks. + + Accepts a plain string or a list of content blocks. Empty/whitespace-only + text is dropped. cache_control on a block is preserved. + """ + blocks: List[Dict[str, Any]] = [] + if isinstance(content, str): + if content.strip(): + blocks.append({"type": "text", "text": content}) + return blocks + if isinstance(content, list): + for b in content: + if not isinstance(b, dict): + continue + if b.get("type") == "text" and (b.get("text") or "").strip(): + new_block: Dict[str, Any] = {"type": "text", "text": b["text"]} + if "cache_control" in b: + new_block["cache_control"] = copy.deepcopy(b["cache_control"]) + blocks.append(new_block) + return blocks + + +def normalize_system_messages_for_anthropic( + messages: List[Any], + model: str, +) -> Tuple[List[Any], List[Dict[str, Any]]]: + """Normalize ``role: "system"`` entries inside the Anthropic messages array. + + Anthropic (Claude Opus 4.8+) accepts ``{"role": "system"}`` entries inside + ``messages`` (mid-conversation system messages). Downstream Bedrock paths do + not all support that shape, so this normalizes it: + + - A *leading* system entry (before the first user/assistant message) is + semantically the initial system prompt; it is always hoisted into the + top-level ``system`` field. + - A *mid-conversation* system entry is left in place only if the model + advertises ``supports_mid_conversation_system``; otherwise it is hoisted + out (demoted) so it still reaches the model via the top-level system field. + + Illegally placed mid system is left untouched on the pass-through path so + Bedrock returns its precise 400 — we do not reorder messages. + + Returns ``(new_messages, hoisted_system_blocks)``. The caller merges the + hoisted blocks into the top-level ``system``. Input is never mutated. + """ + supports_mid = AnthropicModelInfo._supports_model_capability( + model, "supports_mid_conversation_system" + ) + out: List[Any] = [] + hoisted: List[Dict[str, Any]] = [] + seen_conversation = False + for m in messages: + if isinstance(m, dict) and m.get("role") == "system": + is_leading = not seen_conversation + if is_leading or not supports_mid: + hoisted.extend(_system_content_to_blocks(m.get("content"))) + continue + out.append(m) + continue + seen_conversation = True + out.append(m) + return out, hoisted + + def process_anthropic_headers(headers: Union[httpx.Headers, dict]) -> dict: openai_headers = {} if "anthropic-ratelimit-requests-limit" in headers: diff --git a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py index bf425637b56..5c6b982d180 100644 --- a/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py @@ -400,6 +400,36 @@ class LiteLLMAnthropicMessagesAdapter: new_user_content_list: List[ Union[ChatCompletionTextObject, ChatCompletionImageObject] ] = [] + if m["role"] == "system": + content = m.get("content", "") + if isinstance(content, str): + if content.strip(): + new_messages.append( + ChatCompletionSystemMessage(role="system", content=content) + ) + elif isinstance(content, list): + text_blocks: List[Dict[str, Any]] = [] + for b in content: + if ( + isinstance(b, dict) + and b.get("type") == "text" + and (b.get("text") or "").strip() + ): + sys_text_block: Dict[str, Any] = { + "type": "text", + "text": b["text"], + } + self._add_cache_control_if_applicable( + b, sys_text_block, model or "" + ) + text_blocks.append(sys_text_block) + if text_blocks: + new_messages.append( + ChatCompletionSystemMessage( + role="system", content=text_blocks + ) + ) + continue ## USER MESSAGE ## if m["role"] == "user": ## translate user message diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index a3ac465c463..d9f1373a156 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -23,6 +23,7 @@ from typing import ( import litellm from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.llms.anthropic.common_utils import ( + normalize_system_messages_for_anthropic, strip_empty_text_blocks_from_anthropic_messages, ) from litellm.llms.base_llm.anthropic_messages.transformation import ( @@ -187,7 +188,7 @@ async def anthropic_messages( metadata: Optional[Dict] = None, stop_sequences: Optional[List[str]] = None, stream: Optional[bool] = False, - system: Optional[str] = None, + system: Optional[Union[str, List]] = None, temperature: Optional[float] = None, thinking: Optional[Dict] = None, tool_choice: Optional[Dict] = None, @@ -360,7 +361,7 @@ def anthropic_messages_handler( metadata: Optional[Dict] = None, stop_sequences: Optional[List[str]] = None, stream: Optional[bool] = False, - system: Optional[str] = None, + system: Optional[Union[str, List]] = None, temperature: Optional[float] = None, thinking: Optional[Dict] = None, tool_choice: Optional[Dict] = None, @@ -427,6 +428,22 @@ def anthropic_messages_handler( api_key=litellm_params.api_key, ) + messages, _hoisted_system_blocks = normalize_system_messages_for_anthropic( + messages, model=model + ) + if _hoisted_system_blocks: + if system is None: + system = _hoisted_system_blocks + elif isinstance(system, str): + system = ( + [{"type": "text", "text": system}] + _hoisted_system_blocks + if system.strip() + else _hoisted_system_blocks + ) + elif isinstance(system, list): + system = list(system) + _hoisted_system_blocks + local_vars["system"] = system + # Store agentic loop params in logging object for agentic hooks # This provides original request context needed for follow-up calls if litellm_logging_obj is not None: diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 39d612f252d..45b29d659f8 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -1471,7 +1471,8 @@ "supports_native_structured_output": true, "supports_max_reasoning_effort": true, "supports_output_config": true, - "bedrock_output_config_effort_ceiling": "xhigh" + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_mid_conversation_system": true }, "global.anthropic.claude-opus-4-8": { "cache_creation_input_token_cost": 6.25e-06, @@ -1504,7 +1505,8 @@ "supports_native_structured_output": true, "supports_max_reasoning_effort": true, "supports_output_config": true, - "bedrock_output_config_effort_ceiling": "xhigh" + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_mid_conversation_system": true }, "us.anthropic.claude-opus-4-8": { "cache_creation_input_token_cost": 6.875e-06, @@ -1537,7 +1539,8 @@ "supports_native_structured_output": true, "supports_max_reasoning_effort": true, "supports_output_config": true, - "bedrock_output_config_effort_ceiling": "xhigh" + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_mid_conversation_system": true }, "eu.anthropic.claude-opus-4-8": { "cache_creation_input_token_cost": 6.875e-06, @@ -1570,7 +1573,8 @@ "supports_native_structured_output": true, "supports_max_reasoning_effort": true, "supports_output_config": true, - "bedrock_output_config_effort_ceiling": "xhigh" + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_mid_conversation_system": true }, "au.anthropic.claude-opus-4-8": { "cache_creation_input_token_cost": 6.875e-06, @@ -1603,7 +1607,8 @@ "supports_native_structured_output": true, "supports_max_reasoning_effort": true, "supports_output_config": true, - "bedrock_output_config_effort_ceiling": "xhigh" + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_mid_conversation_system": true }, "jp.anthropic.claude-opus-4-7": { "cache_creation_input_token_cost": 6.875e-06, @@ -2415,7 +2420,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_max_reasoning_effort": true + "supports_max_reasoning_effort": true, + "supports_mid_conversation_system": true }, "azure_ai/claude-opus-4-1": { "cache_creation_input_token_cost": 1.875e-05, @@ -10615,7 +10621,8 @@ "us": 1.1, "fast": 2.0 }, - "supports_output_config": true + "supports_output_config": true, + "supports_mid_conversation_system": true }, "claude-sonnet-4-20250514": { "deprecation_date": "2026-05-14", @@ -34872,7 +34879,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_max_reasoning_effort": true + "supports_max_reasoning_effort": true, + "supports_mid_conversation_system": true }, "vertex_ai/claude-opus-4-8@default": { "cache_creation_input_token_cost": 6.25e-06, @@ -34902,7 +34910,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_max_reasoning_effort": true + "supports_max_reasoning_effort": true, + "supports_mid_conversation_system": true }, "vertex_ai/claude-sonnet-4-5": { "cache_creation_input_token_cost": 3.75e-06, diff --git a/litellm/types/utils.py b/litellm/types/utils.py index de0c621c97b..d50bd212084 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -146,6 +146,7 @@ class ProviderSpecificModelInfo(TypedDict, total=False): supports_low_reasoning_effort: Optional[bool] supports_xhigh_reasoning_effort: Optional[bool] supports_max_reasoning_effort: Optional[bool] + supports_mid_conversation_system: Optional[bool] supports_output_config: Optional[bool] supports_image_size: Optional[bool] bedrock_output_config_effort_ceiling: Optional[ diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index ba8b09498e8..cb9e2224f94 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -1471,7 +1471,8 @@ "supports_native_structured_output": true, "supports_max_reasoning_effort": true, "supports_output_config": true, - "bedrock_output_config_effort_ceiling": "xhigh" + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_mid_conversation_system": true }, "global.anthropic.claude-opus-4-8": { "cache_creation_input_token_cost": 6.25e-06, @@ -1504,7 +1505,8 @@ "supports_native_structured_output": true, "supports_max_reasoning_effort": true, "supports_output_config": true, - "bedrock_output_config_effort_ceiling": "xhigh" + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_mid_conversation_system": true }, "us.anthropic.claude-opus-4-8": { "cache_creation_input_token_cost": 6.875e-06, @@ -1537,7 +1539,8 @@ "supports_native_structured_output": true, "supports_max_reasoning_effort": true, "supports_output_config": true, - "bedrock_output_config_effort_ceiling": "xhigh" + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_mid_conversation_system": true }, "eu.anthropic.claude-opus-4-8": { "cache_creation_input_token_cost": 6.875e-06, @@ -1570,7 +1573,8 @@ "supports_native_structured_output": true, "supports_max_reasoning_effort": true, "supports_output_config": true, - "bedrock_output_config_effort_ceiling": "xhigh" + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_mid_conversation_system": true }, "au.anthropic.claude-opus-4-8": { "cache_creation_input_token_cost": 6.875e-06, @@ -1603,7 +1607,8 @@ "supports_native_structured_output": true, "supports_max_reasoning_effort": true, "supports_output_config": true, - "bedrock_output_config_effort_ceiling": "xhigh" + "bedrock_output_config_effort_ceiling": "xhigh", + "supports_mid_conversation_system": true }, "jp.anthropic.claude-opus-4-7": { "cache_creation_input_token_cost": 6.875e-06, @@ -2415,7 +2420,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_max_reasoning_effort": true + "supports_max_reasoning_effort": true, + "supports_mid_conversation_system": true }, "azure_ai/claude-opus-4-1": { "cache_creation_input_token_cost": 1.875e-05, @@ -10615,7 +10621,8 @@ "us": 1.1, "fast": 2.0 }, - "supports_output_config": true + "supports_output_config": true, + "supports_mid_conversation_system": true }, "claude-sonnet-4-20250514": { "deprecation_date": "2026-05-14", @@ -34912,7 +34919,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_max_reasoning_effort": true + "supports_max_reasoning_effort": true, + "supports_mid_conversation_system": true }, "vertex_ai/claude-opus-4-8@default": { "cache_creation_input_token_cost": 6.25e-06, @@ -34942,7 +34950,8 @@ "supports_tool_choice": true, "supports_vision": true, "supports_xhigh_reasoning_effort": true, - "supports_max_reasoning_effort": true + "supports_max_reasoning_effort": true, + "supports_mid_conversation_system": true }, "vertex_ai/claude-sonnet-4-5": { "cache_creation_input_token_cost": 3.75e-06, diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py index 76aa3a9c6aa..6e1d8437a89 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/adapters/test_anthropic_experimental_pass_through_adapters_transformation.py @@ -2747,3 +2747,87 @@ def test_translate_openai_response_to_anthropic_with_polyfill_both_compaction_an cm = result.get("context_management") assert cm is not None assert cm["applied_edits"][0]["type"] == "compact_20260112" + + +def test_adapter_translates_in_array_system_to_openai_system_message(): + from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( + LiteLLMAnthropicMessagesAdapter, + ) + + adapter = LiteLLMAnthropicMessagesAdapter() + messages = [ + {"role": "user", "content": "Hi"}, + {"role": "system", "content": "Now be a pirate."}, + ] + out = adapter.translate_anthropic_messages_to_openai(messages=messages) + system_msgs = [m for m in out if m.get("role") == "system"] + assert len(system_msgs) == 1 + assert system_msgs[0]["content"] == "Now be a pirate." + + +def test_adapter_translates_in_array_system_list_content(): + from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( + LiteLLMAnthropicMessagesAdapter, + ) + + adapter = LiteLLMAnthropicMessagesAdapter() + messages = [ + {"role": "user", "content": "Hi"}, + { + "role": "system", + "content": [{"type": "text", "text": "Be a pirate."}], + }, + ] + out = adapter.translate_anthropic_messages_to_openai(messages=messages) + system_msgs = [m for m in out if m.get("role") == "system"] + assert len(system_msgs) == 1 + assert system_msgs[0]["content"] == [{"type": "text", "text": "Be a pirate."}] + + +def test_adapter_in_array_system_preserves_cache_control(): + from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( + LiteLLMAnthropicMessagesAdapter, + ) + + adapter = LiteLLMAnthropicMessagesAdapter() + messages = [ + {"role": "user", "content": "Hi"}, + { + "role": "system", + "content": [ + { + "type": "text", + "text": "Be a pirate.", + "cache_control": {"type": "ephemeral"}, + } + ], + }, + ] + out = adapter.translate_anthropic_messages_to_openai( + messages=messages, model="claude-opus-4-8" + ) + system_msgs = [m for m in out if m.get("role") == "system"] + assert len(system_msgs) == 1 + block = system_msgs[0]["content"][0] + assert block["text"] == "Be a pirate." + assert block.get("cache_control") == {"type": "ephemeral"} + + +def test_adapter_in_array_system_drops_empty_content(): + from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( + LiteLLMAnthropicMessagesAdapter, + ) + + adapter = LiteLLMAnthropicMessagesAdapter() + messages = [ + {"role": "user", "content": "Hi"}, + {"role": "system", "content": " "}, # whitespace-only -> dropped + {"role": "system", "content": []}, # empty list -> dropped + { + "role": "system", + "content": [{"type": "text", "text": " "}], + }, # blank text -> dropped + ] + out = adapter.translate_anthropic_messages_to_openai(messages=messages) + system_msgs = [m for m in out if m.get("role") == "system"] + assert system_msgs == [] diff --git a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py index b1e1d789d74..c696554eab5 100644 --- a/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py +++ b/tests/test_litellm/llms/anthropic/experimental_pass_through/messages/test_anthropic_experimental_pass_through_messages_handler.py @@ -673,3 +673,145 @@ async def test_async_wrapper_sets_presanitized_and_sanitizes_once(): assert spy.call_count == 1 assert captured["presanitized"] is True assert [b["type"] for b in captured["messages"][0]["content"]] == ["tool_use"] + + +@pytest.fixture +def local_model_cost_map(monkeypatch): + """Force the bundled backup cost map so Opus 4.8's + ``supports_mid_conversation_system`` flag is read from local data rather + than the network-fetched ``main`` copy, which lacks the flag until this + branch merges.""" + import litellm + + original = litellm.model_cost + monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True") + litellm.model_cost = litellm.get_model_cost_map(url="") + litellm.get_model_info.cache_clear() + try: + yield + finally: + litellm.model_cost = original + litellm.get_model_info.cache_clear() + + +def _capture_base_llm_call(model, messages, system=None): + """Invoke the handler with base_llm mocked; return (optional_params, messages) it received.""" + from litellm.llms.anthropic.experimental_pass_through.messages.handler import ( + anthropic_messages_handler, + ) + + with patch( + "litellm.llms.anthropic.experimental_pass_through.messages.handler." + "base_llm_http_handler" + ) as mock_handler: + mock_handler.anthropic_messages_handler.return_value = MagicMock() + kwargs = dict( + max_tokens=16, + messages=messages, + model=model, + custom_llm_provider="bedrock", + ) + if system is not None: + kwargs["system"] = system + anthropic_messages_handler(**kwargs) + assert mock_handler.anthropic_messages_handler.called + call = mock_handler.anthropic_messages_handler.call_args.kwargs + return ( + call.get("anthropic_messages_optional_request_params", {}), + call.get("messages"), + ) + + +def test_handler_hoists_leading_system_into_top_level_for_invoke(): + opt, msgs = _capture_base_llm_call( + "bedrock/invoke/us.anthropic.claude-opus-4-8", + [ + {"role": "system", "content": "Be a pirate."}, + {"role": "user", "content": "Hi"}, + ], + ) + assert opt.get("system") == [{"type": "text", "text": "Be a pirate."}] + assert all(m.get("role") != "system" for m in msgs) + + +def test_handler_demotes_mid_system_on_unsupported_model_for_invoke(): + opt, msgs = _capture_base_llm_call( + "bedrock/invoke/us.anthropic.claude-opus-4-7", + [ + {"role": "user", "content": "Hi"}, + {"role": "system", "content": "Be a pirate."}, + ], + ) + assert opt.get("system") == [{"type": "text", "text": "Be a pirate."}] + assert all(m.get("role") != "system" for m in msgs) + + +def test_handler_passes_through_mid_system_on_supported_model_for_invoke( + local_model_cost_map, +): + opt, msgs = _capture_base_llm_call( + "bedrock/invoke/us.anthropic.claude-opus-4-8", + [ + {"role": "user", "content": "Hi"}, + {"role": "system", "content": "Be a pirate."}, + ], + ) + assert any(m.get("role") == "system" for m in msgs) + assert not opt.get("system") + + +def test_handler_passes_through_mid_system_on_supported_model_for_mantle( + local_model_cost_map, +): + opt, msgs = _capture_base_llm_call( + "bedrock/mantle/anthropic.claude-opus-4-8", + [ + {"role": "user", "content": "Hi"}, + {"role": "system", "content": "Be a pirate."}, + ], + ) + assert any(m.get("role") == "system" for m in msgs) + assert not opt.get("system") + + +def test_handler_merges_leading_system_after_existing_string_system(): + opt, msgs = _capture_base_llm_call( + "bedrock/invoke/us.anthropic.claude-opus-4-8", + [ + {"role": "system", "content": "In-array sys."}, + {"role": "user", "content": "Hi"}, + ], + system="Top-level sys.", + ) + assert opt.get("system") == [ + {"type": "text", "text": "Top-level sys."}, + {"type": "text", "text": "In-array sys."}, + ] + assert all(m.get("role") != "system" for m in msgs) + + +def test_handler_merges_leading_system_after_empty_string_system(): + opt, msgs = _capture_base_llm_call( + "bedrock/invoke/us.anthropic.claude-opus-4-8", + [ + {"role": "system", "content": "In-array sys."}, + {"role": "user", "content": "Hi"}, + ], + system=" ", + ) + assert opt.get("system") == [{"type": "text", "text": "In-array sys."}] + + +def test_handler_merges_leading_system_after_existing_list_system(): + opt, msgs = _capture_base_llm_call( + "bedrock/invoke/us.anthropic.claude-opus-4-8", + [ + {"role": "system", "content": "In-array sys."}, + {"role": "user", "content": "Hi"}, + ], + system=[{"type": "text", "text": "Top-level sys."}], + ) + assert opt.get("system") == [ + {"type": "text", "text": "Top-level sys."}, + {"type": "text", "text": "In-array sys."}, + ] diff --git a/tests/test_litellm/llms/anthropic/test_anthropic_common_utils.py b/tests/test_litellm/llms/anthropic/test_anthropic_common_utils.py index c437c6a8938..590867b4154 100644 --- a/tests/test_litellm/llms/anthropic/test_anthropic_common_utils.py +++ b/tests/test_litellm/llms/anthropic/test_anthropic_common_utils.py @@ -1382,6 +1382,81 @@ class TestAnthropicThinkingSignatureSelfHeal: assert data["messages"] == [] +def test_normalize_leading_system_is_hoisted(): + from litellm.llms.anthropic.common_utils import ( + normalize_system_messages_for_anthropic, + ) + + messages = [ + {"role": "system", "content": "You are a pirate."}, + {"role": "user", "content": "How are you?"}, + ] + new_messages, hoisted = normalize_system_messages_for_anthropic( + messages, model="claude-opus-4-8" + ) + assert all(m["role"] != "system" for m in new_messages) + assert new_messages == [{"role": "user", "content": "How are you?"}] + assert hoisted == [{"type": "text", "text": "You are a pirate."}] + assert messages[0] == {"role": "system", "content": "You are a pirate."} + + +def test_normalize_multiple_leading_system_preserves_order(): + from litellm.llms.anthropic.common_utils import ( + normalize_system_messages_for_anthropic, + ) + + messages = [ + {"role": "system", "content": "First."}, + {"role": "system", "content": "Second."}, + {"role": "user", "content": "Hi"}, + ] + new_messages, hoisted = normalize_system_messages_for_anthropic( + messages, model="claude-opus-4-8" + ) + assert new_messages == [{"role": "user", "content": "Hi"}] + assert hoisted == [ + {"type": "text", "text": "First."}, + {"type": "text", "text": "Second."}, + ] + + +def test_system_content_to_blocks_string_and_empty(): + from litellm.llms.anthropic.common_utils import _system_content_to_blocks + + assert _system_content_to_blocks("Be brief.") == [ + {"type": "text", "text": "Be brief."} + ] + # empty / whitespace-only string is dropped + assert _system_content_to_blocks(" ") == [] + assert _system_content_to_blocks("") == [] + + +def test_system_content_to_blocks_list_preserves_cache_control_and_drops_empty(): + from litellm.llms.anthropic.common_utils import _system_content_to_blocks + + content = [ + {"type": "text", "text": "Keep me.", "cache_control": {"type": "ephemeral"}}, + {"type": "text", "text": " "}, # empty -> dropped + {"type": "image", "source": {}}, # non-text -> dropped + "not-a-dict", # non-dict -> skipped + ] + out = _system_content_to_blocks(content) + assert out == [ + {"type": "text", "text": "Keep me.", "cache_control": {"type": "ephemeral"}} + ] + + +def test_system_content_to_blocks_deepcopies_cache_control(): + from litellm.llms.anthropic.common_utils import _system_content_to_blocks + + cc = {"type": "ephemeral"} + content = [{"type": "text", "text": "x", "cache_control": cc}] + out = _system_content_to_blocks(content) + # mutating the output's cache_control must NOT reach back into the input + out[0]["cache_control"]["type"] = "MUTATED" + assert cc == {"type": "ephemeral"} + + @pytest.fixture def local_model_cost_map(monkeypatch): """Force the bundled backup cost map so detection doesn't depend on the @@ -1452,6 +1527,110 @@ class TestClaudeOpus48AdaptiveThinking: assert AnthropicModelInfo._is_adaptive_thinking_model(model) is False +def test_supports_mid_conversation_system_resolves_through_route_prefixes( + local_model_cost_map, +): + from litellm.llms.anthropic.common_utils import AnthropicModelInfo + + K = "supports_mid_conversation_system" + # all of these are the post-get_llm_provider shapes for opus-4-8 -> must be True + for m in [ + "invoke/us.anthropic.claude-opus-4-8", + "converse/us.anthropic.claude-opus-4-8", + "mantle/anthropic.claude-opus-4-8", + "mantle/us.anthropic.claude-opus-4-8", + "us.anthropic.claude-opus-4-8", + "claude-opus-4-8", + ]: + assert AnthropicModelInfo._supports_model_capability(m, K) is True, m + # negatives -> must be False + for m in [ + "invoke/us.anthropic.claude-opus-4-7", + "mantle/anthropic.claude-opus-4-7", + "claude-opus-4-7", + "us.anthropic.claude-sonnet-4-5-20250929-v1:0", + ]: + assert AnthropicModelInfo._supports_model_capability(m, K) is False, m + + +def test_normalize_mid_system_passthrough_on_supported_model(local_model_cost_map): + from litellm.llms.anthropic.common_utils import ( + normalize_system_messages_for_anthropic, + ) + + messages = [ + {"role": "user", "content": "Hi"}, + {"role": "system", "content": "Now be a pirate."}, + ] + new_messages, hoisted = normalize_system_messages_for_anthropic( + messages, model="claude-opus-4-8" + ) + # supported model: mid system stays in place, nothing hoisted + assert new_messages == messages + assert hoisted == [] + + +def test_normalize_mid_system_demoted_on_unsupported_model(): + from litellm.llms.anthropic.common_utils import ( + normalize_system_messages_for_anthropic, + ) + + messages = [ + {"role": "user", "content": "Hi"}, + {"role": "system", "content": "Now be a pirate."}, + ] + new_messages, hoisted = normalize_system_messages_for_anthropic( + messages, model="claude-opus-4-7" + ) + # unsupported model: mid system removed and hoisted + assert new_messages == [{"role": "user", "content": "Hi"}] + assert hoisted == [{"type": "text", "text": "Now be a pirate."}] + + +def test_normalize_mid_system_cache_control_preserved_on_demote(): + from litellm.llms.anthropic.common_utils import ( + normalize_system_messages_for_anthropic, + ) + + messages = [ + {"role": "user", "content": "Hi"}, + { + "role": "system", + "content": [ + { + "type": "text", + "text": "Be a pirate.", + "cache_control": {"type": "ephemeral"}, + } + ], + }, + ] + new_messages, hoisted = normalize_system_messages_for_anthropic( + messages, model="claude-opus-4-7" + ) + assert hoisted == [ + { + "type": "text", + "text": "Be a pirate.", + "cache_control": {"type": "ephemeral"}, + } + ] + + +def test_normalize_empty_system_dropped(): + from litellm.llms.anthropic.common_utils import ( + normalize_system_messages_for_anthropic, + ) + + messages = [ + {"role": "system", "content": " "}, + {"role": "user", "content": "Hi"}, + ] + new_messages, hoisted = normalize_system_messages_for_anthropic( + messages, model="claude-opus-4-8" + ) + assert new_messages == [{"role": "user", "content": "Hi"}] + assert hoisted == [] def test_create_anthropic_model_list_response_shape(): from litellm.llms.anthropic.common_utils import ( create_anthropic_model_list_response, diff --git a/tests/test_litellm/test_utils.py b/tests/test_litellm/test_utils.py index 44e0b55ee3b..57270c3bfff 100644 --- a/tests/test_litellm/test_utils.py +++ b/tests/test_litellm/test_utils.py @@ -864,6 +864,7 @@ def test_aaamodel_prices_and_context_window_json_is_valid(): "supports_sampling_params": {"type": "boolean"}, "supports_service_tier": {"type": "boolean"}, "supports_preset": {"type": "boolean"}, + "supports_mid_conversation_system": {"type": "boolean"}, "supports_output_config": {"type": "boolean"}, "bedrock_output_config_effort_ceiling": { "type": "string",