diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py index 84fb7a1f45e..7c4986ca3fe 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py @@ -1,4 +1,4 @@ -from collections.abc import AsyncIterator +from collections.abc import AsyncIterator, Mapping, Sequence from typing import Any, Final import httpx @@ -163,14 +163,55 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): "Operator note (not from the user): the following was originally a mid-conversation system-role reminder." ) - def _system_role_message_as_user(self, message: dict) -> dict: + def _system_role_message_as_user(self, message: Mapping) -> Mapping: return { - **message, "role": "user", "content": self._as_system_content_blocks(self._CONVERTED_SYSTEM_NOTE) + self._as_system_content_blocks(message.get("content")), } + @staticmethod + def _opens_with_tool_results(message: object) -> bool: + if not isinstance(message, dict) or message.get("role") != "user": + return False + content: Final = message.get("content") + return ( + isinstance(content, list) + and len(content) > 0 + and isinstance(content[0], dict) + and content[0].get("type") == "tool_result" + ) + + def _system_run_before(self, messages: Sequence, index: int) -> Sequence: + start: Final = next( + (j + 1 for j in range(index - 1, -1, -1) if not self._is_system_role_message(messages[j])), + 0, + ) + return messages[start:index] + + def _system_run_end(self, messages: Sequence, index: int) -> int: + return next( + (j for j in range(index, len(messages)) if not self._is_system_role_message(messages[j])), + len(messages), + ) + + def _reordered_around_tool_results(self, messages: Sequence, index: int) -> tuple: + message: Final = messages[index] + if self._opens_with_tool_results(message): + return (message, *self._system_run_before(messages, index)) + if not self._is_system_role_message(message): + return (message,) + run_end: Final = self._system_run_end(messages, index) + follower: Final = messages[run_end] if run_end < len(messages) else None + return () if self._opens_with_tool_results(follower) else (message,) + + def _system_turns_after_tool_results(self, messages: Sequence) -> tuple: + return tuple( + message + for index in range(len(messages)) + for message in self._reordered_around_tool_results(messages, index) + ) + def _normalize_system_role_messages(self, anthropic_messages_request: dict, model: str) -> None: """Normalize ``role: "system"`` entries in ``messages`` per the Anthropic ``/v1/messages`` contract, which the first-party API, Bedrock Invoke, @@ -188,7 +229,13 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): so without the flag a mid-conversation entry is converted to a user turn in place (prefixed with an operator note) rather than hoisted: hoisting would mutate the ``system`` prefix and likewise collapse the cache, while - the in-place conversion keeps everything before it byte-identical. + the in-place conversion keeps everything before it byte-identical. Like + the hoist, the conversion carries only the entry's content. A run of + entries wedged between an assistant ``tool_use`` turn and its + ``tool_result`` turn is placed after that turn instead, since a user + turn in between would split the tool call from its result ("tool_use + ids were found without tool_result blocks immediately after") while + consecutive user turns merge upstream. Billing-header system blocks are stripped from the top-level ``system`` field regardless of whether anything was hoisted. @@ -214,7 +261,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig): ) else [ self._system_role_message_as_user(m) if self._is_system_role_message(m) else m - for m in messages[leading_count:] + for m in self._system_turns_after_tool_results(messages[leading_count:]) ] ) if hoisted or remaining != messages: diff --git a/tests/e2e/coverage_registry/llm_conversational.yaml b/tests/e2e/coverage_registry/llm_conversational.yaml index 82bee39b9b2..61ce30e3b81 100644 --- a/tests/e2e/coverage_registry/llm_conversational.yaml +++ b/tests/e2e/coverage_registry/llm_conversational.yaml @@ -53,12 +53,12 @@ - {id: llm.messages.anthropic.prompt_cache_5m.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Prompt caching via Messages API"} - {id: llm.messages.anthropic.thinking.nonstream.works, module: llm, tier: P1, subject_endpoint: messages, route: anthropic, capability: thinking, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Extended thinking via Messages API"} - {id: llm.messages.bedrock_invoke.mid_conversation_system.nonstream.cache_hit, module: llm, tier: P0, subject_endpoint: messages, route: bedrock_invoke, capability: mid_conversation_system, streaming: nonstream, assertions: [works, cache_hit], source: "llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py", rationale: "Flagged Claude 4.8+/5 must keep mid-conversation system reminders in messages; hoisting mutates the system prefix and collapses the prompt cache (#32578/#32831/#32882)", fail_before_fix: proven} -- {id: llm.messages.bedrock_invoke.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: bedrock_invoke, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py", rationale: "Claude <= 4.7 rejects role system inside messages; unflagged models must hoist reminders into top-level system or every Claude Code session 400s (#32831)", fail_before_fix: proven} +- {id: llm.messages.bedrock_invoke.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: bedrock_invoke, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py", rationale: "Claude <= 4.7 rejects role system inside messages; unflagged models must convert reminders to user turns in place (hoisting collapses the prompt cache) or every Claude Code session 400s (#32831)", fail_before_fix: proven} - {id: llm.messages.bedrock_invoke.web_search_server_tool.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: bedrock_invoke, capability: web_search_server_tool, streaming: nonstream, assertions: [works], source: "llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py", rationale: "Bedrock hosts no web_search server tool, so this only works because interception rewrites it before the upstream call and the agentic loop feeds the results back in native shape; a regression that short-circuits or forwards it instead yields raw text or AWS's 400", fail_before_fix: unproven} - {id: llm.messages.azure_foundry.mid_conversation_system.nonstream.cache_hit, module: llm, tier: P0, subject_endpoint: messages, route: azure_foundry, capability: mid_conversation_system, streaming: nonstream, assertions: [works, cache_hit], source: "llms/azure_ai/anthropic/messages_transformation.py", rationale: "Azure Foundry serves Claude on the native Anthropic contract, so flagged 4.8+/5 must keep mid-conversation system reminders in messages; hoisting mutates the system prefix and collapses the prompt cache (customer RCA gap)", fail_before_fix: proven} -- {id: llm.messages.azure_foundry.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: azure_foundry, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/azure_ai/anthropic/messages_transformation.py", rationale: "Azure Foundry Claude <= 4.7 rejects role system inside messages; unflagged models must hoist reminders into top-level system or every Claude Code session 400s (customer RCA gap)", fail_before_fix: proven} +- {id: llm.messages.azure_foundry.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: azure_foundry, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/azure_ai/anthropic/messages_transformation.py", rationale: "Azure Foundry Claude <= 4.7 rejects role system inside messages; unflagged models must convert reminders to user turns in place (hoisting collapses the prompt cache) or every Claude Code session 400s (customer RCA gap)", fail_before_fix: proven} - {id: llm.messages.vertex.mid_conversation_system.nonstream.cache_hit, module: llm, tier: P0, subject_endpoint: messages, route: vertex, capability: mid_conversation_system, streaming: nonstream, assertions: [works, cache_hit], source: "llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py", rationale: "Vertex serves Claude on the native Anthropic contract, so flagged 4.8+/5 must keep mid-conversation system reminders in messages; hoisting mutates the system prefix and collapses the prompt cache (customer RCA gap)", fail_before_fix: proven} -- {id: llm.messages.vertex.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: vertex, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py", rationale: "Vertex Claude <= 4.7 rejects role system inside messages; unflagged models must hoist reminders into top-level system or every Claude Code session 400s (customer RCA gap)", fail_before_fix: proven} +- {id: llm.messages.vertex.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: vertex, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py", rationale: "Vertex Claude <= 4.7 rejects role system inside messages; unflagged models must convert reminders to user turns in place (hoisting collapses the prompt cache) or every Claude Code session 400s (customer RCA gap)", fail_before_fix: proven} - {id: llm.responses.openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "response_api_endpoints/endpoints.py:26", rationale: "Core endpoint; OpenAI Responses native"} - {id: llm.responses.openai.input_validation.nonstream.works, module: llm, tier: P1, subject_endpoint: responses, route: openai, capability: input_validation, streaming: nonstream, assertions: [works], source: "vendor strategy ยง9.9 / LIT-4778", rationale: "Responses missing/empty input and missing model are rejected"} - {id: llm.responses.openai.basic.stream.works, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: basic, streaming: stream, assertions: [works], source: "response_api_endpoints/endpoints.py:26", rationale: "Streaming via /v1/responses"} diff --git a/tests/e2e/llm_translation/test_messages_mid_conversation_system_e2e.py b/tests/e2e/llm_translation/test_messages_mid_conversation_system_e2e.py index bd59df959c4..234a647907d 100644 --- a/tests/e2e/llm_translation/test_messages_mid_conversation_system_e2e.py +++ b/tests/e2e/llm_translation/test_messages_mid_conversation_system_e2e.py @@ -6,8 +6,8 @@ and the 5 family) must keep a mid-conversation system reminder in place inside ``messages`` so the top-level ``system`` prefix stays byte-identical and the prompt cache written on turn one is read back in full on turn two. Models without the flag (Claude 4.7 and older) reject the role inside ``messages`` -outright, so the proxy must convert the reminder to a user turn in place -field and the call must still return a completion instead of a provider 400. +outright, so the proxy must convert the reminder to a user turn in place and +the call must still return a completion instead of a provider 400. The conversation shape mirrors what Claude Code sends mid-session: a cached system prompt, a user turn carrying its own ``cache_control`` breakpoint, a @@ -49,8 +49,9 @@ CACHE_PRIMING_INTERVAL_SECONDS = 3.0 def _cacheable_system_block(marker: str) -> TextBlock: - """A system prompt comfortably above Sonnet's 1024-token minimum cacheable - size, unique per run so no other run's cache entry can satisfy the read.""" + """A system prompt comfortably above the 4096-token minimum cacheable size + of Haiku 4.5 (the smallest model here), unique per run so no other run's + cache entry can satisfy the read.""" text = " ".join( f"Reference paragraph {index} for run {marker}." for index in range(300) ) @@ -118,10 +119,12 @@ def _prime_prompt_cache( ) -> PrimedCache: """Send first-turn calls (fresh cache-marked user turn each attempt, identical system prefix) until one both reads the system prefix back from - cache and writes its own user-turn chunk, proving the cache is live in both - directions. Only the pre-reminder turn is ever retried here, so retries can - never warm a mutated-prefix cache entry and mask the regression the second - turn asserts on.""" + cache and writes its own user-turn chunk, then re-send that exact turn until + its own chunk reads back too, proving the cache is live in both directions + before the reminder turn goes out (a freshly written entry can take a few + seconds to become readable). Only the pre-reminder turn is ever retried + here, so retries can never warm a mutated-prefix cache entry and mask the + regression the second turn asserts on.""" deadline = time.monotonic() + CACHE_PRIMING_DEADLINE_SECONDS while True: user_text = _first_turn_user_text(unique_marker()) @@ -132,19 +135,37 @@ def _prime_prompt_cache( ) usage = unwrap(_post_messages(client, key, body)).usage if usage.cache_read_input_tokens > 0 and usage.cache_creation_input_tokens > 0: - return PrimedCache( + primed = PrimedCache( first_user_text=user_text, prefix_read_tokens=usage.cache_read_input_tokens, first_turn_creation_tokens=usage.cache_creation_input_tokens, ) + if _first_turn_reads_back(client, key, body, primed.full_prefix_tokens, deadline): + return primed if time.monotonic() >= deadline: pytest.fail( - f"{model}: prompt cache never became readable within " + f"{model}: prompt cache never became readable in full within " f"{CACHE_PRIMING_DEADLINE_SECONDS}s (last usage: {usage})" ) time.sleep(CACHE_PRIMING_INTERVAL_SECONDS) +def _first_turn_reads_back( + client: EndpointsClient, + key: str, + body: RichMessagesRequest, + full_prefix_tokens: int, + deadline: float, +) -> bool: + while True: + usage = unwrap(_post_messages(client, key, body)).usage + if usage.cache_read_input_tokens >= full_prefix_tokens: + return True + if time.monotonic() >= deadline: + return False + time.sleep(CACHE_PRIMING_INTERVAL_SECONDS) + + #: Kept in sync with the copy in test_messages_mid_conversation_system_native_providers_e2e.py; #: the e2e suites stay self-contained rather than importing across test modules. MID_CONVERSATION_CACHE_SKIP_REASON = ( diff --git a/tests/e2e/llm_translation/test_messages_mid_conversation_system_native_providers_e2e.py b/tests/e2e/llm_translation/test_messages_mid_conversation_system_native_providers_e2e.py index bf4e662e02e..ba8b77787e0 100644 --- a/tests/e2e/llm_translation/test_messages_mid_conversation_system_native_providers_e2e.py +++ b/tests/e2e/llm_translation/test_messages_mid_conversation_system_native_providers_e2e.py @@ -128,10 +128,12 @@ def _prime_prompt_cache( ) -> PrimedCache: """Send first-turn calls (fresh cache-marked user turn each attempt, identical system prefix) until one both reads the system prefix back from - cache and writes its own user-turn chunk, proving the cache is live in both - directions. Only the pre-reminder turn is ever retried here, so retries can - never warm a mutated-prefix cache entry and mask the regression the second - turn asserts on.""" + cache and writes its own user-turn chunk, then re-send that exact turn until + its own chunk reads back too, proving the cache is live in both directions + before the reminder turn goes out (a freshly written entry can take a few + seconds to become readable). Only the pre-reminder turn is ever retried + here, so retries can never warm a mutated-prefix cache entry and mask the + regression the second turn asserts on.""" deadline = time.monotonic() + CACHE_PRIMING_DEADLINE_SECONDS while True: user_text = _first_turn_user_text(unique_marker()) @@ -142,19 +144,37 @@ def _prime_prompt_cache( ) usage = unwrap(_post_messages(client, key, body)).usage if usage.cache_read_input_tokens > 0 and usage.cache_creation_input_tokens > 0: - return PrimedCache( + primed = PrimedCache( first_user_text=user_text, prefix_read_tokens=usage.cache_read_input_tokens, first_turn_creation_tokens=usage.cache_creation_input_tokens, ) + if _first_turn_reads_back(client, key, body, primed.full_prefix_tokens, deadline): + return primed if time.monotonic() >= deadline: pytest.fail( - f"{model}: prompt cache never became readable within " + f"{model}: prompt cache never became readable in full within " f"{CACHE_PRIMING_DEADLINE_SECONDS}s (last usage: {usage})" ) time.sleep(CACHE_PRIMING_INTERVAL_SECONDS) +def _first_turn_reads_back( + client: EndpointsClient, + key: str, + body: RichMessagesRequest, + full_prefix_tokens: int, + deadline: float, +) -> bool: + while True: + usage = unwrap(_post_messages(client, key, body)).usage + if usage.cache_read_input_tokens >= full_prefix_tokens: + return True + if time.monotonic() >= deadline: + return False + time.sleep(CACHE_PRIMING_INTERVAL_SECONDS) + + #: Why the flagged-model cache checks are skipped rather than failing. The #: assertions below are correct and must be restored unchanged when the bug is #: fixed; they are the regression guard for a real billing cost. diff --git a/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py b/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py index 84df706b722..b55ca073bc9 100644 --- a/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py +++ b/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py @@ -2176,6 +2176,112 @@ def test_bedrock_invoke_transform_converts_mid_conversation_system_for_older_cla assert result["system"] == [{"type": "text", "text": "Base."}] +def test_bedrock_invoke_transform_moves_converted_system_after_tool_result_turn(local_model_cost_map): + """A reminder wedged between an assistant ``tool_use`` turn and the user + ``tool_result`` turn cannot become a user turn in that position: the API + requires the result right after the call ("tool_use ids were found without + tool_result blocks immediately after"). The converted turn goes after the + tool-result turn instead, where consecutive user turns merge upstream.""" + from litellm.types.router import GenericLiteLLMParams + + cfg = AmazonAnthropicClaudeMessagesConfig() + tool_use_turn = { + "role": "assistant", + "content": [{"type": "tool_use", "id": "toolu_01", "name": "read_file", "input": {"path": "big1.txt"}}], + } + tool_result_turn = { + "role": "user", + "content": [ + {"type": "tool_result", "tool_use_id": "toolu_01", "content": "first 100 lines"}, + {"type": "text", "text": "keep going"}, + ], + } + messages = [ + {"role": "user", "content": "read the file"}, + tool_use_turn, + {"role": "system", "content": "[Truncated: PARTIAL view of big1.txt]"}, + {"role": "system", "content": "low"}, + tool_result_turn, + ] + + result = cfg.transform_anthropic_messages_request( + model="us.anthropic.claude-opus-4-7", + messages=copy.deepcopy(messages), + anthropic_messages_optional_request_params={"max_tokens": 256, "stream": False}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert result["messages"] == [ + {"role": "user", "content": "read the file"}, + tool_use_turn, + tool_result_turn, + { + "role": "user", + "content": [ + { + "type": "text", + "text": ( + "Operator note (not from the user): the following was " + "originally a mid-conversation system-role reminder." + ), + }, + {"type": "text", "text": "[Truncated: PARTIAL view of big1.txt]"}, + ], + }, + { + "role": "user", + "content": [ + { + "type": "text", + "text": ( + "Operator note (not from the user): the following was " + "originally a mid-conversation system-role reminder." + ), + }, + {"type": "text", "text": "low"}, + ], + }, + ] + + +def test_bedrock_invoke_transform_converted_system_carries_only_its_content(local_model_cost_map): + """Hoisting only ever kept a system entry's content, so the in-place + conversion must not forward the entry's other keys either ("messages.2.name: + Extra inputs are not permitted").""" + from litellm.types.router import GenericLiteLLMParams + + cfg = AmazonAnthropicClaudeMessagesConfig() + messages = [ + {"role": "user", "content": "read the file"}, + {"role": "assistant", "content": "reading"}, + {"role": "system", "content": "[Truncated: PARTIAL view of big1.txt]", "name": "ops"}, + {"role": "user", "content": "continue"}, + ] + + result = cfg.transform_anthropic_messages_request( + model="us.anthropic.claude-opus-4-7", + messages=copy.deepcopy(messages), + anthropic_messages_optional_request_params={"max_tokens": 256, "stream": False}, + litellm_params=GenericLiteLLMParams(), + headers={}, + ) + + assert result["messages"][2] == { + "role": "user", + "content": [ + { + "type": "text", + "text": ( + "Operator note (not from the user): the following was " + "originally a mid-conversation system-role reminder." + ), + }, + {"type": "text", "text": "[Truncated: PARTIAL view of big1.txt]"}, + ], + } + + def test_bedrock_invoke_transform_converts_system_for_unmapped_model(local_model_cost_map): """A model with no cost-map entry and no fallback-generalization rule gets the unsupported-model treatment: the safe default converts the reminder to