diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py
index 84fb7a1f45e..7c4986ca3fe 100644
--- a/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py
+++ b/litellm/llms/anthropic/experimental_pass_through/messages/transformation.py
@@ -1,4 +1,4 @@
-from collections.abc import AsyncIterator
+from collections.abc import AsyncIterator, Mapping, Sequence
from typing import Any, Final
import httpx
@@ -163,14 +163,55 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
"Operator note (not from the user): the following was originally a mid-conversation system-role reminder."
)
- def _system_role_message_as_user(self, message: dict) -> dict:
+ def _system_role_message_as_user(self, message: Mapping) -> Mapping:
return {
- **message,
"role": "user",
"content": self._as_system_content_blocks(self._CONVERTED_SYSTEM_NOTE)
+ self._as_system_content_blocks(message.get("content")),
}
+ @staticmethod
+ def _opens_with_tool_results(message: object) -> bool:
+ if not isinstance(message, dict) or message.get("role") != "user":
+ return False
+ content: Final = message.get("content")
+ return (
+ isinstance(content, list)
+ and len(content) > 0
+ and isinstance(content[0], dict)
+ and content[0].get("type") == "tool_result"
+ )
+
+ def _system_run_before(self, messages: Sequence, index: int) -> Sequence:
+ start: Final = next(
+ (j + 1 for j in range(index - 1, -1, -1) if not self._is_system_role_message(messages[j])),
+ 0,
+ )
+ return messages[start:index]
+
+ def _system_run_end(self, messages: Sequence, index: int) -> int:
+ return next(
+ (j for j in range(index, len(messages)) if not self._is_system_role_message(messages[j])),
+ len(messages),
+ )
+
+ def _reordered_around_tool_results(self, messages: Sequence, index: int) -> tuple:
+ message: Final = messages[index]
+ if self._opens_with_tool_results(message):
+ return (message, *self._system_run_before(messages, index))
+ if not self._is_system_role_message(message):
+ return (message,)
+ run_end: Final = self._system_run_end(messages, index)
+ follower: Final = messages[run_end] if run_end < len(messages) else None
+ return () if self._opens_with_tool_results(follower) else (message,)
+
+ def _system_turns_after_tool_results(self, messages: Sequence) -> tuple:
+ return tuple(
+ message
+ for index in range(len(messages))
+ for message in self._reordered_around_tool_results(messages, index)
+ )
+
def _normalize_system_role_messages(self, anthropic_messages_request: dict, model: str) -> None:
"""Normalize ``role: "system"`` entries in ``messages`` per the Anthropic
``/v1/messages`` contract, which the first-party API, Bedrock Invoke,
@@ -188,7 +229,13 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
so without the flag a mid-conversation entry is converted to a user turn
in place (prefixed with an operator note) rather than hoisted: hoisting
would mutate the ``system`` prefix and likewise collapse the cache, while
- the in-place conversion keeps everything before it byte-identical.
+ the in-place conversion keeps everything before it byte-identical. Like
+ the hoist, the conversion carries only the entry's content. A run of
+ entries wedged between an assistant ``tool_use`` turn and its
+ ``tool_result`` turn is placed after that turn instead, since a user
+ turn in between would split the tool call from its result ("tool_use
+ ids were found without tool_result blocks immediately after") while
+ consecutive user turns merge upstream.
Billing-header system blocks are stripped from the top-level ``system``
field regardless of whether anything was hoisted.
@@ -214,7 +261,7 @@ class AnthropicMessagesConfig(BaseAnthropicMessagesConfig):
)
else [
self._system_role_message_as_user(m) if self._is_system_role_message(m) else m
- for m in messages[leading_count:]
+ for m in self._system_turns_after_tool_results(messages[leading_count:])
]
)
if hoisted or remaining != messages:
diff --git a/tests/e2e/coverage_registry/llm_conversational.yaml b/tests/e2e/coverage_registry/llm_conversational.yaml
index 82bee39b9b2..61ce30e3b81 100644
--- a/tests/e2e/coverage_registry/llm_conversational.yaml
+++ b/tests/e2e/coverage_registry/llm_conversational.yaml
@@ -53,12 +53,12 @@
- {id: llm.messages.anthropic.prompt_cache_5m.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: anthropic, capability: prompt_cache_5m, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Prompt caching via Messages API"}
- {id: llm.messages.anthropic.thinking.nonstream.works, module: llm, tier: P1, subject_endpoint: messages, route: anthropic, capability: thinking, streaming: nonstream, assertions: [works], source: "model_prices json", rationale: "Extended thinking via Messages API"}
- {id: llm.messages.bedrock_invoke.mid_conversation_system.nonstream.cache_hit, module: llm, tier: P0, subject_endpoint: messages, route: bedrock_invoke, capability: mid_conversation_system, streaming: nonstream, assertions: [works, cache_hit], source: "llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py", rationale: "Flagged Claude 4.8+/5 must keep mid-conversation system reminders in messages; hoisting mutates the system prefix and collapses the prompt cache (#32578/#32831/#32882)", fail_before_fix: proven}
-- {id: llm.messages.bedrock_invoke.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: bedrock_invoke, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py", rationale: "Claude <= 4.7 rejects role system inside messages; unflagged models must hoist reminders into top-level system or every Claude Code session 400s (#32831)", fail_before_fix: proven}
+- {id: llm.messages.bedrock_invoke.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: bedrock_invoke, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py", rationale: "Claude <= 4.7 rejects role system inside messages; unflagged models must convert reminders to user turns in place (hoisting collapses the prompt cache) or every Claude Code session 400s (#32831)", fail_before_fix: proven}
- {id: llm.messages.bedrock_invoke.web_search_server_tool.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: bedrock_invoke, capability: web_search_server_tool, streaming: nonstream, assertions: [works], source: "llms/bedrock/messages/invoke_transformations/anthropic_claude3_transformation.py", rationale: "Bedrock hosts no web_search server tool, so this only works because interception rewrites it before the upstream call and the agentic loop feeds the results back in native shape; a regression that short-circuits or forwards it instead yields raw text or AWS's 400", fail_before_fix: unproven}
- {id: llm.messages.azure_foundry.mid_conversation_system.nonstream.cache_hit, module: llm, tier: P0, subject_endpoint: messages, route: azure_foundry, capability: mid_conversation_system, streaming: nonstream, assertions: [works, cache_hit], source: "llms/azure_ai/anthropic/messages_transformation.py", rationale: "Azure Foundry serves Claude on the native Anthropic contract, so flagged 4.8+/5 must keep mid-conversation system reminders in messages; hoisting mutates the system prefix and collapses the prompt cache (customer RCA gap)", fail_before_fix: proven}
-- {id: llm.messages.azure_foundry.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: azure_foundry, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/azure_ai/anthropic/messages_transformation.py", rationale: "Azure Foundry Claude <= 4.7 rejects role system inside messages; unflagged models must hoist reminders into top-level system or every Claude Code session 400s (customer RCA gap)", fail_before_fix: proven}
+- {id: llm.messages.azure_foundry.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: azure_foundry, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/azure_ai/anthropic/messages_transformation.py", rationale: "Azure Foundry Claude <= 4.7 rejects role system inside messages; unflagged models must convert reminders to user turns in place (hoisting collapses the prompt cache) or every Claude Code session 400s (customer RCA gap)", fail_before_fix: proven}
- {id: llm.messages.vertex.mid_conversation_system.nonstream.cache_hit, module: llm, tier: P0, subject_endpoint: messages, route: vertex, capability: mid_conversation_system, streaming: nonstream, assertions: [works, cache_hit], source: "llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py", rationale: "Vertex serves Claude on the native Anthropic contract, so flagged 4.8+/5 must keep mid-conversation system reminders in messages; hoisting mutates the system prefix and collapses the prompt cache (customer RCA gap)", fail_before_fix: proven}
-- {id: llm.messages.vertex.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: vertex, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py", rationale: "Vertex Claude <= 4.7 rejects role system inside messages; unflagged models must hoist reminders into top-level system or every Claude Code session 400s (customer RCA gap)", fail_before_fix: proven}
+- {id: llm.messages.vertex.mid_conversation_system.nonstream.works, module: llm, tier: P0, subject_endpoint: messages, route: vertex, capability: mid_conversation_system, streaming: nonstream, assertions: [works], source: "llms/vertex_ai/vertex_ai_partner_models/anthropic/experimental_pass_through/transformation.py", rationale: "Vertex Claude <= 4.7 rejects role system inside messages; unflagged models must convert reminders to user turns in place (hoisting collapses the prompt cache) or every Claude Code session 400s (customer RCA gap)", fail_before_fix: proven}
- {id: llm.responses.openai.basic.nonstream.works, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: basic, streaming: nonstream, assertions: [works], source: "response_api_endpoints/endpoints.py:26", rationale: "Core endpoint; OpenAI Responses native"}
- {id: llm.responses.openai.input_validation.nonstream.works, module: llm, tier: P1, subject_endpoint: responses, route: openai, capability: input_validation, streaming: nonstream, assertions: [works], source: "vendor strategy ยง9.9 / LIT-4778", rationale: "Responses missing/empty input and missing model are rejected"}
- {id: llm.responses.openai.basic.stream.works, module: llm, tier: P0, subject_endpoint: responses, route: openai, capability: basic, streaming: stream, assertions: [works], source: "response_api_endpoints/endpoints.py:26", rationale: "Streaming via /v1/responses"}
diff --git a/tests/e2e/llm_translation/test_messages_mid_conversation_system_e2e.py b/tests/e2e/llm_translation/test_messages_mid_conversation_system_e2e.py
index bd59df959c4..234a647907d 100644
--- a/tests/e2e/llm_translation/test_messages_mid_conversation_system_e2e.py
+++ b/tests/e2e/llm_translation/test_messages_mid_conversation_system_e2e.py
@@ -6,8 +6,8 @@ and the 5 family) must keep a mid-conversation system reminder in place inside
``messages`` so the top-level ``system`` prefix stays byte-identical and the
prompt cache written on turn one is read back in full on turn two. Models
without the flag (Claude 4.7 and older) reject the role inside ``messages``
-outright, so the proxy must convert the reminder to a user turn in place
-field and the call must still return a completion instead of a provider 400.
+outright, so the proxy must convert the reminder to a user turn in place and
+the call must still return a completion instead of a provider 400.
The conversation shape mirrors what Claude Code sends mid-session: a cached
system prompt, a user turn carrying its own ``cache_control`` breakpoint, a
@@ -49,8 +49,9 @@ CACHE_PRIMING_INTERVAL_SECONDS = 3.0
def _cacheable_system_block(marker: str) -> TextBlock:
- """A system prompt comfortably above Sonnet's 1024-token minimum cacheable
- size, unique per run so no other run's cache entry can satisfy the read."""
+ """A system prompt comfortably above the 4096-token minimum cacheable size
+ of Haiku 4.5 (the smallest model here), unique per run so no other run's
+ cache entry can satisfy the read."""
text = " ".join(
f"Reference paragraph {index} for run {marker}." for index in range(300)
)
@@ -118,10 +119,12 @@ def _prime_prompt_cache(
) -> PrimedCache:
"""Send first-turn calls (fresh cache-marked user turn each attempt,
identical system prefix) until one both reads the system prefix back from
- cache and writes its own user-turn chunk, proving the cache is live in both
- directions. Only the pre-reminder turn is ever retried here, so retries can
- never warm a mutated-prefix cache entry and mask the regression the second
- turn asserts on."""
+ cache and writes its own user-turn chunk, then re-send that exact turn until
+ its own chunk reads back too, proving the cache is live in both directions
+ before the reminder turn goes out (a freshly written entry can take a few
+ seconds to become readable). Only the pre-reminder turn is ever retried
+ here, so retries can never warm a mutated-prefix cache entry and mask the
+ regression the second turn asserts on."""
deadline = time.monotonic() + CACHE_PRIMING_DEADLINE_SECONDS
while True:
user_text = _first_turn_user_text(unique_marker())
@@ -132,19 +135,37 @@ def _prime_prompt_cache(
)
usage = unwrap(_post_messages(client, key, body)).usage
if usage.cache_read_input_tokens > 0 and usage.cache_creation_input_tokens > 0:
- return PrimedCache(
+ primed = PrimedCache(
first_user_text=user_text,
prefix_read_tokens=usage.cache_read_input_tokens,
first_turn_creation_tokens=usage.cache_creation_input_tokens,
)
+ if _first_turn_reads_back(client, key, body, primed.full_prefix_tokens, deadline):
+ return primed
if time.monotonic() >= deadline:
pytest.fail(
- f"{model}: prompt cache never became readable within "
+ f"{model}: prompt cache never became readable in full within "
f"{CACHE_PRIMING_DEADLINE_SECONDS}s (last usage: {usage})"
)
time.sleep(CACHE_PRIMING_INTERVAL_SECONDS)
+def _first_turn_reads_back(
+ client: EndpointsClient,
+ key: str,
+ body: RichMessagesRequest,
+ full_prefix_tokens: int,
+ deadline: float,
+) -> bool:
+ while True:
+ usage = unwrap(_post_messages(client, key, body)).usage
+ if usage.cache_read_input_tokens >= full_prefix_tokens:
+ return True
+ if time.monotonic() >= deadline:
+ return False
+ time.sleep(CACHE_PRIMING_INTERVAL_SECONDS)
+
+
#: Kept in sync with the copy in test_messages_mid_conversation_system_native_providers_e2e.py;
#: the e2e suites stay self-contained rather than importing across test modules.
MID_CONVERSATION_CACHE_SKIP_REASON = (
diff --git a/tests/e2e/llm_translation/test_messages_mid_conversation_system_native_providers_e2e.py b/tests/e2e/llm_translation/test_messages_mid_conversation_system_native_providers_e2e.py
index bf4e662e02e..ba8b77787e0 100644
--- a/tests/e2e/llm_translation/test_messages_mid_conversation_system_native_providers_e2e.py
+++ b/tests/e2e/llm_translation/test_messages_mid_conversation_system_native_providers_e2e.py
@@ -128,10 +128,12 @@ def _prime_prompt_cache(
) -> PrimedCache:
"""Send first-turn calls (fresh cache-marked user turn each attempt,
identical system prefix) until one both reads the system prefix back from
- cache and writes its own user-turn chunk, proving the cache is live in both
- directions. Only the pre-reminder turn is ever retried here, so retries can
- never warm a mutated-prefix cache entry and mask the regression the second
- turn asserts on."""
+ cache and writes its own user-turn chunk, then re-send that exact turn until
+ its own chunk reads back too, proving the cache is live in both directions
+ before the reminder turn goes out (a freshly written entry can take a few
+ seconds to become readable). Only the pre-reminder turn is ever retried
+ here, so retries can never warm a mutated-prefix cache entry and mask the
+ regression the second turn asserts on."""
deadline = time.monotonic() + CACHE_PRIMING_DEADLINE_SECONDS
while True:
user_text = _first_turn_user_text(unique_marker())
@@ -142,19 +144,37 @@ def _prime_prompt_cache(
)
usage = unwrap(_post_messages(client, key, body)).usage
if usage.cache_read_input_tokens > 0 and usage.cache_creation_input_tokens > 0:
- return PrimedCache(
+ primed = PrimedCache(
first_user_text=user_text,
prefix_read_tokens=usage.cache_read_input_tokens,
first_turn_creation_tokens=usage.cache_creation_input_tokens,
)
+ if _first_turn_reads_back(client, key, body, primed.full_prefix_tokens, deadline):
+ return primed
if time.monotonic() >= deadline:
pytest.fail(
- f"{model}: prompt cache never became readable within "
+ f"{model}: prompt cache never became readable in full within "
f"{CACHE_PRIMING_DEADLINE_SECONDS}s (last usage: {usage})"
)
time.sleep(CACHE_PRIMING_INTERVAL_SECONDS)
+def _first_turn_reads_back(
+ client: EndpointsClient,
+ key: str,
+ body: RichMessagesRequest,
+ full_prefix_tokens: int,
+ deadline: float,
+) -> bool:
+ while True:
+ usage = unwrap(_post_messages(client, key, body)).usage
+ if usage.cache_read_input_tokens >= full_prefix_tokens:
+ return True
+ if time.monotonic() >= deadline:
+ return False
+ time.sleep(CACHE_PRIMING_INTERVAL_SECONDS)
+
+
#: Why the flagged-model cache checks are skipped rather than failing. The
#: assertions below are correct and must be restored unchanged when the bug is
#: fixed; they are the regression guard for a real billing cost.
diff --git a/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py b/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py
index 84df706b722..b55ca073bc9 100644
--- a/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py
+++ b/tests/test_litellm/llms/bedrock/messages/invoke_transformations/test_anthropic_claude3_transformation.py
@@ -2176,6 +2176,112 @@ def test_bedrock_invoke_transform_converts_mid_conversation_system_for_older_cla
assert result["system"] == [{"type": "text", "text": "Base."}]
+def test_bedrock_invoke_transform_moves_converted_system_after_tool_result_turn(local_model_cost_map):
+ """A reminder wedged between an assistant ``tool_use`` turn and the user
+ ``tool_result`` turn cannot become a user turn in that position: the API
+ requires the result right after the call ("tool_use ids were found without
+ tool_result blocks immediately after"). The converted turn goes after the
+ tool-result turn instead, where consecutive user turns merge upstream."""
+ from litellm.types.router import GenericLiteLLMParams
+
+ cfg = AmazonAnthropicClaudeMessagesConfig()
+ tool_use_turn = {
+ "role": "assistant",
+ "content": [{"type": "tool_use", "id": "toolu_01", "name": "read_file", "input": {"path": "big1.txt"}}],
+ }
+ tool_result_turn = {
+ "role": "user",
+ "content": [
+ {"type": "tool_result", "tool_use_id": "toolu_01", "content": "first 100 lines"},
+ {"type": "text", "text": "keep going"},
+ ],
+ }
+ messages = [
+ {"role": "user", "content": "read the file"},
+ tool_use_turn,
+ {"role": "system", "content": "[Truncated: PARTIAL view of big1.txt]"},
+ {"role": "system", "content": "low"},
+ tool_result_turn,
+ ]
+
+ result = cfg.transform_anthropic_messages_request(
+ model="us.anthropic.claude-opus-4-7",
+ messages=copy.deepcopy(messages),
+ anthropic_messages_optional_request_params={"max_tokens": 256, "stream": False},
+ litellm_params=GenericLiteLLMParams(),
+ headers={},
+ )
+
+ assert result["messages"] == [
+ {"role": "user", "content": "read the file"},
+ tool_use_turn,
+ tool_result_turn,
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": (
+ "Operator note (not from the user): the following was "
+ "originally a mid-conversation system-role reminder."
+ ),
+ },
+ {"type": "text", "text": "[Truncated: PARTIAL view of big1.txt]"},
+ ],
+ },
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": (
+ "Operator note (not from the user): the following was "
+ "originally a mid-conversation system-role reminder."
+ ),
+ },
+ {"type": "text", "text": "low"},
+ ],
+ },
+ ]
+
+
+def test_bedrock_invoke_transform_converted_system_carries_only_its_content(local_model_cost_map):
+ """Hoisting only ever kept a system entry's content, so the in-place
+ conversion must not forward the entry's other keys either ("messages.2.name:
+ Extra inputs are not permitted")."""
+ from litellm.types.router import GenericLiteLLMParams
+
+ cfg = AmazonAnthropicClaudeMessagesConfig()
+ messages = [
+ {"role": "user", "content": "read the file"},
+ {"role": "assistant", "content": "reading"},
+ {"role": "system", "content": "[Truncated: PARTIAL view of big1.txt]", "name": "ops"},
+ {"role": "user", "content": "continue"},
+ ]
+
+ result = cfg.transform_anthropic_messages_request(
+ model="us.anthropic.claude-opus-4-7",
+ messages=copy.deepcopy(messages),
+ anthropic_messages_optional_request_params={"max_tokens": 256, "stream": False},
+ litellm_params=GenericLiteLLMParams(),
+ headers={},
+ )
+
+ assert result["messages"][2] == {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": (
+ "Operator note (not from the user): the following was "
+ "originally a mid-conversation system-role reminder."
+ ),
+ },
+ {"type": "text", "text": "[Truncated: PARTIAL view of big1.txt]"},
+ ],
+ }
+
+
def test_bedrock_invoke_transform_converts_system_for_unmapped_model(local_model_cost_map):
"""A model with no cost-map entry and no fallback-generalization rule gets
the unsupported-model treatment: the safe default converts the reminder to