From b8f38829af90901d294f48950a555e6bd8bdf06c Mon Sep 17 00:00:00 2001 From: Charles Cheng Date: Fri, 19 Jun 2026 18:10:58 +0800 Subject: [PATCH] refactor: address Greptile P2 review items for cache_control /v1/messages fix - Clarify apply_to_anthropic_messages_request() Returns: docstring: the no-injection early-return path returns original references (no deep copy), while the injection path returns deep-copied values; the previous wording overstated the no-copy path's mutation safety. - Add explanatory note in _supports_native_anthropic_messages() docstring explaining why get_llm_provider() is called unconditionally even when custom_llm_provider is set: routing prefixes like converse/ make naive prefix-stripping unsafe (wrong native-support answer for Bedrock Converse models), so the optimisation is deferred to a follow-up. - NON_CACHEABLE_TOOL_TYPES in anthropic_cache_control_hook is already imported from litellm.constants.ANTHROPIC_NON_CACHEABLE_TOOL_TYPES (shared with transformation.py), so no divergence risk. --- .../anthropic_cache_control_hook.py | 17 ++++++++--------- .../messages/handler.py | 7 +++++++ 2 files changed, 15 insertions(+), 9 deletions(-) diff --git a/litellm/integrations/anthropic_cache_control_hook.py b/litellm/integrations/anthropic_cache_control_hook.py index 3424eb456ff..e3b833b4be4 100644 --- a/litellm/integrations/anthropic_cache_control_hook.py +++ b/litellm/integrations/anthropic_cache_control_hook.py @@ -10,6 +10,7 @@ import copy from typing import TYPE_CHECKING, Any, Dict, List, Optional, Tuple, Union, cast from litellm._logging import verbose_logger +from litellm.constants import ANTHROPIC_NON_CACHEABLE_TOOL_TYPES from litellm.integrations.custom_logger import CustomLogger from litellm.integrations.custom_prompt_management import CustomPromptManagement from litellm.integrations.prompt_management_base import PromptManagementClient @@ -31,13 +32,9 @@ else: # breakpoints: "A maximum of 4 blocks with cache_control may be provided." MAX_CACHE_CONTROL_BLOCKS = 4 -# Tool types Anthropic rejects `cache_control` on. Mirrors the exclusion in -# litellm/llms/anthropic/chat/transformation.py so the native /v1/messages path -# never marks a tool the API would 400 on. -NON_CACHEABLE_TOOL_TYPES = ( - "tool_search_tool_regex_20251119", - "tool_search_tool_bm25_20251119", -) +# Re-export for any existing internal callers while the canonical definition +# lives in litellm.constants (shared with transformation.py). +NON_CACHEABLE_TOOL_TYPES = ANTHROPIC_NON_CACHEABLE_TOOL_TYPES class AnthropicCacheControlHook(CustomPromptManagement): @@ -287,8 +284,10 @@ class AnthropicCacheControlHook(CustomPromptManagement): downstream provider transforms still receive them. Returns: - A tuple of (messages, system, tools) with cache control applied. The - inputs are deep-copied, so the caller's originals are never mutated. + A tuple of (messages, system, tools). When there are no injection + points, the original references are returned as-is (no copy is + made). When injection occurs, the inputs are deep-copied before + modification so the caller's originals are not mutated. """ injection_points: list[CacheControlInjectionPoint] = non_default_params.pop( "cache_control_injection_points", [] diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 86e2c66df6a..6b13be94268 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -197,6 +197,13 @@ def _supports_native_anthropic_messages( provider selected purely by endpoint (e.g. an ``/anthropic`` ``api_base``) resolves the same way it will in the handler, instead of being misread as a fallback and having its cache injection skipped. + + Note: ``get_llm_provider`` is called unconditionally here even when + ``custom_llm_provider`` is already set. Skipping it is non-trivial because + model strings may carry routing prefixes (e.g. ``converse/``) that affect + both the resolved model name and whether ``get_provider_anthropic_messages_config`` + returns a native config — plain prefix-stripping produces wrong answers for + those cases. Left as a potential optimisation for a follow-up. """ from litellm.types.utils import LlmProviders