From cd33993f4906207f78ad621c6ae9112f0b6d924f Mon Sep 17 00:00:00 2001 From: Krrish Dholakia Date: Tue, 17 Feb 2026 22:41:09 -0800 Subject: [PATCH] refactor: remove cc cost optimization template can't deliver on promise right now --- .../guardrail_hooks/claude_code/__init__.py | 24 +---- .../claude_code/prompt_cache.py | 91 ------------------- litellm/types/guardrails.py | 1 - policy_templates.json | 44 --------- 4 files changed, 2 insertions(+), 158 deletions(-) delete mode 100644 litellm/proxy/guardrails/guardrail_hooks/claude_code/prompt_cache.py diff --git a/litellm/proxy/guardrails/guardrail_hooks/claude_code/__init__.py b/litellm/proxy/guardrails/guardrail_hooks/claude_code/__init__.py index d10dda9b5f3..028cd17381a 100644 --- a/litellm/proxy/guardrails/guardrail_hooks/claude_code/__init__.py +++ b/litellm/proxy/guardrails/guardrail_hooks/claude_code/__init__.py @@ -1,10 +1,9 @@ """ Claude Code guardrail integrations for LiteLLM. -Two focused policy-enforcement guardrails for Claude Code deployments: +Focused policy-enforcement guardrail for Claude Code deployments: -1. claude_code_prompt_cache — auto-inject Anthropic prompt-caching headers -2. claude_code_block_expensive_flags — block expensive API flags (fast mode, etc.) +- claude_code_block_expensive_flags — block expensive API flags (fast mode, etc.) Hosted tool blocking is handled by the provider-agnostic 'block_hosted_tools' guardrail (see guardrail_hooks/block_hosted_tools/). @@ -16,7 +15,6 @@ import litellm from litellm.types.guardrails import SupportedGuardrailIntegrations from .block_expensive_flags import ClaudeCodeBlockExpensiveFlagsGuardrail -from .prompt_cache import ClaudeCodePromptCacheGuardrail if TYPE_CHECKING: from litellm.types.guardrails import Guardrail, LitellmParams @@ -27,21 +25,6 @@ if TYPE_CHECKING: # ------------------------------------------------------------------ # -def _init_prompt_cache( - litellm_params: "LitellmParams", guardrail: "Guardrail" -) -> ClaudeCodePromptCacheGuardrail: - guardrail_name = guardrail.get("guardrail_name") - if not guardrail_name: - raise ValueError("ClaudeCodePromptCacheGuardrail requires a guardrail_name") - instance = ClaudeCodePromptCacheGuardrail( - guardrail_name=guardrail_name, - event_hook=litellm_params.mode, - default_on=litellm_params.default_on, - ) - litellm.logging_callback_manager.add_litellm_callback(instance) - return instance - - def _init_block_expensive_flags( litellm_params: "LitellmParams", guardrail: "Guardrail" ) -> ClaudeCodeBlockExpensiveFlagsGuardrail: @@ -62,16 +45,13 @@ def _init_block_expensive_flags( # ------------------------------------------------------------------ # guardrail_initializer_registry = { - SupportedGuardrailIntegrations.CLAUDE_CODE_PROMPT_CACHE.value: _init_prompt_cache, SupportedGuardrailIntegrations.CLAUDE_CODE_BLOCK_EXPENSIVE_FLAGS.value: _init_block_expensive_flags, } guardrail_class_registry = { - SupportedGuardrailIntegrations.CLAUDE_CODE_PROMPT_CACHE.value: ClaudeCodePromptCacheGuardrail, SupportedGuardrailIntegrations.CLAUDE_CODE_BLOCK_EXPENSIVE_FLAGS.value: ClaudeCodeBlockExpensiveFlagsGuardrail, } __all__ = [ - "ClaudeCodePromptCacheGuardrail", "ClaudeCodeBlockExpensiveFlagsGuardrail", ] diff --git a/litellm/proxy/guardrails/guardrail_hooks/claude_code/prompt_cache.py b/litellm/proxy/guardrails/guardrail_hooks/claude_code/prompt_cache.py deleted file mode 100644 index 37ea876f3cf..00000000000 --- a/litellm/proxy/guardrails/guardrail_hooks/claude_code/prompt_cache.py +++ /dev/null @@ -1,91 +0,0 @@ -""" -Claude Code - Prompt Cache Injection Guardrail - -Automatically injects cache_control: {type: ephemeral} into system messages -so Anthropic can cache the prompt prefix, reducing costs on repeated calls. - -Only runs when the request targets an Anthropic API model. Uses the existing -AnthropicCacheControlHook._safe_insert_cache_control_in_message utility so -the injection logic is not duplicated. -""" - -from typing import TYPE_CHECKING, Literal, Optional - -from litellm._logging import verbose_proxy_logger -from litellm.integrations.anthropic_cache_control_hook import \ - AnthropicCacheControlHook -from litellm.integrations.custom_guardrail import (CustomGuardrail, - log_guardrail_information) -from litellm.types.guardrails import GuardrailEventHooks -from litellm.types.llms.openai import ChatCompletionCachedContent -from litellm.types.utils import GenericGuardrailAPIInputs -from litellm.utils import get_llm_provider - -if TYPE_CHECKING: - from litellm.litellm_core_utils.litellm_logging import \ - Logging as LiteLLMLoggingObj - - -def _is_anthropic_model(model: Optional[str]) -> bool: - """Return True when the model resolves to the Anthropic provider.""" - if not model: - return False - try: - _, custom_llm_provider, _, _ = get_llm_provider(model=model) - return custom_llm_provider == "anthropic" - except Exception: - return False - - -class ClaudeCodePromptCacheGuardrail(CustomGuardrail): - """ - Guardrail that injects Anthropic prompt-caching headers into system messages. - - Targets only Anthropic API models — the provider is detected from the - model string so this guardrail is safe to apply globally without - restricting other providers. - """ - - def __init__(self, **kwargs): - if "supported_event_hooks" not in kwargs: - kwargs["supported_event_hooks"] = [GuardrailEventHooks.pre_call] - super().__init__(**kwargs) - verbose_proxy_logger.debug("ClaudeCodePromptCacheGuardrail initialized") - - @log_guardrail_information - async def apply_guardrail( - self, - inputs: GenericGuardrailAPIInputs, - request_data: dict, - input_type: Literal["request", "response"], - logging_obj: Optional["LiteLLMLoggingObj"] = None, - ) -> GenericGuardrailAPIInputs: - if input_type != "request": - return inputs - - model: Optional[str] = request_data.get("model") or inputs.get("model") # type: ignore[assignment] - if not _is_anthropic_model(model): - verbose_proxy_logger.debug( - f"ClaudeCodePromptCacheGuardrail: skipping non-Anthropic model '{model}'" - ) - return inputs - - messages = request_data.get("messages") or [] - if not messages: - return inputs - - control = ChatCompletionCachedContent(type="ephemeral") - modified = [] - for msg in messages: - if msg.get("role") == "system": - msg = AnthropicCacheControlHook._safe_insert_cache_control_in_message( - message=msg, # type: ignore[arg-type] - control=control, - ) - modified.append(msg) - - request_data["messages"] = modified - verbose_proxy_logger.debug( - "ClaudeCodePromptCacheGuardrail: injected cache_control into system messages" - ) - return inputs diff --git a/litellm/types/guardrails.py b/litellm/types/guardrails.py index c43df4f6b83..5bd8ab5376f 100644 --- a/litellm/types/guardrails.py +++ b/litellm/types/guardrails.py @@ -65,7 +65,6 @@ class SupportedGuardrailIntegrations(Enum): QUALIFIRE = "qualifire" CUSTOM_CODE = "custom_code" BLOCK_HOSTED_TOOLS = "block_hosted_tools" - CLAUDE_CODE_PROMPT_CACHE = "claude_code_prompt_cache" CLAUDE_CODE_BLOCK_EXPENSIVE_FLAGS = "claude_code_block_expensive_flags" diff --git a/policy_templates.json b/policy_templates.json index 14cbdad62bf..f718074d5d2 100644 --- a/policy_templates.json +++ b/policy_templates.json @@ -1106,49 +1106,5 @@ "guardrails_add": ["mcp-security-block"], "guardrails_remove": [] } - }, - { - "id": "claude-code-cost-optimization", - "title": "Claude Code Cost Optimization", - "description": "Reduces Claude Code API spend by blocking expensive inference modes (fast/turbo, inference_geo, and extended thinking) and automatically injecting prompt caching headers into system messages to maximize cache hit rates.", - "icon": "CurrencyDollarIcon", - "iconColor": "text-green-500", - "iconBg": "bg-green-50", - "guardrails": [ - "claude-code-inject-prompt-cache", - "claude-code-block-expensive-flags" - ], - "complexity": "Low", - "guardrailDefinitions": [ - { - "guardrail_name": "claude-code-inject-prompt-cache", - "litellm_params": { - "guardrail": "claude_code_prompt_cache", - "mode": "pre_call" - }, - "guardrail_info": { - "description": "Automatically adds cache_control: {type: ephemeral} to system messages so Anthropic can cache the system prompt prefix. Only applies to Anthropic API models. Reduces cost on repeated calls that share the same system prompt." - } - }, - { - "guardrail_name": "claude-code-block-expensive-flags", - "litellm_params": { - "guardrail": "claude_code_block_expensive_flags", - "mode": "pre_call" - }, - "guardrail_info": { - "description": "Blocks expensive Anthropic API flags: fast/turbo inference (speed=fast, ~6x pricing), inference geo-routing (inference_geo, 1.1x pricing), and extended thinking (thinking.type=enabled). Also blocks Anthropic-hosted tools inherited from the hosted tools policy." - } - } - ], - "templateData": { - "policy_name": "claude-code-cost-optimization", - "description": "Reduces Claude Code API spend by blocking fast mode, inference_geo, extended thinking, and Anthropic-hosted tools, while auto-injecting prompt caching into system messages.", - "guardrails_add": [ - "claude-code-inject-prompt-cache", - "claude-code-block-expensive-flags" - ], - "guardrails_remove": [] - } } ]