diff --git a/docs/my-website/docs/providers/anthropic.md b/docs/my-website/docs/providers/anthropic.md
index 2227b7a6b51..2a7804bfda0 100644
--- a/docs/my-website/docs/providers/anthropic.md
+++ b/docs/my-website/docs/providers/anthropic.md
@@ -225,22 +225,336 @@ print(response)
| claude-instant-1.2 | `completion('claude-instant-1.2', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
| claude-instant-1 | `completion('claude-instant-1', messages)` | `os.environ['ANTHROPIC_API_KEY']` |
-## Passing Extra Headers to Anthropic API
+## **Prompt Caching**
-Pass `extra_headers: dict` to `litellm.completion`
+Use Anthropic Prompt Caching
-```python
-from litellm import completion
-messages = [{"role": "user", "content": "What is Anthropic?"}]
-response = completion(
- model="claude-3-5-sonnet-20240620",
- messages=messages,
- extra_headers={"anthropic-beta": "max-tokens-3-5-sonnet-2024-07-15"}
+
+[Relevant Anthropic API Docs](https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching)
+
+### Caching - Large Context Caching
+
+This example demonstrates basic Prompt Caching usage, caching the full text of the legal agreement as a prefix while keeping the user instruction uncached.
+
+
+
+
+```python
+response = await litellm.acompletion(
+ model="anthropic/claude-3-5-sonnet-20240620",
+ messages=[
+ {
+ "role": "system",
+ "content": [
+ {
+ "type": "text",
+ "text": "You are an AI assistant tasked with analyzing legal documents.",
+ },
+ {
+ "type": "text",
+ "text": "Here is the full text of a complex legal agreement",
+ "cache_control": {"type": "ephemeral"},
+ },
+ ],
+ },
+ {
+ "role": "user",
+ "content": "what are the key terms and conditions in this agreement?",
+ },
+ ],
+ extra_headers={
+ "anthropic-version": "2023-06-01",
+ "anthropic-beta": "prompt-caching-2024-07-31",
+ },
+)
+
+```
+
+
+
+:::info
+
+LiteLLM Proxy is OpenAI compatible
+
+This is an example using the OpenAI Python SDK sending a request to LiteLLM Proxy
+
+Assuming you have a model=`anthropic/claude-3-5-sonnet-20240620` on the [litellm proxy config.yaml](#usage-with-litellm-proxy)
+
+:::
+
+```python
+import openai
+client = openai.AsyncOpenAI(
+ api_key="anything", # litellm proxy api key
+ base_url="http://0.0.0.0:4000" # litellm proxy base url
+)
+
+
+response = await client.chat.completions.create(
+ model="anthropic/claude-3-5-sonnet-20240620",
+ messages=[
+ {
+ "role": "system",
+ "content": [
+ {
+ "type": "text",
+ "text": "You are an AI assistant tasked with analyzing legal documents.",
+ },
+ {
+ "type": "text",
+ "text": "Here is the full text of a complex legal agreement",
+ "cache_control": {"type": "ephemeral"},
+ },
+ ],
+ },
+ {
+ "role": "user",
+ "content": "what are the key terms and conditions in this agreement?",
+ },
+ ],
+ extra_headers={
+ "anthropic-version": "2023-06-01",
+ "anthropic-beta": "prompt-caching-2024-07-31",
+ },
+)
+
+```
+
+
+
+
+### Caching - Tools definitions
+
+In this example, we demonstrate caching tool definitions.
+
+The cache_control parameter is placed on the final tool
+
+
+
+
+```python
+import litellm
+
+response = await litellm.acompletion(
+ model="anthropic/claude-3-5-sonnet-20240620",
+ messages = [{"role": "user", "content": "What's the weather like in Boston today?"}]
+ tools = [
+ {
+ "type": "function",
+ "function": {
+ "name": "get_current_weather",
+ "description": "Get the current weather in a given location",
+ "parameters": {
+ "type": "object",
+ "properties": {
+ "location": {
+ "type": "string",
+ "description": "The city and state, e.g. San Francisco, CA",
+ },
+ "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]},
+ },
+ "required": ["location"],
+ },
+ "cache_control": {"type": "ephemeral"}
+ },
+ }
+ ],
+ extra_headers={
+ "anthropic-version": "2023-06-01",
+ "anthropic-beta": "prompt-caching-2024-07-31",
+ },
)
```
-## Advanced
+
+
-## Usage - Function Calling
+:::info
+
+LiteLLM Proxy is OpenAI compatible
+
+This is an example using the OpenAI Python SDK sending a request to LiteLLM Proxy
+
+Assuming you have a model=`anthropic/claude-3-5-sonnet-20240620` on the [litellm proxy config.yaml](#usage-with-litellm-proxy)
+
+:::
+
+```python
+import openai
+client = openai.AsyncOpenAI(
+ api_key="anything", # litellm proxy api key
+ base_url="http://0.0.0.0:4000" # litellm proxy base url
+)
+
+response = await client.chat.completions.create(
+ model="anthropic/claude-3-5-sonnet-20240620",
+ messages = [{"role": "user", "content": "What's the weather like in Boston today?"}]
+ tools = [
+ {
+ "type": "function",
+ "function": {
+ "name": "get_current_weather",
+ "description": "Get the current weather in a given location",
+ "parameters": {
+ "type": "object",
+ "properties": {
+ "location": {
+ "type": "string",
+ "description": "The city and state, e.g. San Francisco, CA",
+ },
+ "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]},
+ },
+ "required": ["location"],
+ },
+ "cache_control": {"type": "ephemeral"}
+ },
+ }
+ ],
+ extra_headers={
+ "anthropic-version": "2023-06-01",
+ "anthropic-beta": "prompt-caching-2024-07-31",
+ },
+)
+```
+
+
+
+
+
+### Caching - Continuing Multi-Turn Convo
+
+In this example, we demonstrate how to use Prompt Caching in a multi-turn conversation.
+
+The cache_control parameter is placed on the system message to designate it as part of the static prefix.
+
+The conversation history (previous messages) is included in the messages array. The final turn is marked with cache-control, for continuing in followups. The second-to-last user message is marked for caching with the cache_control parameter, so that this checkpoint can read from the previous cache.
+
+
+
+
+```python
+import litellm
+
+response = await litellm.acompletion(
+ model="anthropic/claude-3-5-sonnet-20240620",
+ messages=[
+ # System Message
+ {
+ "role": "system",
+ "content": [
+ {
+ "type": "text",
+ "text": "Here is the full text of a complex legal agreement"
+ * 400,
+ "cache_control": {"type": "ephemeral"},
+ }
+ ],
+ },
+ # marked for caching with the cache_control parameter, so that this checkpoint can read from the previous cache.
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What are the key terms and conditions in this agreement?",
+ "cache_control": {"type": "ephemeral"},
+ }
+ ],
+ },
+ {
+ "role": "assistant",
+ "content": "Certainly! the key terms and conditions are the following: the contract is 1 year long for $10/mo",
+ },
+ # The final turn is marked with cache-control, for continuing in followups.
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What are the key terms and conditions in this agreement?",
+ "cache_control": {"type": "ephemeral"},
+ }
+ ],
+ },
+ ],
+ extra_headers={
+ "anthropic-version": "2023-06-01",
+ "anthropic-beta": "prompt-caching-2024-07-31",
+ },
+)
+```
+
+
+
+:::info
+
+LiteLLM Proxy is OpenAI compatible
+
+This is an example using the OpenAI Python SDK sending a request to LiteLLM Proxy
+
+Assuming you have a model=`anthropic/claude-3-5-sonnet-20240620` on the [litellm proxy config.yaml](#usage-with-litellm-proxy)
+
+:::
+
+```python
+import openai
+client = openai.AsyncOpenAI(
+ api_key="anything", # litellm proxy api key
+ base_url="http://0.0.0.0:4000" # litellm proxy base url
+)
+
+response = await client.chat.completions.create(
+ model="anthropic/claude-3-5-sonnet-20240620",
+ messages=[
+ # System Message
+ {
+ "role": "system",
+ "content": [
+ {
+ "type": "text",
+ "text": "Here is the full text of a complex legal agreement"
+ * 400,
+ "cache_control": {"type": "ephemeral"},
+ }
+ ],
+ },
+ # marked for caching with the cache_control parameter, so that this checkpoint can read from the previous cache.
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What are the key terms and conditions in this agreement?",
+ "cache_control": {"type": "ephemeral"},
+ }
+ ],
+ },
+ {
+ "role": "assistant",
+ "content": "Certainly! the key terms and conditions are the following: the contract is 1 year long for $10/mo",
+ },
+ # The final turn is marked with cache-control, for continuing in followups.
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What are the key terms and conditions in this agreement?",
+ "cache_control": {"type": "ephemeral"},
+ }
+ ],
+ },
+ ],
+ extra_headers={
+ "anthropic-version": "2023-06-01",
+ "anthropic-beta": "prompt-caching-2024-07-31",
+ },
+)
+```
+
+
+
+
+## **Function/Tool Calling**
:::info
@@ -429,6 +743,20 @@ resp = litellm.completion(
print(f"\nResponse: {resp}")
```
+## **Passing Extra Headers to Anthropic API**
+
+Pass `extra_headers: dict` to `litellm.completion`
+
+```python
+from litellm import completion
+messages = [{"role": "user", "content": "What is Anthropic?"}]
+response = completion(
+ model="claude-3-5-sonnet-20240620",
+ messages=messages,
+ extra_headers={"anthropic-beta": "max-tokens-3-5-sonnet-2024-07-15"}
+)
+```
+
## Usage - "Assistant Pre-fill"
You can "put words in Claude's mouth" by including an `assistant` role message as the last item in the `messages` array.
diff --git a/litellm/llms/anthropic.py b/litellm/llms/anthropic.py
index 6f05aa226e0..cf58163461a 100644
--- a/litellm/llms/anthropic.py
+++ b/litellm/llms/anthropic.py
@@ -35,6 +35,7 @@ from litellm.types.llms.anthropic import (
AnthropicResponseContentBlockText,
AnthropicResponseContentBlockToolUse,
AnthropicResponseUsageBlock,
+ AnthropicSystemMessageContent,
ContentBlockDelta,
ContentBlockStart,
ContentBlockStop,
@@ -759,6 +760,7 @@ class AnthropicChatCompletion(BaseLLM):
## CALCULATING USAGE
prompt_tokens = completion_response["usage"]["input_tokens"]
completion_tokens = completion_response["usage"]["output_tokens"]
+ _usage = completion_response["usage"]
total_tokens = prompt_tokens + completion_tokens
model_response.created = int(time.time())
@@ -768,6 +770,11 @@ class AnthropicChatCompletion(BaseLLM):
completion_tokens=completion_tokens,
total_tokens=total_tokens,
)
+
+ if "cache_creation_input_tokens" in _usage:
+ usage["cache_creation_input_tokens"] = _usage["cache_creation_input_tokens"]
+ if "cache_read_input_tokens" in _usage:
+ usage["cache_read_input_tokens"] = _usage["cache_read_input_tokens"]
setattr(model_response, "usage", usage) # type: ignore
return model_response
@@ -901,6 +908,7 @@ class AnthropicChatCompletion(BaseLLM):
# Separate system prompt from rest of message
system_prompt_indices = []
system_prompt = ""
+ anthropic_system_message_list = None
for idx, message in enumerate(messages):
if message["role"] == "system":
valid_content: bool = False
@@ -908,8 +916,23 @@ class AnthropicChatCompletion(BaseLLM):
system_prompt += message["content"]
valid_content = True
elif isinstance(message["content"], list):
- for content in message["content"]:
- system_prompt += content.get("text", "")
+ for _content in message["content"]:
+ anthropic_system_message_content = (
+ AnthropicSystemMessageContent(
+ type=_content.get("type"),
+ text=_content.get("text"),
+ )
+ )
+ if "cache_control" in _content:
+ anthropic_system_message_content["cache_control"] = (
+ _content["cache_control"]
+ )
+
+ if anthropic_system_message_list is None:
+ anthropic_system_message_list = []
+ anthropic_system_message_list.append(
+ anthropic_system_message_content
+ )
valid_content = True
if valid_content:
@@ -919,6 +942,10 @@ class AnthropicChatCompletion(BaseLLM):
messages.pop(idx)
if len(system_prompt) > 0:
optional_params["system"] = system_prompt
+
+ # Handling anthropic API Prompt Caching
+ if anthropic_system_message_list is not None:
+ optional_params["system"] = anthropic_system_message_list
# Format rest of message according to anthropic guidelines
try:
messages = prompt_factory(
@@ -954,6 +981,8 @@ class AnthropicChatCompletion(BaseLLM):
else: # assume openai tool call
new_tool = tool["function"]
new_tool["input_schema"] = new_tool.pop("parameters") # rename key
+ if "cache_control" in tool:
+ new_tool["cache_control"] = tool["cache_control"]
anthropic_tools.append(new_tool)
optional_params["tools"] = anthropic_tools
diff --git a/litellm/llms/prompt_templates/factory.py b/litellm/llms/prompt_templates/factory.py
index 4e552b3b078..f81515e98d6 100644
--- a/litellm/llms/prompt_templates/factory.py
+++ b/litellm/llms/prompt_templates/factory.py
@@ -1224,6 +1224,19 @@ def convert_to_anthropic_tool_invoke(
return anthropic_tool_invoke
+def add_cache_control_to_content(
+ anthropic_content_element: Union[
+ dict, AnthropicMessagesImageParam, AnthropicMessagesTextParam
+ ],
+ orignal_content_element: dict,
+):
+ if "cache_control" in orignal_content_element:
+ anthropic_content_element["cache_control"] = orignal_content_element[
+ "cache_control"
+ ]
+ return anthropic_content_element
+
+
def anthropic_messages_pt(
messages: list,
model: str,
@@ -1264,18 +1277,31 @@ def anthropic_messages_pt(
image_chunk = convert_to_anthropic_image_obj(
m["image_url"]["url"]
)
- user_content.append(
- AnthropicMessagesImageParam(
- type="image",
- source=AnthropicImageParamSource(
- type="base64",
- media_type=image_chunk["media_type"],
- data=image_chunk["data"],
- ),
- )
+
+ _anthropic_content_element = AnthropicMessagesImageParam(
+ type="image",
+ source=AnthropicImageParamSource(
+ type="base64",
+ media_type=image_chunk["media_type"],
+ data=image_chunk["data"],
+ ),
)
+
+ anthropic_content_element = add_cache_control_to_content(
+ anthropic_content_element=_anthropic_content_element,
+ orignal_content_element=m,
+ )
+ user_content.append(anthropic_content_element)
elif m.get("type", "") == "text":
- user_content.append({"type": "text", "text": m["text"]})
+ _anthropic_text_content_element = {
+ "type": "text",
+ "text": m["text"],
+ }
+ anthropic_content_element = add_cache_control_to_content(
+ anthropic_content_element=_anthropic_text_content_element,
+ orignal_content_element=m,
+ )
+ user_content.append(anthropic_content_element)
elif (
messages[msg_i]["role"] == "tool"
or messages[msg_i]["role"] == "function"
@@ -1306,6 +1332,10 @@ def anthropic_messages_pt(
anthropic_message = AnthropicMessagesTextParam(
type="text", text=m.get("text")
)
+ anthropic_message = add_cache_control_to_content(
+ anthropic_content_element=anthropic_message,
+ orignal_content_element=m,
+ )
assistant_content.append(anthropic_message)
elif (
"content" in messages[msg_i]
@@ -1313,9 +1343,17 @@ def anthropic_messages_pt(
and len(messages[msg_i]["content"])
> 0 # don't pass empty text blocks. anthropic api raises errors.
):
- assistant_content.append(
- {"type": "text", "text": messages[msg_i]["content"]}
+
+ _anthropic_text_content_element = {
+ "type": "text",
+ "text": messages[msg_i]["content"],
+ }
+
+ anthropic_content_element = add_cache_control_to_content(
+ anthropic_content_element=_anthropic_text_content_element,
+ orignal_content_element=messages[msg_i],
)
+ assistant_content.append(anthropic_content_element)
if messages[msg_i].get(
"tool_calls", []
diff --git a/litellm/tests/test_anthropic_prompt_caching.py b/litellm/tests/test_anthropic_prompt_caching.py
new file mode 100644
index 00000000000..87bfc23f841
--- /dev/null
+++ b/litellm/tests/test_anthropic_prompt_caching.py
@@ -0,0 +1,321 @@
+import json
+import os
+import sys
+import traceback
+
+from dotenv import load_dotenv
+
+load_dotenv()
+import io
+import os
+
+sys.path.insert(
+ 0, os.path.abspath("../..")
+) # Adds the parent directory to the system path
+
+import os
+from unittest.mock import AsyncMock, MagicMock, patch
+
+import pytest
+
+import litellm
+from litellm import RateLimitError, Timeout, completion, completion_cost, embedding
+from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
+from litellm.llms.prompt_templates.factory import anthropic_messages_pt
+
+# litellm.num_retries =3
+litellm.cache = None
+litellm.success_callback = []
+user_message = "Write a short poem about the sky"
+messages = [{"content": user_message, "role": "user"}]
+
+
+def logger_fn(user_model_dict):
+ print(f"user_model_dict: {user_model_dict}")
+
+
+@pytest.fixture(autouse=True)
+def reset_callbacks():
+ print("\npytest fixture - resetting callbacks")
+ litellm.success_callback = []
+ litellm._async_success_callback = []
+ litellm.failure_callback = []
+ litellm.callbacks = []
+
+
+@pytest.mark.asyncio
+async def test_litellm_anthropic_prompt_caching_tools():
+ # Arrange: Set up the MagicMock for the httpx.AsyncClient
+ mock_response = AsyncMock()
+
+ def return_val():
+ return {
+ "id": "msg_01XFDUDYJgAACzvnptvVoYEL",
+ "type": "message",
+ "role": "assistant",
+ "content": [{"type": "text", "text": "Hello!"}],
+ "model": "claude-3-5-sonnet-20240620",
+ "stop_reason": "end_turn",
+ "stop_sequence": None,
+ "usage": {"input_tokens": 12, "output_tokens": 6},
+ }
+
+ mock_response.json = return_val
+
+ litellm.set_verbose = True
+ with patch(
+ "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
+ return_value=mock_response,
+ ) as mock_post:
+ # Act: Call the litellm.acompletion function
+ response = await litellm.acompletion(
+ api_key="mock_api_key",
+ model="anthropic/claude-3-5-sonnet-20240620",
+ messages=[
+ {"role": "user", "content": "What's the weather like in Boston today?"}
+ ],
+ tools=[
+ {
+ "type": "function",
+ "function": {
+ "name": "get_current_weather",
+ "description": "Get the current weather in a given location",
+ "parameters": {
+ "type": "object",
+ "properties": {
+ "location": {
+ "type": "string",
+ "description": "The city and state, e.g. San Francisco, CA",
+ },
+ "unit": {
+ "type": "string",
+ "enum": ["celsius", "fahrenheit"],
+ },
+ },
+ "required": ["location"],
+ },
+ "cache_control": {"type": "ephemeral"},
+ },
+ }
+ ],
+ extra_headers={
+ "anthropic-version": "2023-06-01",
+ "anthropic-beta": "prompt-caching-2024-07-31",
+ },
+ )
+
+ # Print what was called on the mock
+ print("call args=", mock_post.call_args)
+
+ expected_url = "https://api.anthropic.com/v1/messages"
+ expected_headers = {
+ "accept": "application/json",
+ "content-type": "application/json",
+ "anthropic-version": "2023-06-01",
+ "anthropic-beta": "prompt-caching-2024-07-31",
+ "x-api-key": "mock_api_key",
+ }
+
+ expected_json = {
+ "messages": [
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What's the weather like in Boston today?",
+ }
+ ],
+ }
+ ],
+ "tools": [
+ {
+ "name": "get_current_weather",
+ "description": "Get the current weather in a given location",
+ "cache_control": {"type": "ephemeral"},
+ "input_schema": {
+ "type": "object",
+ "properties": {
+ "location": {
+ "type": "string",
+ "description": "The city and state, e.g. San Francisco, CA",
+ },
+ "unit": {
+ "type": "string",
+ "enum": ["celsius", "fahrenheit"],
+ },
+ },
+ "required": ["location"],
+ },
+ }
+ ],
+ "max_tokens": 4096,
+ "model": "claude-3-5-sonnet-20240620",
+ }
+
+ mock_post.assert_called_once_with(
+ expected_url, json=expected_json, headers=expected_headers, timeout=600.0
+ )
+
+
+@pytest.mark.asyncio()
+async def test_anthropic_api_prompt_caching_basic():
+ litellm.set_verbose = True
+ response = await litellm.acompletion(
+ model="anthropic/claude-3-5-sonnet-20240620",
+ messages=[
+ # System Message
+ {
+ "role": "system",
+ "content": [
+ {
+ "type": "text",
+ "text": "Here is the full text of a complex legal agreement"
+ * 400,
+ "cache_control": {"type": "ephemeral"},
+ }
+ ],
+ },
+ # marked for caching with the cache_control parameter, so that this checkpoint can read from the previous cache.
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What are the key terms and conditions in this agreement?",
+ "cache_control": {"type": "ephemeral"},
+ }
+ ],
+ },
+ {
+ "role": "assistant",
+ "content": "Certainly! the key terms and conditions are the following: the contract is 1 year long for $10/mo",
+ },
+ # The final turn is marked with cache-control, for continuing in followups.
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What are the key terms and conditions in this agreement?",
+ "cache_control": {"type": "ephemeral"},
+ }
+ ],
+ },
+ ],
+ temperature=0.2,
+ max_tokens=10,
+ extra_headers={
+ "anthropic-version": "2023-06-01",
+ "anthropic-beta": "prompt-caching-2024-07-31",
+ },
+ )
+
+ print("response=", response)
+
+ assert "cache_read_input_tokens" in response.usage
+ assert "cache_creation_input_tokens" in response.usage
+
+ # Assert either a cache entry was created or cache was read - changes depending on the anthropic api ttl
+ assert (response.usage.cache_read_input_tokens > 0) or (
+ response.usage.cache_creation_input_tokens > 0
+ )
+
+
+@pytest.mark.asyncio
+async def test_litellm_anthropic_prompt_caching_system():
+ # https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching#prompt-caching-examples
+ # LArge Context Caching Example
+ mock_response = AsyncMock()
+
+ def return_val():
+ return {
+ "id": "msg_01XFDUDYJgAACzvnptvVoYEL",
+ "type": "message",
+ "role": "assistant",
+ "content": [{"type": "text", "text": "Hello!"}],
+ "model": "claude-3-5-sonnet-20240620",
+ "stop_reason": "end_turn",
+ "stop_sequence": None,
+ "usage": {"input_tokens": 12, "output_tokens": 6},
+ }
+
+ mock_response.json = return_val
+
+ litellm.set_verbose = True
+ with patch(
+ "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
+ return_value=mock_response,
+ ) as mock_post:
+ # Act: Call the litellm.acompletion function
+ response = await litellm.acompletion(
+ api_key="mock_api_key",
+ model="anthropic/claude-3-5-sonnet-20240620",
+ messages=[
+ {
+ "role": "system",
+ "content": [
+ {
+ "type": "text",
+ "text": "You are an AI assistant tasked with analyzing legal documents.",
+ },
+ {
+ "type": "text",
+ "text": "Here is the full text of a complex legal agreement",
+ "cache_control": {"type": "ephemeral"},
+ },
+ ],
+ },
+ {
+ "role": "user",
+ "content": "what are the key terms and conditions in this agreement?",
+ },
+ ],
+ extra_headers={
+ "anthropic-version": "2023-06-01",
+ "anthropic-beta": "prompt-caching-2024-07-31",
+ },
+ )
+
+ # Print what was called on the mock
+ print("call args=", mock_post.call_args)
+
+ expected_url = "https://api.anthropic.com/v1/messages"
+ expected_headers = {
+ "accept": "application/json",
+ "content-type": "application/json",
+ "anthropic-version": "2023-06-01",
+ "anthropic-beta": "prompt-caching-2024-07-31",
+ "x-api-key": "mock_api_key",
+ }
+
+ expected_json = {
+ "system": [
+ {
+ "type": "text",
+ "text": "You are an AI assistant tasked with analyzing legal documents.",
+ },
+ {
+ "type": "text",
+ "text": "Here is the full text of a complex legal agreement",
+ "cache_control": {"type": "ephemeral"},
+ },
+ ],
+ "messages": [
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "what are the key terms and conditions in this agreement?",
+ }
+ ],
+ }
+ ],
+ "max_tokens": 4096,
+ "model": "claude-3-5-sonnet-20240620",
+ }
+
+ mock_post.assert_called_once_with(
+ expected_url, json=expected_json, headers=expected_headers, timeout=600.0
+ )
diff --git a/litellm/tests/test_completion.py b/litellm/tests/test_completion.py
index 83031aba085..796dd3da34a 100644
--- a/litellm/tests/test_completion.py
+++ b/litellm/tests/test_completion.py
@@ -14,7 +14,7 @@ sys.path.insert(
) # Adds the parent directory to the system path
import os
-from unittest.mock import MagicMock, patch
+from unittest.mock import AsyncMock, MagicMock, patch
import pytest
@@ -3449,7 +3449,6 @@ def response_format_tests(response: litellm.ModelResponse):
assert isinstance(response.usage.total_tokens, int) # type: ignore
-@pytest.mark.parametrize("sync_mode", [True, False])
@pytest.mark.parametrize(
"model",
[
@@ -3463,6 +3462,7 @@ def response_format_tests(response: litellm.ModelResponse):
"cohere.command-text-v14",
],
)
+@pytest.mark.parametrize("sync_mode", [True, False])
@pytest.mark.asyncio
async def test_completion_bedrock_httpx_models(sync_mode, model):
litellm.set_verbose = True
diff --git a/litellm/tests/test_prompt_factory.py b/litellm/tests/test_prompt_factory.py
index f7a715a2204..93e92a7926a 100644
--- a/litellm/tests/test_prompt_factory.py
+++ b/litellm/tests/test_prompt_factory.py
@@ -260,3 +260,56 @@ def test_anthropic_messages_tool_call():
translated_messages[-1]["content"][0]["tool_use_id"]
== "bc8cb4b6-88c4-4138-8993-3a9d9cd51656"
)
+
+
+def test_anthropic_cache_controls_pt():
+ "see anthropic docs for this: https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching#continuing-a-multi-turn-conversation"
+ messages = [
+ # marked for caching with the cache_control parameter, so that this checkpoint can read from the previous cache.
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What are the key terms and conditions in this agreement?",
+ "cache_control": {"type": "ephemeral"},
+ }
+ ],
+ },
+ {
+ "role": "assistant",
+ "content": "Certainly! the key terms and conditions are the following: the contract is 1 year long for $10/mo",
+ },
+ # The final turn is marked with cache-control, for continuing in followups.
+ {
+ "role": "user",
+ "content": [
+ {
+ "type": "text",
+ "text": "What are the key terms and conditions in this agreement?",
+ "cache_control": {"type": "ephemeral"},
+ }
+ ],
+ },
+ {
+ "role": "assistant",
+ "content": "Certainly! the key terms and conditions are the following: the contract is 1 year long for $10/mo",
+ "cache_control": {"type": "ephemeral"},
+ },
+ ]
+
+ translated_messages = anthropic_messages_pt(
+ messages, model="claude-3-5-sonnet-20240620", llm_provider="anthropic"
+ )
+
+ for i, msg in enumerate(translated_messages):
+ if i == 0:
+ assert msg["content"][0]["cache_control"] == {"type": "ephemeral"}
+ elif i == 1:
+ assert "cache_controls" not in msg["content"][0]
+ elif i == 2:
+ assert msg["content"][0]["cache_control"] == {"type": "ephemeral"}
+ elif i == 3:
+ assert msg["content"][0]["cache_control"] == {"type": "ephemeral"}
+
+ print("translated_messages: ", translated_messages)
diff --git a/litellm/types/llms/anthropic.py b/litellm/types/llms/anthropic.py
index 36bcb6cc736..f14aa20c733 100644
--- a/litellm/types/llms/anthropic.py
+++ b/litellm/types/llms/anthropic.py
@@ -15,9 +15,10 @@ class AnthropicMessagesTool(TypedDict, total=False):
input_schema: Required[dict]
-class AnthropicMessagesTextParam(TypedDict):
+class AnthropicMessagesTextParam(TypedDict, total=False):
type: Literal["text"]
text: str
+ cache_control: Optional[dict]
class AnthropicMessagesToolUseParam(TypedDict):
@@ -54,9 +55,10 @@ class AnthropicImageParamSource(TypedDict):
data: str
-class AnthropicMessagesImageParam(TypedDict):
+class AnthropicMessagesImageParam(TypedDict, total=False):
type: Literal["image"]
source: AnthropicImageParamSource
+ cache_control: Optional[dict]
class AnthropicMessagesToolResultContent(TypedDict):
@@ -92,6 +94,12 @@ class AnthropicMetadata(TypedDict, total=False):
user_id: str
+class AnthropicSystemMessageContent(TypedDict, total=False):
+ type: str
+ text: str
+ cache_control: Optional[dict]
+
+
class AnthropicMessagesRequest(TypedDict, total=False):
model: Required[str]
messages: Required[
@@ -106,7 +114,7 @@ class AnthropicMessagesRequest(TypedDict, total=False):
metadata: AnthropicMetadata
stop_sequences: List[str]
stream: bool
- system: str
+ system: Union[str, List]
temperature: float
tool_choice: AnthropicMessagesToolChoice
tools: List[AnthropicMessagesTool]
diff --git a/litellm/types/llms/openai.py b/litellm/types/llms/openai.py
index 0d67d5d602c..5d2c416f9cd 100644
--- a/litellm/types/llms/openai.py
+++ b/litellm/types/llms/openai.py
@@ -361,7 +361,7 @@ class ChatCompletionToolMessage(TypedDict):
class ChatCompletionSystemMessage(TypedDict, total=False):
role: Required[Literal["system"]]
- content: Required[str]
+ content: Required[Union[str, List]]
name: str