test: move offline anthropic prompt caching tests to tests/unit (#45616)

The two offline tests in tests/local_testing/test_anthropic_prompt_caching.py failed on main because
#24071 started passing logging_obj to client.post, so assert_called_with on the patched
AsyncHTTPHandler.post no longer matched. The request bodies themselves were still correct

Both now live in tests/unit/llms/anthropic/chat and assert the body and headers that reach the
wire through respx instead of patching our own HTTP handler. The coverage allowlist entry keeps the
five tests that still need provider credentials
This commit is contained in:
yuneng-jiang 2026-10-09 10:53:08 -07:00 • committed by GitHub
parent 61e5f2dd3c
commit 6afdf482de
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
3 changed files with 112 additions and 219 deletions

View file

@ -38,10 +38,8 @@ test_paths:
Mixed file that no job has ever run in full. Every CircleCI job that globs
tests/local_testing deselects it by name ("caching") or keeps only another keyword, so only
its router test ran, and that test now lives in tests/unit/router_utils/pre_call_checks.
Five of the seven left need ANTHROPIC_API_KEY or Vertex credentials. The other two,
test_litellm_anthropic_prompt_caching_tools and test_litellm_anthropic_prompt_caching_system,
are offline mocks that already fail on main against a stale expected request body; they need
that fixed before they can move to tests/unit
Its two offline tests now live in tests/unit/llms/anthropic/chat, and the five left all need
ANTHROPIC_API_KEY or Vertex credentials
paths:
- tests/local_testing/test_anthropic_prompt_caching.py
- reason: >-

View file

@ -36,123 +36,6 @@ def reset_callbacks():
litellm.callbacks = []
@pytest.mark.asyncio
async def test_litellm_anthropic_prompt_caching_tools():
# Arrange: Set up the MagicMock for the httpx.AsyncClient
mock_response = AsyncMock()
def return_val():
return {
"id": "msg_01XFDUDYJgAACzvnptvVoYEL",
"type": "message",
"role": "assistant",
"content": [{"type": "text", "text": "Hello!"}],
"model": "claude-sonnet-4-5-20250929",
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 12, "output_tokens": 6},
}
mock_response.json = return_val
mock_response.headers = {"key": "value"}
litellm.set_verbose = True
with patch(
"litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
return_value=mock_response,
) as mock_post:
# Act: Call the litellm.acompletion function
response = await litellm.acompletion(
api_key="mock_api_key",
model="anthropic/claude-sonnet-4-5-20250929",
messages=[
{"role": "user", "content": "What's the weather like in Boston today?"}
],
tools=[
{
"type": "function",
"function": {
"name": "get_current_weather",
"description": "Get the current weather in a given location",
"parameters": {
"type": "object",
"properties": {
"location": {
"type": "string",
"description": "The city and state, e.g. San Francisco, CA",
},
"unit": {
"type": "string",
"enum": ["celsius", "fahrenheit"],
},
},
"required": ["location"],
},
"cache_control": {"type": "ephemeral"},
},
}
],
extra_headers={
"anthropic-version": "2023-06-01",
},
)
# Print what was called on the mock
print("call args=", mock_post.call_args)
expected_url = "https://api.anthropic.com/v1/messages"
# Note: anthropic-beta header for prompt-caching is no longer required
# Anthropic now supports prompt caching automatically when cache_control is used
expected_headers = {
"accept": "application/json",
"content-type": "application/json",
"anthropic-version": "2023-06-01",
"x-api-key": "mock_api_key",
}
expected_json = {
"messages": [
{
"role": "user",
"content": [
{
"type": "text",
"text": "What's the weather like in Boston today?",
}
],
}
],
"tools": [
{
"name": "get_current_weather",
"description": "Get the current weather in a given location",
"cache_control": {"type": "ephemeral"},
"input_schema": {
"type": "object",
"properties": {
"location": {
"type": "string",
"description": "The city and state, e.g. San Francisco, CA",
},
"unit": {
"type": "string",
"enum": ["celsius", "fahrenheit"],
},
},
"required": ["location"],
},
"type": "custom",
}
],
"max_tokens": 64000,
"model": "claude-sonnet-4-5-20250929",
}
mock_post.assert_called_once_with(
expected_url, json=expected_json, headers=expected_headers, timeout=600.0
)
@pytest.fixture
def anthropic_messages():
return [
@ -494,101 +377,3 @@ async def test_anthropic_api_prompt_caching_streaming():
assert (
is_cache_read_input_tokens_in_usage and is_cache_creation_input_tokens_in_usage
)
@pytest.mark.asyncio
async def test_litellm_anthropic_prompt_caching_system():
# https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching#prompt-caching-examples
# LArge Context Caching Example
mock_response = AsyncMock()
def return_val():
return {
"id": "msg_01XFDUDYJgAACzvnptvVoYEL",
"type": "message",
"role": "assistant",
"content": [{"type": "text", "text": "Hello!"}],
"model": "claude-sonnet-4-5-20250929",
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 12, "output_tokens": 6},
}
mock_response.json = return_val
mock_response.headers = {"key": "value"}
litellm.set_verbose = True
with patch(
"litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post",
return_value=mock_response,
) as mock_post:
# Act: Call the litellm.acompletion function
response = await litellm.acompletion(
api_key="mock_api_key",
model="anthropic/claude-sonnet-4-5-20250929",
messages=[
{
"role": "system",
"content": [
{
"type": "text",
"text": "You are an AI assistant tasked with analyzing legal documents.",
},
{
"type": "text",
"text": "Here is the full text of a complex legal agreement",
"cache_control": {"type": "ephemeral"},
},
],
},
{
"role": "user",
"content": "what are the key terms and conditions in this agreement?",
},
],
extra_headers={
"anthropic-version": "2023-06-01",
},
)
# Print what was called on the mock
print("call args=", mock_post.call_args)
expected_url = "https://api.anthropic.com/v1/messages"
expected_headers = {
"accept": "application/json",
"content-type": "application/json",
"anthropic-version": "2023-06-01",
"x-api-key": "mock_api_key",
}
expected_json = {
"system": [
{
"type": "text",
"text": "You are an AI assistant tasked with analyzing legal documents.",
},
{
"type": "text",
"text": "Here is the full text of a complex legal agreement",
"cache_control": {"type": "ephemeral"},
},
],
"messages": [
{
"role": "user",
"content": [
{
"type": "text",
"text": "what are the key terms and conditions in this agreement?",
}
],
}
],
"max_tokens": 64000,
"model": "claude-sonnet-4-5-20250929",
}
mock_post.assert_called_once_with(
expected_url, json=expected_json, headers=expected_headers, timeout=600.0
)

View file

@ -7968,3 +7968,113 @@ def test_calculate_usage_sums_cache_tokens_across_compaction_iterations():
assert usage.completion_tokens == 250
assert usage.prompt_tokens_details.cache_creation_tokens == 60
assert usage.prompt_tokens_details.cached_tokens == 17020
PROMPT_CACHING_MODEL: Final = "claude-sonnet-5-5"
EPHEMERAL: Final = {"type": "ephemeral"}
WEATHER_PARAMETERS: Final = {
"type": "object",
"properties": {
"location": {"type": "string", "description": "The city and state, e.g. San Francisco, CA"},
"unit": {"type": "string", "enum": ["celsius", "fahrenheit"]},
},
"required": ["location"],
}
async def _send_prompt_caching_request(
respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch, **params: object
) -> httpx.Request:
monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True")
route: Final = respx_mock.post("https://api.anthropic.com/v1/messages").mock(
return_value=httpx.Response(
200,
json={
"id": "msg_01XFDUDYJgAACzvnptvVoYEL",
"type": "message",
"role": "assistant",
"content": [{"type": "text", "text": "Hello!"}],
"model": PROMPT_CACHING_MODEL,
"stop_reason": "end_turn",
"stop_sequence": None,
"usage": {"input_tokens": 12, "output_tokens": 6},
},
)
)
await litellm.acompletion(
api_key="mock_api_key",
model=f"anthropic/{PROMPT_CACHING_MODEL}",
extra_headers={"anthropic-version": "2023-06-01"},
**params,
)
assert route.call_count == 1
request: Final = route.calls.last.request
assert request.headers["x-api-key"] == "mock_api_key"
assert request.headers["anthropic-version"] == "2023-06-01"
assert "anthropic-beta" not in request.headers
return request
async def test_litellm_anthropic_prompt_caching_tools(
respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
) -> None:
request: Final = await _send_prompt_caching_request(
respx_mock,
monkeypatch,
messages=[{"role": "user", "content": "What's the weather like in Boston today?"}],
tools=[
{
"type": "function",
"function": {
"name": "get_current_weather",
"description": "Get the current weather in a given location",
"parameters": WEATHER_PARAMETERS,
"cache_control": EPHEMERAL,
},
}
],
)
assert json.loads(request.content) == {
"model": PROMPT_CACHING_MODEL,
"messages": [
{"role": "user", "content": [{"type": "text", "text": "What's the weather like in Boston today?"}]}
],
"tools": [
{
"name": "get_current_weather",
"description": "Get the current weather in a given location",
"input_schema": WEATHER_PARAMETERS,
"type": "custom",
"cache_control": EPHEMERAL,
}
],
"max_tokens": litellm.get_max_tokens(PROMPT_CACHING_MODEL),
}
async def test_litellm_anthropic_prompt_caching_system(
respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch
) -> None:
system_blocks: Final = [
{"type": "text", "text": "You are an AI assistant tasked with analyzing legal documents."},
{"type": "text", "text": "Here is the full text of a complex legal agreement", "cache_control": EPHEMERAL},
]
request: Final = await _send_prompt_caching_request(
respx_mock,
monkeypatch,
messages=[
{"role": "system", "content": system_blocks},
{"role": "user", "content": "what are the key terms and conditions in this agreement?"},
],
)
assert json.loads(request.content) == {
"model": PROMPT_CACHING_MODEL,
"system": system_blocks,
"messages": [
{
"role": "user",
"content": [{"type": "text", "text": "what are the key terms and conditions in this agreement?"}],
}
],
"max_tokens": litellm.get_max_tokens(PROMPT_CACHING_MODEL),
}