From 6afdf482ded69e41e54c3e8fd92e058c373f6bf7 Mon Sep 17 00:00:00 2001 From: yuneng-jiang Date: Fri, 9 Oct 2026 10:53:08 -0700 Subject: [PATCH] test: move offline anthropic prompt caching tests to tests/unit (#45616) The two offline tests in tests/local_testing/test_anthropic_prompt_caching.py failed on main because #24071 started passing logging_obj to client.post, so assert_called_with on the patched AsyncHTTPHandler.post no longer matched. The request bodies themselves were still correct Both now live in tests/unit/llms/anthropic/chat and assert the body and headers that reach the wire through respx instead of patching our own HTTP handler. The coverage allowlist entry keeps the five tests that still need provider credentials --- .github/ci-coverage-allowlist.yml | 6 +- .../test_anthropic_prompt_caching.py | 215 ------------------ .../test_anthropic_chat_transformation.py | 110 +++++++++ 3 files changed, 112 insertions(+), 219 deletions(-) diff --git a/.github/ci-coverage-allowlist.yml b/.github/ci-coverage-allowlist.yml index 837832795b1..7f62ac3bf53 100644 --- a/.github/ci-coverage-allowlist.yml +++ b/.github/ci-coverage-allowlist.yml @@ -38,10 +38,8 @@ test_paths: Mixed file that no job has ever run in full. Every CircleCI job that globs tests/local_testing deselects it by name ("caching") or keeps only another keyword, so only its router test ran, and that test now lives in tests/unit/router_utils/pre_call_checks. - Five of the seven left need ANTHROPIC_API_KEY or Vertex credentials. The other two, - test_litellm_anthropic_prompt_caching_tools and test_litellm_anthropic_prompt_caching_system, - are offline mocks that already fail on main against a stale expected request body; they need - that fixed before they can move to tests/unit + Its two offline tests now live in tests/unit/llms/anthropic/chat, and the five left all need + ANTHROPIC_API_KEY or Vertex credentials paths: - tests/local_testing/test_anthropic_prompt_caching.py - reason: >- diff --git a/tests/local_testing/test_anthropic_prompt_caching.py b/tests/local_testing/test_anthropic_prompt_caching.py index 24ffe1b093e..169e6762ad0 100644 --- a/tests/local_testing/test_anthropic_prompt_caching.py +++ b/tests/local_testing/test_anthropic_prompt_caching.py @@ -36,123 +36,6 @@ def reset_callbacks(): litellm.callbacks = [] -@pytest.mark.asyncio -async def test_litellm_anthropic_prompt_caching_tools(): - # Arrange: Set up the MagicMock for the httpx.AsyncClient - mock_response = AsyncMock() - - def return_val(): - return { - "id": "msg_01XFDUDYJgAACzvnptvVoYEL", - "type": "message", - "role": "assistant", - "content": [{"type": "text", "text": "Hello!"}], - "model": "claude-sonnet-4-5-20250929", - "stop_reason": "end_turn", - "stop_sequence": None, - "usage": {"input_tokens": 12, "output_tokens": 6}, - } - - mock_response.json = return_val - mock_response.headers = {"key": "value"} - - litellm.set_verbose = True - with patch( - "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", - return_value=mock_response, - ) as mock_post: - # Act: Call the litellm.acompletion function - response = await litellm.acompletion( - api_key="mock_api_key", - model="anthropic/claude-sonnet-4-5-20250929", - messages=[ - {"role": "user", "content": "What's the weather like in Boston today?"} - ], - tools=[ - { - "type": "function", - "function": { - "name": "get_current_weather", - "description": "Get the current weather in a given location", - "parameters": { - "type": "object", - "properties": { - "location": { - "type": "string", - "description": "The city and state, e.g. San Francisco, CA", - }, - "unit": { - "type": "string", - "enum": ["celsius", "fahrenheit"], - }, - }, - "required": ["location"], - }, - "cache_control": {"type": "ephemeral"}, - }, - } - ], - extra_headers={ - "anthropic-version": "2023-06-01", - }, - ) - - # Print what was called on the mock - print("call args=", mock_post.call_args) - - expected_url = "https://api.anthropic.com/v1/messages" - # Note: anthropic-beta header for prompt-caching is no longer required - # Anthropic now supports prompt caching automatically when cache_control is used - expected_headers = { - "accept": "application/json", - "content-type": "application/json", - "anthropic-version": "2023-06-01", - "x-api-key": "mock_api_key", - } - - expected_json = { - "messages": [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": "What's the weather like in Boston today?", - } - ], - } - ], - "tools": [ - { - "name": "get_current_weather", - "description": "Get the current weather in a given location", - "cache_control": {"type": "ephemeral"}, - "input_schema": { - "type": "object", - "properties": { - "location": { - "type": "string", - "description": "The city and state, e.g. San Francisco, CA", - }, - "unit": { - "type": "string", - "enum": ["celsius", "fahrenheit"], - }, - }, - "required": ["location"], - }, - "type": "custom", - } - ], - "max_tokens": 64000, - "model": "claude-sonnet-4-5-20250929", - } - - mock_post.assert_called_once_with( - expected_url, json=expected_json, headers=expected_headers, timeout=600.0 - ) - - @pytest.fixture def anthropic_messages(): return [ @@ -494,101 +377,3 @@ async def test_anthropic_api_prompt_caching_streaming(): assert ( is_cache_read_input_tokens_in_usage and is_cache_creation_input_tokens_in_usage ) - - -@pytest.mark.asyncio -async def test_litellm_anthropic_prompt_caching_system(): - # https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching#prompt-caching-examples - # LArge Context Caching Example - mock_response = AsyncMock() - - def return_val(): - return { - "id": "msg_01XFDUDYJgAACzvnptvVoYEL", - "type": "message", - "role": "assistant", - "content": [{"type": "text", "text": "Hello!"}], - "model": "claude-sonnet-4-5-20250929", - "stop_reason": "end_turn", - "stop_sequence": None, - "usage": {"input_tokens": 12, "output_tokens": 6}, - } - - mock_response.json = return_val - mock_response.headers = {"key": "value"} - - litellm.set_verbose = True - with patch( - "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", - return_value=mock_response, - ) as mock_post: - # Act: Call the litellm.acompletion function - response = await litellm.acompletion( - api_key="mock_api_key", - model="anthropic/claude-sonnet-4-5-20250929", - messages=[ - { - "role": "system", - "content": [ - { - "type": "text", - "text": "You are an AI assistant tasked with analyzing legal documents.", - }, - { - "type": "text", - "text": "Here is the full text of a complex legal agreement", - "cache_control": {"type": "ephemeral"}, - }, - ], - }, - { - "role": "user", - "content": "what are the key terms and conditions in this agreement?", - }, - ], - extra_headers={ - "anthropic-version": "2023-06-01", - }, - ) - - # Print what was called on the mock - print("call args=", mock_post.call_args) - - expected_url = "https://api.anthropic.com/v1/messages" - expected_headers = { - "accept": "application/json", - "content-type": "application/json", - "anthropic-version": "2023-06-01", - "x-api-key": "mock_api_key", - } - - expected_json = { - "system": [ - { - "type": "text", - "text": "You are an AI assistant tasked with analyzing legal documents.", - }, - { - "type": "text", - "text": "Here is the full text of a complex legal agreement", - "cache_control": {"type": "ephemeral"}, - }, - ], - "messages": [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": "what are the key terms and conditions in this agreement?", - } - ], - } - ], - "max_tokens": 64000, - "model": "claude-sonnet-4-5-20250929", - } - - mock_post.assert_called_once_with( - expected_url, json=expected_json, headers=expected_headers, timeout=600.0 - ) diff --git a/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py index 74d68d76813..c3a3dad92c8 100644 --- a/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -7968,3 +7968,113 @@ def test_calculate_usage_sums_cache_tokens_across_compaction_iterations(): assert usage.completion_tokens == 250 assert usage.prompt_tokens_details.cache_creation_tokens == 60 assert usage.prompt_tokens_details.cached_tokens == 17020 + + +PROMPT_CACHING_MODEL: Final = "claude-sonnet-5-5" +EPHEMERAL: Final = {"type": "ephemeral"} +WEATHER_PARAMETERS: Final = { + "type": "object", + "properties": { + "location": {"type": "string", "description": "The city and state, e.g. San Francisco, CA"}, + "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, + }, + "required": ["location"], +} + + +async def _send_prompt_caching_request( + respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch, **params: object +) -> httpx.Request: + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + route: Final = respx_mock.post("https://api.anthropic.com/v1/messages").mock( + return_value=httpx.Response( + 200, + json={ + "id": "msg_01XFDUDYJgAACzvnptvVoYEL", + "type": "message", + "role": "assistant", + "content": [{"type": "text", "text": "Hello!"}], + "model": PROMPT_CACHING_MODEL, + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 12, "output_tokens": 6}, + }, + ) + ) + await litellm.acompletion( + api_key="mock_api_key", + model=f"anthropic/{PROMPT_CACHING_MODEL}", + extra_headers={"anthropic-version": "2023-06-01"}, + **params, + ) + assert route.call_count == 1 + request: Final = route.calls.last.request + assert request.headers["x-api-key"] == "mock_api_key" + assert request.headers["anthropic-version"] == "2023-06-01" + assert "anthropic-beta" not in request.headers + return request + + +async def test_litellm_anthropic_prompt_caching_tools( + respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch +) -> None: + request: Final = await _send_prompt_caching_request( + respx_mock, + monkeypatch, + messages=[{"role": "user", "content": "What's the weather like in Boston today?"}], + tools=[ + { + "type": "function", + "function": { + "name": "get_current_weather", + "description": "Get the current weather in a given location", + "parameters": WEATHER_PARAMETERS, + "cache_control": EPHEMERAL, + }, + } + ], + ) + assert json.loads(request.content) == { + "model": PROMPT_CACHING_MODEL, + "messages": [ + {"role": "user", "content": [{"type": "text", "text": "What's the weather like in Boston today?"}]} + ], + "tools": [ + { + "name": "get_current_weather", + "description": "Get the current weather in a given location", + "input_schema": WEATHER_PARAMETERS, + "type": "custom", + "cache_control": EPHEMERAL, + } + ], + "max_tokens": litellm.get_max_tokens(PROMPT_CACHING_MODEL), + } + + +async def test_litellm_anthropic_prompt_caching_system( + respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch +) -> None: + system_blocks: Final = [ + {"type": "text", "text": "You are an AI assistant tasked with analyzing legal documents."}, + {"type": "text", "text": "Here is the full text of a complex legal agreement", "cache_control": EPHEMERAL}, + ] + request: Final = await _send_prompt_caching_request( + respx_mock, + monkeypatch, + messages=[ + {"role": "system", "content": system_blocks}, + {"role": "user", "content": "what are the key terms and conditions in this agreement?"}, + ], + ) + assert json.loads(request.content) == { + "model": PROMPT_CACHING_MODEL, + "system": system_blocks, + "messages": [ + { + "role": "user", + "content": [{"type": "text", "text": "what are the key terms and conditions in this agreement?"}], + } + ], + "max_tokens": litellm.get_max_tokens(PROMPT_CACHING_MODEL), + }