diff --git a/.github/ci-coverage-allowlist.yml b/.github/ci-coverage-allowlist.yml index 837832795b1..7f62ac3bf53 100644 --- a/.github/ci-coverage-allowlist.yml +++ b/.github/ci-coverage-allowlist.yml @@ -38,10 +38,8 @@ test_paths: Mixed file that no job has ever run in full. Every CircleCI job that globs tests/local_testing deselects it by name ("caching") or keeps only another keyword, so only its router test ran, and that test now lives in tests/unit/router_utils/pre_call_checks. - Five of the seven left need ANTHROPIC_API_KEY or Vertex credentials. The other two, - test_litellm_anthropic_prompt_caching_tools and test_litellm_anthropic_prompt_caching_system, - are offline mocks that already fail on main against a stale expected request body; they need - that fixed before they can move to tests/unit + Its two offline tests now live in tests/unit/llms/anthropic/chat, and the five left all need + ANTHROPIC_API_KEY or Vertex credentials paths: - tests/local_testing/test_anthropic_prompt_caching.py - reason: >- diff --git a/tests/local_testing/test_anthropic_prompt_caching.py b/tests/local_testing/test_anthropic_prompt_caching.py index 24ffe1b093e..169e6762ad0 100644 --- a/tests/local_testing/test_anthropic_prompt_caching.py +++ b/tests/local_testing/test_anthropic_prompt_caching.py @@ -36,123 +36,6 @@ def reset_callbacks(): litellm.callbacks = [] -@pytest.mark.asyncio -async def test_litellm_anthropic_prompt_caching_tools(): - # Arrange: Set up the MagicMock for the httpx.AsyncClient - mock_response = AsyncMock() - - def return_val(): - return { - "id": "msg_01XFDUDYJgAACzvnptvVoYEL", - "type": "message", - "role": "assistant", - "content": [{"type": "text", "text": "Hello!"}], - "model": "claude-sonnet-4-5-20250929", - "stop_reason": "end_turn", - "stop_sequence": None, - "usage": {"input_tokens": 12, "output_tokens": 6}, - } - - mock_response.json = return_val - mock_response.headers = {"key": "value"} - - litellm.set_verbose = True - with patch( - "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", - return_value=mock_response, - ) as mock_post: - # Act: Call the litellm.acompletion function - response = await litellm.acompletion( - api_key="mock_api_key", - model="anthropic/claude-sonnet-4-5-20250929", - messages=[ - {"role": "user", "content": "What's the weather like in Boston today?"} - ], - tools=[ - { - "type": "function", - "function": { - "name": "get_current_weather", - "description": "Get the current weather in a given location", - "parameters": { - "type": "object", - "properties": { - "location": { - "type": "string", - "description": "The city and state, e.g. San Francisco, CA", - }, - "unit": { - "type": "string", - "enum": ["celsius", "fahrenheit"], - }, - }, - "required": ["location"], - }, - "cache_control": {"type": "ephemeral"}, - }, - } - ], - extra_headers={ - "anthropic-version": "2023-06-01", - }, - ) - - # Print what was called on the mock - print("call args=", mock_post.call_args) - - expected_url = "https://api.anthropic.com/v1/messages" - # Note: anthropic-beta header for prompt-caching is no longer required - # Anthropic now supports prompt caching automatically when cache_control is used - expected_headers = { - "accept": "application/json", - "content-type": "application/json", - "anthropic-version": "2023-06-01", - "x-api-key": "mock_api_key", - } - - expected_json = { - "messages": [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": "What's the weather like in Boston today?", - } - ], - } - ], - "tools": [ - { - "name": "get_current_weather", - "description": "Get the current weather in a given location", - "cache_control": {"type": "ephemeral"}, - "input_schema": { - "type": "object", - "properties": { - "location": { - "type": "string", - "description": "The city and state, e.g. San Francisco, CA", - }, - "unit": { - "type": "string", - "enum": ["celsius", "fahrenheit"], - }, - }, - "required": ["location"], - }, - "type": "custom", - } - ], - "max_tokens": 64000, - "model": "claude-sonnet-4-5-20250929", - } - - mock_post.assert_called_once_with( - expected_url, json=expected_json, headers=expected_headers, timeout=600.0 - ) - - @pytest.fixture def anthropic_messages(): return [ @@ -494,101 +377,3 @@ async def test_anthropic_api_prompt_caching_streaming(): assert ( is_cache_read_input_tokens_in_usage and is_cache_creation_input_tokens_in_usage ) - - -@pytest.mark.asyncio -async def test_litellm_anthropic_prompt_caching_system(): - # https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching#prompt-caching-examples - # LArge Context Caching Example - mock_response = AsyncMock() - - def return_val(): - return { - "id": "msg_01XFDUDYJgAACzvnptvVoYEL", - "type": "message", - "role": "assistant", - "content": [{"type": "text", "text": "Hello!"}], - "model": "claude-sonnet-4-5-20250929", - "stop_reason": "end_turn", - "stop_sequence": None, - "usage": {"input_tokens": 12, "output_tokens": 6}, - } - - mock_response.json = return_val - mock_response.headers = {"key": "value"} - - litellm.set_verbose = True - with patch( - "litellm.llms.custom_httpx.http_handler.AsyncHTTPHandler.post", - return_value=mock_response, - ) as mock_post: - # Act: Call the litellm.acompletion function - response = await litellm.acompletion( - api_key="mock_api_key", - model="anthropic/claude-sonnet-4-5-20250929", - messages=[ - { - "role": "system", - "content": [ - { - "type": "text", - "text": "You are an AI assistant tasked with analyzing legal documents.", - }, - { - "type": "text", - "text": "Here is the full text of a complex legal agreement", - "cache_control": {"type": "ephemeral"}, - }, - ], - }, - { - "role": "user", - "content": "what are the key terms and conditions in this agreement?", - }, - ], - extra_headers={ - "anthropic-version": "2023-06-01", - }, - ) - - # Print what was called on the mock - print("call args=", mock_post.call_args) - - expected_url = "https://api.anthropic.com/v1/messages" - expected_headers = { - "accept": "application/json", - "content-type": "application/json", - "anthropic-version": "2023-06-01", - "x-api-key": "mock_api_key", - } - - expected_json = { - "system": [ - { - "type": "text", - "text": "You are an AI assistant tasked with analyzing legal documents.", - }, - { - "type": "text", - "text": "Here is the full text of a complex legal agreement", - "cache_control": {"type": "ephemeral"}, - }, - ], - "messages": [ - { - "role": "user", - "content": [ - { - "type": "text", - "text": "what are the key terms and conditions in this agreement?", - } - ], - } - ], - "max_tokens": 64000, - "model": "claude-sonnet-4-5-20250929", - } - - mock_post.assert_called_once_with( - expected_url, json=expected_json, headers=expected_headers, timeout=600.0 - ) diff --git a/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py b/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py index 74d68d76813..c3a3dad92c8 100644 --- a/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py +++ b/tests/unit/llms/anthropic/chat/test_anthropic_chat_transformation.py @@ -7968,3 +7968,113 @@ def test_calculate_usage_sums_cache_tokens_across_compaction_iterations(): assert usage.completion_tokens == 250 assert usage.prompt_tokens_details.cache_creation_tokens == 60 assert usage.prompt_tokens_details.cached_tokens == 17020 + + +PROMPT_CACHING_MODEL: Final = "claude-sonnet-5-5" +EPHEMERAL: Final = {"type": "ephemeral"} +WEATHER_PARAMETERS: Final = { + "type": "object", + "properties": { + "location": {"type": "string", "description": "The city and state, e.g. San Francisco, CA"}, + "unit": {"type": "string", "enum": ["celsius", "fahrenheit"]}, + }, + "required": ["location"], +} + + +async def _send_prompt_caching_request( + respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch, **params: object +) -> httpx.Request: + monkeypatch.setenv("DISABLE_AIOHTTP_TRANSPORT", "True") + route: Final = respx_mock.post("https://api.anthropic.com/v1/messages").mock( + return_value=httpx.Response( + 200, + json={ + "id": "msg_01XFDUDYJgAACzvnptvVoYEL", + "type": "message", + "role": "assistant", + "content": [{"type": "text", "text": "Hello!"}], + "model": PROMPT_CACHING_MODEL, + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 12, "output_tokens": 6}, + }, + ) + ) + await litellm.acompletion( + api_key="mock_api_key", + model=f"anthropic/{PROMPT_CACHING_MODEL}", + extra_headers={"anthropic-version": "2023-06-01"}, + **params, + ) + assert route.call_count == 1 + request: Final = route.calls.last.request + assert request.headers["x-api-key"] == "mock_api_key" + assert request.headers["anthropic-version"] == "2023-06-01" + assert "anthropic-beta" not in request.headers + return request + + +async def test_litellm_anthropic_prompt_caching_tools( + respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch +) -> None: + request: Final = await _send_prompt_caching_request( + respx_mock, + monkeypatch, + messages=[{"role": "user", "content": "What's the weather like in Boston today?"}], + tools=[ + { + "type": "function", + "function": { + "name": "get_current_weather", + "description": "Get the current weather in a given location", + "parameters": WEATHER_PARAMETERS, + "cache_control": EPHEMERAL, + }, + } + ], + ) + assert json.loads(request.content) == { + "model": PROMPT_CACHING_MODEL, + "messages": [ + {"role": "user", "content": [{"type": "text", "text": "What's the weather like in Boston today?"}]} + ], + "tools": [ + { + "name": "get_current_weather", + "description": "Get the current weather in a given location", + "input_schema": WEATHER_PARAMETERS, + "type": "custom", + "cache_control": EPHEMERAL, + } + ], + "max_tokens": litellm.get_max_tokens(PROMPT_CACHING_MODEL), + } + + +async def test_litellm_anthropic_prompt_caching_system( + respx_mock: respx.MockRouter, monkeypatch: pytest.MonkeyPatch +) -> None: + system_blocks: Final = [ + {"type": "text", "text": "You are an AI assistant tasked with analyzing legal documents."}, + {"type": "text", "text": "Here is the full text of a complex legal agreement", "cache_control": EPHEMERAL}, + ] + request: Final = await _send_prompt_caching_request( + respx_mock, + monkeypatch, + messages=[ + {"role": "system", "content": system_blocks}, + {"role": "user", "content": "what are the key terms and conditions in this agreement?"}, + ], + ) + assert json.loads(request.content) == { + "model": PROMPT_CACHING_MODEL, + "system": system_blocks, + "messages": [ + { + "role": "user", + "content": [{"type": "text", "text": "what are the key terms and conditions in this agreement?"}], + } + ], + "max_tokens": litellm.get_max_tokens(PROMPT_CACHING_MODEL), + }