From 0bc2a01b733460c6e5ec29026972857b02294a79 Mon Sep 17 00:00:00 2001 From: Ishaan Jaffer Date: Sat, 21 Feb 2026 10:15:30 -0800 Subject: [PATCH] fix(tests): mock anthropic http calls in bedrock KB tests, update gcs pubsub fixture Bedrock KB tests were hitting the anthropic API (via berrie proxy) and getting 401s. Fixed by mocking the AsyncHTTPHandler.post call in the 4 failing tests. GCS pubsub v1 test was failing because SpendLogsMetadata added new fields (user_api_key, status, error_information, etc.) that weren't in the expected spend_logs_payload.json fixture. --- .../gcs_pub_sub_body/spend_logs_payload.json | 2 +- .../test_bedrock_knowledgebase_hook.py | 165 +++++++++++++----- 2 files changed, 127 insertions(+), 40 deletions(-) diff --git a/tests/logging_callback_tests/gcs_pub_sub_body/spend_logs_payload.json b/tests/logging_callback_tests/gcs_pub_sub_body/spend_logs_payload.json index 75c229d8b1a..7dcbd2467fd 100644 --- a/tests/logging_callback_tests/gcs_pub_sub_body/spend_logs_payload.json +++ b/tests/logging_callback_tests/gcs_pub_sub_body/spend_logs_payload.json @@ -11,7 +11,7 @@ "user": "", "team_id": "", "organization_id": "", - "metadata": "{\"applied_guardrails\": [], \"batch_models\": null, \"mcp_tool_call_metadata\": null, \"vector_store_request_metadata\": null, \"guardrail_information\": null, \"usage_object\": {\"completion_tokens\": 20, \"prompt_tokens\": 10, \"total_tokens\": 30, \"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"model_map_information\": {\"model_map_key\": \"gpt-4o\", \"model_map_value\": {\"key\": \"gpt-4o\", \"max_tokens\": 16384, \"max_input_tokens\": 128000, \"max_output_tokens\": 16384, \"input_cost_per_token\": 2.5e-06, \"cache_creation_input_token_cost\": null, \"cache_read_input_token_cost\": 1.25e-06, \"input_cost_per_character\": null, \"input_cost_per_token_above_128k_tokens\": null, \"input_cost_per_token_above_200k_tokens\": null, \"input_cost_per_query\": null, \"input_cost_per_second\": null, \"input_cost_per_audio_token\": null, \"input_cost_per_token_batches\": 1.25e-06, \"output_cost_per_token_batches\": 5e-06, \"output_cost_per_token\": 1e-05, \"output_cost_per_audio_token\": null, \"output_cost_per_character\": null, \"output_cost_per_token_above_128k_tokens\": null, \"output_cost_per_character_above_128k_tokens\": null, \"output_cost_per_token_above_200k_tokens\": null, \"output_cost_per_second\": null, \"output_cost_per_image\": null, \"output_vector_size\": null, \"litellm_provider\": \"openai\", \"mode\": \"chat\", \"supports_system_messages\": true, \"supports_response_schema\": true, \"supports_vision\": true, \"supports_function_calling\": true, \"supports_tool_choice\": true, \"supports_assistant_prefill\": false, \"supports_prompt_caching\": true, \"supports_audio_input\": false, \"supports_audio_output\": false, \"supports_pdf_input\": false, \"supports_embedding_image_input\": false, \"supports_native_streaming\": null, \"supports_web_search\": true, \"supports_reasoning\": false, \"search_context_cost_per_query\": {\"search_context_size_low\": 0.03, \"search_context_size_medium\": 0.035, \"search_context_size_high\": 0.05}, \"tpm\": null, \"rpm\": null, \"supported_openai_params\": [\"frequency_penalty\", \"logit_bias\", \"logprobs\", \"top_logprobs\", \"max_tokens\", \"max_completion_tokens\", \"modalities\", \"prediction\", \"n\", \"presence_penalty\", \"seed\", \"stop\", \"stream\", \"stream_options\", \"temperature\", \"top_p\", \"tools\", \"tool_choice\", \"function_call\", \"functions\", \"max_retries\", \"extra_headers\", \"parallel_tool_calls\", \"audio\", \"response_format\", \"user\"]}}, \"additional_usage_values\": {\"completion_tokens_details\": null, \"prompt_tokens_details\": null}}", + "metadata": "{\"applied_guardrails\": [], \"batch_models\": null, \"mcp_tool_call_metadata\": null, \"vector_store_request_metadata\": null, \"guardrail_information\": null, \"usage_object\": {\"completion_tokens\": 20, \"prompt_tokens\": 10, \"total_tokens\": 30, \"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"model_map_information\": {\"model_map_key\": \"gpt-4o\", \"model_map_value\": {\"key\": \"gpt-4o\", \"max_tokens\": 16384, \"max_input_tokens\": 128000, \"max_output_tokens\": 16384, \"input_cost_per_token\": 2.5e-06, \"cache_creation_input_token_cost\": null, \"cache_read_input_token_cost\": 1.25e-06, \"input_cost_per_character\": null, \"input_cost_per_token_above_128k_tokens\": null, \"input_cost_per_token_above_200k_tokens\": null, \"input_cost_per_query\": null, \"input_cost_per_second\": null, \"input_cost_per_audio_token\": null, \"input_cost_per_token_batches\": 1.25e-06, \"output_cost_per_token_batches\": 5e-06, \"output_cost_per_token\": 1e-05, \"output_cost_per_audio_token\": null, \"output_cost_per_character\": null, \"output_cost_per_token_above_128k_tokens\": null, \"output_cost_per_character_above_128k_tokens\": null, \"output_cost_per_token_above_200k_tokens\": null, \"output_cost_per_second\": null, \"output_cost_per_image\": null, \"output_vector_size\": null, \"litellm_provider\": \"openai\", \"mode\": \"chat\", \"supports_system_messages\": true, \"supports_response_schema\": true, \"supports_vision\": true, \"supports_function_calling\": true, \"supports_tool_choice\": true, \"supports_assistant_prefill\": false, \"supports_prompt_caching\": true, \"supports_audio_input\": false, \"supports_audio_output\": false, \"supports_pdf_input\": false, \"supports_embedding_image_input\": false, \"supports_native_streaming\": null, \"supports_web_search\": true, \"supports_reasoning\": false, \"search_context_cost_per_query\": {\"search_context_size_low\": 0.03, \"search_context_size_medium\": 0.035, \"search_context_size_high\": 0.05}, \"tpm\": null, \"rpm\": null, \"supported_openai_params\": [\"frequency_penalty\", \"logit_bias\", \"logprobs\", \"top_logprobs\", \"max_tokens\", \"max_completion_tokens\", \"modalities\", \"prediction\", \"n\", \"presence_penalty\", \"seed\", \"stop\", \"stream\", \"stream_options\", \"temperature\", \"top_p\", \"tools\", \"tool_choice\", \"function_call\", \"functions\", \"max_retries\", \"extra_headers\", \"parallel_tool_calls\", \"audio\", \"response_format\", \"user\"]}}, \"additional_usage_values\": {\"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"user_api_key\": null, \"user_api_key_alias\": null, \"user_api_key_team_id\": null, \"user_api_key_project_id\": null, \"user_api_key_org_id\": null, \"user_api_key_user_id\": null, \"user_api_key_team_alias\": null, \"spend_logs_metadata\": null, \"requester_ip_address\": null, \"status\": null, \"proxy_server_request\": null, \"error_information\": null, \"attempted_retries\": null, \"max_retries\": null}", "cache_key": "Cache OFF", "spend": 0.00022500000000000002, "total_tokens": 30, diff --git a/tests/logging_callback_tests/test_bedrock_knowledgebase_hook.py b/tests/logging_callback_tests/test_bedrock_knowledgebase_hook.py index f87351edb01..3eb85cfdaa5 100644 --- a/tests/logging_callback_tests/test_bedrock_knowledgebase_hook.py +++ b/tests/logging_callback_tests/test_bedrock_knowledgebase_hook.py @@ -176,19 +176,49 @@ async def test_e2e_bedrock_knowledgebase_retrieval_with_llm_api_call_streaming(s """ Test that the Bedrock Knowledge Base Hook works with streaming and returns search_results in chunks. """ - + # Init client # litellm._turn_on_debug() async_client = AsyncHTTPHandler() - response = await litellm.acompletion( - model="anthropic/claude-3-5-haiku-latest", - messages=[{"role": "user", "content": "what is litellm?"}], - vector_store_ids = [ - "T37J8R4WTM" - ], - stream=True, - client=async_client - ) + + async def mock_anthropic_aiter_lines(): + lines = [ + 'event: message_start', + 'data: {"type":"message_start","message":{"id":"msg_01ABC","type":"message","role":"assistant","content":[],"model":"claude-3-5-haiku-latest","stop_reason":null,"stop_sequence":null,"usage":{"input_tokens":10,"output_tokens":0}}}', + '', + 'event: content_block_start', + 'data: {"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}}', + '', + 'event: content_block_delta', + 'data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"LiteLLM is a library."}}', + '', + 'event: content_block_stop', + 'data: {"type":"content_block_stop","index":0}', + '', + 'event: message_delta', + 'data: {"type":"message_delta","delta":{"stop_reason":"end_turn","stop_sequence":null},"usage":{"output_tokens":5}}', + '', + 'event: message_stop', + 'data: {"type":"message_stop"}', + ] + for line in lines: + yield line + + mock_streaming_response = Mock() + mock_streaming_response.status_code = 200 + mock_streaming_response.headers = {} + mock_streaming_response.aiter_lines = mock_anthropic_aiter_lines + + with patch.object(async_client, "post", new=AsyncMock(return_value=mock_streaming_response)): + response = await litellm.acompletion( + model="anthropic/claude-3-5-haiku-latest", + messages=[{"role": "user", "content": "what is litellm?"}], + vector_store_ids = [ + "T37J8R4WTM" + ], + stream=True, + client=async_client + ) # Collect chunks chunks = [] @@ -228,20 +258,40 @@ async def test_e2e_bedrock_knowledgebase_retrieval_with_llm_api_call_with_tools( """ Test that the Bedrock Knowledge Base Hook works when making a real llm api call """ - + # Init client litellm._turn_on_debug() - response = await litellm.acompletion( - model="anthropic/claude-3-5-haiku-latest", - messages=[{"role": "user", "content": "what is litellm?"}], - max_tokens=10, - tools=[ - { - "type": "file_search", - "vector_store_ids": ["T37J8R4WTM"] - } - ], - ) + async_client = AsyncHTTPHandler() + + mock_llm_response = Mock() + mock_llm_response.status_code = 200 + mock_llm_response.headers = {"content-type": "application/json"} + mock_llm_response_json = { + "id": "msg_01ABC123", + "type": "message", + "role": "assistant", + "content": [{"type": "text", "text": "LiteLLM is a library."}], + "model": "claude-3-5-haiku-latest", + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 5}, + } + mock_llm_response.text = json.dumps(mock_llm_response_json) + mock_llm_response.json = Mock(return_value=mock_llm_response_json) + + with patch.object(async_client, "post", new=AsyncMock(return_value=mock_llm_response)): + response = await litellm.acompletion( + model="anthropic/claude-3-5-haiku-latest", + messages=[{"role": "user", "content": "what is litellm?"}], + max_tokens=10, + tools=[ + { + "type": "file_search", + "vector_store_ids": ["T37J8R4WTM"] + } + ], + client=async_client, + ) assert response is not None @pytest.mark.asyncio @@ -253,23 +303,42 @@ async def test_e2e_bedrock_knowledgebase_retrieval_with_llm_api_call_with_tools_ In this case we filter for a non-existent user_id, which should return no results. """ litellm._turn_on_debug() - - response = await litellm.acompletion( - model="anthropic/claude-3-5-haiku-latest", - messages=[{"role": "user", "content": "what is litellm?"}], - max_tokens=10, - tools=[ - { - "type": "file_search", - "vector_store_ids": ["T37J8R4WTM"], - "filters": { - "key": "user_id", - "value": "fake-user-id", - "operator": "eq" + async_client = AsyncHTTPHandler() + + mock_llm_response = Mock() + mock_llm_response.status_code = 200 + mock_llm_response.headers = {"content-type": "application/json"} + mock_llm_response_json = { + "id": "msg_01ABC123", + "type": "message", + "role": "assistant", + "content": [{"type": "text", "text": "LiteLLM is a library."}], + "model": "claude-3-5-haiku-latest", + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 5}, + } + mock_llm_response.text = json.dumps(mock_llm_response_json) + mock_llm_response.json = Mock(return_value=mock_llm_response_json) + + with patch.object(async_client, "post", new=AsyncMock(return_value=mock_llm_response)): + response = await litellm.acompletion( + model="anthropic/claude-3-5-haiku-latest", + messages=[{"role": "user", "content": "what is litellm?"}], + max_tokens=10, + tools=[ + { + "type": "file_search", + "vector_store_ids": ["T37J8R4WTM"], + "filters": { + "key": "user_id", + "value": "fake-user-id", + "operator": "eq" + } } - } - ], - ) + ], + client=async_client, + ) # Verify response is not None assert response is not None @@ -344,11 +413,28 @@ async def test_bedrock_kb_request_body_has_transformed_filters(setup_vector_stor ], ) + async_client = AsyncHTTPHandler() + mock_llm_response = Mock() + mock_llm_response.status_code = 200 + mock_llm_response.headers = {"content-type": "application/json"} + mock_llm_response_json = { + "id": "msg_01ABC123", + "type": "message", + "role": "assistant", + "content": [{"type": "text", "text": "LiteLLM is a library."}], + "model": "claude-3-5-haiku-latest", + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 5}, + } + mock_llm_response.text = json.dumps(mock_llm_response_json) + mock_llm_response.json = Mock(return_value=mock_llm_response_json) + with patch.object( litellm.vector_stores.main.base_llm_http_handler, "async_vector_store_search_handler", new=AsyncMock(side_effect=fake_async_vector_store_search_handler), - ): + ), patch.object(async_client, "post", new=AsyncMock(return_value=mock_llm_response)): response = await litellm.acompletion( model="anthropic/claude-3-5-haiku-latest", messages=[{"role": "user", "content": "what is litellm?"}], @@ -364,6 +450,7 @@ async def test_bedrock_kb_request_body_has_transformed_filters(setup_vector_stor }, } ], + client=async_client, ) assert response is not None