diff --git a/tests/logging_callback_tests/gcs_pub_sub_body/spend_logs_payload.json b/tests/logging_callback_tests/gcs_pub_sub_body/spend_logs_payload.json index 75c229d8b1a..7dcbd2467fd 100644 --- a/tests/logging_callback_tests/gcs_pub_sub_body/spend_logs_payload.json +++ b/tests/logging_callback_tests/gcs_pub_sub_body/spend_logs_payload.json @@ -11,7 +11,7 @@ "user": "", "team_id": "", "organization_id": "", - "metadata": "{\"applied_guardrails\": [], \"batch_models\": null, \"mcp_tool_call_metadata\": null, \"vector_store_request_metadata\": null, \"guardrail_information\": null, \"usage_object\": {\"completion_tokens\": 20, \"prompt_tokens\": 10, \"total_tokens\": 30, \"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"model_map_information\": {\"model_map_key\": \"gpt-4o\", \"model_map_value\": {\"key\": \"gpt-4o\", \"max_tokens\": 16384, \"max_input_tokens\": 128000, \"max_output_tokens\": 16384, \"input_cost_per_token\": 2.5e-06, \"cache_creation_input_token_cost\": null, \"cache_read_input_token_cost\": 1.25e-06, \"input_cost_per_character\": null, \"input_cost_per_token_above_128k_tokens\": null, \"input_cost_per_token_above_200k_tokens\": null, \"input_cost_per_query\": null, \"input_cost_per_second\": null, \"input_cost_per_audio_token\": null, \"input_cost_per_token_batches\": 1.25e-06, \"output_cost_per_token_batches\": 5e-06, \"output_cost_per_token\": 1e-05, \"output_cost_per_audio_token\": null, \"output_cost_per_character\": null, \"output_cost_per_token_above_128k_tokens\": null, \"output_cost_per_character_above_128k_tokens\": null, \"output_cost_per_token_above_200k_tokens\": null, \"output_cost_per_second\": null, \"output_cost_per_image\": null, \"output_vector_size\": null, \"litellm_provider\": \"openai\", \"mode\": \"chat\", \"supports_system_messages\": true, \"supports_response_schema\": true, \"supports_vision\": true, \"supports_function_calling\": true, \"supports_tool_choice\": true, \"supports_assistant_prefill\": false, \"supports_prompt_caching\": true, \"supports_audio_input\": false, \"supports_audio_output\": false, \"supports_pdf_input\": false, \"supports_embedding_image_input\": false, \"supports_native_streaming\": null, \"supports_web_search\": true, \"supports_reasoning\": false, \"search_context_cost_per_query\": {\"search_context_size_low\": 0.03, \"search_context_size_medium\": 0.035, \"search_context_size_high\": 0.05}, \"tpm\": null, \"rpm\": null, \"supported_openai_params\": [\"frequency_penalty\", \"logit_bias\", \"logprobs\", \"top_logprobs\", \"max_tokens\", \"max_completion_tokens\", \"modalities\", \"prediction\", \"n\", \"presence_penalty\", \"seed\", \"stop\", \"stream\", \"stream_options\", \"temperature\", \"top_p\", \"tools\", \"tool_choice\", \"function_call\", \"functions\", \"max_retries\", \"extra_headers\", \"parallel_tool_calls\", \"audio\", \"response_format\", \"user\"]}}, \"additional_usage_values\": {\"completion_tokens_details\": null, \"prompt_tokens_details\": null}}", + "metadata": "{\"applied_guardrails\": [], \"batch_models\": null, \"mcp_tool_call_metadata\": null, \"vector_store_request_metadata\": null, \"guardrail_information\": null, \"usage_object\": {\"completion_tokens\": 20, \"prompt_tokens\": 10, \"total_tokens\": 30, \"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"model_map_information\": {\"model_map_key\": \"gpt-4o\", \"model_map_value\": {\"key\": \"gpt-4o\", \"max_tokens\": 16384, \"max_input_tokens\": 128000, \"max_output_tokens\": 16384, \"input_cost_per_token\": 2.5e-06, \"cache_creation_input_token_cost\": null, \"cache_read_input_token_cost\": 1.25e-06, \"input_cost_per_character\": null, \"input_cost_per_token_above_128k_tokens\": null, \"input_cost_per_token_above_200k_tokens\": null, \"input_cost_per_query\": null, \"input_cost_per_second\": null, \"input_cost_per_audio_token\": null, \"input_cost_per_token_batches\": 1.25e-06, \"output_cost_per_token_batches\": 5e-06, \"output_cost_per_token\": 1e-05, \"output_cost_per_audio_token\": null, \"output_cost_per_character\": null, \"output_cost_per_token_above_128k_tokens\": null, \"output_cost_per_character_above_128k_tokens\": null, \"output_cost_per_token_above_200k_tokens\": null, \"output_cost_per_second\": null, \"output_cost_per_image\": null, \"output_vector_size\": null, \"litellm_provider\": \"openai\", \"mode\": \"chat\", \"supports_system_messages\": true, \"supports_response_schema\": true, \"supports_vision\": true, \"supports_function_calling\": true, \"supports_tool_choice\": true, \"supports_assistant_prefill\": false, \"supports_prompt_caching\": true, \"supports_audio_input\": false, \"supports_audio_output\": false, \"supports_pdf_input\": false, \"supports_embedding_image_input\": false, \"supports_native_streaming\": null, \"supports_web_search\": true, \"supports_reasoning\": false, \"search_context_cost_per_query\": {\"search_context_size_low\": 0.03, \"search_context_size_medium\": 0.035, \"search_context_size_high\": 0.05}, \"tpm\": null, \"rpm\": null, \"supported_openai_params\": [\"frequency_penalty\", \"logit_bias\", \"logprobs\", \"top_logprobs\", \"max_tokens\", \"max_completion_tokens\", \"modalities\", \"prediction\", \"n\", \"presence_penalty\", \"seed\", \"stop\", \"stream\", \"stream_options\", \"temperature\", \"top_p\", \"tools\", \"tool_choice\", \"function_call\", \"functions\", \"max_retries\", \"extra_headers\", \"parallel_tool_calls\", \"audio\", \"response_format\", \"user\"]}}, \"additional_usage_values\": {\"completion_tokens_details\": null, \"prompt_tokens_details\": null}, \"user_api_key\": null, \"user_api_key_alias\": null, \"user_api_key_team_id\": null, \"user_api_key_project_id\": null, \"user_api_key_org_id\": null, \"user_api_key_user_id\": null, \"user_api_key_team_alias\": null, \"spend_logs_metadata\": null, \"requester_ip_address\": null, \"status\": null, \"proxy_server_request\": null, \"error_information\": null, \"attempted_retries\": null, \"max_retries\": null}", "cache_key": "Cache OFF", "spend": 0.00022500000000000002, "total_tokens": 30, diff --git a/tests/logging_callback_tests/test_bedrock_knowledgebase_hook.py b/tests/logging_callback_tests/test_bedrock_knowledgebase_hook.py index f87351edb01..3eb85cfdaa5 100644 --- a/tests/logging_callback_tests/test_bedrock_knowledgebase_hook.py +++ b/tests/logging_callback_tests/test_bedrock_knowledgebase_hook.py @@ -176,19 +176,49 @@ async def test_e2e_bedrock_knowledgebase_retrieval_with_llm_api_call_streaming(s """ Test that the Bedrock Knowledge Base Hook works with streaming and returns search_results in chunks. """ - + # Init client # litellm._turn_on_debug() async_client = AsyncHTTPHandler() - response = await litellm.acompletion( - model="anthropic/claude-3-5-haiku-latest", - messages=[{"role": "user", "content": "what is litellm?"}], - vector_store_ids = [ - "T37J8R4WTM" - ], - stream=True, - client=async_client - ) + + async def mock_anthropic_aiter_lines(): + lines = [ + 'event: message_start', + 'data: {"type":"message_start","message":{"id":"msg_01ABC","type":"message","role":"assistant","content":[],"model":"claude-3-5-haiku-latest","stop_reason":null,"stop_sequence":null,"usage":{"input_tokens":10,"output_tokens":0}}}', + '', + 'event: content_block_start', + 'data: {"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}}', + '', + 'event: content_block_delta', + 'data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"LiteLLM is a library."}}', + '', + 'event: content_block_stop', + 'data: {"type":"content_block_stop","index":0}', + '', + 'event: message_delta', + 'data: {"type":"message_delta","delta":{"stop_reason":"end_turn","stop_sequence":null},"usage":{"output_tokens":5}}', + '', + 'event: message_stop', + 'data: {"type":"message_stop"}', + ] + for line in lines: + yield line + + mock_streaming_response = Mock() + mock_streaming_response.status_code = 200 + mock_streaming_response.headers = {} + mock_streaming_response.aiter_lines = mock_anthropic_aiter_lines + + with patch.object(async_client, "post", new=AsyncMock(return_value=mock_streaming_response)): + response = await litellm.acompletion( + model="anthropic/claude-3-5-haiku-latest", + messages=[{"role": "user", "content": "what is litellm?"}], + vector_store_ids = [ + "T37J8R4WTM" + ], + stream=True, + client=async_client + ) # Collect chunks chunks = [] @@ -228,20 +258,40 @@ async def test_e2e_bedrock_knowledgebase_retrieval_with_llm_api_call_with_tools( """ Test that the Bedrock Knowledge Base Hook works when making a real llm api call """ - + # Init client litellm._turn_on_debug() - response = await litellm.acompletion( - model="anthropic/claude-3-5-haiku-latest", - messages=[{"role": "user", "content": "what is litellm?"}], - max_tokens=10, - tools=[ - { - "type": "file_search", - "vector_store_ids": ["T37J8R4WTM"] - } - ], - ) + async_client = AsyncHTTPHandler() + + mock_llm_response = Mock() + mock_llm_response.status_code = 200 + mock_llm_response.headers = {"content-type": "application/json"} + mock_llm_response_json = { + "id": "msg_01ABC123", + "type": "message", + "role": "assistant", + "content": [{"type": "text", "text": "LiteLLM is a library."}], + "model": "claude-3-5-haiku-latest", + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 5}, + } + mock_llm_response.text = json.dumps(mock_llm_response_json) + mock_llm_response.json = Mock(return_value=mock_llm_response_json) + + with patch.object(async_client, "post", new=AsyncMock(return_value=mock_llm_response)): + response = await litellm.acompletion( + model="anthropic/claude-3-5-haiku-latest", + messages=[{"role": "user", "content": "what is litellm?"}], + max_tokens=10, + tools=[ + { + "type": "file_search", + "vector_store_ids": ["T37J8R4WTM"] + } + ], + client=async_client, + ) assert response is not None @pytest.mark.asyncio @@ -253,23 +303,42 @@ async def test_e2e_bedrock_knowledgebase_retrieval_with_llm_api_call_with_tools_ In this case we filter for a non-existent user_id, which should return no results. """ litellm._turn_on_debug() - - response = await litellm.acompletion( - model="anthropic/claude-3-5-haiku-latest", - messages=[{"role": "user", "content": "what is litellm?"}], - max_tokens=10, - tools=[ - { - "type": "file_search", - "vector_store_ids": ["T37J8R4WTM"], - "filters": { - "key": "user_id", - "value": "fake-user-id", - "operator": "eq" + async_client = AsyncHTTPHandler() + + mock_llm_response = Mock() + mock_llm_response.status_code = 200 + mock_llm_response.headers = {"content-type": "application/json"} + mock_llm_response_json = { + "id": "msg_01ABC123", + "type": "message", + "role": "assistant", + "content": [{"type": "text", "text": "LiteLLM is a library."}], + "model": "claude-3-5-haiku-latest", + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 5}, + } + mock_llm_response.text = json.dumps(mock_llm_response_json) + mock_llm_response.json = Mock(return_value=mock_llm_response_json) + + with patch.object(async_client, "post", new=AsyncMock(return_value=mock_llm_response)): + response = await litellm.acompletion( + model="anthropic/claude-3-5-haiku-latest", + messages=[{"role": "user", "content": "what is litellm?"}], + max_tokens=10, + tools=[ + { + "type": "file_search", + "vector_store_ids": ["T37J8R4WTM"], + "filters": { + "key": "user_id", + "value": "fake-user-id", + "operator": "eq" + } } - } - ], - ) + ], + client=async_client, + ) # Verify response is not None assert response is not None @@ -344,11 +413,28 @@ async def test_bedrock_kb_request_body_has_transformed_filters(setup_vector_stor ], ) + async_client = AsyncHTTPHandler() + mock_llm_response = Mock() + mock_llm_response.status_code = 200 + mock_llm_response.headers = {"content-type": "application/json"} + mock_llm_response_json = { + "id": "msg_01ABC123", + "type": "message", + "role": "assistant", + "content": [{"type": "text", "text": "LiteLLM is a library."}], + "model": "claude-3-5-haiku-latest", + "stop_reason": "end_turn", + "stop_sequence": None, + "usage": {"input_tokens": 10, "output_tokens": 5}, + } + mock_llm_response.text = json.dumps(mock_llm_response_json) + mock_llm_response.json = Mock(return_value=mock_llm_response_json) + with patch.object( litellm.vector_stores.main.base_llm_http_handler, "async_vector_store_search_handler", new=AsyncMock(side_effect=fake_async_vector_store_search_handler), - ): + ), patch.object(async_client, "post", new=AsyncMock(return_value=mock_llm_response)): response = await litellm.acompletion( model="anthropic/claude-3-5-haiku-latest", messages=[{"role": "user", "content": "what is litellm?"}], @@ -364,6 +450,7 @@ async def test_bedrock_kb_request_body_has_transformed_filters(setup_vector_stor }, } ], + client=async_client, ) assert response is not None