diff --git a/tests/llm_translation/test_anthropic_completion.py b/tests/llm_translation/test_anthropic_completion.py index 7a478e494b1..0f896c93683 100644 --- a/tests/llm_translation/test_anthropic_completion.py +++ b/tests/llm_translation/test_anthropic_completion.py @@ -1,10 +1,8 @@ # What is this? ## Unit tests for Anthropic Adapter -import asyncio import os import sys -import traceback from dotenv import load_dotenv @@ -13,28 +11,21 @@ import litellm.types.utils from litellm.llms.anthropic.chat import ModelResponseIterator load_dotenv() -import io -import os sys.path.insert( 0, os.path.abspath("../..") ) # Adds the parent directory to the system path -from typing import Optional -from unittest.mock import MagicMock, patch +from unittest.mock import patch import pytest import litellm from litellm import ( AnthropicConfig, - Router, - adapter_completion, ) -from litellm.types.llms.anthropic import AnthropicResponse -from litellm.types.utils import GenericStreamingChunk, ChatCompletionToolCallChunk +from litellm.types.utils import ChatCompletionToolCallChunk from litellm.types.llms.openai import ChatCompletionToolCallFunctionChunk from litellm.llms.anthropic.common_utils import process_anthropic_headers -from litellm.llms.anthropic.chat.handler import AnthropicChatCompletion from httpx import Headers from base_llm_unit_tests import BaseLLMChatTest, BaseAnthropicChatTest @@ -360,7 +351,6 @@ def test_process_anthropic_headers_with_no_matching_headers(): ) def test_anthropic_tool_use(tool_type, tool_config, message_content): """Test Anthropic tool use with computer use and web fetch tools.""" - from litellm import completion litellm._turn_on_debug() @@ -951,7 +941,6 @@ def test_anthropic_citations_api(): """ Test the citations API """ - from litellm import completion try: resp = completion( @@ -997,7 +986,6 @@ def test_anthropic_citations_api(): def test_anthropic_citations_api_streaming(): - from litellm import completion resp = completion( model="claude-sonnet-4-5-20250929", @@ -1044,7 +1032,6 @@ def test_anthropic_citations_api_streaming(): ], ) def test_anthropic_thinking_output(model): - from litellm import completion litellm._turn_on_debug() @@ -1110,45 +1097,6 @@ def test_anthropic_thinking_output_stream(model): pytest.skip("Model is timing out") -def test_anthropic_custom_headers(): - from litellm import completion - from litellm.llms.custom_httpx.http_handler import HTTPHandler - - client = HTTPHandler() - - tools = [ - { - "type": "computer_20241022", - "function": { - "name": "get_current_weather", - "parameters": { - "display_height_px": 100, - "display_width_px": 100, - "display_number": 1, - }, - }, - } - ] - - with patch.object(client, "post") as mock_post: - try: - resp = completion( - model="claude-sonnet-4-5-20250929", - headers={"anthropic-beta": "computer-use-2025-01-24"}, - messages=[ - {"role": "user", "content": "What is the capital of France?"} - ], - client=client, - tools=tools, - ) - except Exception as e: - print(f"Error: {e}") - - mock_post.assert_called_once() - headers = mock_post.call_args[1]["headers"] - assert "computer-use-2025-01-24" in headers["anthropic-beta"] - - @pytest.mark.parametrize( "model", [ @@ -1398,7 +1346,6 @@ def test_anthropic_mcp_server_tool_use(spec: str): os.getenv("ZAPIER_CI_CD_MCP_TOKEN") is None, reason="ZAPIER_CI_CD_MCP_TOKEN not set" ) def test_anthropic_mcp_server_responses_api(model: str): - from litellm import responses litellm._turn_on_debug() tools = [ @@ -1528,7 +1475,6 @@ def test_anthropic_tool_cache_control(): def test_anthropic_streaming(): - from litellm import completion request_data = { "messages": [ diff --git a/tests/llm_translation/test_openai.py b/tests/llm_translation/test_openai.py index 1fec7665daa..26b21143dc4 100644 --- a/tests/llm_translation/test_openai.py +++ b/tests/llm_translation/test_openai.py @@ -1,8 +1,6 @@ -import json import os import sys -from datetime import datetime -from unittest.mock import AsyncMock, patch +from unittest.mock import patch from typing import Optional sys.path.insert( @@ -10,13 +8,10 @@ sys.path.insert( ) # Adds the parent directory to the system path -import httpx import pytest import litellm -from litellm import Choices, Message, ModelResponse from base_llm_unit_tests import BaseLLMChatTest -import asyncio from litellm.types.llms.openai import ( ChatCompletionAnnotation, ChatCompletionAnnotationURLCitation, @@ -69,67 +64,6 @@ def test_openai_prediction_param(): ) -@pytest.mark.asyncio -async def test_openai_prediction_param_mock(): - """ - Tests that prediction parameter is correctly passed to the API - """ - litellm.set_verbose = True - - code = """ - /// - /// Represents a user with a first name, last name, and username. - /// - public class User - { - /// - /// Gets or sets the user's first name. - /// - public string FirstName { get; set; } - - /// - /// Gets or sets the user's last name. - /// - public string LastName { get; set; } - - /// - /// Gets or sets the user's username. - /// - public string Username { get; set; } - } - """ - from openai import AsyncOpenAI - - client = AsyncOpenAI(api_key="fake-api-key") - - with patch.object( - client.chat.completions.with_raw_response, "create" - ) as mock_client: - try: - await litellm.acompletion( - model="gpt-4o-mini", - messages=[ - { - "role": "user", - "content": "Replace the Username property with an Email property. Respond only with code, and with no markdown formatting.", - }, - {"role": "user", "content": code}, - ], - prediction={"type": "content", "content": code}, - client=client, - ) - except Exception as e: - print(f"Error: {e}") - - mock_client.assert_called_once() - request_body = mock_client.call_args.kwargs - - # Verify the request contains the prediction parameter - assert "prediction" in request_body - # verify prediction is correctly sent to the API - assert request_body["prediction"] == {"type": "content", "content": code} - - @pytest.mark.asyncio async def test_openai_prediction_param_with_caching(): """ @@ -207,76 +141,6 @@ async def test_openai_prediction_param_with_caching(): assert completion_response_3.id != completion_response_1.id -@pytest.mark.asyncio() -async def test_vision_with_custom_model(): - """ - Tests that an OpenAI compatible endpoint when sent an image will receive the image in the request - - """ - import base64 - import requests - from openai import AsyncOpenAI - - client = AsyncOpenAI(api_key="fake-api-key") - - litellm.set_verbose = True - api_base = "https://my-custom.api.openai.com" - - # Fetch and encode a test image - url = "https://dummyimage.com/100/100/fff&text=Test+image" - response = requests.get(url) - file_data = response.content - encoded_file = base64.b64encode(file_data).decode("utf-8") - base64_image = f"data:image/png;base64,{encoded_file}" - - with patch.object( - client.chat.completions.with_raw_response, "create" - ) as mock_client: - try: - response = await litellm.acompletion( - model="openai/my-custom-model", - max_tokens=10, - api_base=api_base, # use the mock api - messages=[ - { - "role": "user", - "content": [ - {"type": "text", "text": "What's in this image?"}, - { - "type": "image_url", - "image_url": {"url": base64_image}, - }, - ], - } - ], - client=client, - ) - except Exception as e: - print(f"Error: {e}") - - mock_client.assert_called_once() - request_body = mock_client.call_args.kwargs - - print("request_body: ", request_body) - - assert request_body["messages"] == [ - { - "role": "user", - "content": [ - {"type": "text", "text": "What's in this image?"}, - { - "type": "image_url", - "image_url": { - "url": "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAGQAAABkBAMAAACCzIhnAAAAG1BMVEURAAD///+ln5/h39/Dv79qX18uHx+If39MPz9oMSdmAAAACXBIWXMAAA7EAAAOxAGVKw4bAAABDElEQVRYhe2SzWqEMBRGPyQTfQxJsc5jBKGzFmlslyFIZxsCQ7sUaWd87EanpdpIrbtC71mE/NyTm9wEIAiCIAiC+N/otQBxU2Sf/aeh4enqptHXri+/yxIq63jlKCw6cXssnr3ObdzdGYFYCJ2IzHKXLygHXCB98Gm4DE+ZZemu5EisQSyZTmyg+AuzQbkezCuIy7EI0k9Ig3FtruwydY+qniqtV5yQyo8qpUIl2fc90KVzJWohWf2qu75vlw52rdfjVDHg8vLWwixW7PChqLkSyUadwfSS0uQZhEvRuIkS53uJvrK8cGWYaPwpGt8efvw+vlo8TPMzcmP8w7lrNypc1RsNgiAIgiD+Iu/RyDYhCaWrgQAAAABJRU5ErkJggg==" - }, - }, - ], - }, - ] - assert request_body["model"] == "my-custom-model" - assert request_body["max_tokens"] == 10 - - class TestOpenAIChatCompletion(BaseLLMChatTest): def get_base_completion_call_args(self) -> dict: return {"model": "gpt-4o-mini"} @@ -299,67 +163,6 @@ class TestOpenAIChatCompletion(BaseLLMChatTest): pass -@patch("litellm.main.openai_chat_completions._get_openai_client") -def test_openai_max_retries_0(mock_get_openai_client): - import litellm - - litellm.set_verbose = True - response = litellm.completion( - model="gpt-4o-mini", - messages=[{"role": "user", "content": "hi"}], - max_retries=0, - ) - - mock_get_openai_client.assert_called_once() - assert mock_get_openai_client.call_args.kwargs["max_retries"] == 0 - - -@patch("litellm.main.openai_chat_completions._get_openai_client") -def test_openai_image_generation_forwards_organization(mock_get_openai_client): - """Ensure organization flows to OpenAI client for image generation.""" - - class _DummyImages: - def generate(self, **kwargs): # type: ignore - class _Resp: - def model_dump(self_inner): # minimal OpenAI ImagesResponse shape - return { - "created": 123, - "data": [{"url": "http://example.com/image.png"}], - "usage": { - "input_tokens": 0, - "output_tokens": 0, - "total_tokens": 0, - }, - } - - return _Resp() - - class _DummyClient: - def __init__(self): - self.api_key = "sk-test" - - class _BaseURL: - _uri_reference = "https://api.openai.com/v1" - - self._base_url = _BaseURL() - self.images = _DummyImages() - - mock_get_openai_client.return_value = _DummyClient() - - org = "org_test_123" - resp = litellm.image_generation( - model="gpt-image-1", - prompt="A cute baby sea otter", - organization=org, - ) - - # Assert organization forwarded into OpenAI client factory - assert mock_get_openai_client.call_args.kwargs.get("organization") == org - - # Basic sanity on response shape - assert hasattr(resp, "data") and len(resp.data) == 1 - - @pytest.mark.parametrize("model", ["o1", "o3-mini"]) def test_o1_parallel_tool_calls(model): litellm.completion( @@ -692,128 +495,6 @@ async def test_openai_gpt5_reasoning(): assert response.choices[0].message.content is not None -@pytest.mark.asyncio -async def test_openai_safety_identifier_parameter(): - """Test that safety_identifier parameter is correctly passed to the OpenAI API.""" - from openai import AsyncOpenAI - - litellm.set_verbose = True - client = AsyncOpenAI(api_key="fake-api-key") - - with patch.object( - client.chat.completions.with_raw_response, "create" - ) as mock_client: - try: - await litellm.acompletion( - model="openai/gpt-4o", - messages=[{"role": "user", "content": "Hello, how are you?"}], - safety_identifier="user_code_123456", - client=client, - ) - except Exception as e: - print(f"Error: {e}") - - mock_client.assert_called_once() - request_body = mock_client.call_args.kwargs - - # Verify the request contains the safety_identifier parameter - assert "safety_identifier" in request_body - # Verify safety_identifier is correctly sent to the API - assert request_body["safety_identifier"] == "user_code_123456" - - -def test_openai_safety_identifier_parameter_sync(): - """Test that safety_identifier parameter is correctly passed to the OpenAI API.""" - from openai import OpenAI - - litellm.set_verbose = True - client = OpenAI(api_key="fake-api-key") - - with patch.object( - client.chat.completions.with_raw_response, "create" - ) as mock_client: - try: - litellm.completion( - model="openai/gpt-4o", - messages=[{"role": "user", "content": "Hello, how are you?"}], - safety_identifier="user_code_123456", - client=client, - ) - except Exception as e: - print(f"Error: {e}") - - mock_client.assert_called_once() - request_body = mock_client.call_args.kwargs - - # Verify the request contains the safety_identifier parameter - assert "safety_identifier" in request_body - # Verify safety_identifier is correctly sent to the API - assert request_body["safety_identifier"] == "user_code_123456" - - -@pytest.mark.asyncio -async def test_openai_service_tier_parameter(): - """Test that service_tier parameter is correctly passed to the OpenAI API.""" - from openai import AsyncOpenAI - - litellm.set_verbose = True - client = AsyncOpenAI(api_key="fake-api-key") - - with patch.object( - client.chat.completions.with_raw_response, "create" - ) as mock_client: - try: - await litellm.acompletion( - model="openai/gpt-4o", - messages=[{"role": "user", "content": "Hello, how are you?"}], - service_tier="priority", - client=client, - ) - except Exception as e: - print(f"Error: {e}") - - mock_client.assert_called_once() - request_body = mock_client.call_args.kwargs - - # Verify the request contains the service_tier parameter - assert "service_tier" in request_body, "service_tier should be in request body" - # Verify service_tier is correctly sent to the API - assert ( - request_body["service_tier"] == "priority" - ), "service_tier should be 'priority'" - - -def test_openai_service_tier_parameter_sync(): - """Test that service_tier parameter is correctly passed to the OpenAI API.""" - from openai import OpenAI - - litellm.set_verbose = True - client = OpenAI(api_key="fake-api-key") - - with patch.object( - client.chat.completions.with_raw_response, "create" - ) as mock_client: - try: - litellm.completion( - model="openai/gpt-4o", - messages=[{"role": "user", "content": "Hello, how are you?"}], - service_tier="priority", - client=client, - ) - except Exception as e: - print(f"Error: {e}") - - mock_client.assert_called_once() - request_body = mock_client.call_args.kwargs - - # Verify the request contains the service_tier parameter - assert "service_tier" in request_body, "service_tier should be in request body" - # Verify service_tier is correctly sent to the API - assert ( - request_body["service_tier"] == "priority" - ), "service_tier should be 'priority'" - - def test_gpt_5_reasoning_streaming(): litellm._turn_on_debug() response = litellm.completion( diff --git a/tests/llm_translation/test_openai_o1.py b/tests/llm_translation/test_openai_o1.py index fccb1c6f1e3..54b1734884e 100644 --- a/tests/llm_translation/test_openai_o1.py +++ b/tests/llm_translation/test_openai_o1.py @@ -1,19 +1,16 @@ -import json import os import sys -from datetime import datetime -from unittest.mock import AsyncMock, patch, MagicMock +from unittest.mock import patch sys.path.insert( 0, os.path.abspath("../..") ) # Adds the parent directory to the system path -import httpx import pytest import litellm -from litellm import Choices, Message, ModelResponse +from litellm import ModelResponse from base_llm_unit_tests import BaseLLMChatTest, BaseOSeriesModelsTest @@ -78,7 +75,6 @@ async def test_o1_handle_tool_calling_optional_params( - max_tokens is translated to 'max_completion_tokens' - role 'system' is translated to 'user' """ - from openai import AsyncOpenAI from litellm.utils import ProviderConfigManager from litellm.types.utils import LlmProviders @@ -94,47 +90,10 @@ async def test_o1_handle_tool_calling_optional_params( assert expected_tool_calling_support == ("tools" in supported_params) -@pytest.mark.asyncio -@pytest.mark.parametrize("model", ["gpt-4", "gpt-4-0613"]) -async def test_o1_max_completion_tokens(model: str): - """ - Tests that: - - max_completion_tokens is passed directly to OpenAI chat completion models - """ - from openai import AsyncOpenAI - - litellm.set_verbose = True - - client = AsyncOpenAI(api_key="fake-api-key") - - with patch.object( - client.chat.completions.with_raw_response, "create" - ) as mock_client: - try: - await litellm.acompletion( - model=model, - max_completion_tokens=10, - messages=[{"role": "user", "content": "Hello!"}], - client=client, - ) - except Exception as e: - print(f"Error: {e}") - - mock_client.assert_called_once() - request_body = mock_client.call_args.kwargs - - print("request_body: ", request_body) - - assert request_body["model"] == model - assert request_body["max_completion_tokens"] == 10 - assert request_body["messages"] == [{"role": "user", "content": "Hello!"}] - - def test_litellm_responses(): """ ensures that type of completion_tokens_details is correctly handled / returned """ - from litellm import ModelResponse from litellm.types.utils import CompletionTokensDetails response = ModelResponse( diff --git a/tests/llm_translation/test_optional_params.py b/tests/llm_translation/test_optional_params.py index 93acf016833..6ebbe39452b 100644 --- a/tests/llm_translation/test_optional_params.py +++ b/tests/llm_translation/test_optional_params.py @@ -1,11 +1,7 @@ #### What this tests #### # This tests if get_optional_params works as expected -import asyncio -import inspect import os import sys -import time -import traceback import pytest @@ -15,7 +11,6 @@ from unittest.mock import MagicMock, patch import litellm from litellm.litellm_core_utils.prompt_templates.factory import map_system_message_pt from litellm.types.completion import ( - ChatCompletionMessageParam, ChatCompletionSystemMessageParam, ChatCompletionUserMessageParam, ) @@ -74,36 +69,6 @@ def test_get_requester_metadata_returns_none_for_empty(): assert get_requester_metadata(metadata) is None -@patch("litellm.main.openai_chat_completions.completion") -def test_requester_metadata_forwarded_to_openai(mock_completion): - mock_completion.return_value = MagicMock() - metadata = { - "requester_metadata": { - "custom_meta_key": "value", - "hidden_params": "secret", - "int_value": 123, - } - } - - original_api_key = litellm.api_key - litellm.api_key = "sk-test" - original_preview_flag = litellm.enable_preview_features - litellm.enable_preview_features = True - - try: - litellm.completion( - model="gpt-4o", - messages=[{"role": "user", "content": "hi"}], - metadata=metadata, - ) - finally: - litellm.api_key = original_api_key - litellm.enable_preview_features = original_preview_flag - - sent_metadata = mock_completion.call_args.kwargs["optional_params"]["metadata"] - assert sent_metadata == {"custom_meta_key": "value"} - - def test_get_optional_params_with_allowed_openai_params(): """ Test if use can dynamically pass in allowed_openai_params to override default behavior @@ -707,26 +672,6 @@ def test_bedrock_optional_params_embeddings_provider_specific_params(): assert len(optional_params) == 1 -def test_get_optional_params_num_retries(): - """ - Relevant issue - https://github.com/BerriAI/litellm/issues/5124 - """ - with patch( - "litellm.main.get_optional_params", - new=MagicMock(return_value={"max_retries": 0}), - ) as mock_client: - _ = litellm.completion( - model="gpt-3.5-turbo", - messages=[{"role": "user", "content": "Hello world"}], - num_retries=10, - ) - - mock_client.assert_called() - - print(f"mock_client.call_args: {mock_client.call_args}") - assert mock_client.call_args.kwargs["max_retries"] == 10 - - @pytest.mark.parametrize( "provider", [ @@ -1101,7 +1046,7 @@ def test_together_ai_model_params(): def test_forward_user_param(): - from litellm.utils import get_supported_openai_params, get_optional_params + from litellm.utils import get_optional_params model = "claude-3-5-sonnet-20240620" optional_params = get_optional_params( @@ -1895,8 +1840,7 @@ def test_optional_params_image_gen_with_aspect_ratio(): def test_optional_params_responses_api_allowed_openai_params(): - from litellm import responses - from unittest.mock import patch, MagicMock + from unittest.mock import patch from litellm.llms.custom_httpx.http_handler import HTTPHandler client = HTTPHandler() @@ -1987,43 +1931,6 @@ def test_validate_openai_optional_params_disable_stop_sequence_limit(): litellm.disable_stop_sequence_limit = original_value -def test_validate_openai_optional_params_integration(): - """ - Test that validate_openai_optional_params is properly integrated in the completion flow. - """ - # Test that completion with more than 4 stop sequences works without error - try: - with patch("litellm.llms.openai.openai.OpenAI") as mock_client: - mock_response = MagicMock() - mock_response.choices = [MagicMock()] - mock_response.choices[0].message.content = "Test response" - mock_response.model = "gpt-3.5-turbo" - mock_response.id = "test-id" - mock_response.created = 1234567890 - mock_response.usage = MagicMock() - mock_response.usage.prompt_tokens = 10 - mock_response.usage.completion_tokens = 5 - mock_response.usage.total_tokens = 15 - - mock_client.return_value.chat.completions.create.return_value = ( - mock_response - ) - - # Call completion with more than 4 stop sequences - response = litellm.completion( - model="gpt-3.5-turbo", - messages=[{"role": "user", "content": "Hello"}], - stop=["stop1", "stop2", "stop3", "stop4", "stop5", "stop6"], - mock_response="Test response", # This will use mock - ) - - # Verify the call was made (stop sequences should be truncated internally) - assert response is not None - except Exception as e: - # Should not raise an exception - pytest.fail(f"validate_openai_optional_params integration failed: {e}") - - def test_drop_store_param_for_anthropic(): """ Test that the OpenAI-specific `store` parameter is correctly dropped diff --git a/tests/llm_translation/test_perplexity_reasoning.py b/tests/llm_translation/test_perplexity_reasoning.py index 2ea28b76696..6af199f625e 100644 --- a/tests/llm_translation/test_perplexity_reasoning.py +++ b/tests/llm_translation/test_perplexity_reasoning.py @@ -1,7 +1,5 @@ -import json import os import sys -from unittest.mock import patch, MagicMock import pytest @@ -10,7 +8,6 @@ sys.path.insert( ) # Adds the parent directory to the system path import litellm -from litellm import completion from litellm.utils import get_optional_params @@ -53,93 +50,6 @@ class TestPerplexityReasoning: assert "reasoning_effort" in optional_params assert optional_params["reasoning_effort"] == reasoning_effort - @pytest.mark.parametrize( - "model", - [ - "perplexity/sonar-reasoning", - "perplexity/sonar-reasoning-pro", - ], - ) - def test_perplexity_reasoning_effort_mock_completion(self, model): - """ - Test that reasoning_effort is correctly passed in actual completion call (mocked) - """ - from openai import OpenAI - from openai.types.chat.chat_completion import ChatCompletion - - litellm.set_verbose = True - - # Mock successful response with reasoning content - response_object = { - "id": "cmpl-test", - "object": "chat.completion", - "created": 1677652288, - "model": model.split("/")[1], - "choices": [ - { - "index": 0, - "message": { - "role": "assistant", - "content": "This is a test response from the reasoning model.", - "reasoning_content": "Let me think about this step by step...", - }, - "finish_reason": "stop", - } - ], - "usage": { - "prompt_tokens": 9, - "completion_tokens": 20, - "total_tokens": 29, - "completion_tokens_details": {"reasoning_tokens": 15}, - }, - } - - pydantic_obj = ChatCompletion(**response_object) - - def _return_pydantic_obj(*args, **kwargs): - new_response = MagicMock() - new_response.headers = {"content-type": "application/json"} - new_response.parse.return_value = pydantic_obj - return new_response - - openai_client = OpenAI(api_key="fake-api-key") - - with patch.object( - openai_client.chat.completions.with_raw_response, - "create", - side_effect=_return_pydantic_obj, - ) as mock_client: - - response = completion( - model=model, - messages=[ - { - "role": "user", - "content": "Hello, please think about this carefully.", - } - ], - reasoning_effort="high", - client=openai_client, - ) - - # Verify the call was made - assert mock_client.called - - # Get the request data from the mock call - call_args = mock_client.call_args - request_data = call_args.kwargs - - # Verify reasoning_effort was included in the request - assert "reasoning_effort" in request_data - assert request_data["reasoning_effort"] == "high" - - # Verify response structure - assert response.choices[0].message.content is not None - assert ( - response.choices[0].message.content - == "This is a test response from the reasoning model." - ) - def test_perplexity_reasoning_models_support_reasoning(self): """ Test that Perplexity Sonar reasoning models are correctly identified as supporting reasoning