diff --git a/tests/llm_translation/base_llm_unit_tests.py b/tests/llm_translation/base_llm_unit_tests.py index fef1d23d867..791d417d4cb 100644 --- a/tests/llm_translation/base_llm_unit_tests.py +++ b/tests/llm_translation/base_llm_unit_tests.py @@ -2,33 +2,25 @@ import httpx import json import pytest import sys -from typing import Any, Dict, List -from unittest.mock import MagicMock, Mock, patch import os from litellm._uuid import uuid import time import base64 -import inspect sys.path.insert( 0, os.path.abspath("../..") ) # Adds the parent directory to the system path import litellm -from litellm.exceptions import BadRequestError -from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler from litellm.utils import ( CustomStreamWrapper, - get_supported_openai_params, get_optional_params, ProviderConfigManager, ) from litellm.main import stream_chunk_builder -from typing import Union from litellm.types.utils import Usage, ModelResponse # test_example.py from abc import ABC, abstractmethod -from openai import OpenAI sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", ".."))) @@ -191,7 +183,6 @@ class BaseLLMChatTest(ABC): print(response) print(json.dumps(response, indent=4, default=str)) - @pytest.mark.flaky(retries=3, delay=1) def test_tool_call_with_empty_enum_property(self): litellm._turn_on_debug() from litellm.utils import supports_function_calling @@ -291,7 +282,7 @@ class BaseLLMChatTest(ABC): def test_pydantic_model_input(self): litellm.set_verbose = True - from litellm import completion, Message + from litellm import Message base_completion_call_args = self.get_base_completion_call_args() messages = [Message(content="Hello, how are you?", role="user")] @@ -483,7 +474,6 @@ class BaseLLMChatTest(ABC): {"type": "text"}, ], ) - @pytest.mark.flaky(retries=6, delay=1) def test_json_response_format(self, response_format): """ Test that the JSON response format is supported by the LLM API @@ -525,7 +515,6 @@ class BaseLLMChatTest(ABC): {"type": "text"}, ], ) - @pytest.mark.flaky(retries=6, delay=1) def test_response_format_type_text_with_tool_calls_no_tool_choice( self, response_format ): @@ -603,7 +592,6 @@ class BaseLLMChatTest(ABC): print(f"translated_params={translated_params}") - @pytest.mark.flaky(retries=6, delay=1) def test_json_response_pydantic_obj(self): litellm._turn_on_debug() from pydantic import BaseModel @@ -643,16 +631,12 @@ class BaseLLMChatTest(ABC): except litellm.InternalServerError: pytest.skip("Model is overloaded") - @pytest.mark.flaky(retries=6, delay=1) def test_json_response_pydantic_obj_nested_obj(self): litellm.set_verbose = True - from pydantic import BaseModel - from litellm.utils import supports_response_schema os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" litellm.model_cost = litellm.get_model_cost_map(url="") - @pytest.mark.flaky(retries=6, delay=1) def test_json_response_nested_pydantic_obj(self): from pydantic import BaseModel from litellm.utils import supports_response_schema @@ -696,7 +680,6 @@ class BaseLLMChatTest(ABC): except litellm.InternalServerError: pytest.skip("Model is overloaded") - @pytest.mark.flaky(retries=6, delay=1) def test_json_response_nested_json_schema(self): """ PROD Test: ensure nested json schema sent to proxy works as expected. @@ -747,7 +730,6 @@ class BaseLLMChatTest(ABC): except litellm.InternalServerError: pytest.skip("Model is overloaded") - @pytest.mark.flaky(retries=6, delay=1) def test_audio_input(self): """ Test that audio input is supported by the LLM API @@ -785,7 +767,6 @@ class BaseLLMChatTest(ABC): print(completion.choices[0].message) - @pytest.mark.flaky(retries=6, delay=1) def test_json_response_format_stream(self): """ Test that the JSON response format with streaming is supported by the LLM API @@ -848,11 +829,6 @@ class BaseLLMChatTest(ABC): ], } - @abstractmethod - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - @pytest.mark.parametrize("detail", [None, "low", "high"]) @pytest.mark.parametrize( "image_url", @@ -865,7 +841,6 @@ class BaseLLMChatTest(ABC): "https://awsmp-logos.s3.amazonaws.com/seller-xw5kijmvmzasy/c233c9ade2ccb5491072ae232c814942.png", ], ) - @pytest.mark.flaky(retries=4, delay=2) def test_image_url(self, detail, image_url): litellm.set_verbose = True from litellm.utils import supports_vision @@ -962,7 +937,6 @@ class BaseLLMChatTest(ABC): assert response is not None - @pytest.mark.flaky(retries=4, delay=1) def test_prompt_caching(self): _skip_live_prompt_caching_test() print("test_prompt_caching") @@ -1081,13 +1055,12 @@ class BaseLLMChatTest(ABC): return url - @pytest.mark.flaky(retries=3, delay=1) def test_empty_tools(self): """ Related Issue: https://github.com/BerriAI/litellm/issues/9080 """ try: - from litellm import completion, ModelResponse + from litellm import completion litellm.set_verbose = True litellm._turn_on_debug() @@ -1117,7 +1090,6 @@ class BaseLLMChatTest(ABC): except Exception as e: pytest.fail(f"Error occurred: {e}") - @pytest.mark.flaky(retries=3, delay=1) def test_basic_tool_calling(self): try: from litellm import completion, ModelResponse @@ -1240,10 +1212,8 @@ class BaseLLMChatTest(ABC): except Exception as e: pytest.fail(f"Error occurred: {e}") - @pytest.mark.flaky(retries=3, delay=1) @pytest.mark.asyncio async def test_completion_cost(self): - from litellm import completion_cost litellm._turn_on_debug() @@ -1471,114 +1441,6 @@ class BaseLLMChatTest(ABC): pytest.fail(f"Error: {e}") -class BaseOSeriesModelsTest(ABC): # test across azure/openai - @abstractmethod - def get_base_completion_call_args(self): - pass - - @abstractmethod - def get_client(self) -> OpenAI: - pass - - def test_reasoning_effort(self): - """Test that reasoning_effort is passed correctly to the model""" - - from litellm import completion - - client = self.get_client() - - completion_args = self.get_base_completion_call_args() - - with patch.object( - client.chat.completions.with_raw_response, "create" - ) as mock_client: - try: - completion( - **completion_args, - reasoning_effort="low", - messages=[{"role": "user", "content": "Hello!"}], - client=client, - ) - except Exception as e: - print(f"Error: {e}") - - mock_client.assert_called_once() - request_body = mock_client.call_args.kwargs - print("request_body: ", request_body) - assert request_body["reasoning_effort"] == "low" - - def test_developer_role_translation(self): - """Test that developer role is translated correctly to system role for non-OpenAI providers""" - from litellm import completion - - client = self.get_client() - - completion_args = self.get_base_completion_call_args() - - with patch.object( - client.chat.completions.with_raw_response, "create" - ) as mock_client: - try: - completion( - **completion_args, - reasoning_effort="low", - messages=[ - {"role": "developer", "content": "Be a good bot!"}, - {"role": "user", "content": "Hello!"}, - ], - client=client, - ) - except Exception as e: - print(f"Error: {e}") - - mock_client.assert_called_once() - request_body = mock_client.call_args.kwargs - print("request_body: ", request_body) - assert ( - request_body["messages"][0]["role"] == "developer" - ), "Got={} instead of system".format(request_body["messages"][0]["role"]) - assert request_body["messages"][0]["content"] == "Be a good bot!" - - def test_completion_o_series_models_temperature(self): - """ - Test that temperature is not passed to O-series models - """ - try: - from litellm import completion - - client = self.get_client() - - completion_args = self.get_base_completion_call_args() - - with patch.object( - client.chat.completions.with_raw_response, "create" - ) as mock_client: - try: - completion( - **completion_args, - temperature=0.0, - messages=[ - { - "role": "user", - "content": "Hello, world!", - } - ], - drop_params=True, - client=client, - ) - except Exception as e: - print(f"Error: {e}") - - mock_client.assert_called_once() - request_body = mock_client.call_args.kwargs - print("request_body: ", request_body) - assert ( - "temperature" not in request_body - ), "temperature should not be in the request body" - except Exception as e: - pytest.fail(f"Error occurred: {e}") - - class BaseAnthropicChatTest(ABC): """ Ensures consistent result across anthropic model usage @@ -1681,7 +1543,6 @@ class BaseAnthropicChatTest(ABC): print(response) def test_completion_thinking_with_max_tokens(self): - from pydantic import BaseModel litellm._turn_on_debug() @@ -1697,7 +1558,6 @@ class BaseAnthropicChatTest(ABC): print(response) def test_completion_thinking_without_max_tokens(self): - from pydantic import BaseModel litellm._turn_on_debug() diff --git a/tests/llm_translation/test_azure_o_series.py b/tests/llm_translation/test_azure_o_series.py index ed259a620ad..e6db69c7b27 100644 --- a/tests/llm_translation/test_azure_o_series.py +++ b/tests/llm_translation/test_azure_o_series.py @@ -8,10 +8,10 @@ sys.path.insert( import litellm -from base_llm_unit_tests import BaseLLMChatTest, BaseOSeriesModelsTest +from base_llm_unit_tests import BaseLLMChatTest -class TestAzureOpenAIO3Mini(BaseOSeriesModelsTest, BaseLLMChatTest): +class TestAzureOpenAIO3Mini(BaseLLMChatTest): def get_base_completion_call_args(self): # Clear the LLM client cache to prevent test pollution from cached clients litellm.in_memory_llm_clients_cache.flush_cache() @@ -31,10 +31,6 @@ class TestAzureOpenAIO3Mini(BaseOSeriesModelsTest, BaseLLMChatTest): api_version="2024-02-15-preview", ) - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - def test_basic_tool_calling(self): pass @@ -73,21 +69,3 @@ class TestAzureOpenAIO3Mini(BaseOSeriesModelsTest, BaseLLMChatTest): assert fake_stream is False -class TestAzureOpenAIO3(BaseOSeriesModelsTest): - def get_base_completion_call_args(self): - return { - "model": "azure/o3-mini", - "api_key": "my-fake-o1-key", - "api_base": "https://openai-gpt-4-test-v-1.openai.azure.com", - } - - def get_client(self): - from openai import AzureOpenAI - - return AzureOpenAI( - api_key="my-fake-o1-key", - base_url="https://openai-gpt-4-test-v-1.openai.azure.com", - api_version="2024-02-15-preview", - ) - - diff --git a/tests/llm_translation/test_bedrock_completion.py b/tests/llm_translation/test_bedrock_completion.py index c684749f3d3..86284f7c1eb 100644 --- a/tests/llm_translation/test_bedrock_completion.py +++ b/tests/llm_translation/test_bedrock_completion.py @@ -1903,10 +1903,6 @@ class TestBedrockConverseChatCrossRegion(BaseLLMChatTest): "model": "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0", } - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - def test_prompt_caching(self): """ Remove override once we have access to Bedrock prompt caching @@ -1956,11 +1952,6 @@ class TestBedrockConverseChatNormal(BaseLLMChatTest): "aws_region_name": "us-east-1", } - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - - class TestBedrockConverseNovaTestSuite(BaseLLMChatTest): def get_base_completion_call_args(self) -> dict: os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" @@ -1971,10 +1962,6 @@ class TestBedrockConverseNovaTestSuite(BaseLLMChatTest): "aws_region_name": "us-east-1", } - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - def test_prompt_caching(self): """ TODO: Ensure this test passes our base llm test suite diff --git a/tests/llm_translation/test_bedrock_gpt_oss.py b/tests/llm_translation/test_bedrock_gpt_oss.py index 0a595ad7114..09f009675db 100644 --- a/tests/llm_translation/test_bedrock_gpt_oss.py +++ b/tests/llm_translation/test_bedrock_gpt_oss.py @@ -19,10 +19,6 @@ class TestBedrockGPTOSS(BaseLLMChatTest): "model": "bedrock/converse/openai.gpt-oss-20b-1:0", } - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - def test_function_calling_with_tool_response(self): """Bedrock GPT-OSS intermittently emits truncated toolUse.input deltas on the live endpoint, which makes the inherited live integration test flaky. diff --git a/tests/llm_translation/test_bedrock_invoke_tests.py b/tests/llm_translation/test_bedrock_invoke_tests.py index 901b43542f7..b74bfb1dc57 100644 --- a/tests/llm_translation/test_bedrock_invoke_tests.py +++ b/tests/llm_translation/test_bedrock_invoke_tests.py @@ -18,21 +18,12 @@ class TestBedrockInvokeClaudeJson(BaseLLMChatTest): "model": "bedrock/invoke/us.anthropic.claude-haiku-4-5-20251001-v1:0", } - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - - class TestBedrockInvokeNovaJson(BaseLLMChatTest): def get_base_completion_call_args(self) -> dict: return { "model": "bedrock/invoke/us.amazon.nova-micro-v1:0", } - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - @pytest.fixture(autouse=True) def skip_non_json_tests(self, request): if not "json" in request.function.__name__.lower(): diff --git a/tests/llm_translation/test_bedrock_llama.py b/tests/llm_translation/test_bedrock_llama.py index b18928747eb..543a5842dd5 100644 --- a/tests/llm_translation/test_bedrock_llama.py +++ b/tests/llm_translation/test_bedrock_llama.py @@ -10,9 +10,6 @@ import litellm class TestBedrockTestSuite(BaseLLMChatTest): - def test_tool_call_no_arguments(self, tool_call_no_arguments): - pass - def get_base_completion_call_args(self) -> dict: litellm._turn_on_debug() return { diff --git a/tests/llm_translation/test_bedrock_moonshot.py b/tests/llm_translation/test_bedrock_moonshot.py index 5bf4c20694c..2b79fdc82a5 100644 --- a/tests/llm_translation/test_bedrock_moonshot.py +++ b/tests/llm_translation/test_bedrock_moonshot.py @@ -36,10 +36,6 @@ class TestBedrockMoonshotInvoke(BaseLLMChatTest): "model": "bedrock/invoke/moonshot.kimi-k2-thinking", } - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly.""" - pass - # --------------------------------------------------------------------- # The overrides below replace inherited BaseLLMChatTest tests that would # otherwise make live AWS Bedrock calls. The live versions were diff --git a/tests/llm_translation/test_bedrock_nova_json.py b/tests/llm_translation/test_bedrock_nova_json.py index 7531891c4ef..c7a42b047bc 100644 --- a/tests/llm_translation/test_bedrock_nova_json.py +++ b/tests/llm_translation/test_bedrock_nova_json.py @@ -22,10 +22,6 @@ class TestBedrockNovaJson(BaseLLMChatTest): def test_json_response_nested_json_schema(self): pass - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - def test_prompt_caching(self): """ Remove override once we have access to Bedrock prompt caching diff --git a/tests/llm_translation/test_databricks.py b/tests/llm_translation/test_databricks.py index d2d51b099c7..fec82b45043 100644 --- a/tests/llm_translation/test_databricks.py +++ b/tests/llm_translation/test_databricks.py @@ -63,8 +63,3 @@ class TestDatabricksCompletion(BaseLLMChatTest, BaseAnthropicChatTest): def test_pdf_handling(self, pdf_messages): pytest.skip("Databricks does not support PDF handling") - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pytest.skip("Databricks is openai compatible") - - diff --git a/tests/llm_translation/test_deepseek_completion.py b/tests/llm_translation/test_deepseek_completion.py index cf31b882c14..11914826996 100644 --- a/tests/llm_translation/test_deepseek_completion.py +++ b/tests/llm_translation/test_deepseek_completion.py @@ -11,11 +11,6 @@ class TestDeepSeekChatCompletion(BaseLLMChatTest): "model": "deepseek/deepseek-reasoner", } - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - - def test_completion_cost_deepseek(): litellm.set_verbose = True model_name = "deepseek/deepseek-chat" diff --git a/tests/llm_translation/test_groq.py b/tests/llm_translation/test_groq.py index cf4be9e801e..3543c4fe9d1 100644 --- a/tests/llm_translation/test_groq.py +++ b/tests/llm_translation/test_groq.py @@ -23,10 +23,6 @@ class TestGroq(BaseLLMChatTest): "model": "groq/llama-3.3-70b-versatile", } - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - def test_tool_call_with_empty_enum_property(self): pass diff --git a/tests/llm_translation/test_mistral_api.py b/tests/llm_translation/test_mistral_api.py index 8cf704fbe89..6709838fa6c 100644 --- a/tests/llm_translation/test_mistral_api.py +++ b/tests/llm_translation/test_mistral_api.py @@ -34,6 +34,3 @@ class TestMistralCompletion(BaseLLMChatTest): litellm.set_verbose = True return {"model": "mistral/mistral-medium-latest"} - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass diff --git a/tests/llm_translation/test_openai.py b/tests/llm_translation/test_openai.py index 26b21143dc4..625fe56077c 100644 --- a/tests/llm_translation/test_openai.py +++ b/tests/llm_translation/test_openai.py @@ -145,10 +145,6 @@ class TestOpenAIChatCompletion(BaseLLMChatTest): def get_base_completion_call_args(self) -> dict: return {"model": "gpt-4o-mini"} - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - def test_prompt_caching(self): """ Test that prompt caching works correctly. diff --git a/tests/llm_translation/test_openai_o1.py b/tests/llm_translation/test_openai_o1.py index 54b1734884e..3bbb2fc9aa9 100644 --- a/tests/llm_translation/test_openai_o1.py +++ b/tests/llm_translation/test_openai_o1.py @@ -11,7 +11,7 @@ import pytest import litellm from litellm import ModelResponse -from base_llm_unit_tests import BaseLLMChatTest, BaseOSeriesModelsTest +from base_llm_unit_tests import BaseLLMChatTest @pytest.mark.parametrize("model", ["o1"]) @@ -110,7 +110,7 @@ def test_litellm_responses(): assert isinstance(response.usage.completion_tokens_details, CompletionTokensDetails) -class TestOpenAIO1(BaseOSeriesModelsTest, BaseLLMChatTest): +class TestOpenAIO1(BaseLLMChatTest): def get_base_completion_call_args(self): return { "model": "o1", @@ -121,16 +121,12 @@ class TestOpenAIO1(BaseOSeriesModelsTest, BaseLLMChatTest): return OpenAI(api_key="fake-api-key") - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - def test_prompt_caching(self): """Temporary override. o1 prompt caching is not working.""" pass -class TestOpenAIO3(BaseOSeriesModelsTest, BaseLLMChatTest): +class TestOpenAIO3(BaseLLMChatTest): def get_base_completion_call_args(self): return { "model": "o3-mini", @@ -141,10 +137,6 @@ class TestOpenAIO3(BaseOSeriesModelsTest, BaseLLMChatTest): return OpenAI(api_key="fake-api-key") - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - def test_prompt_caching(self): """Override, as o3 prompt caching is flaky""" pass diff --git a/tests/llm_translation/test_router_llm_translation_tests.py b/tests/llm_translation/test_router_llm_translation_tests.py index 26456ab0a35..dd579715c7e 100644 --- a/tests/llm_translation/test_router_llm_translation_tests.py +++ b/tests/llm_translation/test_router_llm_translation_tests.py @@ -45,10 +45,6 @@ class TestRouterLLMTranslation(BaseLLMChatTest): def get_base_completion_call_args(self) -> dict: return {"model": "gpt-4o-mini"} - def test_tool_call_no_arguments(self, tool_call_no_arguments): - """Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833""" - pass - def test_prompt_caching(self): """ Works locally but CI/CD is failing this test. Temporary skip to push out a new release.