test(base_llm_unit_tests): delete BaseOSeriesModelsTest and dead tool_call_no_arguments scaffolding per CI audit

Per the keep/drop audit (6):

- BaseOSeriesModelsTest (all 3 methods: test_reasoning_effort,
  test_developer_role_translation, test_completion_o_series_models_temperature)
  deleted: each patches client.chat.completions.with_raw_response.create then
  asserts called-once plus a kwarg, textbook patterns a/c. The real o-series
  transforms (max_tokens to max_completion_tokens, system to user) stay
  covered by the keepers in test_openai_o1.py. Subclasses rebased onto
  BaseLLMChatTest alone; TestAzureOpenAIO3 deleted outright since
  BaseOSeriesModelsTest was its only base, leaving zero tests
- BaseLLMChatTest.test_tool_call_no_arguments deleted: abstract with a pass
  body, forcing every subclass to carry an override. The 18 trivial pass/skip
  overrides across 15 files are deleted with it; the 3 substantive overrides
  (anthropic, gemini, huggingface) remain as ordinary tests. The
  tool_call_no_arguments fixture stays for those
- All 14 @pytest.mark.flaky lines stripped (8d.3): every marker sat on a
  live-call method; provider flake is the harness's problem (it owns retry
  policy), and the deterministic survivors must not paper over real failures
  with retries
This commit is contained in:
mateo-berri 2026-06-11 18:59:41 +00:00
parent e86143fa62
commit 417b57e8ef
15 changed files with 7 additions and 239 deletions

View file

@ -2,33 +2,25 @@ import httpx
import json
import pytest
import sys
from typing import Any, Dict, List
from unittest.mock import MagicMock, Mock, patch
import os
from litellm._uuid import uuid
import time
import base64
import inspect
sys.path.insert(
0, os.path.abspath("../..")
) # Adds the parent directory to the system path
import litellm
from litellm.exceptions import BadRequestError
from litellm.llms.custom_httpx.http_handler import AsyncHTTPHandler, HTTPHandler
from litellm.utils import (
CustomStreamWrapper,
get_supported_openai_params,
get_optional_params,
ProviderConfigManager,
)
from litellm.main import stream_chunk_builder
from typing import Union
from litellm.types.utils import Usage, ModelResponse
# test_example.py
from abc import ABC, abstractmethod
from openai import OpenAI
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")))
@ -191,7 +183,6 @@ class BaseLLMChatTest(ABC):
print(response)
print(json.dumps(response, indent=4, default=str))
@pytest.mark.flaky(retries=3, delay=1)
def test_tool_call_with_empty_enum_property(self):
litellm._turn_on_debug()
from litellm.utils import supports_function_calling
@ -291,7 +282,7 @@ class BaseLLMChatTest(ABC):
def test_pydantic_model_input(self):
litellm.set_verbose = True
from litellm import completion, Message
from litellm import Message
base_completion_call_args = self.get_base_completion_call_args()
messages = [Message(content="Hello, how are you?", role="user")]
@ -483,7 +474,6 @@ class BaseLLMChatTest(ABC):
{"type": "text"},
],
)
@pytest.mark.flaky(retries=6, delay=1)
def test_json_response_format(self, response_format):
"""
Test that the JSON response format is supported by the LLM API
@ -525,7 +515,6 @@ class BaseLLMChatTest(ABC):
{"type": "text"},
],
)
@pytest.mark.flaky(retries=6, delay=1)
def test_response_format_type_text_with_tool_calls_no_tool_choice(
self, response_format
):
@ -603,7 +592,6 @@ class BaseLLMChatTest(ABC):
print(f"translated_params={translated_params}")
@pytest.mark.flaky(retries=6, delay=1)
def test_json_response_pydantic_obj(self):
litellm._turn_on_debug()
from pydantic import BaseModel
@ -643,16 +631,12 @@ class BaseLLMChatTest(ABC):
except litellm.InternalServerError:
pytest.skip("Model is overloaded")
@pytest.mark.flaky(retries=6, delay=1)
def test_json_response_pydantic_obj_nested_obj(self):
litellm.set_verbose = True
from pydantic import BaseModel
from litellm.utils import supports_response_schema
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
@pytest.mark.flaky(retries=6, delay=1)
def test_json_response_nested_pydantic_obj(self):
from pydantic import BaseModel
from litellm.utils import supports_response_schema
@ -696,7 +680,6 @@ class BaseLLMChatTest(ABC):
except litellm.InternalServerError:
pytest.skip("Model is overloaded")
@pytest.mark.flaky(retries=6, delay=1)
def test_json_response_nested_json_schema(self):
"""
PROD Test: ensure nested json schema sent to proxy works as expected.
@ -747,7 +730,6 @@ class BaseLLMChatTest(ABC):
except litellm.InternalServerError:
pytest.skip("Model is overloaded")
@pytest.mark.flaky(retries=6, delay=1)
def test_audio_input(self):
"""
Test that audio input is supported by the LLM API
@ -785,7 +767,6 @@ class BaseLLMChatTest(ABC):
print(completion.choices[0].message)
@pytest.mark.flaky(retries=6, delay=1)
def test_json_response_format_stream(self):
"""
Test that the JSON response format with streaming is supported by the LLM API
@ -848,11 +829,6 @@ class BaseLLMChatTest(ABC):
],
}
@abstractmethod
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
@pytest.mark.parametrize("detail", [None, "low", "high"])
@pytest.mark.parametrize(
"image_url",
@ -865,7 +841,6 @@ class BaseLLMChatTest(ABC):
"https://awsmp-logos.s3.amazonaws.com/seller-xw5kijmvmzasy/c233c9ade2ccb5491072ae232c814942.png",
],
)
@pytest.mark.flaky(retries=4, delay=2)
def test_image_url(self, detail, image_url):
litellm.set_verbose = True
from litellm.utils import supports_vision
@ -962,7 +937,6 @@ class BaseLLMChatTest(ABC):
assert response is not None
@pytest.mark.flaky(retries=4, delay=1)
def test_prompt_caching(self):
_skip_live_prompt_caching_test()
print("test_prompt_caching")
@ -1081,13 +1055,12 @@ class BaseLLMChatTest(ABC):
return url
@pytest.mark.flaky(retries=3, delay=1)
def test_empty_tools(self):
"""
Related Issue: https://github.com/BerriAI/litellm/issues/9080
"""
try:
from litellm import completion, ModelResponse
from litellm import completion
litellm.set_verbose = True
litellm._turn_on_debug()
@ -1117,7 +1090,6 @@ class BaseLLMChatTest(ABC):
except Exception as e:
pytest.fail(f"Error occurred: {e}")
@pytest.mark.flaky(retries=3, delay=1)
def test_basic_tool_calling(self):
try:
from litellm import completion, ModelResponse
@ -1240,10 +1212,8 @@ class BaseLLMChatTest(ABC):
except Exception as e:
pytest.fail(f"Error occurred: {e}")
@pytest.mark.flaky(retries=3, delay=1)
@pytest.mark.asyncio
async def test_completion_cost(self):
from litellm import completion_cost
litellm._turn_on_debug()
@ -1471,114 +1441,6 @@ class BaseLLMChatTest(ABC):
pytest.fail(f"Error: {e}")
class BaseOSeriesModelsTest(ABC): # test across azure/openai
@abstractmethod
def get_base_completion_call_args(self):
pass
@abstractmethod
def get_client(self) -> OpenAI:
pass
def test_reasoning_effort(self):
"""Test that reasoning_effort is passed correctly to the model"""
from litellm import completion
client = self.get_client()
completion_args = self.get_base_completion_call_args()
with patch.object(
client.chat.completions.with_raw_response, "create"
) as mock_client:
try:
completion(
**completion_args,
reasoning_effort="low",
messages=[{"role": "user", "content": "Hello!"}],
client=client,
)
except Exception as e:
print(f"Error: {e}")
mock_client.assert_called_once()
request_body = mock_client.call_args.kwargs
print("request_body: ", request_body)
assert request_body["reasoning_effort"] == "low"
def test_developer_role_translation(self):
"""Test that developer role is translated correctly to system role for non-OpenAI providers"""
from litellm import completion
client = self.get_client()
completion_args = self.get_base_completion_call_args()
with patch.object(
client.chat.completions.with_raw_response, "create"
) as mock_client:
try:
completion(
**completion_args,
reasoning_effort="low",
messages=[
{"role": "developer", "content": "Be a good bot!"},
{"role": "user", "content": "Hello!"},
],
client=client,
)
except Exception as e:
print(f"Error: {e}")
mock_client.assert_called_once()
request_body = mock_client.call_args.kwargs
print("request_body: ", request_body)
assert (
request_body["messages"][0]["role"] == "developer"
), "Got={} instead of system".format(request_body["messages"][0]["role"])
assert request_body["messages"][0]["content"] == "Be a good bot!"
def test_completion_o_series_models_temperature(self):
"""
Test that temperature is not passed to O-series models
"""
try:
from litellm import completion
client = self.get_client()
completion_args = self.get_base_completion_call_args()
with patch.object(
client.chat.completions.with_raw_response, "create"
) as mock_client:
try:
completion(
**completion_args,
temperature=0.0,
messages=[
{
"role": "user",
"content": "Hello, world!",
}
],
drop_params=True,
client=client,
)
except Exception as e:
print(f"Error: {e}")
mock_client.assert_called_once()
request_body = mock_client.call_args.kwargs
print("request_body: ", request_body)
assert (
"temperature" not in request_body
), "temperature should not be in the request body"
except Exception as e:
pytest.fail(f"Error occurred: {e}")
class BaseAnthropicChatTest(ABC):
"""
Ensures consistent result across anthropic model usage
@ -1681,7 +1543,6 @@ class BaseAnthropicChatTest(ABC):
print(response)
def test_completion_thinking_with_max_tokens(self):
from pydantic import BaseModel
litellm._turn_on_debug()
@ -1697,7 +1558,6 @@ class BaseAnthropicChatTest(ABC):
print(response)
def test_completion_thinking_without_max_tokens(self):
from pydantic import BaseModel
litellm._turn_on_debug()

View file

@ -8,10 +8,10 @@ sys.path.insert(
import litellm
from base_llm_unit_tests import BaseLLMChatTest, BaseOSeriesModelsTest
from base_llm_unit_tests import BaseLLMChatTest
class TestAzureOpenAIO3Mini(BaseOSeriesModelsTest, BaseLLMChatTest):
class TestAzureOpenAIO3Mini(BaseLLMChatTest):
def get_base_completion_call_args(self):
# Clear the LLM client cache to prevent test pollution from cached clients
litellm.in_memory_llm_clients_cache.flush_cache()
@ -31,10 +31,6 @@ class TestAzureOpenAIO3Mini(BaseOSeriesModelsTest, BaseLLMChatTest):
api_version="2024-02-15-preview",
)
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
def test_basic_tool_calling(self):
pass
@ -73,21 +69,3 @@ class TestAzureOpenAIO3Mini(BaseOSeriesModelsTest, BaseLLMChatTest):
assert fake_stream is False
class TestAzureOpenAIO3(BaseOSeriesModelsTest):
def get_base_completion_call_args(self):
return {
"model": "azure/o3-mini",
"api_key": "my-fake-o1-key",
"api_base": "https://openai-gpt-4-test-v-1.openai.azure.com",
}
def get_client(self):
from openai import AzureOpenAI
return AzureOpenAI(
api_key="my-fake-o1-key",
base_url="https://openai-gpt-4-test-v-1.openai.azure.com",
api_version="2024-02-15-preview",
)

View file

@ -1903,10 +1903,6 @@ class TestBedrockConverseChatCrossRegion(BaseLLMChatTest):
"model": "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0",
}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
def test_prompt_caching(self):
"""
Remove override once we have access to Bedrock prompt caching
@ -1956,11 +1952,6 @@ class TestBedrockConverseChatNormal(BaseLLMChatTest):
"aws_region_name": "us-east-1",
}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
class TestBedrockConverseNovaTestSuite(BaseLLMChatTest):
def get_base_completion_call_args(self) -> dict:
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
@ -1971,10 +1962,6 @@ class TestBedrockConverseNovaTestSuite(BaseLLMChatTest):
"aws_region_name": "us-east-1",
}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
def test_prompt_caching(self):
"""
TODO: Ensure this test passes our base llm test suite

View file

@ -19,10 +19,6 @@ class TestBedrockGPTOSS(BaseLLMChatTest):
"model": "bedrock/converse/openai.gpt-oss-20b-1:0",
}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
def test_function_calling_with_tool_response(self):
"""Bedrock GPT-OSS intermittently emits truncated toolUse.input deltas on
the live endpoint, which makes the inherited live integration test flaky.

View file

@ -18,21 +18,12 @@ class TestBedrockInvokeClaudeJson(BaseLLMChatTest):
"model": "bedrock/invoke/us.anthropic.claude-haiku-4-5-20251001-v1:0",
}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
class TestBedrockInvokeNovaJson(BaseLLMChatTest):
def get_base_completion_call_args(self) -> dict:
return {
"model": "bedrock/invoke/us.amazon.nova-micro-v1:0",
}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
@pytest.fixture(autouse=True)
def skip_non_json_tests(self, request):
if not "json" in request.function.__name__.lower():

View file

@ -10,9 +10,6 @@ import litellm
class TestBedrockTestSuite(BaseLLMChatTest):
def test_tool_call_no_arguments(self, tool_call_no_arguments):
pass
def get_base_completion_call_args(self) -> dict:
litellm._turn_on_debug()
return {

View file

@ -36,10 +36,6 @@ class TestBedrockMoonshotInvoke(BaseLLMChatTest):
"model": "bedrock/invoke/moonshot.kimi-k2-thinking",
}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly."""
pass
# ---------------------------------------------------------------------
# The overrides below replace inherited BaseLLMChatTest tests that would
# otherwise make live AWS Bedrock calls. The live versions were

View file

@ -22,10 +22,6 @@ class TestBedrockNovaJson(BaseLLMChatTest):
def test_json_response_nested_json_schema(self):
pass
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
def test_prompt_caching(self):
"""
Remove override once we have access to Bedrock prompt caching

View file

@ -63,8 +63,3 @@ class TestDatabricksCompletion(BaseLLMChatTest, BaseAnthropicChatTest):
def test_pdf_handling(self, pdf_messages):
pytest.skip("Databricks does not support PDF handling")
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pytest.skip("Databricks is openai compatible")

View file

@ -11,11 +11,6 @@ class TestDeepSeekChatCompletion(BaseLLMChatTest):
"model": "deepseek/deepseek-reasoner",
}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
def test_completion_cost_deepseek():
litellm.set_verbose = True
model_name = "deepseek/deepseek-chat"

View file

@ -23,10 +23,6 @@ class TestGroq(BaseLLMChatTest):
"model": "groq/llama-3.3-70b-versatile",
}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
def test_tool_call_with_empty_enum_property(self):
pass

View file

@ -34,6 +34,3 @@ class TestMistralCompletion(BaseLLMChatTest):
litellm.set_verbose = True
return {"model": "mistral/mistral-medium-latest"}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass

View file

@ -145,10 +145,6 @@ class TestOpenAIChatCompletion(BaseLLMChatTest):
def get_base_completion_call_args(self) -> dict:
return {"model": "gpt-4o-mini"}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
def test_prompt_caching(self):
"""
Test that prompt caching works correctly.

View file

@ -11,7 +11,7 @@ import pytest
import litellm
from litellm import ModelResponse
from base_llm_unit_tests import BaseLLMChatTest, BaseOSeriesModelsTest
from base_llm_unit_tests import BaseLLMChatTest
@pytest.mark.parametrize("model", ["o1"])
@ -110,7 +110,7 @@ def test_litellm_responses():
assert isinstance(response.usage.completion_tokens_details, CompletionTokensDetails)
class TestOpenAIO1(BaseOSeriesModelsTest, BaseLLMChatTest):
class TestOpenAIO1(BaseLLMChatTest):
def get_base_completion_call_args(self):
return {
"model": "o1",
@ -121,16 +121,12 @@ class TestOpenAIO1(BaseOSeriesModelsTest, BaseLLMChatTest):
return OpenAI(api_key="fake-api-key")
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
def test_prompt_caching(self):
"""Temporary override. o1 prompt caching is not working."""
pass
class TestOpenAIO3(BaseOSeriesModelsTest, BaseLLMChatTest):
class TestOpenAIO3(BaseLLMChatTest):
def get_base_completion_call_args(self):
return {
"model": "o3-mini",
@ -141,10 +137,6 @@ class TestOpenAIO3(BaseOSeriesModelsTest, BaseLLMChatTest):
return OpenAI(api_key="fake-api-key")
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
def test_prompt_caching(self):
"""Override, as o3 prompt caching is flaky"""
pass

View file

@ -45,10 +45,6 @@ class TestRouterLLMTranslation(BaseLLMChatTest):
def get_base_completion_call_args(self) -> dict:
return {"model": "gpt-4o-mini"}
def test_tool_call_no_arguments(self, tool_call_no_arguments):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
def test_prompt_caching(self):
"""
Works locally but CI/CD is failing this test. Temporary skip to push out a new release.