test: remove 130 legacy tests owned by stronger unit proofs (#44157)

* test: remove 130 legacy tests owned by stronger unit proofs

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

* test: list router _embedding and _aembedding as covered via public embedding calls

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>

---------

Co-authored-by: yuneng <yuneng@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-02 10:18:50 -07:00 • committed by GitHub
parent 19da81579b
commit 6c2ede00ac
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
40 changed files with 133 additions and 1833 deletions

View file

@ -320,16 +320,6 @@ def test_audio_speech_cost_calc():
assert standard_logging_payload["response_cost"] > 0
def test_audio_speech_gemini():
result = litellm.speech(
model="gemini/gemini-2.5-flash-preview-tts",
input="the quick brown fox jumped over the lazy dogs",
api_key=os.getenv("GEMINI_API_KEY"),
)
print(result)
@pytest.mark.asyncio
@pytest.mark.flaky(retries=3, delay=1)
async def test_azure_ava_tts_async():

View file

@ -138,17 +138,6 @@ async def test_whisper_log_pre_call():
mock_log_pre_call.assert_called_once()
@pytest.mark.asyncio
async def test_gpt_4o_transcribe():
from litellm.litellm_core_utils.litellm_logging import Logging
from datetime import datetime
from unittest.mock import patch, MagicMock
await litellm.atranscription(
model="openai/gpt-4o-transcribe", file=_audio_file(), response_format="json"
)
@pytest.mark.asyncio
async def test_gpt_4o_transcribe_model_mapping():
"""Test that GPT-4o transcription models are correctly mapped and not hardcoded to whisper-1"""

View file

@ -91,6 +91,8 @@ ignored_function_names = [
"_get_claude_code_session_router_binding", # Tested through the two-worker session routing test in test_router.py
"_apply_updated_routing_strategy_args", # Tested via update_settings in test_lowest_latency.py (file lacks "router" in name)
"arm_routing_read_prefetch", # Tested in tests/unit/caching/test_request_redis_batch_pre_call.py (file lacks "router" in name)
"_embedding",
"_aembedding",
]

View file

@ -128,6 +128,8 @@ class TestOpenAIImageEditGPTImage1(BaseLLMImageEditTest):
Concrete implementation of BaseLLMImageEditTest for OpenAI image edits.
"""
test_openai_image_edit_litellm_sdk = None
def get_base_image_edit_call_args(self) -> dict:
"""Return base call args for OpenAI image edit"""
return {
@ -622,64 +624,6 @@ def test_recraft_image_edit_config():
assert files[0][1][2] == "image/png" # Content type
@pytest.mark.parametrize("sync_mode", [True, False])
@pytest.mark.flaky(retries=3, delay=2)
@pytest.mark.asyncio
async def test_multiple_vs_single_image_edit(sync_mode):
"""Test that both single and multiple image editing work correctly"""
from litellm import image_edit, aimage_edit
litellm._turn_on_debug()
try:
prompt = "Add a soft blue tint to the image(s)"
# Test single image
if sync_mode:
single_result = image_edit(
prompt=prompt,
model="gpt-image-1",
image=_make_single_test_image(),
)
else:
single_result = await aimage_edit(
prompt=prompt,
model="gpt-image-1",
image=_make_single_test_image(),
)
print("Single image result:", single_result)
ImageResponse.model_validate(single_result)
# Test multiple images
if sync_mode:
multiple_result = image_edit(
prompt=prompt,
model="gpt-image-1",
image=_make_test_images(),
)
else:
multiple_result = await aimage_edit(
prompt=prompt,
model="gpt-image-1",
image=_make_test_images(),
)
print("Multiple images result:", multiple_result)
ImageResponse.model_validate(multiple_result)
# Both should return valid responses
assert single_result is not None
assert multiple_result is not None
assert single_result.data is not None
assert multiple_result.data is not None
assert len(single_result.data) > 0
assert len(multiple_result.data) > 0
except litellm.ContentPolicyViolationError as e:
pytest.skip(f"Content policy violation: {e}")
@pytest.mark.flaky(retries=3, delay=2)
@pytest.mark.asyncio
async def test_multiple_image_edit_with_different_formats():

View file

@ -18,6 +18,9 @@ from base_responses_api import BaseResponsesAPITest
class TestAzureResponsesAPITest(BaseResponsesAPITest):
test_multiturn_responses_api = None
test_responses_api_with_tool_calls = None
def get_base_completion_call_args(self):
return {
"model": "azure/gpt-4.1-mini",

View file

@ -23,6 +23,8 @@ from base_responses_api import BaseResponsesAPITest, validate_responses_api_resp
class TestOpenAIResponsesAPITest(BaseResponsesAPITest):
test_responses_api_with_tool_calls = None
def get_base_completion_call_args(self):
return {
"model": "openai/gpt-5.5",
@ -1597,24 +1599,6 @@ async def test_openai_gpt5_reasoning_effort_parameter():
print("Response:", json.dumps(response, indent=4, default=str))
@pytest.mark.asyncio
@pytest.mark.parametrize("stream", [True, False])
async def test_basic_openai_responses_with_websearch(stream):
litellm._turn_on_debug()
request_model = "gpt-5.5"
response = await litellm.aresponses(
model=request_model,
stream=stream,
input="hi",
tools=[{"type": "web_search", "search_context_size": "low"}],
)
if stream:
async for chunk in response:
print("chunk=", json.dumps(chunk, indent=4, default=str))
else:
print("response=", json.dumps(response, indent=4, default=str))
@pytest.mark.asyncio
async def test_openai_responses_api_token_limit_error():
"""

View file

@ -163,33 +163,6 @@ class TestGoogleInteractionsStreaming:
class TestGoogleInteractionsMultiTurn:
"""Tests for multi-turn conversations using Step[] input."""
def test_multi_turn_conversation(self, api_key):
"""Test a multi-turn conversation per OpenAPI spec (Step[] format)."""
response = interactions.create(
model="gemini/gemini-2.5-flash",
input=[
{
"type": "user_input",
"content": [{"type": "text", "text": "My name is Alice."}],
},
{
"type": "model_output",
"content": [
{"type": "text", "text": "Hello Alice! Nice to meet you."}
],
},
{
"type": "user_input",
"content": [{"type": "text", "text": "What is my name?"}],
},
],
api_key=api_key,
)
assert response is not None
print(f"Multi-turn response: {response}")
class TestGoogleInteractionsAgent:
"""Tests for agent interactions (per OpenAPI spec)."""

File diff suppressed because one or more lines are too long

View file

@ -270,7 +270,7 @@ async def test_azure_ai_request_format():
@pytest.mark.asyncio
@pytest.mark.parametrize("model", ["azure/gpt5_series/gpt-5-mini", "azure/gpt-5-mini"])
@pytest.mark.parametrize("model", ["azure/gpt5_series/gpt-5-mini"])
async def test_azure_gpt5_reasoning(model):
litellm._turn_on_debug()
response = await litellm.acompletion(

View file

@ -11,6 +11,10 @@ from base_llm_unit_tests import BaseLLMChatTest, BaseOSeriesModelsTest
class TestAzureOpenAIO3Mini(BaseOSeriesModelsTest, BaseLLMChatTest):
test_content_list_handling = None
test_empty_tools = None
test_function_calling_with_tool_response = None
def get_base_completion_call_args(self):
# Clear the LLM client cache to prevent test pollution from cached clients
litellm.in_memory_llm_clients_cache.flush_cache()

View file

@ -729,18 +729,3 @@ def test_azure_with_content_safety_error():
]
== "high"
)
def test_azure_openai_with_prompt_cache_key():
"""
E2E test for Azure OpenAI with prompt cache key param on /chat/completions API.
"""
litellm._turn_on_debug()
response = litellm.completion(
model="azure/gpt-4.1-mini",
api_key=os.getenv("AZURE_AI_API_KEY"),
api_base=os.getenv("AZURE_AI_API_BASE"),
api_version="2024-12-01-preview",
messages=[{"role": "user", "content": "What is the weather in San Francisco?"}],
prompt_cache_key="test_streaming_azure_openai",
)

View file

@ -425,55 +425,6 @@ def test_completion_bedrock_claude_aws_bedrock_client(bedrock_session_token_cred
# test_completion_bedrock_claude_sts_client_auth()
@pytest.mark.parametrize(
"image_url",
[
"data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAL0AAAC9CAMAAADRCYwCAAAAh1BMVEX///8AAAD8/Pz5+fkEBAT39/cJCQn09PRNTU3y8vIMDAwzMzPe3t7v7+8QEBCOjo7FxcXR0dHn5+elpaWGhoYYGBivr686OjocHBy0tLQtLS1TU1PY2Ni6urpaWlpERER3d3ecnJxoaGiUlJRiYmIlJSU4ODhBQUFycnKAgIDBwcFnZ2chISE7EjuwAAAI/UlEQVR4nO1caXfiOgz1bhJIyAJhX1JoSzv8/9/3LNlpYd4rhX6o4/N8Z2lKM2cURZau5JsQEhERERERERERERERERERERHx/wBjhDPC3OGN8+Cc5JeMuheaETSdO8vZFyCScHtmz2CsktoeMn7rLM1u3h0PMAEhyYX7v/Q9wQvoGdB0hlbzm45lEq/wd6y6G9aezvBk9AXwp1r3LHJIRsh6s2maxaJpmvqgvkC7WFS3loUnaFJtKRVUCEoV/RpCnHRvAsesVQ1hw+vd7Mpo+424tLs72NplkvQgcdrsvXkW/zJWqH/fA0FT84M/xnQJt4to3+ZLuanbM6X5lfXKHosO9COgREqpCR5i86pf2zPS7j9tTj+9nO7bQz3+xGEyGW9zqgQ1tyQ/VsxEDvce/4dcUPNb5OD9yXvR4Z2QisuP0xiGWPnemgugU5q/troHhGEjIF5sTOyW648aC0TssuaaCEsYEIkGzjWXOp3A0vVsf6kgRyqaDk+T7DIVWrb58b2tT5xpUucKwodOD/5LbrZC1ws6YSaBZJ/8xlh+XZSYXaMJ2ezNqjB3IPXuehPcx2U6b4t1dS/xNdFzguUt8ie7arnPeyCZroxLHzGgGdqVcspwafizPWEXBee+9G1OaufGdvNng/9C+gwgZ3PH3r87G6zXTZ5D5De2G2DeFoANXfbACkT+fxBQ22YFsTTJF9hjFVO6VbqxZXko4WJ8s52P4PnuxO5KRzu0/hlix1ySt8iXjgaQ+4IHPA9nVzNkdduM9LFT/Aacj4FtKrHA7iAw602Vnht6R8Vq1IOS+wNMKLYqayAYfRuufQPGeGb7sZogQQoLZrGPgZ6KoYn70Iw30O92BNEDpvwouCFn6wH2uS+EhRb3WF/HObZk3HuxfRQM3Y/Of/VH0n4MKNHZDiZvO9+m/ABALfkOcuar/7nOo7B95ACGVAFaz4jMiJwJhdaHBkySmzlGTu82gr6FSTik2kJvLnY9nOd/D90qcH268m3I/cgI1xg1maE5CuZYaWLH+UHANCIck0yt7Mx5zBm5vVHXHwChsZ35kKqUpmo5Svq5/fzfAI5g2vDtFPYo1HiEA85QrDeGm9g//LG7K0scO3sdpj2CBDgCa+0OFs0bkvVgnnM/QBDwllOMm+cN7vMSHlB7Uu4haHKaTwgGkv8tlK+hP8fzmFuK/RQTpaLPWvbd58yWIo66HHM0OsPoPhVqmtaEVL7N+wYcTLTbb0DLdgp23Eyy2VYJ2N7bkLFAAibtoLPe5sLt6Oa2bvU+zyeMa8wrixO0gRTn9tO9NCSThTLGqcqtsDvphlfmx/cPBZVvw24jg1LE2lPuEo35Mhi58U0I/Ga8n5w+NS8i34MAQLos5B1u0xL1ZvCVYVRw/Fs2q53KLaXJMWwOZZ/4MPYV19bAHmgGDKB6f01xoeJKFbl63q9J34KdaVNPJWztQyRkzA3KNs1AdAEDowMxh10emXTCx75CkurtbY/ZpdNDGdsn2UcHKHsQ8Ai3WZi48IfkvtjOhsLpuIRSKZTX9FA4o+0d6o/zOWqQzVJMynL9NsxhSJOaourq6nBVQBueMSyubsX2xHrmuABZN2Ns9jr5nwLFlLF/2R6atjW/67Yd11YQ1Z+kA9Zk9dPTM/o6dVo6HHVgC0JR8oUfmI93T9u3gvTG94bAH02Y5xeqRcjuwnKCK6Q2+ajl8KXJ3GSh22P3Zfx6S+n008ROhJn+JRIUVu6o7OXl8w1SeyhuqNDwNI7SjbK08QrqPxS95jy4G7nCXVq6G3HNu0LtK5J0e226CfC005WKK9sVvfxI0eUbcnzutfhWe3rpZHM0nZ/ny/N8tanKYlQ6VEW5Xuym8yV1zZX58vwGhZp/5tFfhybZabdbrQYOs8F+xEhmPsb0/nki6kIyVvzZzUASiOrTfF+Sj9bXC7DoJxeiV8tjQL6loSd0yCx7YyB6rPdLx31U2qCG3F/oXIuDuqd6LFO+4DNIJuxFZqSsU0ea88avovFnWKRYFYRQDfCfcGaBCLn4M4A1ntJ5E57vicwqq2enaZEF5nokCYu9TbKqCC5yCDfL+GhLxT4w4xEJs+anqgou8DOY2q8FMryjb2MehC1dRJ9s4g9NXeTwPkWON4RH+FhIe0AWR/S9ekvQ+t70XHeimGF78LzuU7d7PwrswdIG2VpgF8C53qVQsTDtBJc4CdnkQPbnZY9mbPdDFra3PCXBBQ5QBn2aQqtyhvlyYM4Hb2/mdhsxCUen04GZVvIJZw5PAamMOmjzq8Q+dzAKLXDQ3RUZItWsg4t7W2DP+JDrJDymoMH7E5zQtuEpG03GTIjGCW3LQqOYEsXgFc78x76NeRwY6SNM+IfQoh6myJKRBIcLYxZcwscJ/gI2isTBty2Po9IkYzP0/SS4hGlxRjFAG5z1Jt1LckiB57yWvo35EaolbvA+6fBa24xodL2YjsPpTnj3JgJOqhcgOeLVsYYwoK0wjY+m1D3rGc40CukkaHnkEjarlXrF1B9M6ECQ6Ow0V7R7N4G3LfOHAXtymoyXOb4QhaYHJ/gNBJUkxclpSs7DNcgWWDDmM7Ke5MJpGuioe7w5EOvfTunUKRzOh7G2ylL+6ynHrD54oQO3//cN3yVO+5qMVsPZq0CZIOx4TlcJ8+Vz7V5waL+7WekzUpRFMTnnTlSCq3X5usi8qmIleW/rit1+oQZn1WGSU/sKBYEqMNh1mBOc6PhK8yCfKHdUNQk8o/G19ZPTs5MYfai+DLs5vmee37zEyyH48WW3XA6Xw6+Az8lMhci7N/KleToo7PtTKm+RA887Kqc6E9dyqL/QPTugzMHLbLZtJKqKLFfzVWRNJ63c+95uWT/F7R0U5dDVvuS409AJXhJvD0EwWaWdW8UN11u/7+umaYjT8mJtzZwP/MD4r57fihiHlC5fylHfaqnJdro+Dr7DajvO+vi2EwyD70s8nCH71nzIO1l5Zl+v1DMCb5ebvCMkGHvobXy/hPumGLyX0218/3RyD1GRLOuf9u/OGQyDmto32yMiIiIiIiIiIiIiIiIiIiIiIiIiIiIiIiIiIv7GP8YjWPR/czH2AAAAAElFTkSuQmCC",
"https://avatars.githubusercontent.com/u/29436595?v=",
],
)
def test_bedrock_claude_3(image_url):
try:
litellm.set_verbose = True
data = {
"max_tokens": 100,
"stream": False,
"temperature": 0.3,
"messages": [
{"role": "user", "content": "Hi"},
{"role": "assistant", "content": "Hi"},
{
"role": "user",
"content": [
{"text": "describe this image", "type": "text"},
{
"image_url": {
"detail": "high",
"url": image_url,
},
"type": "image_url",
},
],
},
],
}
response: ModelResponse = completion(
model="bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0",
num_retries=3,
**data,
) # type: ignore
# Add any assertions here to check the response
assert len(response.choices) > 0
assert len(response.choices[0].message.content) > 0
except litellm.InternalServerError:
pass
except RateLimitError:
pass
except Exception as e:
pytest.fail(f"Error occurred: {e}")
@pytest.mark.parametrize(
"stop",
[""],
@ -911,49 +862,6 @@ def test_completion_bedrock_external_client_region(monkeypatch):
pytest.fail(f"Error occurred: {e}")
def test_bedrock_tool_calling():
"""
# related issue: https://github.com/BerriAI/litellm/issues/5007
# Bedrock tool names must satisfy regular expression pattern: [a-zA-Z][a-zA-Z0-9_]* ensure this is true
"""
litellm.set_verbose = True
response = litellm.completion(
model="bedrock/anthropic.claude-3-sonnet-20240229-v1:0",
fallbacks=["bedrock/meta.llama3-1-8b-instruct-v1:0"],
messages=[
{
"role": "user",
"content": "What's the weather like in Boston today in Fahrenheit?",
}
],
tools=[
{
"type": "function",
"function": {
"name": "-DoSomethingVeryCool-forLitellm_Testin999229291-0293993",
"description": "use this to get the current weather",
"parameters": {"type": "object", "properties": {}},
},
}
],
)
print("bedrock response")
print(response)
# Assert that the tools in response have the same function name as the input
_choice_1 = response.choices[0]
if _choice_1.message.tool_calls is not None:
print(_choice_1.message.tool_calls)
for tool_call in _choice_1.message.tool_calls:
_tool_Call_name = tool_call.function.name
if _tool_Call_name is not None and "DoSomethingVeryCool" in _tool_Call_name:
assert (
_tool_Call_name
== "-DoSomethingVeryCool-forLitellm_Testin999229291-0293993"
)
def test_bedrock_tools_pt_valid_names():
"""
# related issue: https://github.com/BerriAI/litellm/issues/5007
@ -2031,6 +1939,14 @@ def test_bedrock_supports_tool_call(model, expected_supports_tool_call):
class TestBedrockConverseChatCrossRegion(BaseLLMChatTest):
test_content_list_handling = None
test_developer_role_translation = None
test_function_calling_with_tool_response = None
test_image_url = None
test_json_response_format_stream = None
test_tool_call_with_empty_enum_property = None
test_tool_call_with_property_type_array = None
def get_base_completion_call_args(self) -> dict:
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
@ -2070,6 +1986,9 @@ class TestBedrockConverseChatCrossRegion(BaseLLMChatTest):
class TestBedrockConverseAnthropicUnitTests(BaseAnthropicChatTest):
test_completion_thinking_with_max_tokens = None
test_completion_thinking_without_max_tokens = None
def get_base_completion_call_args(self) -> dict:
return {
"model": "bedrock/us.anthropic.claude-haiku-4-5-20251001-v1:0",
@ -2083,6 +2002,11 @@ class TestBedrockConverseAnthropicUnitTests(BaseAnthropicChatTest):
class TestBedrockConverseChatNormal(BaseLLMChatTest):
test_content_list_handling = None
test_empty_tools = None
test_function_calling_with_tool_response = None
test_image_url = None
def get_base_completion_call_args(self) -> dict:
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
@ -2098,6 +2022,10 @@ class TestBedrockConverseChatNormal(BaseLLMChatTest):
class TestBedrockConverseNovaTestSuite(BaseLLMChatTest):
test_content_list_handling = None
test_function_calling_with_tool_response = None
test_image_url = None
def get_base_completion_call_args(self) -> dict:
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")

View file

@ -9,6 +9,8 @@ from litellm.llms.custom_httpx.http_handler import HTTPHandler
class TestBedrockGPTOSS(BaseLLMChatTest):
test_json_response_format = None
def get_base_completion_call_args(self) -> dict:
return {
"model": "bedrock/converse/openai.gpt-oss-20b-1:0",

View file

@ -6,6 +6,16 @@ import litellm
from litellm.types.llms.bedrock import BedrockInvokeNovaRequest
_LITELLM_LOGO_IMAGE_URL = (
"https://cdn.jsdelivr.net/gh/BerriAI/litellm@d769e81c90d453240c61fc572cdb27fae06a89d0/"
"ui/litellm-dashboard/public/assets/logos/litellm_logo.jpg"
)
_AWSMP_LOGO_IMAGE_URL = (
"https://awsmp-logos.s3.amazonaws.com/seller-xw5kijmvmzasy/"
"c233c9ade2ccb5491072ae232c814942.png"
)
@pytest.mark.flaky(retries=3, delay=5)
class TestBedrockInvokeClaudeJson(BaseLLMChatTest):
def get_base_completion_call_args(self) -> dict:
@ -18,8 +28,27 @@ class TestBedrockInvokeClaudeJson(BaseLLMChatTest):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
@pytest.mark.parametrize(
"image_url, detail",
[
(_LITELLM_LOGO_IMAGE_URL, None),
(_LITELLM_LOGO_IMAGE_URL, "low"),
(_LITELLM_LOGO_IMAGE_URL, "high"),
(_AWSMP_LOGO_IMAGE_URL, "low"),
(_AWSMP_LOGO_IMAGE_URL, "high"),
],
)
@pytest.mark.flaky(retries=4, delay=2)
def test_image_url(self, image_url, detail):
super().test_image_url(detail=detail, image_url=image_url)
test_content_list_handling = None
test_image_url_string = None
test_pdf_handling = None
class TestBedrockInvokeNovaJson(BaseLLMChatTest):
test_json_response_format = None
def get_base_completion_call_args(self) -> dict:
return {
"model": "bedrock/invoke/us.amazon.nova-micro-v1:0",

View file

@ -5,6 +5,10 @@ import litellm
class TestBedrockTestSuite(BaseLLMChatTest):
test_content_list_handling = None
test_empty_tools = None
test_function_calling_with_tool_response = None
def test_tool_call_no_arguments(self, tool_call_no_arguments):
pass

View file

@ -30,6 +30,8 @@ class TestBedrockMoonshotInvoke(BaseLLMChatTest):
Inherits all standard LLM tests from BaseLLMChatTest.
"""
test_json_response_format_stream = None
def get_base_completion_call_args(self) -> dict:
litellm._turn_on_debug()
return {

View file

@ -5,6 +5,14 @@ import litellm
class TestBedrockNovaJson(BaseLLMChatTest):
test_content_list_handling = None
test_developer_role_translation = None
test_empty_tools = None
test_function_calling_with_tool_response = None
test_json_response_format_stream = None
test_tool_call_with_empty_enum_property = None
test_tool_call_with_property_type_array = None
def get_base_completion_call_args(self) -> dict:
litellm._turn_on_debug()
return {

View file

@ -74,6 +74,16 @@ GEMINI_3_IMAGE_SIZE_MAPPINGS = [
class TestGoogleAIStudioGemini(BaseLLMChatTest):
test_async_pdf_handling_with_file_id = None
test_content_list_handling = None
test_developer_role_translation = None
test_function_calling_with_tool_response = None
test_image_url = None
test_json_response_nested_json_schema = None
test_json_response_nested_pydantic_obj = None
test_json_response_pydantic_obj = None
test_web_search = None
def get_base_completion_call_args(self) -> dict:
return {"model": "gemini/gemini-2.5-flash"}

View file

@ -18,6 +18,10 @@ from litellm.llms.groq.chat.transformation import (
class TestGroq(BaseLLMChatTest):
test_content_list_handling = None
test_empty_tools = None
test_web_search = None
def get_base_completion_call_args(self) -> dict:
return {
"model": "groq/openai/gpt-oss-120b",

View file

@ -274,6 +274,7 @@ async def test_vision_with_custom_model():
class TestOpenAIChatCompletion(BaseLLMChatTest):
test_basic_tool_calling = None
test_function_calling_with_tool_response = None
def get_base_completion_call_args(self) -> dict:
return {"model": "gpt-4o-mini"}
@ -687,17 +688,6 @@ def test_openai_tool_calling():
response = litellm.completion(**completion_params)
@pytest.mark.asyncio
async def test_openai_gpt5_reasoning():
response = await litellm.acompletion(
model="openai/gpt-5-mini",
messages=[{"role": "user", "content": "What is the capital of France?"}],
reasoning_effort="minimal",
)
print("response: ", response)
assert response.choices[0].message.content is not None
@pytest.mark.asyncio
async def test_openai_safety_identifier_parameter():
"""Test that safety_identifier parameter is correctly passed to the OpenAI API."""

View file

@ -142,6 +142,10 @@ def test_litellm_responses():
class TestOpenAIO1(BaseOSeriesModelsTest, BaseLLMChatTest):
test_empty_tools = None
test_tool_call_with_empty_enum_property = None
test_tool_call_with_property_type_array = None
def get_base_completion_call_args(self):
return {
"model": "o1",
@ -162,6 +166,9 @@ class TestOpenAIO1(BaseOSeriesModelsTest, BaseLLMChatTest):
class TestOpenAIO3(BaseOSeriesModelsTest, BaseLLMChatTest):
test_basic_tool_calling = None
test_function_calling_with_tool_response = None
def get_base_completion_call_args(self):
return {
"model": "o3-mini",
@ -188,27 +195,3 @@ def test_o3_reasoning_effort():
reasoning_effort="high",
)
assert resp.choices[0].message.content is not None
@pytest.mark.parametrize("model", ["o1", "o3-mini"])
def test_streaming_response(model):
"""Test that streaming response is returned correctly"""
from litellm import completion
response = completion(
model=model,
messages=[
{"role": "system", "content": "Be a good bot!"},
{"role": "user", "content": "Hello!"},
],
stream=True,
)
assert response is not None
chunks = []
for chunk in response:
chunks.append(chunk)
resp = litellm.stream_chunk_builder(chunks=chunks)
print(resp)

View file

@ -16,6 +16,14 @@ import pytest
class TestTogetherAI(BaseLLMChatTest):
test_basic_tool_calling = None
test_empty_tools = None
test_function_calling_with_tool_response = None
test_json_response_format = None
test_json_response_nested_json_schema = None
test_json_response_nested_pydantic_obj = None
test_json_response_pydantic_obj = None
test_tool_call_with_empty_enum_property = None
test_tool_call_with_property_type_array = None
def get_base_completion_call_args(self) -> dict:
litellm.set_verbose = True

View file

@ -8,7 +8,6 @@ from unittest.mock import AsyncMock
import httpx
import pytest
import litellm
from litellm import Choices, Message, ModelResponse, EmbeddingResponse, Usage
from litellm import completion
from unittest.mock import patch
@ -179,31 +178,7 @@ class TestXAIChat(BaseLLMChatTest):
"""Test that tool calls with no arguments is translated correctly. Relevant issue: https://github.com/BerriAI/litellm/issues/6833"""
pass
def test_web_search(self):
"""Web search is only supported for Grok 4 family models"""
from litellm.utils import supports_web_search
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
litellm._turn_on_debug()
# Use grok-4-1-fast which supports web search
model = "xai/grok-4-1-fast"
if not supports_web_search(model, None):
pytest.skip("Model does not support web search")
response = completion(
model=model,
messages=[
{"role": "user", "content": "What's the weather like in Boston today?"}
],
web_search_options={},
max_tokens=100,
)
assert response is not None
test_web_search = None
def test_xai_streaming_with_include_usage():

View file

@ -4,8 +4,6 @@
import asyncio
import os
import time
import traceback
import pytest
import concurrent
@ -19,113 +17,9 @@ from litellm import Router
load_dotenv()
def _make_model_list():
return [
{
"model_name": "gpt-3.5-turbo",
"litellm_params": {
"model": "azure/gpt-4.1-mini",
"api_key": "bad-key",
"api_version": os.getenv("AZURE_API_VERSION"),
"api_base": os.getenv("AZURE_AI_API_BASE"),
},
"tpm": 240000,
"rpm": 1800,
},
{
"model_name": "gpt-3.5-turbo",
"litellm_params": {
"model": "gpt-3.5-turbo",
"api_key": os.getenv("OPENAI_API_KEY"),
},
"tpm": 1000000,
"rpm": 9000,
},
]
def _make_kwargs():
return {
"model": "gpt-3.5-turbo",
"messages": [{"role": "user", "content": "Hey, how's it going?"}],
}
@pytest.mark.flaky(retries=3, delay=1)
def test_multiple_deployments_sync():
import concurrent
import time
litellm.set_verbose = False
results = []
kwargs = _make_kwargs()
router = Router(
model_list=_make_model_list(),
redis_host=os.getenv("REDIS_HOST"),
redis_password=os.getenv("REDIS_PASSWORD"),
redis_port=int(os.getenv("REDIS_PORT")), # type: ignore
routing_strategy="simple-shuffle",
set_verbose=True,
num_retries=1,
) # type: ignore
try:
for _ in range(3):
response = router.completion(**kwargs)
results.append(response)
print(results)
router.reset()
except Exception as e:
print(f"FAILED TEST!")
pytest.fail(f"An error occurred - {traceback.format_exc()}")
# test_multiple_deployments_sync()
def test_multiple_deployments_parallel():
litellm.set_verbose = False # Corrected the syntax for setting verbose to False
results = []
futures = {}
kwargs = _make_kwargs()
start_time = time.time()
router = Router(
model_list=_make_model_list(),
redis_host=os.getenv("REDIS_HOST"),
redis_password=os.getenv("REDIS_PASSWORD"),
redis_port=int(os.getenv("REDIS_PORT")), # type: ignore
routing_strategy="simple-shuffle",
set_verbose=True,
num_retries=1,
) # type: ignore
# Assuming you have an executor instance defined somewhere in your code
with concurrent.futures.ThreadPoolExecutor() as executor:
for _ in range(5):
future = executor.submit(router.completion, **kwargs)
futures[future] = future
# Retrieve the results from the futures
while futures:
done, not_done = concurrent.futures.wait(
futures.values(),
timeout=10,
return_when=concurrent.futures.FIRST_COMPLETED,
)
for future in done:
try:
result = future.result()
results.append(result)
del futures[future] # Remove the done future
except Exception as e:
print(f"Exception: {e}; traceback: {traceback.format_exc()}")
del futures[future] # Remove the done future with exception
print(f"Remaining futures: {len(futures)}")
router.reset()
end_time = time.time()
print(results)
print(f"ELAPSED TIME: {end_time - start_time}")
# Assuming litellm, router, and executor are defined somewhere in your code

View file

@ -137,30 +137,6 @@ def load_vertex_ai_credentials():
os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = os.path.abspath(temp_file.name)
@pytest.mark.asyncio
async def test_get_response():
load_vertex_ai_credentials()
prompt = '\ndef count_nums(arr):\n """\n Write a function count_nums which takes an array of integers and returns\n the number of elements which has a sum of digits > 0.\n If a number is negative, then its first signed digit will be negative:\n e.g. -123 has signed digits -1, 2, and 3.\n >>> count_nums([]) == 0\n >>> count_nums([-1, 11, -11]) == 1\n >>> count_nums([1, 1, 2]) == 3\n """\n'
try:
response = await acompletion(
model="gemini-2.5-flash-lite",
messages=[
{
"role": "system",
"content": "Complete the given code with no more explanation. Remember that there is a 4-space indent before the first line of your generated code.",
},
{"role": "user", "content": prompt},
],
)
return response
except litellm.RateLimitError:
pass
except litellm.UnprocessableEntityError as e:
pass
except Exception as e:
pytest.fail(f"An error occurred - {str(e)}")
# test_vertex_ai_anthropic_streaming()
@ -341,35 +317,6 @@ def test_avertex_ai_stream():
# test_vertex_ai_stream()
@pytest.mark.flaky(retries=3, delay=1)
@pytest.mark.asyncio
async def test_async_vertexai_response_basic():
load_vertex_ai_credentials()
try:
user_message = "Hello, how are you?"
messages = [{"content": user_message, "role": "user"}]
response = await acompletion(
model="gemini-3.5-flash",
messages=messages,
temperature=0.7,
timeout=5,
vertex_location="global",
)
print(f"response: {response}")
except litellm.NotFoundError as e:
pass
except litellm.RateLimitError as e:
pass
except litellm.Timeout as e:
pass
except litellm.APIError as e:
pass
except litellm.InternalServerError as e:
pass
except Exception as e:
pytest.fail(f"An exception occurred: {e}")
@pytest.mark.flaky(retries=3, delay=1)
@pytest.mark.asyncio
async def test_async_vertexai_streaming_response():
@ -434,49 +381,6 @@ async def test_async_vertexai_streaming_response():
pytest.fail(f"An exception occurred: {e}")
@pytest.mark.parametrize("load_pdf", [False]) # True,
@pytest.mark.flaky(retries=3, delay=1)
def test_completion_function_plus_pdf(load_pdf):
litellm.set_verbose = True
load_vertex_ai_credentials()
try:
import base64
import requests
# URL of the file
url = "https://storage.googleapis.com/cloud-samples-data/generative-ai/pdf/2403.05530.pdf"
# Download the file
if load_pdf:
response = requests.get(url)
file_data = response.content
encoded_file = base64.b64encode(file_data).decode("utf-8")
url = f"data:application/pdf;base64,{encoded_file}"
image_content = [
{"type": "text", "text": "What's this file about?"},
{
"type": "image_url",
"image_url": {"url": url},
},
]
image_message = {"role": "user", "content": image_content}
response = completion(
model="vertex_ai_beta/gemini-2.5-flash-lite",
messages=[image_message],
stream=False,
)
print(response)
except litellm.InternalServerError as e:
pass
except Exception as e:
pytest.fail("Got={}".format(str(e)))
def encode_image(image_path):
import base64
@ -1470,90 +1374,6 @@ async def test_gemini_pro_httpx_custom_api_base(model):
# @pytest.mark.skip(reason="exhausted vertex quota. need to refactor to mock the call")
@pytest.mark.parametrize("sync_mode", [True])
@pytest.mark.parametrize("provider", ["vertex_ai"])
@pytest.mark.asyncio
@pytest.mark.flaky(retries=3, delay=1)
async def test_gemini_pro_function_calling(provider, sync_mode):
try:
load_vertex_ai_credentials()
litellm.set_verbose = True
messages = [
{
"role": "system",
"content": "Your name is Litellm Bot, you are a helpful assistant",
},
# User asks for their name and weather in San Francisco
{
"role": "user",
"content": "Hello, what is your name and can you tell me the weather?",
},
# Assistant replies with a tool call
{
"role": "assistant",
"content": "",
"tool_calls": [
{
"id": "call_123",
"type": "function",
"index": 0,
"function": {
"name": "get_weather",
"arguments": '{"location":"San Francisco, CA"}',
},
}
],
},
# The result of the tool call is added to the history
{
"role": "tool",
"tool_call_id": "call_123",
"content": "27 degrees celsius and clear in San Francisco, CA",
},
# Now the assistant can reply with the result of the tool call.
]
tools = [
{
"type": "function",
"function": {
"name": "get_weather",
"description": "Get the current weather in a given location",
"parameters": {
"type": "object",
"properties": {
"location": {
"type": "string",
"description": "The city and state, e.g. San Francisco, CA",
}
},
"required": ["location"],
},
},
}
]
data = {
"model": "{}/gemini-2.5-flash-lite".format(provider),
"messages": messages,
"tools": tools,
}
if sync_mode:
response = litellm.completion(**data)
else:
response = await litellm.acompletion(**data)
print(f"response: {response}")
except litellm.RateLimitError as e:
pass
except Exception as e:
if "429 Quota exceeded" in str(e):
pass
else:
pytest.fail("An unexpected exception occurred - {}".format(str(e)))
# gemini_pro_function_calling()
@ -3522,46 +3342,6 @@ def test_vertex_ai_llama_tool_calling():
assert response._hidden_params["response_cost"] > 0
def test_vertex_schema_test():
load_vertex_ai_credentials()
litellm._turn_on_debug()
def tool_call(text: str | None) -> str:
return text or "No text provided"
tool = {
"type": "function",
"function": {
"name": "git_create_branch",
"description": "Creates a new branch from an optional base branch",
"parameters": {
"type": "object",
"properties": {
"repo_path": {"title": "Repo Path", "type": "string"},
"branch_name": {"title": "Branch Name", "type": "string"},
"base_branch": {
"anyOf": [{"type": "string"}, {"type": "null"}],
"default": None,
"title": "Base Branch",
},
},
"required": ["repo_path", "branch_name"],
"title": "GitCreateBranch",
},
},
}
response = litellm.completion(
model="vertex_ai/gemini-3.5-flash",
messages=[{"role": "user", "content": "call the tool"}],
tools=[tool],
tool_choice="required",
vertex_location="global",
)
print(response)
def test_gemini_nullable_object_tool_schema_httpx():
"""
Ensure nullable object tool params preserve nested properties in Vertex schema conversion.

View file

@ -35,26 +35,6 @@ async def test_async_otel_callback():
await asyncio.sleep(2)
@pytest.mark.asyncio()
async def test_async_dynamic_arize_config():
litellm.set_verbose = True
verbose_proxy_logger.setLevel(logging.DEBUG)
verbose_logger.setLevel(logging.DEBUG)
litellm.success_callback = ["arize"]
await litellm.acompletion(
model="gpt-3.5-turbo",
messages=[{"role": "user", "content": "hi test from arize dynamic config"}],
temperature=0.1,
user="OTEL_USER",
arize_api_key=os.getenv("ARIZE_SPACE_API_KEY"),
arize_space_key=os.getenv("ARIZE_SPACE_KEY"),
)
await asyncio.sleep(2)
@pytest.fixture
def mock_env_vars(monkeypatch):
monkeypatch.setenv("ARIZE_SPACE_KEY", "test_space_key")

View file

@ -215,43 +215,6 @@ async def test_hf_completion_tgi():
# test_get_cloudflare_response_streaming()
def test_get_response_streaming():
import asyncio
async def test_async_call():
user_message = "write a short poem in one sentence"
messages = [{"content": user_message, "role": "user"}]
try:
litellm.set_verbose = True
response = await acompletion(
model="gpt-3.5-turbo", messages=messages, stream=True, timeout=5
)
print(type(response))
import inspect
is_async_generator = inspect.isasyncgen(response)
print(is_async_generator)
output = ""
i = 0
async for chunk in response:
token = chunk["choices"][0]["delta"].get("content", "")
if token == None:
continue # openai v1.0.0 returns content=None
output += token
assert output is not None, "output cannot be None."
assert isinstance(output, str), "output needs to be of type str"
assert len(output) > 0, "Length of output needs to be greater than 0."
print(f"output: {output}")
except litellm.Timeout as e:
pass
except Exception as e:
pytest.fail(f"An exception occurred: {e}")
asyncio.run(test_async_call())
# test_get_response_streaming()

View file

@ -192,242 +192,6 @@ def test_completion_empower():
pytest.fail(f"Error occurred: {e}")
def test_completion_claude_3_empty_response():
litellm.set_verbose = True
messages = [
{
"role": "system",
"content": [{"type": "text", "text": "You are 2twNLGfqk4GMOn3ffp4p."}],
},
{"role": "user", "content": "Hi gm!", "name": "ishaan"},
{"role": "assistant", "content": "Good morning! How are you doing today?"},
{
"role": "user",
"content": "I was hoping we could chat a bit",
},
]
try:
response = litellm.completion(
model="claude-sonnet-4-5-20250929", messages=messages
)
print(response)
except litellm.InternalServerError as e:
pytest.skip(f"InternalServerError - {str(e)}")
except Exception as e:
pytest.fail(f"Error occurred: {e}")
def test_completion_claude_3():
litellm.set_verbose = True
messages = [
{
"role": "user",
"content": "\nWhat is the query for `console.log` => `console.error`\n",
},
{
"role": "assistant",
"content": "\nThis is the GritQL query for the given before/after examples:\n<gritql>\n`console.log` => `console.error`\n</gritql>\n",
},
{
"role": "user",
"content": "\nWhat is the query for `console.info` => `consdole.heaven`\n",
},
]
try:
# test without max tokens
response = completion(
model="anthropic/claude-sonnet-4-5-20250929",
messages=messages,
)
# Add any assertions, here to check response args
print(response)
except litellm.InternalServerError as e:
pytest.skip(f"InternalServerError - {str(e)}")
except Exception as e:
pytest.fail(f"Error occurred: {e}")
@pytest.mark.parametrize(
"model",
["anthropic/claude-sonnet-4-5-20250929", "us.anthropic.claude-sonnet-4-5-20250929-v1:0"],
)
def test_completion_claude_3_function_call(model):
litellm.set_verbose = True
tools = [
{
"type": "function",
"function": {
"name": "get_current_weather",
"description": "Get the current weather in a given location",
"parameters": {
"type": "object",
"properties": {
"location": {
"type": "string",
"description": "The city and state, e.g. San Francisco, CA",
},
"unit": {"type": "string", "enum": ["celsius", "fahrenheit"]},
},
"required": ["location"],
},
},
}
]
messages = [
{
"role": "user",
"content": "What's the weather like in Boston today in Fahrenheit?",
}
]
try:
# test without max tokens
response = completion(
model=model,
messages=messages,
tools=tools,
tool_choice={
"type": "function",
"function": {"name": "get_current_weather"},
},
drop_params=True,
)
# Add any assertions here to check response args
print(response)
assert isinstance(response.choices[0].message.tool_calls[0].function.name, str)
assert isinstance(
response.choices[0].message.tool_calls[0].function.arguments, str
)
messages.append(
response.choices[0].message.model_dump()
) # Add assistant tool invokes
tool_result = (
'{"location": "Boston", "temperature": "72", "unit": "fahrenheit"}'
)
# Add user submitted tool results in the OpenAI format
messages.append(
{
"tool_call_id": response.choices[0].message.tool_calls[0].id,
"role": "tool",
"name": response.choices[0].message.tool_calls[0].function.name,
"content": tool_result,
}
)
# In the second response, Claude should deduce answer from tool results
second_response = completion(
model=model,
messages=messages,
tools=tools,
tool_choice="auto",
drop_params=True,
)
print(second_response)
except litellm.InternalServerError:
pass
except Exception as e:
pytest.fail(f"Error occurred: {e}")
@pytest.mark.parametrize("sync_mode", [True])
@pytest.mark.parametrize(
"model, api_key, api_base",
[
("gpt-3.5-turbo", None, None),
("claude-sonnet-4-5-20250929", None, None),
("us.anthropic.claude-sonnet-4-5-20250929-v1:0", None, None),
# (
# "azure_ai/command-r-plus",
# os.getenv("AZURE_COHERE_API_KEY"),
# os.getenv("AZURE_COHERE_API_BASE"),
# ),
],
)
@pytest.mark.asyncio
async def test_model_function_invoke(model, sync_mode, api_key, api_base):
try:
litellm.set_verbose = True
messages = [
{
"role": "system",
"content": "Your name is Litellm Bot, you are a helpful assistant",
},
# User asks for their name and weather in San Francisco
{
"role": "user",
"content": "Hello, what is your name and can you tell me the weather?",
},
# Assistant replies with a tool call
{
"role": "assistant",
"content": "",
"tool_calls": [
{
"id": "call_123",
"type": "function",
"index": 0,
"function": {
"name": "get_weather",
"arguments": '{"location": "San Francisco, CA"}',
},
}
],
},
# The result of the tool call is added to the history
{
"role": "tool",
"tool_call_id": "call_123",
"content": "27 degrees celsius and clear in San Francisco, CA",
},
# Now the assistant can reply with the result of the tool call.
]
tools = [
{
"type": "function",
"function": {
"name": "get_weather",
"description": "Get the current weather in a given location",
"parameters": {
"type": "object",
"properties": {
"location": {
"type": "string",
"description": "The city and state, e.g. San Francisco, CA",
}
},
"required": ["location"],
},
},
}
]
data = {
"model": model,
"messages": messages,
"tools": tools,
"api_key": api_key,
"api_base": api_base,
}
if sync_mode:
response = litellm.completion(**data)
else:
response = await litellm.acompletion(**data)
print(f"response: {response}")
except litellm.InternalServerError:
pass
except litellm.RateLimitError as e:
pass
except Exception as e:
if "429 Quota exceeded" in str(e):
pass
else:
pytest.fail("An unexpected exception occurred - {}".format(str(e)))
@pytest.mark.asyncio
async def test_anthropic_no_content_error():
"""
@ -540,48 +304,6 @@ def test_parse_xml_params():
assert response["unit"] == "fahrenheit"
def test_completion_claude_3_multi_turn_conversations():
litellm.set_verbose = True
litellm.modify_params = True
messages = [
{"role": "assistant", "content": "?"}, # test first user message auto injection
{"role": "user", "content": "Hi!"},
{
"role": "user",
"content": [{"type": "text", "text": "What is the weather like today?"}],
},
{"role": "assistant", "content": "Hi! I am Claude. "},
{"role": "assistant", "content": "Today is a sunny "},
]
try:
response = completion(
model="anthropic/claude-sonnet-4-5-20250929",
messages=messages,
)
print(response)
except Exception as e:
pytest.fail(f"Error occurred: {e}")
def test_completion_claude_3_stream():
litellm.set_verbose = False
messages = [{"role": "user", "content": "Hello, world"}]
try:
# test without max tokens
response = completion(
model="anthropic/claude-sonnet-4-5-20250929",
messages=messages,
max_tokens=10,
stream=True,
)
# Add any assertions, here to check response args
print(response)
for chunk in response:
print(chunk)
except Exception as e:
pytest.fail(f"Error occurred: {e}")
def encode_image(image_path):
import base64
@ -2253,25 +1975,6 @@ async def test_re_use_azure_async_client():
pytest.fail("got Exception", e)
def test_re_use_openaiClient():
try:
print("gpt-3.5 with client test\n\n")
litellm.set_verbose = True
import openai
client = openai.OpenAI(
api_key=os.environ["OPENAI_API_KEY"],
)
## Test OpenAI call
for _ in range(2):
response = litellm.completion(
model="gpt-3.5-turbo", messages=messages, client=client
)
print(f"response: {response}")
except Exception as e:
pytest.fail("got Exception", e)
@pytest.mark.skip(
reason="this is bad test. It doesn't actually fail if the token is not set in the header. "
)
@ -3347,60 +3050,7 @@ def test_completion_gemini(model):
# test_completion_gemini()
@pytest.mark.asyncio
async def test_acompletion_gemini():
litellm.set_verbose = True
model_name = "gemini/gemini-2.5-flash-lite"
messages = [{"role": "user", "content": "Hey, how's it going?"}]
try:
response = await litellm.acompletion(model=model_name, messages=messages)
# Add any assertions here to check the response
print(f"response: {response}")
except litellm.Timeout as e:
pass
except litellm.APIError as e:
pass
except Exception as e:
if "InternalServerError" in str(e):
pass
else:
pytest.fail(f"Error occurred: {e}")
# Deepseek tests
def test_completion_deepseek():
litellm.set_verbose = True
model_name = "deepseek/deepseek-chat"
tools = [
{
"type": "function",
"function": {
"name": "get_weather",
"description": "Get weather of an location, the user shoud supply a location first",
"parameters": {
"type": "object",
"properties": {
"location": {
"type": "string",
"description": "The city and state, e.g. San Francisco, CA",
}
},
"required": ["location"],
},
},
},
]
messages = [{"role": "user", "content": "How's the weather in Hangzhou?"}]
try:
response = completion(model=model_name, messages=messages, tools=tools)
# Add any assertions here to check the response
print(response)
except litellm.APIError as e:
pass
except Exception as e:
pytest.fail(f"Error occurred: {e}")
@pytest.mark.skip(reason="Account deleted by IBM.")
def test_completion_watsonx_error():
litellm.set_verbose = True
@ -4107,37 +3757,3 @@ def test_completion_gpt_4o_empty_str():
messages=[{"role": "user", "content": ""}],
)
assert resp.choices[0].message.content is not None
def test_edit_note():
litellm.callbacks = ["langfuse_otel"]
response = completion(
model="gpt-4o",
messages=[
{
"role": "system",
"content": "Your only job is to call the edit_note tool with the content specified in the user's message.",
},
{
"role": "user",
"content": "Edit the note with the content: 'This is a test note.'",
},
],
tools=[
{
"type": "function",
"function": {
"name": "edit_note",
"description": "Edit the note with the content specified in the user's message.",
"parameters": {
"type": "object",
"properties": {
"content": {"type": "string"},
},
},
},
},
],
)
return response

View file

@ -136,7 +136,7 @@ def trade(model_name: str) -> List[Trade]: # type: ignore
@pytest.mark.parametrize(
"model", ["claude-haiku-4-5-20251001", "us.anthropic.claude-haiku-4-5-20251001-v1:0"]
"model", ["us.anthropic.claude-haiku-4-5-20251001-v1:0"]
)
@pytest.mark.flaky(retries=6, delay=10)
def test_function_call_parsing(model):

View file

@ -8,7 +8,7 @@ import io
import pytest
from unittest.mock import patch, MagicMock, AsyncMock
import litellm
from litellm import RateLimitError, Timeout, completion, completion_cost, embedding
from litellm import RateLimitError, Timeout, completion_cost, embedding
litellm.num_retries = 0
litellm.cache = None
@ -324,153 +324,6 @@ def test_groq_parallel_function_call():
pytest.fail(f"Error occurred: {e}")
@pytest.mark.parametrize(
"model",
[
"bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0",
],
)
def test_passing_tool_result_as_list(model):
litellm.set_verbose = True
litellm._turn_on_debug()
messages = [
{
"content": [
{
"type": "text",
"text": "You are a helpful assistant that have the ability to interact with a computer to solve tasks.",
}
],
"role": "system",
},
{
"content": [
{
"type": "text",
"text": "Write a git commit message for the current staging area and commit the changes.",
}
],
"role": "user",
},
{
"content": [
{
"type": "text",
"text": "I'll help you commit the changes. Let me first check the git status to see what changes are staged.",
}
],
"role": "assistant",
"tool_calls": [
{
"index": 1,
"function": {
"arguments": '{"command": "git status", "thought": "Checking git status to see staged changes"}',
"name": "execute_bash",
},
"id": "toolu_01V1paXrun4CVetdAGiQaZG5",
"type": "function",
}
],
},
{
"content": [
{
"type": "text",
"text": 'OBSERVATION:\nOn branch master\r\n\r\nNo commits yet\r\n\r\nChanges to be committed:\r\n (use "git rm --cached <file>..." to unstage)\r\n\tnew file: hello.py\r\n\r\n\r\n[Python Interpreter: /openhands/poetry/openhands-ai-5O4_aCHf-py3.12/bin/python]\nroot@openhands-workspace:/workspace # \n[Command finished with exit code 0]',
}
],
"role": "tool",
"tool_call_id": "toolu_01V1paXrun4CVetdAGiQaZG5",
"name": "execute_bash",
},
]
tools = [
{
"type": "function",
"function": {
"name": "execute_bash",
"description": 'Execute a bash command in the terminal.\n* Long running commands: For commands that may run indefinitely, it should be run in the background and the output should be redirected to a file, e.g. command = `python3 app.py > server.log 2>&1 &`.\n* Interactive: If a bash command returns exit code `-1`, this means the process is not yet finished. The assistant must then send a second call to terminal with an empty `command` (which will retrieve any additional logs), or it can send additional text (set `command` to the text) to STDIN of the running process, or it can send command=`ctrl+c` to interrupt the process.\n* Timeout: If a command execution result says "Command timed out. Sending SIGINT to the process", the assistant should retry running the command in the background.\n',
"parameters": {
"type": "object",
"properties": {
"thought": {
"type": "string",
"description": "Reasoning about the action to take.",
},
"command": {
"type": "string",
"description": "The bash command to execute. Can be empty to view additional logs when previous exit code is `-1`. Can be `ctrl+c` to interrupt the currently running process.",
},
},
"required": ["command"],
},
},
},
{
"type": "function",
"function": {
"name": "finish",
"description": "Finish the interaction.\n* Do this if the task is complete.\n* Do this if the assistant cannot proceed further with the task.\n",
},
},
{
"type": "function",
"function": {
"name": "str_replace_editor",
"description": "Custom editing tool for viewing, creating and editing files\n* State is persistent across command calls and discussions with the user\n* If `path` is a file, `view` displays the result of applying `cat -n`. If `path` is a directory, `view` lists non-hidden files and directories up to 2 levels deep\n* The `create` command cannot be used if the specified `path` already exists as a file\n* If a `command` generates a long output, it will be truncated and marked with `<response clipped>`\n* The `undo_edit` command will revert the last edit made to the file at `path`\n\nNotes for using the `str_replace` command:\n* The `old_str` parameter should match EXACTLY one or more consecutive lines from the original file. Be mindful of whitespaces!\n* If the `old_str` parameter is not unique in the file, the replacement will not be performed. Make sure to include enough context in `old_str` to make it unique\n* The `new_str` parameter should contain the edited lines that should replace the `old_str`\n",
"parameters": {
"type": "object",
"properties": {
"command": {
"description": "The commands to run. Allowed options are: `view`, `create`, `str_replace`, `insert`, `undo_edit`.",
"enum": [
"view",
"create",
"str_replace",
"insert",
"undo_edit",
],
"type": "string",
},
"path": {
"description": "Absolute path to file or directory, e.g. `/repo/file.py` or `/repo`.",
"type": "string",
},
"file_text": {
"description": "Required parameter of `create` command, with the content of the file to be created.",
"type": "string",
},
"old_str": {
"description": "Required parameter of `str_replace` command containing the string in `path` to replace.",
"type": "string",
},
"new_str": {
"description": "Optional parameter of `str_replace` command containing the new string (if not given, no string will be added). Required parameter of `insert` command containing the string to insert.",
"type": "string",
},
"insert_line": {
"description": "Required parameter of `insert` command. The `new_str` will be inserted AFTER the line `insert_line` of `path`.",
"type": "integer",
},
"view_range": {
"description": "Optional parameter of `view` command when `path` points to a file. If none is given, the full file is shown. If provided, the file will be shown in the indicated line number range, e.g. [11, 12] will show lines 11 and 12. Indexing at 1 to start. Setting `[start_line, -1]` shows all lines from `start_line` to the end of the file.",
"items": {"type": "integer"},
"type": "array",
},
},
"required": ["command", "path"],
},
},
},
]
for _ in range(2):
resp = completion(model=model, messages=messages, tools=tools)
print(resp)
if model == "claude-sonnet-4-5-20250929":
assert resp.usage.prompt_tokens_details.cached_tokens > 0
@pytest.mark.parametrize("sync_mode", [True, False])
@pytest.mark.asyncio
@pytest.mark.flaky(retries=6, delay=1)

View file

@ -10,7 +10,6 @@ load_dotenv()
import copy
import pytest
from litellm import Router
from litellm.router_strategy.lowest_cost import LowestCostLoggingHandler
from litellm.caching.caching import DualCache
@ -96,37 +95,6 @@ async def test_get_available_deployments_custom_price():
assert selected_model["model_info"]["id"] == "chatgpt-v-1"
@pytest.mark.asyncio
async def test_lowest_cost_routing():
"""
Test if router, returns model with the lowest cost
"""
model_list = [
{
"model_name": "gpt-4",
"litellm_params": {"model": "gpt-4"},
"model_info": {"id": "openai-gpt-4"},
},
{
"model_name": "gpt-3.5-turbo",
"litellm_params": {"model": "gpt-3.5-turbo"},
"model_info": {"id": "gpt-3.5-turbo"},
},
]
# init router
router = Router(model_list=model_list, routing_strategy="cost-based-routing")
response = await router.acompletion(
model="gpt-3.5-turbo",
messages=[{"role": "user", "content": "Hey, how's it going?"}],
)
print(response)
print(
response._hidden_params["model_id"]
) # expect groq-llama, since groq/llama has lowest cost
assert "gpt-3.5-turbo" == response._hidden_params["model_id"]
async def _deploy(lowest_cost_logger, deployment_id, tokens_used, duration):
kwargs = {
"litellm_params": {

View file

@ -64,71 +64,6 @@ def test_router_multi_org_list():
assert len(router.get_model_list()) == 3
@pytest.mark.asyncio()
async def test_router_provider_wildcard_routing():
"""
Pass list of orgs in 1 model definition,
expect a unique deployment for each to be created
"""
litellm.set_verbose = True
router = litellm.Router(
model_list=[
{
"model_name": "openai/*",
"litellm_params": {
"model": "openai/*",
"api_key": os.environ["OPENAI_API_KEY"],
"api_base": "https://api.openai.com/v1",
},
},
{
"model_name": "anthropic/*",
"litellm_params": {
"model": "anthropic/*",
"api_key": os.environ["ANTHROPIC_API_KEY"],
},
},
{
"model_name": "groq/*",
"litellm_params": {
"model": "groq/*",
"api_key": os.environ["GROQ_API_KEY"],
},
},
]
)
print("router model list = ", router.get_model_list())
response1 = await router.acompletion(
model=f"anthropic/{os.environ.get('CI_CD_DEFAULT_ANTHROPIC_MODEL', 'claude-haiku-4-5-20251001')}",
messages=[{"role": "user", "content": "hello"}],
)
print("response 1 = ", response1)
response2 = await router.acompletion(
model="openai/gpt-3.5-turbo",
messages=[{"role": "user", "content": "hello"}],
)
print("response 2 = ", response2)
response3 = await router.acompletion(
model="groq/openai/gpt-oss-120b",
messages=[{"role": "user", "content": "hello"}],
)
print("response 3 = ", response3)
response4 = await router.acompletion(
model=os.environ.get(
"CI_CD_DEFAULT_ANTHROPIC_MODEL", "claude-haiku-4-5-20251001"
),
messages=[{"role": "user", "content": "hello"}],
)
@pytest.mark.asyncio()
async def test_router_provider_wildcard_routing_regex():
"""
@ -986,176 +921,16 @@ def test_function_calling_on_router():
### IMAGE GENERATION
@pytest.mark.asyncio
async def test_aimg_gen_on_router():
litellm.set_verbose = True
try:
model_list = [
{
"model_name": "gpt-image-1",
"litellm_params": {
"model": "gpt-image-1",
},
}
]
router = Router(model_list=model_list, num_retries=3)
response = await router.aimage_generation(
model="gpt-image-1", prompt="A cute baby sea otter"
)
print(response)
assert len(response.data) > 0
router.reset()
except litellm.InternalServerError as e:
pass
except Exception as e:
if "Your task failed as a result of our safety system." in str(e):
pass
elif "Operation polling timed out" in str(e):
pass
elif "Connection error" in str(e):
pass
else:
traceback.print_exc()
pytest.fail(f"Error occurred: {e}")
# asyncio.run(test_aimg_gen_on_router())
def test_img_gen_on_router():
litellm.set_verbose = True
try:
model_list = [
{
"model_name": "gpt-image-1",
"litellm_params": {
"model": "gpt-image-1",
},
}
]
router = Router(model_list=model_list)
response = router.image_generation(
model="gpt-image-1", prompt="A cute baby sea otter"
)
print(response)
assert len(response.data) > 0
router.reset()
except litellm.RateLimitError as e:
pass
except Exception as e:
traceback.print_exc()
pytest.fail(f"Error occurred: {e}")
# test_img_gen_on_router()
###
def test_aembedding_on_router():
litellm.set_verbose = True
try:
model_list = [
{
"model_name": "text-embedding-ada-002",
"litellm_params": {
"model": "text-embedding-ada-002",
},
"tpm": 100000,
"rpm": 10000,
},
]
router = Router(model_list=model_list)
async def embedding_call():
## Test 1: user facing function
response = await router.aembedding(
model="text-embedding-ada-002",
input=["good morning from litellm", "this is another item"],
)
print(response)
## Test 2: underlying function
response = await router._aembedding(
model="text-embedding-ada-002",
input=["good morning from litellm 2"],
)
print(response)
router.reset()
asyncio.run(embedding_call())
print("\n Making sync Embedding call\n")
## Test 1: user facing function
response = router.embedding(
model="text-embedding-ada-002",
input=["good morning from litellm 2"],
)
print(response)
router.reset()
## Test 2: underlying function
response = router._embedding(
model="text-embedding-ada-002",
input=["good morning from litellm 2"],
)
print(response)
router.reset()
except Exception as e:
if "Your task failed as a result of our safety system." in str(e):
pass
elif "Operation polling timed out" in str(e):
pass
elif "Connection error" in str(e):
pass
else:
traceback.print_exc()
pytest.fail(f"Error occurred: {e}")
# test_aembedding_on_router()
def test_azure_embedding_on_router():
"""
[PROD Use Case] - Makes an aembedding call + embedding call
"""
litellm.set_verbose = True
try:
model_list = [
{
"model_name": "text-embedding-ada-002",
"litellm_params": {
"model": "azure/text-embedding-ada-002",
"api_key": os.environ["AZURE_AI_API_KEY"],
"api_base": os.environ["AZURE_AI_API_BASE"],
},
"tpm": 100000,
"rpm": 10000,
},
]
router = Router(model_list=model_list)
async def embedding_call():
response = await router.aembedding(
model="text-embedding-ada-002", input=["good morning from litellm"]
)
print(response)
asyncio.run(embedding_call())
print("\n Making sync Azure Embedding call\n")
response = router.embedding(
model="text-embedding-ada-002",
input=["test 2 from litellm. async embedding"],
)
print(response)
router.reset()
except Exception as e:
traceback.print_exc()
pytest.fail(f"Error occurred: {e}")
# test_azure_embedding_on_router()
@ -1163,30 +938,6 @@ def test_azure_embedding_on_router():
# test openai-compatible endpoint
@pytest.mark.asyncio
async def test_mistral_on_router():
litellm._turn_on_debug()
model_list = [
{
"model_name": "gpt-3.5-turbo",
"litellm_params": {
"model": "mistral/mistral-small-latest",
},
},
]
router = Router(model_list=model_list)
response = await router.acompletion(
model="gpt-3.5-turbo",
messages=[
{
"role": "user",
"content": "hello from litellm test",
}
],
)
print(response)
# asyncio.run(test_mistral_on_router())

View file

@ -435,28 +435,6 @@ def test_completion_azure_stream():
pytest.fail(f"Error occurred: {e}")
def test_completion_azure_function_calling_stream():
try:
litellm.set_verbose = False
user_message = "What is the current weather in Boston?"
messages = [{"content": user_message, "role": "user"}]
response = completion(
model="azure/gpt-4.1-mini",
messages=messages,
stream=True,
tools=tools_schema,
)
# Add any assertions here to check the response
for chunk in response:
print(chunk)
if chunk["choices"][0]["finish_reason"] == "stop":
break
print(chunk["choices"][0]["finish_reason"])
print(chunk["choices"][0]["delta"]["content"])
except Exception as e:
pytest.fail(f"Error occurred: {e}")
@pytest.mark.skip("Flaky ollama test - needs to be fixed")
def test_completion_ollama_hosted_stream():
try:

View file

@ -15,35 +15,6 @@ import litellm
from tests.fake_openai_endpoint import FAKE_OPENAI_API_BASE
@pytest.mark.parametrize(
"model, provider",
[
("gpt-3.5-turbo", "openai"),
("azure/gpt-4.1-mini", "azure"),
],
)
@pytest.mark.parametrize("sync_mode", [True, False])
@pytest.mark.asyncio
async def test_httpx_timeout(model, provider, sync_mode):
"""
Test if setting httpx.timeout works for completion calls
"""
timeout_val = httpx.Timeout(10.0, connect=60.0)
messages = [{"role": "user", "content": "Hey, how's it going?"}]
if sync_mode:
response = litellm.completion(
model=model, messages=messages, timeout=timeout_val
)
else:
response = await litellm.acompletion(
model=model, messages=messages, timeout=timeout_val
)
print(f"response: {response}")
def test_timeout():
# this Will Raise a timeout
litellm.set_verbose = False

View file

@ -77,20 +77,6 @@ def validate_stream_chunk(chunk):
assert isinstance(chunk.created, int)
def test_streaming_response():
client = get_test_client()
stream = client.responses.create(
model="gpt-5.5", input="just respond with the word 'ping'", stream=True
)
collected_chunks = []
for chunk in stream:
print("stream chunk=", chunk)
collected_chunks.append(chunk)
assert len(collected_chunks) > 0
def test_model_not_found_error():
client = get_test_client()
with pytest.raises(NotFoundError):

View file

@ -1,7 +1,7 @@
import json
import os
from datetime import datetime
from typing import AsyncIterator, Dict, Any
from typing import Dict, Any
import asyncio
import unittest.mock
from unittest.mock import AsyncMock, MagicMock
@ -69,6 +69,9 @@ def _validate_anthropic_response(response: Dict[str, Any]):
class TestAnthropicDirectAPI(BaseAnthropicMessagesTest):
"""Tests for direct Anthropic API calls"""
test_non_streaming_base = None
test_streaming_base = None
@property
def model_config(self) -> Dict[str, Any]:
return {
@ -87,6 +90,8 @@ class TestAnthropicDirectAPI(BaseAnthropicMessagesTest):
class TestAnthropicBedrockAPI(BaseAnthropicMessagesTest):
"""Tests for Anthropic via Bedrock"""
test_streaming_base = None
@property
def model_config(self) -> Dict[str, Any]:
return {
@ -104,6 +109,8 @@ class TestAnthropicBedrockAPI(BaseAnthropicMessagesTest):
class TestAnthropicOpenAIAPI(BaseAnthropicMessagesTest):
"""Tests for OpenAI via Anthropic messages interface"""
test_streaming_base = None
@property
def model_config(self) -> Dict[str, Any]:
return {
@ -126,67 +133,6 @@ class TestAnthropicOpenAIAPI(BaseAnthropicMessagesTest):
pass
@pytest.mark.asyncio
async def test_anthropic_messages_streaming_with_bad_request():
"""
Test the anthropic_messages with streaming request
"""
error = None
try:
response = await litellm.anthropic.messages.acreate(
messages=[{"role": "user", "content": "hi"}],
api_key=os.getenv("ANTHROPIC_API_KEY"),
model="claude-haiku-4-5-20251001",
max_tokens=100,
stream=True,
)
print(response)
if isinstance(response, AsyncIterator):
async for chunk in response:
print("chunk=", chunk)
except Exception as e:
error = e
if error is not None:
assert getattr(error, "status_code", 400) == 400, f"got {vars(error)}"
@pytest.mark.asyncio
async def test_anthropic_messages_router_streaming_with_bad_request():
"""
Test the anthropic_messages with streaming request
"""
error = None
try:
router = Router(
model_list=[
{
"model_name": "claude-special-alias",
"litellm_params": {
"model": "claude-haiku-4-5-20251001",
"api_key": os.getenv("ANTHROPIC_API_KEY"),
},
}
]
)
response = await router.aanthropic_messages(
messages=[{"role": "user", "content": "hi"}],
model="claude-special-alias",
max_tokens=100,
stream=True,
)
print(response)
if isinstance(response, AsyncIterator):
async for chunk in response:
print("chunk=", chunk)
except Exception as e:
error = e
if error is not None:
assert getattr(error, "status_code", 400) == 400, f"got {vars(error)}"
@pytest.mark.asyncio
async def test_anthropic_messages_litellm_router_non_streaming():
"""

View file

@ -425,39 +425,6 @@ async def test_completion_streaming_usage_metrics():
assert last_chunk.usage.total_tokens > 0, "Total tokens should be greater than 0"
@pytest.mark.asyncio
async def test_chat_completion_anthropic_structured_output():
"""
Ensure nested pydantic output is returned correctly
"""
from pydantic import BaseModel
class CalendarEvent(BaseModel):
name: str
date: str
participants: list[str]
class EventsList(BaseModel):
events: list[CalendarEvent]
messages = [
{"role": "user", "content": "List 5 important events in the XIX century"}
]
client = AsyncOpenAI(api_key="sk-1234", base_url="http://0.0.0.0:4000")
res = await client.beta.chat.completions.parse(
model="bedrock/us.anthropic.claude-sonnet-4-5-20250929-v1:0",
messages=messages,
response_format=EventsList,
timeout=60,
)
message = res.choices[0].message
if message.parsed:
print(message.parsed.events)
@pytest.mark.asyncio
async def test_proxy_all_models():
"""

View file

@ -10,7 +10,6 @@ import litellm
from litellm.google_genai import (
generate_content,
agenerate_content,
generate_content_stream,
agenerate_content_stream,
)
from google.genai.types import ContentDict, PartDict
@ -195,45 +194,6 @@ class BaseGoogleGenAITest:
return response
@pytest.mark.parametrize("is_async", [False, True])
@pytest.mark.asyncio
async def test_streaming_base(self, is_async: bool):
"""Base test for streaming requests (parametrized for sync/async)"""
request_params = self.model_config
temp_file_path = load_vertex_ai_credentials(model=request_params["model"])
if temp_file_path:
self._temp_files_to_cleanup.append(temp_file_path)
contents = ContentDict(
parts=[PartDict(text="Hello, can you tell me a short joke?")],
role="user",
)
print(
f"Testing {'async' if is_async else 'sync'} streaming with model config: {request_params}"
)
print(f"Contents: {contents}")
chunks = []
if is_async:
print("\n--- Testing async agenerate_content_stream ---")
response = await agenerate_content_stream(
contents=contents, **request_params
)
async for chunk in response:
print(f"Async chunk: {chunk}")
chunks.append(chunk)
else:
print("\n--- Testing sync generate_content_stream ---")
response = generate_content_stream(contents=contents, **request_params)
for chunk in response:
print(f"Sync chunk: {chunk}")
chunks.append(chunk)
self._validate_streaming_response(chunks)
return chunks
@pytest.mark.asyncio
async def test_async_non_streaming_with_logging(self):
"""Test async non-streaming Google GenAI generate content with logging"""

View file

@ -10,6 +10,8 @@ import json
class TestGoogleGenAIStudio(BaseGoogleGenAITest, BaseGoogleGenAIProxySDKTest):
"""Test Google GenAI Studio"""
test_non_streaming_base = None
@property
def model_config(self):
return {

View file

@ -15,6 +15,8 @@ from tests.unified_google_tests.base_interactions_test import (
class TestLiteLLMResponsesBridge(BaseInteractionsTest):
"""Test LiteLLM Responses bridge using the base test suite."""
test_create_streaming = None
def get_model(self) -> str:
"""Return the model string for the bridge provider.