test: remove end of life model from tests

This commit is contained in:
Krrish Dholakia 2025-09-09 21:01:45 -07:00
parent e443d01925
commit d05f58721e
7 changed files with 28 additions and 125 deletions

View file

@ -338,7 +338,9 @@ def test_aget_valid_models():
print(valid_models)
# list of openai supported llms on litellm
expected_models = litellm.open_ai_chat_completion_models | litellm.open_ai_text_completion_models
expected_models = (
litellm.open_ai_chat_completion_models | litellm.open_ai_text_completion_models
)
assert set(valid_models) == set(expected_models)
@ -410,7 +412,12 @@ def test_validate_environment_api_key():
def test_validate_environment_api_version():
response_obj = validate_environment(model="azure/openai-deployment", api_key="sk-my-test-key", api_base="https://fake.openai.azure.com/", api_version="2024-02-15")
response_obj = validate_environment(
model="azure/openai-deployment",
api_key="sk-my-test-key",
api_base="https://fake.openai.azure.com/",
api_version="2024-02-15",
)
assert (
response_obj["keys_in_environment"] is True
), f"Missing keys={response_obj['missing_keys']}"
@ -513,7 +520,6 @@ def test_function_to_dict():
("gpt-3.5-turbo", True),
("azure/gpt-4-1106-preview", True),
("groq/gemma-7b-it", True),
("anthropic.claude-instant-v1", False),
("gemini/gemini-1.5-flash", True),
],
)
@ -1690,15 +1696,6 @@ def test_pick_cheapest_chat_model_from_llm_provider():
assert len(pick_cheapest_chat_models_from_llm_provider("unknown", n=1)) == 0
def test_get_potential_model_names():
from litellm.utils import _get_potential_model_names
assert _get_potential_model_names(
model="bedrock/ap-northeast-1/anthropic.claude-instant-v1",
custom_llm_provider="bedrock",
)
@pytest.mark.parametrize("num_retries", [0, 1, 5])
def test_get_num_retries(num_retries):
from litellm.utils import _get_wrapper_num_retries

View file

@ -69,7 +69,7 @@ def test_completion_bedrock_claude_completion_auth():
try:
response = completion(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=messages,
max_tokens=10,
temperature=0.1,
@ -195,7 +195,7 @@ def test_completion_bedrock_claude_external_client_auth():
)
response = completion(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=messages,
max_tokens=10,
temperature=0.1,
@ -232,7 +232,7 @@ def test_completion_bedrock_claude_sts_client_auth():
litellm.set_verbose = True
response = completion(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=messages,
max_tokens=10,
temperature=0.1,
@ -904,7 +904,7 @@ def test_bedrock_ptu():
)
try:
response = litellm.completion(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=[{"role": "user", "content": "What's AWS?"}],
model_id=model_id,
client=client,
@ -1070,7 +1070,7 @@ def test_completion_bedrock_external_client_region():
with patch.object(client, "post", new=Mock()) as mock_client_post:
try:
response = completion(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
messages=messages,
max_tokens=10,
temperature=0.1,

View file

@ -759,6 +759,7 @@ def test_completion_base64(model):
else:
pytest.fail(f"An exception occurred - {str(e)}")
def test_completion_mistral_api():
try:
litellm.set_verbose = True
@ -3190,7 +3191,6 @@ def response_format_tests(response: litellm.ModelResponse):
"bedrock/mistral.mistral-large-2407-v1:0",
"bedrock/cohere.command-r-plus-v1:0",
"anthropic.claude-3-sonnet-20240229-v1:0",
"anthropic.claude-instant-v1",
"mistral.mistral-7b-instruct-v0:2",
# "bedrock/amazon.titan-tg1-large",
"meta.llama3-8b-instruct-v1:0",

View file

@ -319,64 +319,9 @@ def test_cost_openai_image_gen():
assert cost == 0.019922944
def test_cost_bedrock_pricing():
"""
- get pricing specific to region for a model
"""
from litellm import Choices, Message, ModelResponse
from litellm.utils import Usage
litellm.set_verbose = True
input_tokens = litellm.token_counter(
model="bedrock/anthropic.claude-instant-v1",
messages=[{"role": "user", "content": "Hey, how's it going?"}],
)
print(f"input_tokens: {input_tokens}")
output_tokens = litellm.token_counter(
model="bedrock/anthropic.claude-instant-v1",
text="It's all going well",
count_response_tokens=True,
)
print(f"output_tokens: {output_tokens}")
resp = ModelResponse(
id="chatcmpl-e41836bb-bb8b-4df2-8e70-8f3e160155ac",
choices=[
Choices(
finish_reason=None,
index=0,
message=Message(
content="It's all going well",
role="assistant",
),
)
],
created=1700775391,
model="anthropic.claude-instant-v1",
object="chat.completion",
system_fingerprint=None,
usage=Usage(
prompt_tokens=input_tokens,
completion_tokens=output_tokens,
total_tokens=input_tokens + output_tokens,
),
)
resp._hidden_params = {
"custom_llm_provider": "bedrock",
"region_name": "ap-northeast-1",
}
cost = litellm.completion_cost(
model="anthropic.claude-instant-v1",
completion_response=resp,
messages=[{"role": "user", "content": "Hey, how's it going?"}],
)
predicted_cost = input_tokens * 0.00000223 + 0.00000755 * output_tokens
assert cost == predicted_cost
def test_cost_bedrock_pricing_actual_calls():
litellm.set_verbose = True
model = "anthropic.claude-instant-v1"
model = "anthropic.claude-3-5-sonnet-20240620-v1:0"
messages = [{"role": "user", "content": "Hey, how's it going?"}]
response = litellm.completion(
model=model, messages=messages, mock_response="hello cool one"
@ -384,7 +329,7 @@ def test_cost_bedrock_pricing_actual_calls():
print("response", response)
cost = litellm.completion_cost(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
completion_response=response,
messages=[{"role": "user", "content": "Hey, how's it going?"}],
)
@ -864,6 +809,7 @@ def test_vertex_ai_embedding_completion_cost(caplog):
# assert False
@pytest.mark.parametrize("sync_mode", [True, False])
@pytest.mark.asyncio
async def test_completion_cost_hidden_params(sync_mode):
@ -949,7 +895,9 @@ def test_vertex_ai_mistral_predict_cost(usage):
assert predictive_cost > 0
@pytest.mark.parametrize("model", ["openai/tts-1", "azure/tts-1", "openai/gpt-4o-mini-tts"])
@pytest.mark.parametrize(
"model", ["openai/tts-1", "azure/tts-1", "openai/gpt-4o-mini-tts"]
)
def test_completion_cost_tts(model):
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
@ -1225,7 +1173,10 @@ def test_get_model_params_fireworks_ai(model, base_model):
@pytest.mark.parametrize(
"model",
["fireworks_ai/llama-v3p1-405b-instruct", "fireworks_ai/llama4-maverick-instruct-basic"],
[
"fireworks_ai/llama-v3p1-405b-instruct",
"fireworks_ai/llama4-maverick-instruct-basic",
],
)
def test_completion_cost_fireworks_ai(model):
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
@ -2862,6 +2813,7 @@ def test_cost_calculator_with_custom_pricing():
@pytest.mark.asyncio
async def test_cost_calculator_with_custom_pricing_router(model_item, custom_pricing):
from litellm import Router
if custom_pricing == "litellm_params":
model_item["litellm_params"]["input_cost_per_token"] = 0.0000008
model_item["litellm_params"]["output_cost_per_token"] = 0.0000032

View file

@ -105,7 +105,7 @@ async def test_router_timeouts_bedrock():
{
"model_name": "bedrock",
"litellm_params": {
"model": "bedrock/anthropic.claude-instant-v1",
"model": "bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
"timeout": 0.00001,
},
"tpm": 80000,

View file

@ -1317,7 +1317,6 @@ async def test_completion_replicate_llama3_streaming(sync_mode):
# ["bedrock/ai21.jamba-instruct-v1:0", "us-east-1"],
# ["bedrock/cohere.command-r-plus-v1:0", None],
["anthropic.claude-3-sonnet-20240229-v1:0", None],
# ["anthropic.claude-instant-v1", None],
# ["mistral.mistral-7b-instruct-v0:2", None],
["bedrock/amazon.titan-tg1-large", None],
# ["meta.llama3-8b-instruct-v1:0", None],
@ -1545,50 +1544,6 @@ def test_completion_replicate_stream_bad_key():
# test_completion_replicate_stream_bad_key()
def test_completion_bedrock_claude_stream():
try:
litellm.set_verbose = True
response = completion(
model="bedrock/anthropic.claude-instant-v1",
messages=[
{
"role": "user",
"content": "Be as verbose as possible and give as many details as possible, how does a court case get to the Supreme Court?",
}
],
temperature=1,
max_tokens=20,
stream=True,
)
print(response)
complete_response = ""
has_finish_reason = False
# Add any assertions here to check the response
first_chunk_id = None
for idx, chunk in enumerate(response):
# print
if idx == 0:
first_chunk_id = chunk.id
else:
assert (
chunk.id == first_chunk_id
), f"chunk ids do not match: {chunk.id} != first chunk id{first_chunk_id}"
chunk, finished = streaming_format_tests(idx, chunk)
has_finish_reason = finished
complete_response += chunk
if finished:
break
if has_finish_reason is False:
raise Exception("finish reason not set for last chunk")
if complete_response.strip() == "":
raise Exception("Empty response received")
except RateLimitError:
pass
except Exception as e:
pytest.fail(f"Error occurred: {e}")
# test_completion_bedrock_claude_stream()

View file

@ -22,7 +22,6 @@ import litellm
"model, provider",
[
("gpt-3.5-turbo", "openai"),
("anthropic.claude-instant-v1", "bedrock"),
("azure/chatgpt-v-3", "azure"),
],
)
@ -77,7 +76,7 @@ def test_bedrock_timeout():
litellm.set_verbose = True
try:
response = litellm.completion(
model="bedrock/anthropic.claude-instant-v1",
model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0",
timeout=0.01,
messages=[{"role": "user", "content": "hello, write a 20 pg essay"}],
)