diff --git a/tests/litellm_utils_tests/test_utils.py b/tests/litellm_utils_tests/test_utils.py index 0818d0b4b08..8d4fc3ac451 100644 --- a/tests/litellm_utils_tests/test_utils.py +++ b/tests/litellm_utils_tests/test_utils.py @@ -338,7 +338,9 @@ def test_aget_valid_models(): print(valid_models) # list of openai supported llms on litellm - expected_models = litellm.open_ai_chat_completion_models | litellm.open_ai_text_completion_models + expected_models = ( + litellm.open_ai_chat_completion_models | litellm.open_ai_text_completion_models + ) assert set(valid_models) == set(expected_models) @@ -410,7 +412,12 @@ def test_validate_environment_api_key(): def test_validate_environment_api_version(): - response_obj = validate_environment(model="azure/openai-deployment", api_key="sk-my-test-key", api_base="https://fake.openai.azure.com/", api_version="2024-02-15") + response_obj = validate_environment( + model="azure/openai-deployment", + api_key="sk-my-test-key", + api_base="https://fake.openai.azure.com/", + api_version="2024-02-15", + ) assert ( response_obj["keys_in_environment"] is True ), f"Missing keys={response_obj['missing_keys']}" @@ -513,7 +520,6 @@ def test_function_to_dict(): ("gpt-3.5-turbo", True), ("azure/gpt-4-1106-preview", True), ("groq/gemma-7b-it", True), - ("anthropic.claude-instant-v1", False), ("gemini/gemini-1.5-flash", True), ], ) @@ -1690,15 +1696,6 @@ def test_pick_cheapest_chat_model_from_llm_provider(): assert len(pick_cheapest_chat_models_from_llm_provider("unknown", n=1)) == 0 -def test_get_potential_model_names(): - from litellm.utils import _get_potential_model_names - - assert _get_potential_model_names( - model="bedrock/ap-northeast-1/anthropic.claude-instant-v1", - custom_llm_provider="bedrock", - ) - - @pytest.mark.parametrize("num_retries", [0, 1, 5]) def test_get_num_retries(num_retries): from litellm.utils import _get_wrapper_num_retries diff --git a/tests/llm_translation/test_bedrock_completion.py b/tests/llm_translation/test_bedrock_completion.py index 8fece8f86cb..3252e2eaa72 100644 --- a/tests/llm_translation/test_bedrock_completion.py +++ b/tests/llm_translation/test_bedrock_completion.py @@ -69,7 +69,7 @@ def test_completion_bedrock_claude_completion_auth(): try: response = completion( - model="bedrock/anthropic.claude-instant-v1", + model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", messages=messages, max_tokens=10, temperature=0.1, @@ -195,7 +195,7 @@ def test_completion_bedrock_claude_external_client_auth(): ) response = completion( - model="bedrock/anthropic.claude-instant-v1", + model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", messages=messages, max_tokens=10, temperature=0.1, @@ -232,7 +232,7 @@ def test_completion_bedrock_claude_sts_client_auth(): litellm.set_verbose = True response = completion( - model="bedrock/anthropic.claude-instant-v1", + model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", messages=messages, max_tokens=10, temperature=0.1, @@ -904,7 +904,7 @@ def test_bedrock_ptu(): ) try: response = litellm.completion( - model="bedrock/anthropic.claude-instant-v1", + model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", messages=[{"role": "user", "content": "What's AWS?"}], model_id=model_id, client=client, @@ -1070,7 +1070,7 @@ def test_completion_bedrock_external_client_region(): with patch.object(client, "post", new=Mock()) as mock_client_post: try: response = completion( - model="bedrock/anthropic.claude-instant-v1", + model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", messages=messages, max_tokens=10, temperature=0.1, diff --git a/tests/local_testing/test_completion.py b/tests/local_testing/test_completion.py index b65705e51ba..390ea5835ca 100644 --- a/tests/local_testing/test_completion.py +++ b/tests/local_testing/test_completion.py @@ -759,6 +759,7 @@ def test_completion_base64(model): else: pytest.fail(f"An exception occurred - {str(e)}") + def test_completion_mistral_api(): try: litellm.set_verbose = True @@ -3190,7 +3191,6 @@ def response_format_tests(response: litellm.ModelResponse): "bedrock/mistral.mistral-large-2407-v1:0", "bedrock/cohere.command-r-plus-v1:0", "anthropic.claude-3-sonnet-20240229-v1:0", - "anthropic.claude-instant-v1", "mistral.mistral-7b-instruct-v0:2", # "bedrock/amazon.titan-tg1-large", "meta.llama3-8b-instruct-v1:0", diff --git a/tests/local_testing/test_completion_cost.py b/tests/local_testing/test_completion_cost.py index 39d4536d7aa..e8f5de534dc 100644 --- a/tests/local_testing/test_completion_cost.py +++ b/tests/local_testing/test_completion_cost.py @@ -319,64 +319,9 @@ def test_cost_openai_image_gen(): assert cost == 0.019922944 -def test_cost_bedrock_pricing(): - """ - - get pricing specific to region for a model - """ - from litellm import Choices, Message, ModelResponse - from litellm.utils import Usage - - litellm.set_verbose = True - input_tokens = litellm.token_counter( - model="bedrock/anthropic.claude-instant-v1", - messages=[{"role": "user", "content": "Hey, how's it going?"}], - ) - print(f"input_tokens: {input_tokens}") - output_tokens = litellm.token_counter( - model="bedrock/anthropic.claude-instant-v1", - text="It's all going well", - count_response_tokens=True, - ) - print(f"output_tokens: {output_tokens}") - resp = ModelResponse( - id="chatcmpl-e41836bb-bb8b-4df2-8e70-8f3e160155ac", - choices=[ - Choices( - finish_reason=None, - index=0, - message=Message( - content="It's all going well", - role="assistant", - ), - ) - ], - created=1700775391, - model="anthropic.claude-instant-v1", - object="chat.completion", - system_fingerprint=None, - usage=Usage( - prompt_tokens=input_tokens, - completion_tokens=output_tokens, - total_tokens=input_tokens + output_tokens, - ), - ) - resp._hidden_params = { - "custom_llm_provider": "bedrock", - "region_name": "ap-northeast-1", - } - - cost = litellm.completion_cost( - model="anthropic.claude-instant-v1", - completion_response=resp, - messages=[{"role": "user", "content": "Hey, how's it going?"}], - ) - predicted_cost = input_tokens * 0.00000223 + 0.00000755 * output_tokens - assert cost == predicted_cost - - def test_cost_bedrock_pricing_actual_calls(): litellm.set_verbose = True - model = "anthropic.claude-instant-v1" + model = "anthropic.claude-3-5-sonnet-20240620-v1:0" messages = [{"role": "user", "content": "Hey, how's it going?"}] response = litellm.completion( model=model, messages=messages, mock_response="hello cool one" @@ -384,7 +329,7 @@ def test_cost_bedrock_pricing_actual_calls(): print("response", response) cost = litellm.completion_cost( - model="bedrock/anthropic.claude-instant-v1", + model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", completion_response=response, messages=[{"role": "user", "content": "Hey, how's it going?"}], ) @@ -864,6 +809,7 @@ def test_vertex_ai_embedding_completion_cost(caplog): # assert False + @pytest.mark.parametrize("sync_mode", [True, False]) @pytest.mark.asyncio async def test_completion_cost_hidden_params(sync_mode): @@ -949,7 +895,9 @@ def test_vertex_ai_mistral_predict_cost(usage): assert predictive_cost > 0 -@pytest.mark.parametrize("model", ["openai/tts-1", "azure/tts-1", "openai/gpt-4o-mini-tts"]) +@pytest.mark.parametrize( + "model", ["openai/tts-1", "azure/tts-1", "openai/gpt-4o-mini-tts"] +) def test_completion_cost_tts(model): os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" litellm.model_cost = litellm.get_model_cost_map(url="") @@ -1225,7 +1173,10 @@ def test_get_model_params_fireworks_ai(model, base_model): @pytest.mark.parametrize( "model", - ["fireworks_ai/llama-v3p1-405b-instruct", "fireworks_ai/llama4-maverick-instruct-basic"], + [ + "fireworks_ai/llama-v3p1-405b-instruct", + "fireworks_ai/llama4-maverick-instruct-basic", + ], ) def test_completion_cost_fireworks_ai(model): os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" @@ -2862,6 +2813,7 @@ def test_cost_calculator_with_custom_pricing(): @pytest.mark.asyncio async def test_cost_calculator_with_custom_pricing_router(model_item, custom_pricing): from litellm import Router + if custom_pricing == "litellm_params": model_item["litellm_params"]["input_cost_per_token"] = 0.0000008 model_item["litellm_params"]["output_cost_per_token"] = 0.0000032 diff --git a/tests/local_testing/test_router_timeout.py b/tests/local_testing/test_router_timeout.py index c8d7502eee2..8ca58f84455 100644 --- a/tests/local_testing/test_router_timeout.py +++ b/tests/local_testing/test_router_timeout.py @@ -105,7 +105,7 @@ async def test_router_timeouts_bedrock(): { "model_name": "bedrock", "litellm_params": { - "model": "bedrock/anthropic.claude-instant-v1", + "model": "bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", "timeout": 0.00001, }, "tpm": 80000, diff --git a/tests/local_testing/test_streaming.py b/tests/local_testing/test_streaming.py index ab6b8c5df5b..5ce437a49d4 100644 --- a/tests/local_testing/test_streaming.py +++ b/tests/local_testing/test_streaming.py @@ -1317,7 +1317,6 @@ async def test_completion_replicate_llama3_streaming(sync_mode): # ["bedrock/ai21.jamba-instruct-v1:0", "us-east-1"], # ["bedrock/cohere.command-r-plus-v1:0", None], ["anthropic.claude-3-sonnet-20240229-v1:0", None], - # ["anthropic.claude-instant-v1", None], # ["mistral.mistral-7b-instruct-v0:2", None], ["bedrock/amazon.titan-tg1-large", None], # ["meta.llama3-8b-instruct-v1:0", None], @@ -1545,50 +1544,6 @@ def test_completion_replicate_stream_bad_key(): # test_completion_replicate_stream_bad_key() - -def test_completion_bedrock_claude_stream(): - try: - litellm.set_verbose = True - response = completion( - model="bedrock/anthropic.claude-instant-v1", - messages=[ - { - "role": "user", - "content": "Be as verbose as possible and give as many details as possible, how does a court case get to the Supreme Court?", - } - ], - temperature=1, - max_tokens=20, - stream=True, - ) - print(response) - complete_response = "" - has_finish_reason = False - # Add any assertions here to check the response - first_chunk_id = None - for idx, chunk in enumerate(response): - # print - if idx == 0: - first_chunk_id = chunk.id - else: - assert ( - chunk.id == first_chunk_id - ), f"chunk ids do not match: {chunk.id} != first chunk id{first_chunk_id}" - chunk, finished = streaming_format_tests(idx, chunk) - has_finish_reason = finished - complete_response += chunk - if finished: - break - if has_finish_reason is False: - raise Exception("finish reason not set for last chunk") - if complete_response.strip() == "": - raise Exception("Empty response received") - except RateLimitError: - pass - except Exception as e: - pytest.fail(f"Error occurred: {e}") - - # test_completion_bedrock_claude_stream() diff --git a/tests/local_testing/test_timeout.py b/tests/local_testing/test_timeout.py index 2299ab4409d..2f8050affed 100644 --- a/tests/local_testing/test_timeout.py +++ b/tests/local_testing/test_timeout.py @@ -22,7 +22,6 @@ import litellm "model, provider", [ ("gpt-3.5-turbo", "openai"), - ("anthropic.claude-instant-v1", "bedrock"), ("azure/chatgpt-v-3", "azure"), ], ) @@ -77,7 +76,7 @@ def test_bedrock_timeout(): litellm.set_verbose = True try: response = litellm.completion( - model="bedrock/anthropic.claude-instant-v1", + model="bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0", timeout=0.01, messages=[{"role": "user", "content": "hello, write a 20 pg essay"}], )