diff --git a/tests/local_testing/test_amazing_vertex_completion.py b/tests/local_testing/test_amazing_vertex_completion.py index 73a8ff06c86..4219c8f1a71 100644 --- a/tests/local_testing/test_amazing_vertex_completion.py +++ b/tests/local_testing/test_amazing_vertex_completion.py @@ -3773,7 +3773,7 @@ def test_vertex_schema_test(): } response = litellm.completion( - model="vertex_ai/gemini-2.5-flash-preview-05-20", + model="vertex_ai/gemini-2.5-flash-preview", messages=[{"role": "user", "content": "call the tool"}], tools=[tool], tool_choice="required", @@ -3888,7 +3888,7 @@ def test_vertex_ai_gemini_2_5_pro_streaming(): load_vertex_ai_credentials() # litellm._turn_on_debug() response = completion( - model="vertex_ai/gemini-2.5-pro-preview-06-05", + model="vertex_ai/gemini-2.5-pro", messages=[{"role": "user", "content": "Hi!"}], vertex_location="global", stream=True, diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index eba9cb39e80..612662fd2c6 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -177,7 +177,7 @@ def test_generic_cost_per_token_anthropic_prompt_caching(): def test_string_cost_values(): """Test that cost values defined as strings are properly converted to floats.""" from unittest.mock import patch - + # Mock model info with string cost values (as might be read from config.yaml) mock_model_info = { "input_cost_per_token": "3e-7", # String representation of scientific notation @@ -187,51 +187,46 @@ def test_string_cost_values(): "cache_read_input_token_cost": "1.5e-8", # String representation of scientific notation "cache_creation_input_token_cost": "2.5e-8", # String representation of scientific notation } - + # Test usage with various token types usage = Usage( prompt_tokens=1000, completion_tokens=500, total_tokens=1500, prompt_tokens_details=PromptTokensDetailsWrapper( - audio_tokens=100, - cached_tokens=200, - text_tokens=700, - image_tokens=None + audio_tokens=100, cached_tokens=200, text_tokens=700, image_tokens=None ), completion_tokens_details=CompletionTokensDetailsWrapper( audio_tokens=50, reasoning_tokens=None, text_tokens=450, accepted_prediction_tokens=None, - rejected_prediction_tokens=None + rejected_prediction_tokens=None, ), - _cache_creation_input_tokens=150 + _cache_creation_input_tokens=150, ) - + # Mock get_model_info to return our mock model info - with patch('litellm.litellm_core_utils.llm_cost_calc.utils.get_model_info', return_value=mock_model_info): + with patch( + "litellm.litellm_core_utils.llm_cost_calc.utils.get_model_info", + return_value=mock_model_info, + ): prompt_cost, completion_cost = generic_cost_per_token( - model="test-model", - usage=usage, - custom_llm_provider="test-provider" + model="test-model", usage=usage, custom_llm_provider="test-provider" ) - + # Calculate expected costs manually # Prompt cost = text_tokens * input_cost + audio_tokens * audio_cost + cached_tokens * cache_read_cost + cache_creation_tokens * cache_creation_cost expected_prompt_cost = ( - 700 * 3e-7 + # text tokens - 100 * 1e-6 + # audio tokens - 200 * 1.5e-8 + # cached tokens - 150 * 2.5e-8 # cache creation tokens + 700 * 3e-7 # text tokens + + 100 * 1e-6 # audio tokens + + 200 * 1.5e-8 # cached tokens + + 150 * 2.5e-8 # cache creation tokens ) - + # Completion cost = text_tokens * output_cost + audio_tokens * audio_output_cost - expected_completion_cost = ( - 450 * 6e-7 + # text tokens - 50 * 2e-6 # audio tokens - ) - + expected_completion_cost = 450 * 6e-7 + 50 * 2e-6 # text tokens # audio tokens + # Assert costs are calculated correctly assert round(prompt_cost, 12) == round(expected_prompt_cost, 12) assert round(completion_cost, 12) == round(expected_completion_cost, 12) @@ -240,42 +235,42 @@ def test_string_cost_values(): def test_calculate_cost_component_with_string_values(): """Test the calculate_cost_component function directly with string cost values.""" from litellm.litellm_core_utils.llm_cost_calc.utils import calculate_cost_component - + # Test with valid string scientific notation model_info = {"input_cost_per_token": "3e-7"} cost = calculate_cost_component(model_info, "input_cost_per_token", 1000) assert cost == 1000 * 3e-7 - + # Test with valid string decimal notation model_info = {"output_cost_per_token": "0.000001"} cost = calculate_cost_component(model_info, "output_cost_per_token", 500) assert cost == 500 * 0.000001 - + # Test with float value (should work as before) model_info = {"input_cost_per_token": 3e-7} cost = calculate_cost_component(model_info, "input_cost_per_token", 1000) assert cost == 1000 * 3e-7 - + # Test with invalid string value (should return 0.0) model_info = {"input_cost_per_token": "invalid_number"} cost = calculate_cost_component(model_info, "input_cost_per_token", 1000) assert cost == 0.0 - + # Test with None value (should return 0.0) model_info = {"input_cost_per_token": None} cost = calculate_cost_component(model_info, "input_cost_per_token", 1000) assert cost == 0.0 - + # Test with missing key (should return 0.0) model_info = {} cost = calculate_cost_component(model_info, "input_cost_per_token", 1000) assert cost == 0.0 - + # Test with zero usage (should return 0.0) model_info = {"input_cost_per_token": "3e-7"} cost = calculate_cost_component(model_info, "input_cost_per_token", 0) assert cost == 0.0 - + # Test with None usage (should return 0.0) model_info = {"input_cost_per_token": "3e-7"} cost = calculate_cost_component(model_info, "input_cost_per_token", None) @@ -285,47 +280,45 @@ def test_calculate_cost_component_with_string_values(): def test_string_cost_values_edge_cases(): """Test edge cases for string cost value handling.""" from unittest.mock import patch - + # Test with mixed string and float cost values mock_model_info = { "input_cost_per_token": "1e-6", # String - "output_cost_per_token": 2e-6, # Float + "output_cost_per_token": 2e-6, # Float "input_cost_per_audio_token": "invalid", # Invalid string "output_cost_per_audio_token": None, # None value } - + usage = Usage( prompt_tokens=1000, completion_tokens=500, total_tokens=1500, prompt_tokens_details=PromptTokensDetailsWrapper( - audio_tokens=100, - cached_tokens=0, - text_tokens=1000, - image_tokens=None + audio_tokens=100, cached_tokens=0, text_tokens=1000, image_tokens=None ), completion_tokens_details=CompletionTokensDetailsWrapper( audio_tokens=50, reasoning_tokens=None, text_tokens=500, accepted_prediction_tokens=None, - rejected_prediction_tokens=None - ) + rejected_prediction_tokens=None, + ), ) - - with patch('litellm.litellm_core_utils.llm_cost_calc.utils.get_model_info', return_value=mock_model_info): + + with patch( + "litellm.litellm_core_utils.llm_cost_calc.utils.get_model_info", + return_value=mock_model_info, + ): prompt_cost, completion_cost = generic_cost_per_token( - model="test-model", - usage=usage, - custom_llm_provider="test-provider" + model="test-model", usage=usage, custom_llm_provider="test-provider" ) - + # Expected costs: # Prompt: 1000 * 1e-6 + 100 * 0 (invalid string becomes 0) # Completion: 500 * 2e-6 (text_tokens == completion_tokens, so is_text_tokens_total=True, no separate audio cost) expected_prompt_cost = 1000 * 1e-6 expected_completion_cost = 500 * 2e-6 - + assert round(prompt_cost, 12) == round(expected_prompt_cost, 12) assert round(completion_cost, 12) == round(expected_completion_cost, 12) @@ -333,7 +326,7 @@ def test_string_cost_values_edge_cases(): def test_string_cost_values_with_threshold(): """Test that string cost values work correctly with threshold pricing.""" from unittest.mock import patch - + # Mock model info with string cost values including threshold pricing mock_model_info = { "input_cost_per_token": "1e-6", # String base cost @@ -341,24 +334,25 @@ def test_string_cost_values_with_threshold(): "input_cost_per_token_above_200k_tokens": "5e-7", # String threshold cost (lower) "output_cost_per_token_above_200k_tokens": "1e-6", # String threshold cost (lower) } - + # Test usage above threshold usage = Usage( prompt_tokens=250000, # Above 200k threshold completion_tokens=1000, total_tokens=251000, ) - - with patch('litellm.litellm_core_utils.llm_cost_calc.utils.get_model_info', return_value=mock_model_info): + + with patch( + "litellm.litellm_core_utils.llm_cost_calc.utils.get_model_info", + return_value=mock_model_info, + ): prompt_cost, completion_cost = generic_cost_per_token( - model="test-model", - usage=usage, - custom_llm_provider="test-provider" + model="test-model", usage=usage, custom_llm_provider="test-provider" ) - + # Expected costs using threshold pricing (string values converted to float) expected_prompt_cost = 250000 * 5e-7 # threshold cost expected_completion_cost = 1000 * 1e-6 # threshold cost - + assert round(prompt_cost, 12) == round(expected_prompt_cost, 12) assert round(completion_cost, 12) == round(expected_completion_cost, 12)