diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index 2da87ddecb9..3101e501a3d 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -1397,6 +1397,10 @@ def test_image_count_billing_does_not_fill_prompt_token_gap(): os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" litellm.model_cost = litellm.get_model_cost_map(url="") + # multimodalembedding@001 has non-zero input_cost_per_image (0.0001) and + # input_cost_per_token (8e-07), making the image billing assertion non-trivial. + model = "multimodalembedding@001" + usage = Usage( prompt_tokens=10000, completion_tokens=200, @@ -1408,12 +1412,12 @@ def test_image_count_billing_does_not_fill_prompt_token_gap(): ) prompt_cost, completion_cost = generic_cost_per_token( - model="gemini-2.0-flash-001", + model=model, usage=usage, custom_llm_provider="vertex_ai", ) - model_info = litellm.model_cost["gemini-2.0-flash-001"] + model_info = litellm.model_cost[model] input_cost_per_token = model_info["input_cost_per_token"] output_cost_per_token = model_info["output_cost_per_token"] input_cost_per_image = model_info.get("input_cost_per_image", 0) or 0 @@ -1431,7 +1435,6 @@ def test_image_count_billing_does_not_fill_prompt_token_gap(): ) assert completion_cost == pytest.approx(expected_completion_cost) - def test_character_count_billing_does_not_fill_prompt_token_gap(): """ Regression: when character_count pricing is active, gaps between