diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index 166d076741a..f3a91abc830 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -172,44 +172,6 @@ def test_inline_pdf_tokens_costed_when_text_tokens_less_than_prompt_tokens(): f"got cost={cost}. PDF tokens are likely not being costed." ) -def test_inline_pdf_tokens_costed_when_text_both_costed_correctly(): - usage = Usage( - completion_tokens=80, - prompt_tokens=6000, - total_tokens=6080, - prompt_tokens_details=PromptTokensDetailsWrapper( - audio_tokens=500, cached_tokens=None, text_tokens=50, image_tokens=None - ), - ) - response = ModelResponse( - usage= usage, - model = "gemini-2.0-flash-001", - ) - cost = response_cost_calculator( - response_object= response, - model = "gemini-2.0-flash-001", - custom_llm_provider = "vertex_ai", - call_type = "acompletion", - optional_params= {}, - ) - - model_info = litellm.model_cost["gemini-2.0-flash-001"] - input_cost_per_token = model_info["input_cost_per_token"] - input_cost_per_audio_token = model_info["input_cost_per_audio_token"] - output_cost_per_token = model_info["output_cost_per_token"] - - expected_input_cost = ( - 5500 * input_cost_per_token + 500 * input_cost_per_audio_token - ) - expected_output_cost = 80 * output_cost_per_token - expected_cost = expected_input_cost + expected_output_cost - - assert cost == pytest.approx(expected_cost), ( - f"Expected cost={expected_cost}, got cost={cost}. " - f"Unaccounted PDF tokens alongside audio are not being costed." - ) - - def test_inline_pdf_with_audio_tokens_both_costed_correctly(): """ Scenario: User sends a PDF inline + audio + text.