From 5247068418819bae6968e5c214fedecc492632f0 Mon Sep 17 00:00:00 2001 From: Praveen11558 <44603409+Praveen11558@users.noreply.github.com> Date: Mon, 23 Mar 2026 00:01:06 +0530 Subject: [PATCH] Update test_llm_cost_calc_utils.py --- .../llm_cost_calc/test_llm_cost_calc_utils.py | 267 ++++++++++++++++++ 1 file changed, 267 insertions(+) diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index 15967f3fc96..57443657b71 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -1307,3 +1307,270 @@ def test_unaccounted_pdf_tokens_fill_text_tokens(): f"Expected cost={expected_cost}, got cost={cost}. " f"Fully accounted tokens should not be adjusted." ) +def test_double_counting_still_handled(): + """ + Scenario: xAI-style double counting where text_tokens includes cached_tokens. + Provider reports: + - prompt_tokens = 500 + - text_tokens = 500 (includes cached) + - cached_tokens = 200 + Expected: text_tokens recalculated to 300 (500 - 200). + """ + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + usage = Usage( + prompt_tokens=500, + completion_tokens=50, + total_tokens=550, + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=500, + audio_tokens=0, + cached_tokens=200, + image_tokens=0, + ), + ) + response = ModelResponse( + usage=usage, + model="gemini-2.0-flash-001", + ) + + cost = response_cost_calculator( + response_object=response, + model="gemini-2.0-flash-001", + custom_llm_provider="vertex_ai", + call_type="acompletion", + optional_params={}, + ) + + model_info = litellm.model_cost["gemini-2.0-flash-001"] + cache_read_cost = model_info.get("cache_read_input_token_cost", 0) or 0 + + # text_tokens should be recalculated to 300 (500 - 200 cache_hit) + expected_cost = ( + 300 * model_info["input_cost_per_token"] # non-cached text + + 200 * cache_read_cost # cached tokens + + 50 * model_info["output_cost_per_token"] + ) + + assert cost == pytest.approx(expected_cost), ( + f"Expected cost={expected_cost}, got cost={cost}. " + f"Double-counting fix should still work." + ) + + +def test_large_pdf_small_text_message(): + """ + Scenario: A large PDF (~50 pages) with a tiny instruction. + This is the most common real-world case. + """ + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + usage = Usage( + prompt_tokens=52000, + completion_tokens=500, + total_tokens=52500, + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=8, # "Summarize this document" + audio_tokens=0, + cached_tokens=0, + image_tokens=0, + ), + ) + response = ModelResponse( + usage=usage, + model="gemini-2.0-flash-001", + ) + + cost = response_cost_calculator( + response_object=response, + model="gemini-2.0-flash-001", + custom_llm_provider="vertex_ai", + call_type="acompletion", + optional_params={}, + ) + + model_info = litellm.model_cost["gemini-2.0-flash-001"] + + # All 52000 prompt tokens must be costed + expected_cost = ( + 52000 * model_info["input_cost_per_token"] + + 500 * model_info["output_cost_per_token"] + ) + + assert cost == pytest.approx(expected_cost), ( + f"Expected cost={expected_cost}, got cost={cost}. " + f"Large PDF content tokens (51992 unaccounted) are not being costed." + ) + + +def test_image_count_billing_does_not_fill_prompt_token_gap(): + """ + Scenario: User sends an image URL alongside some text. + Provider reports: + - prompt_tokens = 10000 (text + image overhead tokens) + - text_tokens = 50 (just the text instruction) + - image_count = 1 (the image URL, billed via input_cost_per_image) + Expected: The gap (9950 tokens) must NOT be added to text_tokens. + The provider already reflects image-URL tokens inside prompt_tokens + and charges them separately via input_cost_per_image. Gap-filling + would double-bill those tokens (once at text rate, once per-image). + """ + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + usage = Usage( + prompt_tokens=10000, + completion_tokens=200, + total_tokens=10200, + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=50, + image_count=1, + ), + ) + + prompt_cost, completion_cost = generic_cost_per_token( + model="gemini-2.0-flash-001", + usage=usage, + custom_llm_provider="vertex_ai", + ) + + model_info = litellm.model_cost["gemini-2.0-flash-001"] + input_cost_per_token = model_info["input_cost_per_token"] + output_cost_per_token = model_info["output_cost_per_token"] + input_cost_per_image = model_info.get("input_cost_per_image", 0) or 0 + + # text_tokens should stay at 50 — no gap-fill when image_count is active + expected_prompt_cost = ( + 50 * input_cost_per_token # only reported text tokens + + 1 * input_cost_per_image # image billed via flat per-image cost + ) + expected_completion_cost = 200 * output_cost_per_token + + assert prompt_cost == pytest.approx(expected_prompt_cost), ( + f"Expected prompt_cost={expected_prompt_cost}, got {prompt_cost}. " + f"Gap should not be filled when image_count billing is active." + ) + assert completion_cost == pytest.approx(expected_completion_cost) + + +def test_character_count_billing_does_not_fill_prompt_token_gap(): + """ + Regression: when character_count pricing is active, gaps between + accounted token details and prompt_tokens should NOT be converted to + text_tokens, otherwise character-based providers may be over-billed. + """ + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + usage = Usage( + prompt_tokens=200, + completion_tokens=20, + total_tokens=220, + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=100, + character_count=1000, + image_tokens=0, + audio_tokens=0, + cached_tokens=0, + ), + ) + + prompt_cost, completion_cost = generic_cost_per_token( + model="gemini-1.0-pro", + usage=usage, + custom_llm_provider="vertex_ai", + ) + + model_info = litellm.model_cost["gemini-1.0-pro"] + expected_prompt_cost = ( + 100 * model_info["input_cost_per_token"] + + 1000 * model_info["input_cost_per_character"] + ) + expected_completion_cost = 20 * model_info["output_cost_per_token"] + + assert prompt_cost == pytest.approx(expected_prompt_cost), ( + f"Expected prompt_cost={expected_prompt_cost}, got {prompt_cost}. " + f"character_count-based requests should not fill token gaps as text." + ) + assert completion_cost == pytest.approx(expected_completion_cost) + + +def test_video_length_billing_does_not_fill_prompt_token_gap(): + """ + Regression: when video_length_seconds pricing is active, gaps between + accounted token details and prompt_tokens should NOT be converted to + text_tokens, otherwise video-based providers may be over-billed. + """ + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + usage = Usage( + prompt_tokens=150, + completion_tokens=10, + total_tokens=160, + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=50, + video_length_seconds=12.0, + image_tokens=0, + audio_tokens=0, + cached_tokens=0, + ), + ) + + prompt_cost, completion_cost = generic_cost_per_token( + model="gemini-1.0-pro", + usage=usage, + custom_llm_provider="vertex_ai", + ) + + model_info = litellm.model_cost["gemini-1.0-pro"] + expected_prompt_cost = ( + 50 * model_info["input_cost_per_token"] + + 12.0 * model_info["input_cost_per_video_per_second"] + ) + expected_completion_cost = 10 * model_info["output_cost_per_token"] + + assert prompt_cost == pytest.approx(expected_prompt_cost), ( + f"Expected prompt_cost={expected_prompt_cost}, got {prompt_cost}. " + f"video_length_seconds-based requests should not fill token gaps as text." + ) + assert completion_cost == pytest.approx(expected_completion_cost) + + +def test_negative_text_tokens_clamped_to_zero(): + """ + Scenario: Malformed provider response where cached_tokens > prompt_tokens. + Provider reports: + - prompt_tokens = 100 + - text_tokens = 100 (includes cached → triggers double-counting) + - cached_tokens = 200 + Expected: text_tokens should be clamped to 0, not go negative. + Prompt cost must remain non-negative. + """ + os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" + litellm.model_cost = litellm.get_model_cost_map(url="") + + usage = Usage( + prompt_tokens=100, + completion_tokens=10, + total_tokens=110, + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=100, + cached_tokens=200, + ), + ) + + prompt_cost, completion_cost = generic_cost_per_token( + model="gemini-2.0-flash-001", + usage=usage, + custom_llm_provider="vertex_ai", + ) + + # text_tokens = max(0, 100 - 200) = 0, so only cache_read cost applies + assert prompt_cost >= 0, ( + f"Prompt cost must be non-negative, got {prompt_cost}. " + f"Negative text_tokens from double-counting fix is not clamped." + ) + assert completion_cost >= 0