Update test_llm_cost_calc_utils.py

This commit is contained in:
Praveen11558 2026-03-23 00:01:06 +05:30 • committed by GitHub
parent 9103063789
commit 5247068418
No known key found for this signature in database
GPG key ID: B5690EEEBB952194

View file

@ -1307,3 +1307,270 @@ def test_unaccounted_pdf_tokens_fill_text_tokens():
f"Expected cost={expected_cost}, got cost={cost}. "
f"Fully accounted tokens should not be adjusted."
)
def test_double_counting_still_handled():
"""
Scenario: xAI-style double counting where text_tokens includes cached_tokens.
Provider reports:
- prompt_tokens = 500
- text_tokens = 500 (includes cached)
- cached_tokens = 200
Expected: text_tokens recalculated to 300 (500 - 200).
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
prompt_tokens=500,
completion_tokens=50,
total_tokens=550,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=500,
audio_tokens=0,
cached_tokens=200,
image_tokens=0,
),
)
response = ModelResponse(
usage=usage,
model="gemini-2.0-flash-001",
)
cost = response_cost_calculator(
response_object=response,
model="gemini-2.0-flash-001",
custom_llm_provider="vertex_ai",
call_type="acompletion",
optional_params={},
)
model_info = litellm.model_cost["gemini-2.0-flash-001"]
cache_read_cost = model_info.get("cache_read_input_token_cost", 0) or 0
# text_tokens should be recalculated to 300 (500 - 200 cache_hit)
expected_cost = (
300 * model_info["input_cost_per_token"] # non-cached text
+ 200 * cache_read_cost # cached tokens
+ 50 * model_info["output_cost_per_token"]
)
assert cost == pytest.approx(expected_cost), (
f"Expected cost={expected_cost}, got cost={cost}. "
f"Double-counting fix should still work."
)
def test_large_pdf_small_text_message():
"""
Scenario: A large PDF (~50 pages) with a tiny instruction.
This is the most common real-world case.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
prompt_tokens=52000,
completion_tokens=500,
total_tokens=52500,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=8, # "Summarize this document"
audio_tokens=0,
cached_tokens=0,
image_tokens=0,
),
)
response = ModelResponse(
usage=usage,
model="gemini-2.0-flash-001",
)
cost = response_cost_calculator(
response_object=response,
model="gemini-2.0-flash-001",
custom_llm_provider="vertex_ai",
call_type="acompletion",
optional_params={},
)
model_info = litellm.model_cost["gemini-2.0-flash-001"]
# All 52000 prompt tokens must be costed
expected_cost = (
52000 * model_info["input_cost_per_token"]
+ 500 * model_info["output_cost_per_token"]
)
assert cost == pytest.approx(expected_cost), (
f"Expected cost={expected_cost}, got cost={cost}. "
f"Large PDF content tokens (51992 unaccounted) are not being costed."
)
def test_image_count_billing_does_not_fill_prompt_token_gap():
"""
Scenario: User sends an image URL alongside some text.
Provider reports:
- prompt_tokens = 10000 (text + image overhead tokens)
- text_tokens = 50 (just the text instruction)
- image_count = 1 (the image URL, billed via input_cost_per_image)
Expected: The gap (9950 tokens) must NOT be added to text_tokens.
The provider already reflects image-URL tokens inside prompt_tokens
and charges them separately via input_cost_per_image. Gap-filling
would double-bill those tokens (once at text rate, once per-image).
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
prompt_tokens=10000,
completion_tokens=200,
total_tokens=10200,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=50,
image_count=1,
),
)
prompt_cost, completion_cost = generic_cost_per_token(
model="gemini-2.0-flash-001",
usage=usage,
custom_llm_provider="vertex_ai",
)
model_info = litellm.model_cost["gemini-2.0-flash-001"]
input_cost_per_token = model_info["input_cost_per_token"]
output_cost_per_token = model_info["output_cost_per_token"]
input_cost_per_image = model_info.get("input_cost_per_image", 0) or 0
# text_tokens should stay at 50 — no gap-fill when image_count is active
expected_prompt_cost = (
50 * input_cost_per_token # only reported text tokens
+ 1 * input_cost_per_image # image billed via flat per-image cost
)
expected_completion_cost = 200 * output_cost_per_token
assert prompt_cost == pytest.approx(expected_prompt_cost), (
f"Expected prompt_cost={expected_prompt_cost}, got {prompt_cost}. "
f"Gap should not be filled when image_count billing is active."
)
assert completion_cost == pytest.approx(expected_completion_cost)
def test_character_count_billing_does_not_fill_prompt_token_gap():
"""
Regression: when character_count pricing is active, gaps between
accounted token details and prompt_tokens should NOT be converted to
text_tokens, otherwise character-based providers may be over-billed.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
prompt_tokens=200,
completion_tokens=20,
total_tokens=220,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=100,
character_count=1000,
image_tokens=0,
audio_tokens=0,
cached_tokens=0,
),
)
prompt_cost, completion_cost = generic_cost_per_token(
model="gemini-1.0-pro",
usage=usage,
custom_llm_provider="vertex_ai",
)
model_info = litellm.model_cost["gemini-1.0-pro"]
expected_prompt_cost = (
100 * model_info["input_cost_per_token"]
+ 1000 * model_info["input_cost_per_character"]
)
expected_completion_cost = 20 * model_info["output_cost_per_token"]
assert prompt_cost == pytest.approx(expected_prompt_cost), (
f"Expected prompt_cost={expected_prompt_cost}, got {prompt_cost}. "
f"character_count-based requests should not fill token gaps as text."
)
assert completion_cost == pytest.approx(expected_completion_cost)
def test_video_length_billing_does_not_fill_prompt_token_gap():
"""
Regression: when video_length_seconds pricing is active, gaps between
accounted token details and prompt_tokens should NOT be converted to
text_tokens, otherwise video-based providers may be over-billed.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
prompt_tokens=150,
completion_tokens=10,
total_tokens=160,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=50,
video_length_seconds=12.0,
image_tokens=0,
audio_tokens=0,
cached_tokens=0,
),
)
prompt_cost, completion_cost = generic_cost_per_token(
model="gemini-1.0-pro",
usage=usage,
custom_llm_provider="vertex_ai",
)
model_info = litellm.model_cost["gemini-1.0-pro"]
expected_prompt_cost = (
50 * model_info["input_cost_per_token"]
+ 12.0 * model_info["input_cost_per_video_per_second"]
)
expected_completion_cost = 10 * model_info["output_cost_per_token"]
assert prompt_cost == pytest.approx(expected_prompt_cost), (
f"Expected prompt_cost={expected_prompt_cost}, got {prompt_cost}. "
f"video_length_seconds-based requests should not fill token gaps as text."
)
assert completion_cost == pytest.approx(expected_completion_cost)
def test_negative_text_tokens_clamped_to_zero():
"""
Scenario: Malformed provider response where cached_tokens > prompt_tokens.
Provider reports:
- prompt_tokens = 100
- text_tokens = 100 (includes cached → triggers double-counting)
- cached_tokens = 200
Expected: text_tokens should be clamped to 0, not go negative.
Prompt cost must remain non-negative.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
prompt_tokens=100,
completion_tokens=10,
total_tokens=110,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=100,
cached_tokens=200,
),
)
prompt_cost, completion_cost = generic_cost_per_token(
model="gemini-2.0-flash-001",
usage=usage,
custom_llm_provider="vertex_ai",
)
# text_tokens = max(0, 100 - 200) = 0, so only cache_read cost applies
assert prompt_cost >= 0, (
f"Prompt cost must be non-negative, got {prompt_cost}. "
f"Negative text_tokens from double-counting fix is not clamped."
)
assert completion_cost >= 0