Update test_llm_cost_calc_utils.py

This commit is contained in:
Praveen11558 2026-03-23 00:25:10 +05:30 • committed by GitHub
parent 2b9a66ea15
commit 6004f53d1b
No known key found for this signature in database
GPG key ID: B5690EEEBB952194

View file

@ -1452,92 +1452,7 @@ def test_image_count_billing_does_not_fill_prompt_token_gap():
f"Gap should not be filled when image_count billing is active."
)
assert completion_cost == pytest.approx(expected_completion_cost)
def test_character_count_billing_does_not_fill_prompt_token_gap():
"""
Regression: when character_count pricing is active, gaps between
accounted token details and prompt_tokens should NOT be converted to
text_tokens, otherwise character-based providers may be over-billed.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
prompt_tokens=200,
completion_tokens=20,
total_tokens=220,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=100,
character_count=1000,
image_tokens=0,
audio_tokens=0,
cached_tokens=0,
),
)
prompt_cost, completion_cost = generic_cost_per_token(
model="gemini-1.0-pro",
usage=usage,
custom_llm_provider="vertex_ai",
)
model_info = litellm.model_cost["gemini-1.0-pro"]
expected_prompt_cost = (
100 * model_info["input_cost_per_token"]
+ 1000 * model_info["input_cost_per_character"]
)
expected_completion_cost = 20 * model_info["output_cost_per_token"]
assert prompt_cost == pytest.approx(expected_prompt_cost), (
f"Expected prompt_cost={expected_prompt_cost}, got {prompt_cost}. "
f"character_count-based requests should not fill token gaps as text."
)
assert completion_cost == pytest.approx(expected_completion_cost)
def test_video_length_billing_does_not_fill_prompt_token_gap():
"""
Regression: when video_length_seconds pricing is active, gaps between
accounted token details and prompt_tokens should NOT be converted to
text_tokens, otherwise video-based providers may be over-billed.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
prompt_tokens=150,
completion_tokens=10,
total_tokens=160,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=50,
video_length_seconds=12.0,
image_tokens=0,
audio_tokens=0,
cached_tokens=0,
),
)
prompt_cost, completion_cost = generic_cost_per_token(
model="gemini-1.0-pro",
usage=usage,
custom_llm_provider="vertex_ai",
)
model_info = litellm.model_cost["gemini-1.0-pro"]
expected_prompt_cost = (
50 * model_info["input_cost_per_token"]
+ 12.0 * model_info["input_cost_per_video_per_second"]
)
expected_completion_cost = 10 * model_info["output_cost_per_token"]
assert prompt_cost == pytest.approx(expected_prompt_cost), (
f"Expected prompt_cost={expected_prompt_cost}, got {prompt_cost}. "
f"video_length_seconds-based requests should not fill token gaps as text."
)
assert completion_cost == pytest.approx(expected_completion_cost)
def test_negative_text_tokens_clamped_to_zero():
"""
Scenario: Malformed provider response where cached_tokens > prompt_tokens.