mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
Update test_llm_cost_calc_utils.py
This commit is contained in:
parent
f6fc7ee75a
commit
d3de061d6f
1 changed files with 17 additions and 8 deletions
|
|
@ -1441,6 +1441,11 @@ def test_character_count_billing_does_not_fill_prompt_token_gap():
|
||||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||||
|
|
||||||
|
# multimodalembedding@001 has non-zero input_cost_per_character (2e-07),
|
||||||
|
# input_cost_per_token (8e-07), and output_cost_per_token (0) in the cost map,
|
||||||
|
# making the character_count billing assertion non-trivial.
|
||||||
|
model = "multimodalembedding@001"
|
||||||
|
|
||||||
usage = Usage(
|
usage = Usage(
|
||||||
prompt_tokens=200,
|
prompt_tokens=200,
|
||||||
completion_tokens=20,
|
completion_tokens=20,
|
||||||
|
|
@ -1455,17 +1460,17 @@ def test_character_count_billing_does_not_fill_prompt_token_gap():
|
||||||
)
|
)
|
||||||
|
|
||||||
prompt_cost, completion_cost = generic_cost_per_token(
|
prompt_cost, completion_cost = generic_cost_per_token(
|
||||||
model="gemini-2.0-flash-001",
|
model=model,
|
||||||
usage=usage,
|
usage=usage,
|
||||||
custom_llm_provider="vertex_ai",
|
custom_llm_provider="vertex_ai",
|
||||||
)
|
)
|
||||||
|
|
||||||
model_info = litellm.model_cost["gemini-2.0-flash-001"]
|
model_info = litellm.model_cost[model]
|
||||||
expected_prompt_cost = (
|
expected_prompt_cost = (
|
||||||
100 * model_info["input_cost_per_token"]
|
100 * model_info.get("input_cost_per_token", 0)
|
||||||
+ 1000 * model_info.get("input_cost_per_character", 0)
|
+ 1000 * model_info.get("input_cost_per_character", 0)
|
||||||
)
|
)
|
||||||
expected_completion_cost = 20 * model_info["output_cost_per_token"]
|
expected_completion_cost = 20 * model_info.get("output_cost_per_token", 0)
|
||||||
|
|
||||||
assert prompt_cost == pytest.approx(expected_prompt_cost), (
|
assert prompt_cost == pytest.approx(expected_prompt_cost), (
|
||||||
f"Expected prompt_cost={expected_prompt_cost}, got {prompt_cost}. "
|
f"Expected prompt_cost={expected_prompt_cost}, got {prompt_cost}. "
|
||||||
|
|
@ -1483,6 +1488,10 @@ def test_video_length_billing_does_not_fill_prompt_token_gap():
|
||||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||||
|
|
||||||
|
# multimodalembedding@001 has non-zero input_cost_per_video_per_second (0.0005)
|
||||||
|
# and input_cost_per_token (8e-07), making the video billing assertion non-trivial.
|
||||||
|
model = "multimodalembedding@001"
|
||||||
|
|
||||||
usage = Usage(
|
usage = Usage(
|
||||||
prompt_tokens=150,
|
prompt_tokens=150,
|
||||||
completion_tokens=10,
|
completion_tokens=10,
|
||||||
|
|
@ -1497,12 +1506,12 @@ def test_video_length_billing_does_not_fill_prompt_token_gap():
|
||||||
)
|
)
|
||||||
|
|
||||||
prompt_cost, completion_cost = generic_cost_per_token(
|
prompt_cost, completion_cost = generic_cost_per_token(
|
||||||
model="gemini-2.0-flash-001",
|
model=model,
|
||||||
usage=usage,
|
usage=usage,
|
||||||
custom_llm_provider="vertex_ai",
|
custom_llm_provider="vertex_ai",
|
||||||
)
|
)
|
||||||
|
|
||||||
model_info = litellm.model_cost["gemini-2.0-flash-001"]
|
model_info = litellm.model_cost[model]
|
||||||
expected_prompt_cost = (
|
expected_prompt_cost = (
|
||||||
50 * model_info.get("input_cost_per_token", 0)
|
50 * model_info.get("input_cost_per_token", 0)
|
||||||
+ 12.0 * model_info.get("input_cost_per_video_per_second", 0)
|
+ 12.0 * model_info.get("input_cost_per_video_per_second", 0)
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue