test(cost-calc): stop 182 global writes leaking out of the cost-calc suites (#37815)

* test(cost-calc): stop 182 global writes leaking out of the cost-calc suites

Across test_cost_calculator.py and llm_cost_calc/test_llm_cost_calc_utils.py,
58 tests opened by setting LITELLM_LOCAL_MODEL_COST_MAP in os.environ and
replacing litellm.model_cost, and none of them put the env var back. The
second file already had a _local_model_cost_map fixture doing it by hand with
a try/finally, so both idioms sat in the same file.

Keep that fixture, give it monkeypatch, and have every one of those tests ask
for it. The margin and discount tests drop their hand-rolled
copy-then-restore in favour of monkeypatch.setattr, which also puts the
global back when an assertion fails part way through.

Both files also drop a sys.path.insert whose argument resolves outside the
repo, so it was never what made the imports work.

TQ003 1077 -> 1075, TQ004 768 -> 693, TQ005 2836 -> 2731, and the budget
ceilings come down with them.

* fix(test): make the streamed-cost tests load the map they assert against

The local_cost_map fixture set LITELLM_LOCAL_MODEL_COST_MAP but never reloaded
litellm.model_cost, and reading the variable is not what loads the map. So the
three streaming-cost tests billed against whatever map the process happened to
be holding, and their hardcoded prices only held when something else had
already swapped in the checked-in one. This branch stops the cost-calc tests
leaking that map, which left test_main billing at the ambient prices instead.

The fixture now loads the map it names, so the prices these tests assert hold
on their own.
This commit is contained in:
yuneng-jiang 2026-08-21 21:00:04 -07:00 • committed by GitHub
parent 7481649830
commit 0c97eea660
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
4 changed files with 121 additions and 315 deletions

View file

@ -6,13 +6,13 @@
"limit": 742
},
"TQ003": {
"limit": 1078
"limit": 1075
},
"TQ004": {
"limit": 544
"limit": 469
},
"TQ005": {
"limit": 2810
"limit": 2661
},
"TQ006": {
"limit": 34

View file

@ -1,6 +1,4 @@
import json
import os
import sys
import pytest
from fastapi.testclient import TestClient
@ -28,10 +26,6 @@ from litellm.types.utils import (
StandardBuiltInToolsParams,
)
sys.path.insert(
0, os.path.abspath("../../..")
) # Adds the parent directory to the system path
from litellm.litellm_core_utils.llm_cost_calc.utils import (
PromptTokensDetailsResult,
TokenTypeCostBreakdown,
@ -44,13 +38,17 @@ from litellm.litellm_core_utils.llm_cost_calc.utils import (
from litellm.types.utils import CacheCreationTokenDetails, Usage
def test_reasoning_tokens_no_price_set():
@pytest.fixture
def _local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
def test_reasoning_tokens_no_price_set(_local_model_cost_map):
# Use o1 - o1-mini was deprecated/renamed; o1 has same reasoning-token semantics
# (no separate output_cost_per_reasoning_token, so all completion tokens use output_cost_per_token)
model = "o1"
custom_llm_provider = "openai"
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model_cost_map = litellm.model_cost[model]
usage = Usage(
completion_tokens=1578,
@ -87,11 +85,9 @@ def test_reasoning_tokens_no_price_set():
)
def test_reasoning_tokens_gemini():
def test_reasoning_tokens_gemini(_local_model_cost_map):
model = "gemini-2.5-flash"
custom_llm_provider = "gemini"
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
completion_tokens=1578,
@ -132,12 +128,10 @@ def test_reasoning_tokens_gemini():
)
def test_reasoning_tokens_gemini_3_1_flash_lite():
def test_reasoning_tokens_gemini_3_1_flash_lite(_local_model_cost_map):
"""Test cost calculation for gemini-3.1-flash-lite-preview with reasoning tokens"""
model = "gemini-3.1-flash-lite-preview"
custom_llm_provider = "gemini"
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
completion_tokens=1000,
@ -270,11 +264,9 @@ def test_image_tokens_fallback_to_base_cost():
assert round(completion_cost, 12) == round(expected_completion_cost, 12)
def test_video_output_tokens_gemini_omni_flash_preview():
def test_video_output_tokens_gemini_omni_flash_preview(_local_model_cost_map):
"""Video output tokens are billed at output_cost_per_video_token, not the text rate and not zero."""
model = "gemini-omni-flash-preview"
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
text_tokens = 100
video_tokens = 46336
@ -310,11 +302,9 @@ def test_video_output_tokens_gemini_omni_flash_preview():
)
def test_video_input_tokens_gemini_omni_flash_preview():
def test_video_input_tokens_gemini_omni_flash_preview(_local_model_cost_map):
"""Video input tokens are billed at the standard input rate instead of being dropped."""
model = "gemini-omni-flash-preview"
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
completion_tokens=10,
@ -369,12 +359,10 @@ def test_video_tokens_fallback_to_base_cost():
assert round(completion_cost, 12) == round((600 + 1120) * 2e-6, 12)
def test_generic_cost_per_token_above_200k_tokens():
def test_generic_cost_per_token_above_200k_tokens(_local_model_cost_map):
# gemini-2.5-pro-exp-03-25 was removed; gemini-2.5-pro has same above-200k pricing
model = "gemini-2.5-pro"
custom_llm_provider = "vertex_ai"
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model_cost_map = litellm.model_cost[model]
prompt_tokens = 220 * 1e6
@ -420,12 +408,10 @@ def test_get_token_base_cost_picks_highest_crossed_tier():
assert prompt_base_cost == 9e-6
def test_generic_cost_per_token_gpt54_above_272k_tokens():
def test_generic_cost_per_token_gpt54_above_272k_tokens(_local_model_cost_map):
"""GPT-5.4/5.4-pro: prompts >272K input tokens priced at 2x input, 1.5x output."""
model = "gpt-5.4"
custom_llm_provider = "openai"
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model_cost_map = litellm.model_cost[model]
prompt_tokens = 273000 # Above 272K threshold
@ -450,12 +436,10 @@ def test_generic_cost_per_token_gpt54_above_272k_tokens():
assert round(completion_cost, 10) == round(expected_completion, 10)
def test_generic_cost_per_token_minimax_m3_above_512k_tokens():
def test_generic_cost_per_token_minimax_m3_above_512k_tokens(_local_model_cost_map):
"""MiniMax-M3: prompts >512K input tokens priced at 2x input, output, and cache read."""
model = "minimax/MiniMax-M3"
custom_llm_provider = "minimax"
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model_cost_map = litellm.model_cost[model]
prompt_tokens = 600000
@ -493,10 +477,8 @@ def test_generic_cost_per_token_minimax_m3_above_512k_tokens():
"bedrock_mantle/openai.gpt-5.6-luna",
],
)
def test_generic_cost_per_token_bedrock_mantle_gpt56_long_context(model):
def test_generic_cost_per_token_bedrock_mantle_gpt56_long_context(_local_model_cost_map, model):
"""Bedrock GPT-5.6 supports a 1M context window, billed at the long-context rates above 272K."""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model_cost_map = litellm.model_cost[model]
assert model_cost_map["max_input_tokens"] == 1000000
@ -827,12 +809,10 @@ def test_generic_cost_per_token_tiered_pricing_bills_reasoning_at_tier_rate():
litellm.model_cost.pop(model, None)
def test_generic_cost_per_token_gpt55():
def test_generic_cost_per_token_gpt55(_local_model_cost_map):
"""gpt-5.5: base pricing — $5/1M input, $30/1M output, $0.50/1M cached input."""
model = "gpt-5.5"
custom_llm_provider = "openai"
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model_cost_map = litellm.model_cost[model]
@ -867,12 +847,10 @@ def test_generic_cost_per_token_gpt55():
)
def test_generic_cost_per_token_gpt55_pro():
def test_generic_cost_per_token_gpt55_pro(_local_model_cost_map):
"""gpt-5.5-pro: responses-only model — $30/1M input, $180/1M output, $3/1M cached input."""
model = "gpt-5.5-pro"
custom_llm_provider = "openai"
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model_cost_map = litellm.model_cost[model]
@ -919,7 +897,7 @@ def test_generic_cost_per_token_gpt55_pro():
("gpt-5.6-luna", 2e-7, 1.2e-6, 2e-8, 2.5e-7),
],
)
def test_generic_cost_per_token_gpt56(
def test_generic_cost_per_token_gpt56(_local_model_cost_map,
model, input_cost, output_cost, cache_read_cost, cache_write_cost
):
"""gpt-5.6 (sol/terra/luna): base pricing + new cache-write cost.
@ -927,8 +905,6 @@ def test_generic_cost_per_token_gpt56(
Cache writes are billed at 1.25x the uncached input rate for this family.
"""
custom_llm_provider = "openai"
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model_cost_map = litellm.model_cost[model]
@ -989,7 +965,7 @@ def test_gpt_5_6_alias_prices_match_sol(local_model_cost_map):
("gpt-5.6-luna", 2e-7, 9e-7),
],
)
def test_generic_cost_per_token_gpt56_flex_above_272k(
def test_generic_cost_per_token_gpt56_flex_above_272k(_local_model_cost_map,
model, flex_long_input_cost, flex_long_output_cost
):
"""A >272K flex request bills the flex long-context rate, not the standard one.
@ -998,8 +974,6 @@ def test_generic_cost_per_token_gpt56_flex_above_272k(
``*_above_272k_tokens_flex`` keys these requests silently fell back to the
standard long-context price, billing 2x what OpenAI charges.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
prompt_tokens = 300000
completion_tokens = 1000
@ -1038,11 +1012,9 @@ def test_generic_cost_per_token_gpt56_flex_above_272k(
("flex", 300000, 2e-6, 2.5e-6, 2e-7),
],
)
def test_generic_cost_per_token_gpt56_terra_cache_costs_by_tier_and_context(
def test_generic_cost_per_token_gpt56_terra_cache_costs_by_tier_and_context(_local_model_cost_map,
service_tier, prompt_tokens, input_rate, cache_write_rate, cache_read_rate
):
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
cached_tokens = 50000
cache_write_tokens = 40000
@ -1130,7 +1102,7 @@ def test_generic_cost_per_token_gpt56_cyber(
("azure/eu/gpt-5.6-luna", 2.2e-7, 1.32e-6, 2.2e-8),
],
)
def test_generic_cost_per_token_azure_gpt56(
def test_generic_cost_per_token_azure_gpt56(_local_model_cost_map,
model, input_cost, output_cost, cache_read_cost
):
"""Azure gpt-5.6 (global + us/eu regional): Azure prices this family on its own
@ -1138,8 +1110,6 @@ def test_generic_cost_per_token_azure_gpt56(
promotional cut OpenAI applied to gpt-5.6-sol, so these rates deliberately sit
above the openai ones and must not be lowered to match them.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model_cost_map = litellm.model_cost[model]
assert model_cost_map["litellm_provider"] == "azure"
@ -1180,7 +1150,7 @@ def test_generic_cost_per_token_azure_gpt56(
("gpt-5.5-pro-2026-04-23", False, True, False),
],
)
def test_gpt55_reasoning_effort_flags_match_live_openai_api(
def test_gpt55_reasoning_effort_flags_match_live_openai_api(_local_model_cost_map,
model, expected_none, expected_xhigh, expected_minimal
):
"""Pin reasoning_effort capability flags to OpenAI's actual API contract.
@ -1189,8 +1159,6 @@ def test_gpt55_reasoning_effort_flags_match_live_openai_api(
``Unsupported value: 'reasoning_effort' does not support 'minimal' with
this model``. gpt-5.5-pro additionally rejects 'none' and 'low'.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
m = litellm.model_cost[model]
assert (
@ -1211,7 +1179,7 @@ def test_gpt55_reasoning_effort_flags_match_live_openai_api(
("gpt-5.5-pro", "gpt-5.5-pro-2026-04-23"),
],
)
def test_gpt55_dated_variants_match_base_reasoning_effort_capabilities(
def test_gpt55_dated_variants_match_base_reasoning_effort_capabilities(_local_model_cost_map,
base_model, dated_model
):
"""Dated snapshots must carry the same reasoning_effort capability flags as
@ -1223,8 +1191,6 @@ def test_gpt55_dated_variants_match_base_reasoning_effort_capabilities(
behavior between ``gpt-5.5`` and ``gpt-5.5-2026-04-23``. Pinning to a
dated variant must never lose capabilities relative to the base alias.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
base = litellm.model_cost[base_model]
dated = litellm.model_cost[dated_model]
@ -1251,7 +1217,7 @@ def test_gpt55_dated_variants_match_base_reasoning_effort_capabilities(
("azure/gpt-5.5-pro-2026-04-23", "responses", 3e-5, 1.8e-4, 3e-6),
],
)
def test_azure_gpt55_entries_present_with_correct_pricing(
def test_azure_gpt55_entries_present_with_correct_pricing(_local_model_cost_map,
model, expected_mode, expected_input, expected_output, expected_cache_read
):
"""Day-0 Azure entries for GPT-5.5 mirror the OpenAI pricing structure.
@ -1260,8 +1226,6 @@ def test_azure_gpt55_entries_present_with_correct_pricing(
on 2026-04-24): $5/$30 input/output per 1M for chat, $30/$180 for pro.
Cache discount is 10% of input.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
m = litellm.model_cost[model]
assert m["litellm_provider"] == "azure"
@ -1286,12 +1250,10 @@ def test_azure_gpt55_entries_present_with_correct_pricing(
("azure/gpt-5.5-pro", False, False, True),
],
)
def test_azure_gpt55_reasoning_effort_flags_match_live_openai_api(
def test_azure_gpt55_reasoning_effort_flags_match_live_openai_api(_local_model_cost_map,
model, expected_none, expected_minimal, expected_xhigh
):
"""Azure entries pin reasoning_effort flags to OpenAI's actual API contract."""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
m = litellm.model_cost[model]
assert m.get("supports_none_reasoning_effort") is expected_none
@ -1671,11 +1633,9 @@ def test_cache_writing_cost_with_zero_creation_tokens_and_ephemeral_details():
assert round(result, 6) == round(expected, 6)
def test_service_tier_flex_pricing():
def test_service_tier_flex_pricing(_local_model_cost_map):
"""Test that flex service tier uses correct pricing (approximately 50% of standard)."""
# Set up environment for local model cost map
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
# Test with gpt-5-nano which has flex pricing
model = "gpt-5-nano"
@ -1728,11 +1688,9 @@ def test_service_tier_flex_pricing():
), f"Flex total cost mismatch: {flex_total} vs {expected_flex_total}"
def test_service_tier_default_pricing():
def test_service_tier_default_pricing(_local_model_cost_map):
"""Test that when no service tier is provided, standard pricing is used."""
# Set up environment for local model cost map
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
# Test with gpt-5-nano
model = "gpt-5-nano"
@ -1779,11 +1737,9 @@ def test_service_tier_default_pricing():
), f"Standard completion cost mismatch: {default_cost[1]} vs {expected_standard_completion}"
def test_service_tier_fallback_pricing():
def test_service_tier_fallback_pricing(_local_model_cost_map):
"""Test that when service tier is provided but model doesn't have those keys, it falls back to standard pricing."""
# Set up environment for local model cost map
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
# Test with gpt-4 which doesn't have flex pricing keys
model = "gpt-4"
@ -1891,15 +1847,13 @@ def test_service_tier_ultrafast_pricing():
assert completion_cost == pytest.approx(400 * 3e-04)
def test_service_tier_ultrafast_fallback_pricing():
def test_service_tier_ultrafast_fallback_pricing(_local_model_cost_map):
"""Without *_ultrafast keys an ultrafast request bills the standard rate, not zero.
Guards the suffix fallback in _get_cost_per_unit: "_fast" is a substring of
"_ultrafast", so a shortest-first suffix match would strip the wrong suffix
and price the request at 0.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500)
@ -1929,7 +1883,7 @@ def test_service_tier_ultrafast_fallback_pricing():
"gemini-3.1-flash-lite-image",
],
)
def test_gemini_image_generation_cost_with_zero_text_tokens(model: str):
def test_gemini_image_generation_cost_with_zero_text_tokens(_local_model_cost_map, model: str):
"""
Test that image_tokens are correctly costed when text_tokens=0.
@ -1939,8 +1893,6 @@ def test_gemini_image_generation_cost_with_zero_text_tokens(model: str):
https://github.com/BerriAI/litellm/issues/17410
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
custom_llm_provider = "vertex_ai"
@ -1995,13 +1947,11 @@ def test_gemini_image_generation_cost_with_zero_text_tokens(model: str):
), f"Expected completion cost ${expected_completion_cost:.6f}, got ${completion_cost:.6f}"
def test_vertex_image_generation_cost_prefers_token_usage_metadata():
def test_vertex_image_generation_cost_prefers_token_usage_metadata(_local_model_cost_map):
"""
When usage metadata exists on image responses, Vertex image generation cost
should be calculated from token pricing, not flat output_cost_per_image.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "gemini-3.1-flash-image-preview"
model_info = litellm.get_model_info(model=model, custom_llm_provider="vertex_ai")
@ -2040,13 +1990,11 @@ def test_vertex_image_generation_cost_prefers_token_usage_metadata():
assert cost != len(image_response.data) * model_info["output_cost_per_image"]
def test_vertex_image_generation_cost_falls_back_to_flat_image_pricing():
def test_vertex_image_generation_cost_falls_back_to_flat_image_pricing(_local_model_cost_map):
"""
Without usage metadata, Vertex image generation cost should fall back to
output_cost_per_image * number_of_images.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "gemini-3.1-flash-image-preview"
model_info = litellm.get_model_info(model=model, custom_llm_provider="vertex_ai")
@ -2064,13 +2012,11 @@ def test_vertex_image_generation_cost_falls_back_to_flat_image_pricing():
assert round(cost, 10) == round(expected_cost, 10)
def test_gemini_image_generation_cost_prefers_token_usage_metadata():
def test_gemini_image_generation_cost_prefers_token_usage_metadata(_local_model_cost_map):
"""
When usage metadata exists on image responses, Gemini image generation cost
should be calculated from token pricing, not flat output_cost_per_image.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "gemini/gemini-3-pro-image-preview"
model_info = litellm.get_model_info(model=model, custom_llm_provider="gemini")
@ -2109,13 +2055,11 @@ def test_gemini_image_generation_cost_prefers_token_usage_metadata():
assert cost != len(image_response.data) * model_info["output_cost_per_image"]
def test_gemini_image_generation_cost_falls_back_to_flat_image_pricing():
def test_gemini_image_generation_cost_falls_back_to_flat_image_pricing(_local_model_cost_map):
"""
Without usage metadata, Gemini image generation cost should fall back to
output_cost_per_image * number_of_images.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "gemini/gemini-3-pro-image-preview"
model_info = litellm.get_model_info(model=model, custom_llm_provider="gemini")
@ -2212,7 +2156,7 @@ def test_reasoning_tokens_without_text_tokens_gpt5_nano():
), "Bug detected: Cost calculation is using only reasoning_tokens instead of all completion_tokens!"
def test_image_count_prevents_text_tokens_fallback():
def test_image_count_prevents_text_tokens_fallback(_local_model_cost_map):
"""
Test that the text_tokens fallback in generic_cost_per_token does not
override text_tokens=0 when image_count > 0.
@ -2221,8 +2165,6 @@ def test_image_count_prevents_text_tokens_fallback():
When image_count > 0, text_tokens=0 is intentional (image-only request),
not "text_tokens not set by provider."
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
# Simulate Nova image-only embedding: prompt_tokens estimated from
# embedding dimensions (768 for 3072-dim), image_count=1
@ -2256,20 +2198,6 @@ def test_image_count_prevents_text_tokens_fallback():
# ---------------------------------------------------------------------------
@pytest.fixture
def _local_model_cost_map():
prev_env = os.environ.get("LITELLM_LOCAL_MODEL_COST_MAP")
prev_model_cost = litellm.model_cost
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
try:
yield
finally:
litellm.model_cost = prev_model_cost
if prev_env is None:
os.environ.pop("LITELLM_LOCAL_MODEL_COST_MAP", None)
else:
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = prev_env
@pytest.mark.parametrize("model", ["gpt-5.4", "gpt-realtime-2.1", "gpt-realtime-2.1-mini"])
@ -2603,7 +2531,7 @@ def test_threshold_keys_exclude_service_tier_variants():
("cerebras/qwen-3-32b", "cerebras", 250, 0),
],
)
def test_token_type_cost_breakdown_is_provider_agnostic(
def test_token_type_cost_breakdown_is_provider_agnostic(_local_model_cost_map,
model, custom_llm_provider, reasoning_tokens, cached_tokens
):
"""
@ -2615,8 +2543,6 @@ def test_token_type_cost_breakdown_is_provider_agnostic(
there - not the top-level cache_read_input_tokens attribute the old breakdown code
relied on - is what makes Vertex/OpenAI/Azure cache costs show up at all.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
prompt_tokens=1000,
@ -2647,10 +2573,8 @@ def test_token_type_cost_breakdown_is_provider_agnostic(
assert breakdown.cache_read_cost == pytest.approx(cached_tokens * cache_read_rate)
def test_token_type_cost_breakdown_matches_real_gemini_numbers():
def test_token_type_cost_breakdown_matches_real_gemini_numbers(_local_model_cost_map):
"""Hard-coded against the exact gemini-2.5-flash response that exposed the gap."""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
prompt_tokens=209,
@ -2673,9 +2597,7 @@ def test_token_type_cost_breakdown_matches_real_gemini_numbers():
assert breakdown.cache_creation_cost == 0.0
def test_token_type_cost_breakdown_xai_at_exactly_200k_uses_higher_tier_rates():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_token_type_cost_breakdown_xai_at_exactly_200k_uses_higher_tier_rates(_local_model_cost_map):
usage = Usage(
prompt_tokens=200_000,
@ -2697,9 +2619,7 @@ def test_token_type_cost_breakdown_xai_at_exactly_200k_uses_higher_tier_rates():
assert breakdown.cache_read_cost == pytest.approx(50_000 * 4e-07)
def test_token_type_cost_breakdown_xai_just_below_200k_uses_base_tier_rates():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_token_type_cost_breakdown_xai_just_below_200k_uses_base_tier_rates(_local_model_cost_map):
usage = Usage(
prompt_tokens=199_999,
@ -2721,14 +2641,12 @@ def test_token_type_cost_breakdown_xai_just_below_200k_uses_base_tier_rates():
assert breakdown.cache_read_cost == pytest.approx(50_000 * 2e-07)
def test_token_type_cost_breakdown_includes_cache_creation_from_top_level_usage():
def test_token_type_cost_breakdown_includes_cache_creation_from_top_level_usage(_local_model_cost_map):
"""
Bedrock/Anthropic report cache tokens as top-level usage fields; the Usage
constructor maps them onto prompt_tokens_details, so the breakdown must still
pick up both cache-read and cache-creation costs.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "anthropic.claude-3-5-haiku-20241022-v1:0"
usage = Usage(
@ -2752,14 +2670,12 @@ def test_token_type_cost_breakdown_includes_cache_creation_from_top_level_usage(
)
def test_token_type_cost_breakdown_reads_cache_write_tokens():
def test_token_type_cost_breakdown_reads_cache_write_tokens(_local_model_cost_map):
"""
Some OpenAI-compatible providers (e.g. kimi-k2) report cache-write tokens under
`cache_write_tokens` rather than `cache_creation_tokens`. The breakdown must read
it the same way the total-cost normalization does, so the two agree.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "anthropic.claude-3-5-haiku-20241022-v1:0"
usage = Usage(
@ -2780,7 +2696,7 @@ def test_token_type_cost_breakdown_reads_cache_write_tokens():
)
def test_generic_cost_per_token_openai_cache_write_tokens_gpt_5_6():
def test_generic_cost_per_token_openai_cache_write_tokens_gpt_5_6(_local_model_cost_map):
"""
Regression: OpenAI gpt-5.6 reports cache-write tokens under
prompt_tokens_details.cache_write_tokens (not the Anthropic cache_creation_tokens
@ -2788,8 +2704,6 @@ def test_generic_cost_per_token_openai_cache_write_tokens_gpt_5_6():
input rate. Customer report: cache creation tokens were never counted for the
GPT-5.6 series, so cost was undercounted on cache-write requests.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "gpt-5.6"
usage = Usage(
@ -2811,14 +2725,12 @@ def test_generic_cost_per_token_openai_cache_write_tokens_gpt_5_6():
assert prompt_cost > 1000 * info["input_cost_per_token"]
def test_generic_cost_per_token_backs_out_cache_write_tokens_from_text_tokens():
def test_generic_cost_per_token_backs_out_cache_write_tokens_from_text_tokens(_local_model_cost_map):
"""
Regression for #34801: when a provider reports text_tokens covering the whole
prompt alongside cache-write tokens (and no cache reads), the cache-write tokens
must be backed out of the text total instead of being billed twice.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "gpt-5.6"
usage = Usage(
@ -2837,15 +2749,13 @@ def test_generic_cost_per_token_backs_out_cache_write_tokens_from_text_tokens():
assert prompt_cost == pytest.approx(expected_prompt)
def test_token_type_cost_breakdown_reconciles_with_generic_total():
def test_token_type_cost_breakdown_reconciles_with_generic_total(_local_model_cost_map):
"""
Both-ways check: the reasoning subset must sum with the remaining (text) output
cost to exactly the completion total, and the cache-read subset with the remaining
input cost to exactly the prompt total, as computed by generic_cost_per_token.
A mismatch here would mean the breakdown misrepresents what was actually billed.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "gemini-2.5-flash"
custom_llm_provider = "vertex_ai"
@ -2878,9 +2788,7 @@ def test_token_type_cost_breakdown_reconciles_with_generic_total():
assert text_input_cost + breakdown.cache_read_cost == pytest.approx(prompt_cost)
def test_token_type_cost_breakdown_zero_without_special_tokens():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_token_type_cost_breakdown_zero_without_special_tokens(_local_model_cost_map):
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
breakdown = get_token_type_cost_breakdown(
@ -2917,7 +2825,7 @@ def test_token_type_cost_breakdown_zero_without_special_tokens():
),
],
)
def test_token_type_cost_breakdown_openai_responses_api_cache_write_read(
def test_token_type_cost_breakdown_openai_responses_api_cache_write_read(_local_model_cost_map,
raw_usage, expect_read, expect_write
):
"""Regression for #34309: OpenAI Responses API reports cache tokens under
@ -2926,8 +2834,6 @@ def test_token_type_cost_breakdown_openai_responses_api_cache_write_read(
cache_read_cost / cache_creation_cost from the transformed usage."""
from litellm.responses.utils import ResponseAPILoggingUtils
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "gpt-5.6"
usage = ResponseAPILoggingUtils._transform_response_api_usage_to_chat_usage(raw_usage)
@ -2968,15 +2874,13 @@ def test_token_type_cost_breakdown_handles_unknown_model_gracefully():
)
def test_token_type_cost_breakdown_applies_regional_uplift():
def test_token_type_cost_breakdown_applies_regional_uplift(_local_model_cost_map):
"""
Regional OpenAI hosts (eu./us.) apply a flat uplift to every token cost. The
per-type breakdown must apply the same uplift via data_residency so it stays
reconciled with the uplifted input_cost/output_cost totals, instead of being
logged at the base rate.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "gpt-5.4"
custom_llm_provider = "openai"
@ -3024,15 +2928,13 @@ def test_token_type_cost_breakdown_applies_regional_uplift():
assert text_input_cost + eu.cache_read_cost == pytest.approx(prompt_cost)
def test_token_type_cost_breakdown_applies_vertex_regional_uplift():
def test_token_type_cost_breakdown_applies_vertex_regional_uplift(_local_model_cost_map):
"""
Non-global Vertex endpoints apply a flat 1.1x uplift to every token cost. The
per-type breakdown must apply the same uplift via vertex_location so it stays
reconciled with the uplifted input_cost/output_cost totals, instead of being
logged at the global rate.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "claude-haiku-4-5@20251001"
custom_llm_provider = "vertex_ai"
@ -3075,7 +2977,7 @@ def test_token_type_cost_breakdown_applies_vertex_regional_uplift():
assert text_input_cost + regional.cache_read_cost == pytest.approx(prompt_cost)
def test_token_type_cost_breakdown_applies_anthropic_geo_multiplier(monkeypatch):
def test_token_type_cost_breakdown_applies_anthropic_geo_multiplier(_local_model_cost_map, monkeypatch):
"""
Anthropic's regional (geo) uplift lives in provider_specific_entry and is
applied to every token type in the totals, so the per-type breakdown must
@ -3088,7 +2990,6 @@ def test_token_type_cost_breakdown_applies_anthropic_geo_multiplier(monkeypatch)
)
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "claude-test-geo-breakdown-model"
litellm.register_model(
@ -3209,9 +3110,7 @@ GEMINI_DAY0_LAUNCH_PRICING = [
@pytest.mark.parametrize("model,input_cost,output_cost,cache_read_cost", GEMINI_DAY0_LAUNCH_PRICING)
def test_gemini_36_flash_and_35_flash_lite_launch_pricing(model, input_cost, output_cost, cache_read_cost):
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_gemini_36_flash_and_35_flash_lite_launch_pricing(_local_model_cost_map, model, input_cost, output_cost, cache_read_cost):
model_cost_map = litellm.model_cost[model]
assert model_cost_map["input_cost_per_token"] == input_cost
@ -3224,9 +3123,7 @@ def test_gemini_36_flash_and_35_flash_lite_launch_pricing(model, input_cost, out
assert model_cost_map["max_input_tokens"] == 1048576
def test_generic_cost_per_token_gemini_36_flash():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_generic_cost_per_token_gemini_36_flash(_local_model_cost_map):
usage = Usage(
prompt_tokens=1000,
@ -3292,9 +3189,7 @@ def test_gemini_36_flash_batch_introductory_pricing(model, _local_model_cost_map
assert model_cost_map["output_cost_per_token_batches"] == 1.875e-06
def test_generic_cost_per_token_gemini_35_flash_lite():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_generic_cost_per_token_gemini_35_flash_lite(_local_model_cost_map):
usage = Usage(
prompt_tokens=1000,

View file

@ -1,12 +1,6 @@
import os
import sys
import pytest
sys.path.insert(
0, os.path.abspath("../..")
) # Adds the parent directory to the system path
from pydantic import BaseModel
@ -24,6 +18,12 @@ from litellm.types.utils import ModelInfo, ModelResponse, PromptTokensDetailsWra
from litellm.utils import TranscriptionResponse
@pytest.fixture
def _local_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
def test_cost_per_token_duplicate_openai_prefix_matches_model_cost(monkeypatch):
"""
Router/proxy configs may use deployment ids like openai/openai/<model>. Cost lookup must
@ -93,14 +93,12 @@ def test_cost_per_token_non_string_model_does_not_hang():
assert result.get("status") in ("returned", "raised")
def test_completion_cost_uses_response_model_for_dynamic_routing():
def test_completion_cost_uses_response_model_for_dynamic_routing(_local_model_cost_map):
"""
Test that completion_cost uses the model from the response object
when the input model (e.g., azure-model-router) is not in model_cost.
This supports Azure Model Router and similar dynamic routing scenarios.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
# Simulate Azure Model Router: input is generic router, response has actual model
response = ModelResponse(
@ -139,9 +137,7 @@ def test_cost_calculator_with_response_cost_in_additional_headers():
assert result == 1000
def test_baseten_model_api_pricing_entries():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_baseten_model_api_pricing_entries(_local_model_cost_map):
expected_pricing = {
"baseten/nvidia/Nemotron-120B-A12B": (3e-07, 7.5e-07),
@ -165,9 +161,7 @@ def test_baseten_model_api_pricing_entries():
assert model_info["output_cost_per_token"] == output_cost
def test_wandb_model_api_pricing_entries():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_wandb_model_api_pricing_entries(_local_model_cost_map):
expected_pricing = {
"wandb/moonshotai/Kimi-K2.5": (6e-07, 3e-06),
@ -182,9 +176,7 @@ def test_wandb_model_api_pricing_entries():
assert model_info["output_cost_per_token"] == output_cost
def test_openrouter_qwen36_plus_model_info():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_openrouter_qwen36_plus_model_info(_local_model_cost_map):
model_info = litellm.model_cost.get("openrouter/qwen/qwen3.6-plus")
@ -208,9 +200,7 @@ def test_openrouter_qwen36_plus_model_info():
"github_copilot/mai-code-1-flash-internal",
],
)
def test_github_copilot_mai_code_1_flash_pricing(model):
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_github_copilot_mai_code_1_flash_pricing(_local_model_cost_map, model):
model_info = litellm.model_cost.get(model)
@ -238,9 +228,7 @@ def test_github_copilot_mai_code_1_flash_pricing(model):
assert completion_usd == pytest.approx(500 * 4.5e-06)
def test_cost_calculator_with_usage(monkeypatch):
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_cost_calculator_with_usage(_local_model_cost_map, monkeypatch):
usage = Usage(
prompt_tokens=120,
@ -320,11 +308,9 @@ def test_cost_calculator_with_usage(monkeypatch):
assert result == expected_cost, f"Got {result}, Expected {expected_cost}"
def test_transcription_cost_uses_token_pricing():
def test_transcription_cost_uses_token_pricing(_local_model_cost_map):
from litellm import completion_cost
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
usage = Usage(
prompt_tokens=14,
@ -348,11 +334,9 @@ def test_transcription_cost_uses_token_pricing():
assert pytest.approx(cost, rel=1e-6) == expected_cost
def test_transcription_cost_falls_back_to_duration():
def test_transcription_cost_falls_back_to_duration(_local_model_cost_map):
from litellm import completion_cost
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
response = TranscriptionResponse(text="demo text")
response.duration = 10.0
@ -368,14 +352,12 @@ def test_transcription_cost_falls_back_to_duration():
assert pytest.approx(cost, rel=1e-6) == expected_cost
def test_vertex_chirp_3_transcription_cost_from_duration():
def test_vertex_chirp_3_transcription_cost_from_duration(_local_model_cost_map):
"""Regression: the chirp_3 cost map entry shipped with output_cost_per_second 0.0,
and cost_per_second prefers output_cost_per_second whenever it is not None, so
every transcription priced to $0.00 instead of using input_cost_per_second."""
from litellm import completion_cost
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
response = TranscriptionResponse(text="demo text")
response.duration = 18.0
@ -1127,9 +1109,7 @@ def test_tiered_pricing_only_deployment_completion_cost_is_nonzero():
assert cost > 0
def test_azure_realtime_cost_calculator():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_azure_realtime_cost_calculator(_local_model_cost_map):
cost = handle_realtime_stream_cost_calculation(
results=[
@ -1152,7 +1132,7 @@ def test_azure_realtime_cost_calculator():
assert cost > 0
def test_azure_audio_output_cost_calculation():
def test_azure_audio_output_cost_calculation(_local_model_cost_map):
"""
Test that Azure audio models correctly calculate costs for audio output tokens.
@ -1162,8 +1142,6 @@ def test_azure_audio_output_cost_calculation():
"""
from litellm.types.utils import Choices, CompletionTokensDetailsWrapper, Message
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
# Scenario from issue #19764:
# Input: 17 text tokens, 0 audio tokens
@ -1672,7 +1650,7 @@ def test_gemini_25_explicit_caching_cost_direct_usage():
assert expected_actual_cost == total_cost
def test_azure_ai_cache_cost_calculation():
def test_azure_ai_cache_cost_calculation(_local_model_cost_map):
"""
Test that azure_ai provider correctly calculates cache costs using generic_cost_per_token.
@ -1683,8 +1661,6 @@ def test_azure_ai_cache_cost_calculation():
from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_token
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
# Register a custom azure_ai model with cache pricing
test_model_id = "test-azure-ai-claude-model"
@ -1817,15 +1793,13 @@ def test_vertex_uplift_composes_with_above_128k_pricing(monkeypatch):
assert regional_completion == pytest.approx(global_completion * 1.10, rel=1e-9)
def test_cost_discount_vertex_ai():
def test_cost_discount_vertex_ai(monkeypatch):
"""
Test that cost discount is applied correctly for Vertex AI provider
"""
from litellm import completion_cost
from litellm.types.utils import Usage
# Save original config
original_discount_config = litellm.cost_discount_config.copy()
# Create mock response (use a model that exists in model_prices_and_context_window.json)
response = ModelResponse(
@ -1838,7 +1812,7 @@ def test_cost_discount_vertex_ai():
)
# Calculate cost without discount
litellm.cost_discount_config = {}
monkeypatch.setattr(litellm, "cost_discount_config", {})
cost_without_discount = completion_cost(
completion_response=response,
model="vertex_ai/gemini-3-pro-preview",
@ -1846,7 +1820,7 @@ def test_cost_discount_vertex_ai():
)
# Set 5% discount for vertex_ai
litellm.cost_discount_config = {"vertex_ai": 0.05}
monkeypatch.setattr(litellm, "cost_discount_config", {"vertex_ai": 0.05})
# Calculate cost with discount
cost_with_discount = completion_cost(
@ -1855,8 +1829,6 @@ def test_cost_discount_vertex_ai():
custom_llm_provider="vertex_ai",
)
# Restore original config
litellm.cost_discount_config = original_discount_config
# Verify discount is applied (5% off means 95% of original cost)
expected_cost = cost_without_discount * 0.95
@ -1868,15 +1840,13 @@ def test_cost_discount_vertex_ai():
print(f" - Savings: ${cost_without_discount - cost_with_discount:.6f}")
def test_cost_discount_not_applied_to_other_providers():
def test_cost_discount_not_applied_to_other_providers(monkeypatch):
"""
Test that cost discount only applies to configured providers
"""
from litellm import completion_cost
from litellm.types.utils import Usage
# Save original config
original_discount_config = litellm.cost_discount_config.copy()
# Create mock response for OpenAI
response = ModelResponse(
@ -1889,7 +1859,7 @@ def test_cost_discount_not_applied_to_other_providers():
)
# Set discount only for vertex_ai (not openai)
litellm.cost_discount_config = {"vertex_ai": 0.05}
monkeypatch.setattr(litellm, "cost_discount_config", {"vertex_ai": 0.05})
# Calculate cost for OpenAI - should NOT have discount applied
cost_with_selective_discount = completion_cost(
@ -1899,15 +1869,13 @@ def test_cost_discount_not_applied_to_other_providers():
)
# Clear discount config
litellm.cost_discount_config = {}
monkeypatch.setattr(litellm, "cost_discount_config", {})
cost_without_discount = completion_cost(
completion_response=response,
model="gpt-4",
custom_llm_provider="openai",
)
# Restore original config
litellm.cost_discount_config = original_discount_config
# Costs should be the same (no discount applied to OpenAI)
assert cost_with_selective_discount == cost_without_discount
@ -1917,15 +1885,13 @@ def test_cost_discount_not_applied_to_other_providers():
print(f" - Cost remains unchanged: ${cost_with_selective_discount:.6f}")
def test_cost_margin_percentage():
def test_cost_margin_percentage(monkeypatch):
"""
Test that percentage-based cost margin is applied correctly
"""
from litellm import completion_cost
from litellm.types.utils import Usage
# Save original config
original_margin_config = litellm.cost_margin_config.copy()
# Create mock response
response = ModelResponse(
@ -1938,7 +1904,7 @@ def test_cost_margin_percentage():
)
# Calculate cost without margin
litellm.cost_margin_config = {}
monkeypatch.setattr(litellm, "cost_margin_config", {})
cost_without_margin = completion_cost(
completion_response=response,
model="gpt-4",
@ -1946,7 +1912,7 @@ def test_cost_margin_percentage():
)
# Set 10% margin for openai
litellm.cost_margin_config = {"openai": 0.10}
monkeypatch.setattr(litellm, "cost_margin_config", {"openai": 0.10})
# Calculate cost with margin
cost_with_margin = completion_cost(
@ -1955,8 +1921,6 @@ def test_cost_margin_percentage():
custom_llm_provider="openai",
)
# Restore original config
litellm.cost_margin_config = original_margin_config
# Verify margin is applied (10% margin means 110% of original cost)
expected_cost = cost_without_margin * 1.10
@ -1968,15 +1932,13 @@ def test_cost_margin_percentage():
print(f" - Margin added: ${cost_with_margin - cost_without_margin:.6f}")
def test_cost_margin_fixed_amount():
def test_cost_margin_fixed_amount(monkeypatch):
"""
Test that fixed amount cost margin is applied correctly
"""
from litellm import completion_cost
from litellm.types.utils import Usage
# Save original config
original_margin_config = litellm.cost_margin_config.copy()
# Create mock response
response = ModelResponse(
@ -1989,7 +1951,7 @@ def test_cost_margin_fixed_amount():
)
# Calculate cost without margin
litellm.cost_margin_config = {}
monkeypatch.setattr(litellm, "cost_margin_config", {})
cost_without_margin = completion_cost(
completion_response=response,
model="gpt-4",
@ -1997,7 +1959,7 @@ def test_cost_margin_fixed_amount():
)
# Set $0.001 fixed margin for openai
litellm.cost_margin_config = {"openai": {"fixed_amount": 0.001}}
monkeypatch.setattr(litellm, "cost_margin_config", {"openai": {"fixed_amount": 0.001}})
# Calculate cost with margin
cost_with_margin = completion_cost(
@ -2006,8 +1968,6 @@ def test_cost_margin_fixed_amount():
custom_llm_provider="openai",
)
# Restore original config
litellm.cost_margin_config = original_margin_config
# Verify fixed margin is applied
expected_cost = cost_without_margin + 0.001
@ -2019,15 +1979,13 @@ def test_cost_margin_fixed_amount():
print(f" - Margin added: ${cost_with_margin - cost_without_margin:.6f}")
def test_cost_margin_combined():
def test_cost_margin_combined(monkeypatch):
"""
Test that combined percentage and fixed amount margin is applied correctly
"""
from litellm import completion_cost
from litellm.types.utils import Usage
# Save original config
original_margin_config = litellm.cost_margin_config.copy()
# Create mock response
response = ModelResponse(
@ -2040,7 +1998,7 @@ def test_cost_margin_combined():
)
# Calculate cost without margin
litellm.cost_margin_config = {}
monkeypatch.setattr(litellm, "cost_margin_config", {})
cost_without_margin = completion_cost(
completion_response=response,
model="gpt-4",
@ -2048,9 +2006,9 @@ def test_cost_margin_combined():
)
# Set 8% margin + $0.0005 fixed for openai
litellm.cost_margin_config = {
monkeypatch.setattr(litellm, "cost_margin_config", {
"openai": {"percentage": 0.08, "fixed_amount": 0.0005}
}
})
# Calculate cost with margin
cost_with_margin = completion_cost(
@ -2059,8 +2017,6 @@ def test_cost_margin_combined():
custom_llm_provider="openai",
)
# Restore original config
litellm.cost_margin_config = original_margin_config
# Verify combined margin is applied
expected_cost = cost_without_margin * 1.08 + 0.0005
@ -2072,15 +2028,13 @@ def test_cost_margin_combined():
print(f" - Margin added: ${cost_with_margin - cost_without_margin:.6f}")
def test_cost_margin_global():
def test_cost_margin_global(monkeypatch):
"""
Test that global margin is applied when no provider-specific margin is configured
"""
from litellm import completion_cost
from litellm.types.utils import Usage
# Save original config
original_margin_config = litellm.cost_margin_config.copy()
# Create mock response
response = ModelResponse(
@ -2093,7 +2047,7 @@ def test_cost_margin_global():
)
# Calculate cost without margin
litellm.cost_margin_config = {}
monkeypatch.setattr(litellm, "cost_margin_config", {})
cost_without_margin = completion_cost(
completion_response=response,
model="gpt-4",
@ -2101,7 +2055,7 @@ def test_cost_margin_global():
)
# Set 5% global margin (no provider-specific margin)
litellm.cost_margin_config = {"global": 0.05}
monkeypatch.setattr(litellm, "cost_margin_config", {"global": 0.05})
# Calculate cost with global margin
cost_with_global_margin = completion_cost(
@ -2110,8 +2064,6 @@ def test_cost_margin_global():
custom_llm_provider="openai",
)
# Restore original config
litellm.cost_margin_config = original_margin_config
# Verify global margin is applied
expected_cost = cost_without_margin * 1.05
@ -2123,15 +2075,13 @@ def test_cost_margin_global():
print(f" - Margin added: ${cost_with_global_margin - cost_without_margin:.6f}")
def test_cost_margin_provider_overrides_global():
def test_cost_margin_provider_overrides_global(monkeypatch):
"""
Test that provider-specific margin overrides global margin
"""
from litellm import completion_cost
from litellm.types.utils import Usage
# Save original config
original_margin_config = litellm.cost_margin_config.copy()
# Create mock response
response = ModelResponse(
@ -2144,7 +2094,7 @@ def test_cost_margin_provider_overrides_global():
)
# Calculate cost without margin
litellm.cost_margin_config = {}
monkeypatch.setattr(litellm, "cost_margin_config", {})
cost_without_margin = completion_cost(
completion_response=response,
model="gpt-4",
@ -2152,7 +2102,7 @@ def test_cost_margin_provider_overrides_global():
)
# Set 5% global margin and 10% provider-specific margin
litellm.cost_margin_config = {"global": 0.05, "openai": 0.10}
monkeypatch.setattr(litellm, "cost_margin_config", {"global": 0.05, "openai": 0.10})
# Calculate cost - should use provider-specific margin (10%), not global (5%)
cost_with_provider_margin = completion_cost(
@ -2161,8 +2111,6 @@ def test_cost_margin_provider_overrides_global():
custom_llm_provider="openai",
)
# Restore original config
litellm.cost_margin_config = original_margin_config
# Verify provider-specific margin is used (not global)
expected_cost = cost_without_margin * 1.10 # 10% from provider, not 5% from global
@ -2176,16 +2124,13 @@ def test_cost_margin_provider_overrides_global():
print(f" - Margin added: ${cost_with_provider_margin - cost_without_margin:.6f}")
def test_cost_margin_with_discount():
def test_cost_margin_with_discount(monkeypatch):
"""
Test that margin is applied after discount (independent calculation)
"""
from litellm import completion_cost
from litellm.types.utils import Usage
# Save original configs
original_margin_config = litellm.cost_margin_config.copy()
original_discount_config = litellm.cost_discount_config.copy()
# Create mock response
response = ModelResponse(
@ -2198,8 +2143,8 @@ def test_cost_margin_with_discount():
)
# Calculate base cost
litellm.cost_margin_config = {}
litellm.cost_discount_config = {}
monkeypatch.setattr(litellm, "cost_margin_config", {})
monkeypatch.setattr(litellm, "cost_discount_config", {})
base_cost = completion_cost(
completion_response=response,
model="gpt-4",
@ -2207,8 +2152,8 @@ def test_cost_margin_with_discount():
)
# Set 5% discount and 10% margin
litellm.cost_discount_config = {"openai": 0.05}
litellm.cost_margin_config = {"openai": 0.10}
monkeypatch.setattr(litellm, "cost_discount_config", {"openai": 0.05})
monkeypatch.setattr(litellm, "cost_margin_config", {"openai": 0.10})
# Calculate cost with both discount and margin
cost_with_both = completion_cost(
@ -2217,9 +2162,6 @@ def test_cost_margin_with_discount():
custom_llm_provider="openai",
)
# Restore original configs
litellm.cost_margin_config = original_margin_config
litellm.cost_discount_config = original_discount_config
# Verify: discount applied first, then margin
# Base cost -> discount: base * 0.95 -> margin: (base * 0.95) * 1.10
@ -2286,12 +2228,10 @@ def test_azure_image_generation_cost_calculator():
assert cost > 0.079
def test_completion_cost_extracts_service_tier_from_response():
def test_completion_cost_extracts_service_tier_from_response(_local_model_cost_map):
"""Test that completion_cost extracts service_tier from completion_response object."""
from litellm import completion_cost
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
# Test with gpt-5-nano which has flex pricing
model = "gpt-5-nano"
@ -2338,12 +2278,10 @@ def test_completion_cost_extracts_service_tier_from_response():
), f"Flex pricing should be ~50% of standard, got {flex_ratio:.2f}"
def test_completion_cost_extracts_service_tier_from_usage():
def test_completion_cost_extracts_service_tier_from_usage(_local_model_cost_map):
"""Test that completion_cost extracts service_tier from usage object."""
from litellm import completion_cost
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
# Test with gpt-5-nano which has flex pricing
model = "gpt-5-nano"
@ -2397,12 +2335,10 @@ def test_completion_cost_extracts_service_tier_from_usage():
), f"Flex pricing should be ~50% of standard, got {flex_ratio:.2f}"
def test_completion_cost_service_tier_priority():
def test_completion_cost_service_tier_priority(_local_model_cost_map):
"""Test that service_tier extraction follows priority: optional_params > completion_response > usage."""
from litellm import completion_cost
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
# Test with gpt-5-nano which has flex pricing
model = "gpt-5-nano"
@ -2457,12 +2393,10 @@ def test_completion_cost_service_tier_priority():
), "Costs from params and usage should be similar (both flex)"
def test_completion_cost_service_tier_for_bedrock():
def test_completion_cost_service_tier_for_bedrock(_local_model_cost_map):
"""Test that Bedrock cost calculation applies service_tier-specific pricing."""
from litellm import completion_cost
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "bedrock/us-east-1/test-bedrock-service-tier-cost-model"
litellm.register_model(
@ -2507,7 +2441,7 @@ def test_completion_cost_service_tier_for_bedrock():
assert priority_cost > default_cost > flex_cost > 0
def test_completion_cost_service_tier_for_anthropic():
def test_completion_cost_service_tier_for_anthropic(_local_model_cost_map):
"""
Anthropic priority-tier requests must be priced at the priority rate.
@ -2519,8 +2453,6 @@ def test_completion_cost_service_tier_for_anthropic():
from litellm import completion_cost
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "claude-test-service-tier-cost-model"
litellm.register_model(
@ -2561,7 +2493,7 @@ def test_completion_cost_service_tier_for_anthropic():
assert priority_cost == pytest.approx(2 * standard_cost)
def test_completion_cost_anthropic_auto_tier_uses_served_priority_rate():
def test_completion_cost_anthropic_auto_tier_uses_served_priority_rate(_local_model_cost_map):
"""
Proxy billing path regression for LIT-3771.
@ -2574,8 +2506,6 @@ def test_completion_cost_anthropic_auto_tier_uses_served_priority_rate():
from litellm import completion_cost
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "claude-test-auto-tier-cost-model"
litellm.register_model(
@ -2613,7 +2543,7 @@ def test_completion_cost_anthropic_auto_tier_uses_served_priority_rate():
assert cost == pytest.approx(expected_priority)
def test_completion_cost_non_string_service_tier_defers_to_served_tier():
def test_completion_cost_non_string_service_tier_defers_to_served_tier(_local_model_cost_map):
"""
Regression: a non-string request-level ``service_tier`` (reachable via
``allowed_openai_params``/``drop_params``) must not crash cost tracking.
@ -2627,8 +2557,6 @@ def test_completion_cost_non_string_service_tier_defers_to_served_tier():
from litellm import completion_cost
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "claude-test-non-string-tier-cost-model"
litellm.register_model(
@ -2665,7 +2593,7 @@ def test_completion_cost_non_string_service_tier_defers_to_served_tier():
assert cost == pytest.approx(expected_priority)
def test_completion_cost_non_string_response_service_tier_defers_to_served_tier():
def test_completion_cost_non_string_response_service_tier_defers_to_served_tier(_local_model_cost_map):
"""
Regression: a non-string ``service_tier`` on the response object must not
crash cost tracking.
@ -2679,8 +2607,6 @@ def test_completion_cost_non_string_response_service_tier_defers_to_served_tier(
from litellm import completion_cost
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "claude-test-response-non-string-tier-cost-model"
litellm.register_model(
@ -2718,7 +2644,7 @@ def test_completion_cost_non_string_response_service_tier_defers_to_served_tier(
assert cost == pytest.approx(expected_priority)
def test_completion_cost_non_string_usage_service_tier_prices_standard():
def test_completion_cost_non_string_usage_service_tier_prices_standard(_local_model_cost_map):
"""
Regression: a non-string ``service_tier`` on the usage object must not crash
cost tracking.
@ -2729,8 +2655,6 @@ def test_completion_cost_non_string_usage_service_tier_prices_standard():
"""
from litellm import completion_cost
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "claude-test-usage-non-string-tier-cost-model"
litellm.register_model(
@ -2764,7 +2688,7 @@ def test_completion_cost_non_string_usage_service_tier_prices_standard():
assert cost == pytest.approx(expected_standard)
def test_anthropic_cost_per_token_prices_cache_at_served_tier_with_multiplier():
def test_anthropic_cost_per_token_prices_cache_at_served_tier_with_multiplier(_local_model_cost_map):
"""
Regression for the cache/tier interaction in the Anthropic geo/speed path.
@ -2780,8 +2704,6 @@ def test_anthropic_cost_per_token_prices_cache_at_served_tier_with_multiplier():
)
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "claude-test-priority-cache-fast-model"
litellm.register_model(
@ -2837,7 +2759,7 @@ def _register_anthropic_geo_cache_model(model: str) -> None:
)
def test_anthropic_geo_multiplier_applies_to_cache_tokens(monkeypatch):
def test_anthropic_geo_multiplier_applies_to_cache_tokens(_local_model_cost_map, monkeypatch):
"""
Regression: the regional (geo) uplift must scale cache read and cache write
cost too, not just non-cache input and output.
@ -2853,7 +2775,6 @@ def test_anthropic_geo_multiplier_applies_to_cache_tokens(monkeypatch):
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "claude-test-geo-cache-model"
_register_anthropic_geo_cache_model(model)
@ -2882,7 +2803,7 @@ def test_anthropic_geo_multiplier_applies_to_cache_tokens(monkeypatch):
assert geo_completion_cost == pytest.approx(base_completion_cost * 1.1)
def test_anthropic_geo_and_fast_multipliers_compose(monkeypatch):
def test_anthropic_geo_and_fast_multipliers_compose(_local_model_cost_map, monkeypatch):
"""
The ``fast`` speed multiplier stays cache-exclusive (the old explicit
``fast/`` entries kept base cache rates) while the geo multiplier scales the
@ -2895,7 +2816,6 @@ def test_anthropic_geo_and_fast_multipliers_compose(monkeypatch):
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "claude-test-geo-fast-cache-model"
_register_anthropic_geo_cache_model(model)
@ -3100,7 +3020,7 @@ def test_gemini_implicit_caching_cost_calculation():
)
def test_additional_costs_only_for_azure_ai():
def test_additional_costs_only_for_azure_ai(_local_model_cost_map):
"""
Test that _get_additional_costs is only called for azure_ai provider.
@ -3111,8 +3031,6 @@ def test_additional_costs_only_for_azure_ai():
"""
from litellm.cost_calculator import _get_additional_costs
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
# Non-azure_ai providers should return None
result = _get_additional_costs(
@ -3140,7 +3058,7 @@ def test_additional_costs_only_for_azure_ai():
assert result is None, "Vertex AI should have no additional costs"
def test_openrouter_gemini_3_1_flash_lite_preview_pricing():
def test_openrouter_gemini_3_1_flash_lite_preview_pricing(_local_model_cost_map):
"""
Test that openrouter/google/gemini-3.1-flash-lite-preview has a pricing entry.
@ -3150,8 +3068,6 @@ def test_openrouter_gemini_3_1_flash_lite_preview_pricing():
model_prices_and_context_window.json when other Gemini 3.x variants were present.
This caused ValueError: This model isn't mapped yet during router pre-call checks.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model_name = "openrouter/google/gemini-3.1-flash-lite-preview"
model_info = litellm.model_cost.get(model_name)
@ -3164,9 +3080,7 @@ def test_openrouter_gemini_3_1_flash_lite_preview_pricing():
assert model_info["max_output_tokens"] == 65536
def test_gemini_3_1_flash_lite_pricing():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
def test_gemini_3_1_flash_lite_pricing(_local_model_cost_map):
for model_name in (
"gemini-3.1-flash-lite",
@ -3489,7 +3403,7 @@ def test_custom_pricing_without_cache_keys_preserves_legacy_behavior():
assert cost == pytest.approx(expected)
def test_openrouter_gemini_3_1_flash_lite_stable_pricing():
def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map):
"""
Test that openrouter/google/gemini-3.1-flash-lite (stable, no -preview suffix)
has a pricing entry.
@ -3505,8 +3419,6 @@ def test_openrouter_gemini_3_1_flash_lite_stable_pricing():
Pricing matches the existing -preview entry one-for-one (input $0.25/M, output
$1.50/M, cache-read $0.025/M) — Google did not change costs at the GA cutover.
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model_name = "openrouter/google/gemini-3.1-flash-lite"
model_info = litellm.model_cost.get(model_name)
@ -3520,7 +3432,7 @@ def test_openrouter_gemini_3_1_flash_lite_stable_pricing():
assert model_info["max_output_tokens"] == 65536
def test_completion_cost_logs_reasoning_and_cache_breakdown():
def test_completion_cost_logs_reasoning_and_cache_breakdown(_local_model_cost_map):
"""
completion_cost must surface explicit reasoning and cache-read costs into the
cost_breakdown stored on the logging object, so they end up in the spend logs
@ -3531,8 +3443,6 @@ def test_completion_cost_logs_reasoning_and_cache_breakdown():
from litellm.litellm_core_utils.litellm_logging import Logging
from litellm.types.utils import Choices, CompletionTokensDetailsWrapper, Message
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
logging_obj = Logging(
model="gemini-2.5-flash",
@ -3750,13 +3660,11 @@ def test_combine_usage_objects_sums_mirrored_cache_write_fields_once():
assert combined_pair.prompt_tokens_details.cache_creation_tokens == 100
def test_completion_cost_prices_anthropic_shaped_cache_read_tokens():
def test_completion_cost_prices_anthropic_shaped_cache_read_tokens(_local_model_cost_map):
"""Regression: an Anthropic /v1/messages response reports cache reads as top-level
cache_read_input_tokens with input_tokens excluding them. Reading that usage as
Responses API usage dropped the cache tokens and billed the whole prompt at the
uncached input rate, overstating spend on cache hits."""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
response = {
"id": "msg_1",

View file

@ -2789,7 +2789,10 @@ def _priced_at(prompt_tokens, completion_tokens):
@pytest.fixture
def local_cost_map(monkeypatch):
"""The prices these tests assert are the checked-in ones. Setting the environment
variable alone does not reload the map, so pin the map itself."""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
def test_a_streamed_response_bills_the_usage_the_provider_reported(local_cost_map):