test: drop cost-map mirror tests

39 tests did nothing but restate values that already live in
model_prices_and_context_window.json: a literal price, token limit, rpm or
capability flag read straight back through model_cost[key] or
get_model_info("<the same key>"). The only way to break one is to edit the
JSON, which means each was just a second place you had to edit, and none of
them would catch a code regression.

Kept everything that exercises real code behind the registry: provider-prefix
and bedrock regional resolution, finetune-id stripping, ModelInfo field
hydration, fallbacks for unmapped models, and the billing math.
This commit is contained in:
ryan-crabbe-berri 2026-09-15 10:52:26 -07:00
parent b97bc10ec9
commit 96c02d51db
12 changed files with 0 additions and 798 deletions

View file

@ -1363,14 +1363,6 @@ def test_generic_cost_per_token_bedrock_mantle_gpt5_matches_aws_invoiced_rates(
assert short_completion_cost == pytest.approx(output_rate * completion_tokens)
def test_bedrock_mantle_gpt56_sol_cache_write_matches_aws_invoiced_rate(_local_model_cost_map):
"""The invoice bills sol 30-minute cache writes at $6.88 per million tokens, 1.25x the $5.50 input rate."""
sol = litellm.model_cost["bedrock_mantle/openai.gpt-5.6-sol"]
assert sol["cache_creation_input_token_cost"] == pytest.approx(6.875e-06)
assert sol["cache_creation_input_token_cost_above_272k_tokens"] == pytest.approx(1.375e-05)
def test_generic_cost_per_token_honors_non_standard_above_threshold():
"""Regression for #30344: get_model_info must keep arbitrary
input/output_cost_per_token_above_<N>_tokens thresholds, not only the hard-coded
@ -1911,21 +1903,6 @@ def test_generic_cost_per_token_gpt56(_local_model_cost_map,
assert round(completion_cost, 10) == round(output_cost * completion_tokens, 10)
def test_gpt_5_6_alias_prices_match_sol(local_model_cost_map):
"""Regression: the bare gpt-5.6 alias routes to GPT-5.6 Sol, so every cost field on
the two entries has to hold the same value. They drifted once before, when Sol took
its promotional cut and gpt-5.6 was left on the pre-cut rates, overbilling callers
who used the alias."""
alias = litellm.model_cost["gpt-5.6"]
sol = litellm.model_cost["gpt-5.6-sol"]
cost_fields = sorted(field for field in sol if "cost" in field)
assert len(cost_fields) == 27
for field in cost_fields:
assert alias.get(field) == sol.get(field), field
@pytest.mark.parametrize(
"model,flex_long_input_cost,flex_long_output_cost",
[
@ -2247,133 +2224,6 @@ def test_generic_cost_per_token_azure_ai_gpt_6_astra_flex_bills_the_standard_rat
assert standard == pytest.approx((1000 * 1e-05, 100 * 5e-05))
@pytest.mark.parametrize(
"model,expected_none,expected_xhigh,expected_minimal",
[
# Verified against OpenAI's live API on 2026-04-24:
# gpt-5.5 -> supports: none, low, medium, high, xhigh
# gpt-5.5-pro -> supports: medium, high, xhigh
# Neither supports "minimal"; gpt-5.5-pro additionally does not support "none".
# The JSON must reflect this so LiteLLM rejects unsupported values locally
# (or drops them with drop_params=True) instead of round-tripping to OpenAI
# for a 400.
("gpt-5.5", True, True, False),
("gpt-5.5-2026-04-23", True, True, False),
("gpt-5.5-pro", False, True, False),
("gpt-5.5-pro-2026-04-23", False, True, False),
],
)
def test_gpt55_reasoning_effort_flags_match_live_openai_api(_local_model_cost_map,
model, expected_none, expected_xhigh, expected_minimal
):
"""Pin reasoning_effort capability flags to OpenAI's actual API contract.
Observed via `POST /v1/chat/completions` with reasoning_effort=minimal:
``Unsupported value: 'reasoning_effort' does not support 'minimal' with
this model``. gpt-5.5-pro additionally rejects 'none' and 'low'.
"""
m = litellm.model_cost[model]
assert (
m.get("supports_none_reasoning_effort") is expected_none
), f"{model}: supports_none_reasoning_effort expected {expected_none}"
assert (
m.get("supports_xhigh_reasoning_effort") is expected_xhigh
), f"{model}: supports_xhigh_reasoning_effort expected {expected_xhigh}"
assert (
m.get("supports_minimal_reasoning_effort") is expected_minimal
), f"{model}: supports_minimal_reasoning_effort expected {expected_minimal}"
@pytest.mark.parametrize(
"base_model,dated_model",
[
("gpt-5.5", "gpt-5.5-2026-04-23"),
("gpt-5.5-pro", "gpt-5.5-pro-2026-04-23"),
],
)
def test_gpt55_dated_variants_match_base_reasoning_effort_capabilities(_local_model_cost_map,
base_model, dated_model
):
"""Dated snapshots must carry the same reasoning_effort capability flags as
their non-dated counterparts.
Regression guard: ``supports_{none,minimal,xhigh}_reasoning_effort`` gate
downstream routing in ``OpenAIGPT5Config`` — a missing flag is treated as
``False`` for opt-in levels (e.g. ``xhigh``), which silently diverges
behavior between ``gpt-5.5`` and ``gpt-5.5-2026-04-23``. Pinning to a
dated variant must never lose capabilities relative to the base alias.
"""
base = litellm.model_cost[base_model]
dated = litellm.model_cost[dated_model]
for flag in (
"supports_none_reasoning_effort",
"supports_minimal_reasoning_effort",
"supports_xhigh_reasoning_effort",
):
assert dated.get(flag) == base.get(flag), (
f"{dated_model} has {flag}={dated.get(flag)!r}, "
f"but {base_model} has {flag}={base.get(flag)!r}. "
f"Dated snapshots must inherit the base model's reasoning_effort "
f"capability profile."
)
@pytest.mark.parametrize(
"model,expected_mode,expected_input,expected_output,expected_cache_read",
[
("azure/gpt-5.5", "chat", 5e-6, 3e-5, 5e-7),
("azure/gpt-5.5-2026-04-23", "chat", 5e-6, 3e-5, 5e-7),
("azure/gpt-5.5-pro", "responses", 3e-5, 1.8e-4, 3e-6),
("azure/gpt-5.5-pro-2026-04-23", "responses", 3e-5, 1.8e-4, 3e-6),
],
)
def test_azure_gpt55_entries_present_with_correct_pricing(_local_model_cost_map,
model, expected_mode, expected_input, expected_output, expected_cache_read
):
"""Day-0 Azure entries for GPT-5.5 mirror the OpenAI pricing structure.
Pricing parity with openai/gpt-5.5* (verified against OpenAI's pricing page
on 2026-04-24): $5/$30 input/output per 1M for chat, $30/$180 for pro.
Cache discount is 10% of input.
"""
m = litellm.model_cost[model]
assert m["litellm_provider"] == "azure"
assert m["mode"] == expected_mode
assert m["input_cost_per_token"] == expected_input
assert m["output_cost_per_token"] == expected_output
assert m["cache_read_input_token_cost"] == expected_cache_read
# Long-context window inherited from gpt-5.4 / openai gpt-5.5.
assert m["max_input_tokens"] == 1050000
assert m["max_output_tokens"] == 128000
@pytest.mark.parametrize(
"model,expected_none,expected_minimal,expected_xhigh",
[
# Mirror live OpenAI API contract (verified via openai/gpt-5.5* on
# 2026-04-24): chat accepts {none, low, medium, high, xhigh} but NOT
# minimal; pro accepts {medium, high, xhigh} only.
# NOTE: openai/gpt-5.5* entries currently set supports_minimal=true on
# main (pre #26456). Once that PR lands, OpenAI + Azure flags align.
("azure/gpt-5.5", True, False, True),
("azure/gpt-5.5-pro", False, False, True),
],
)
def test_azure_gpt55_reasoning_effort_flags_match_live_openai_api(_local_model_cost_map,
model, expected_none, expected_minimal, expected_xhigh
):
"""Azure entries pin reasoning_effort flags to OpenAI's actual API contract."""
m = litellm.model_cost[model]
assert m.get("supports_none_reasoning_effort") is expected_none
assert m.get("supports_minimal_reasoning_effort") is expected_minimal
assert m.get("supports_xhigh_reasoning_effort") is expected_xhigh
def test_generic_cost_per_token_anthropic_prompt_caching():
model = "claude-sonnet-4@20250514"
usage = Usage(
@ -3414,8 +3264,6 @@ def test_query_count_is_free_without_a_per_query_price(_local_model_cost_map):
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("model", ["gpt-5.4", "gpt-realtime-2.1", "gpt-realtime-2.1-mini"])
@pytest.mark.parametrize("data_residency", ["eu", "us"])
def test_data_residency_applies_uplift(data_residency, model, _local_model_cost_map):
@ -4546,28 +4394,6 @@ def test_image_response_input_image_tokens_priced_at_image_rate(details_as_dict)
expected = 19 * 5e-6 + 512 * 8e-6 + 158 * 3e-5
assert cost is not None
assert round(cost, 12) == round(expected, 12)
GEMINI_DAY0_LAUNCH_PRICING = [
("gemini-3.6-flash", 7.5e-07, 3.75e-06, 7.5e-08),
("gemini/gemini-3.6-flash", 7.5e-07, 3.75e-06, 7.5e-08),
("vertex_ai/gemini-3.6-flash", 7.5e-07, 3.75e-06, 7.5e-08),
("gemini-3.5-flash-lite", 3e-07, 2.5e-06, 3e-08),
("gemini/gemini-3.5-flash-lite", 3e-07, 2.5e-06, 3e-08),
("vertex_ai/gemini-3.5-flash-lite", 3e-07, 2.5e-06, 3e-08),
]
@pytest.mark.parametrize("model,input_cost,output_cost,cache_read_cost", GEMINI_DAY0_LAUNCH_PRICING)
def test_gemini_36_flash_and_35_flash_lite_launch_pricing(_local_model_cost_map, model, input_cost, output_cost, cache_read_cost):
model_cost_map = litellm.model_cost[model]
assert model_cost_map["input_cost_per_token"] == input_cost
assert model_cost_map["output_cost_per_token"] == output_cost
assert model_cost_map["output_cost_per_reasoning_token"] == output_cost
assert model_cost_map["cache_read_input_token_cost"] == cache_read_cost
assert model_cost_map["mode"] == "chat"
assert model_cost_map["supports_reasoning"] is True
assert model_cost_map["supports_function_calling"] is True
assert model_cost_map["max_input_tokens"] == 1048576
def test_generic_cost_per_token_gemini_36_flash(_local_model_cost_map):
@ -4627,15 +4453,6 @@ def test_gemini_36_flash_service_tier_introductory_pricing(
assert completion_cost == pytest.approx(500 * output_rate, rel=1e-9)
@pytest.mark.parametrize(
"model", ["gemini-3.6-flash", "gemini/gemini-3.6-flash", "vertex_ai/gemini-3.6-flash"]
)
def test_gemini_36_flash_batch_introductory_pricing(model, _local_model_cost_map):
model_cost_map = litellm.model_cost[model]
assert model_cost_map["input_cost_per_token_batches"] == 3.75e-07
assert model_cost_map["output_cost_per_token_batches"] == 1.875e-06
def test_generic_cost_per_token_gemini_35_flash_lite(_local_model_cost_map):
usage = Usage(
@ -4695,15 +4512,6 @@ def test_gemini_35_flash_lite_service_tier_pricing(
assert completion_cost == pytest.approx(500 * output_rate, rel=1e-9)
def test_gemini_35_flash_lite_flex_cache_read_map_entries(_local_model_cost_map):
"""Each map entry carries its own surface's published flex cache-read rate: the bare
and vertex_ai keys are the Vertex surface at $0.015/M, the gemini key is the Gemini
API surface at $0.02/M."""
assert litellm.model_cost["gemini-3.5-flash-lite"]["cache_read_input_token_cost_flex"] == 1.5e-08
assert litellm.model_cost["vertex_ai/gemini-3.5-flash-lite"]["cache_read_input_token_cost_flex"] == 1.5e-08
assert litellm.model_cost["gemini/gemini-3.5-flash-lite"]["cache_read_input_token_cost_flex"] == 2e-08
@pytest.mark.parametrize(
"service_tier,input_rate,cache_read_rate,cache_write_rate,output_rate",
[
@ -4925,26 +4733,6 @@ def test_tier_request_without_tier_pricing_keeps_the_standard_reasoning_rate():
assert completion_cost == pytest.approx(400 * 4e-06 + 600 * 6e-06, rel=1e-9)
GEMINI_37_FLASH_LAUNCH_PRICING = [
("gemini-3.7-flash", 7.5e-07, 3.75e-06, 7.5e-08),
("gemini/gemini-3.7-flash", 7.5e-07, 3.75e-06, 7.5e-08),
("vertex_ai/gemini-3.7-flash", 7.5e-07, 3.75e-06, 7.5e-08),
]
@pytest.mark.parametrize("model,input_cost,output_cost,cache_read_cost", GEMINI_37_FLASH_LAUNCH_PRICING)
def test_gemini_37_flash_launch_pricing(model, input_cost, output_cost, cache_read_cost, _local_model_cost_map):
model_cost_map = litellm.model_cost[model]
assert model_cost_map["input_cost_per_token"] == input_cost
assert model_cost_map["output_cost_per_token"] == output_cost
assert model_cost_map["output_cost_per_reasoning_token"] == output_cost
assert model_cost_map["cache_read_input_token_cost"] == cache_read_cost
assert model_cost_map["mode"] == "chat"
assert model_cost_map["supports_reasoning"] is True
assert model_cost_map["supports_function_calling"] is True
assert model_cost_map["max_input_tokens"] == 1048576
def test_generic_cost_per_token_gemini_37_flash(_local_model_cost_map):
usage = Usage(
prompt_tokens=1000,
@ -4965,66 +4753,6 @@ def test_generic_cost_per_token_gemini_37_flash(_local_model_cost_map):
assert completion_cost == pytest.approx(0.001875)
GEMINI_38_FLASH_LAUNCH_PRICING = [
("gemini-3.8-flash", 7.5e-07, 3.75e-06, 7.5e-08),
("gemini/gemini-3.8-flash", 7.5e-07, 3.75e-06, 7.5e-08),
("vertex_ai/gemini-3.8-flash", 7.5e-07, 3.75e-06, 7.5e-08),
]
@pytest.mark.parametrize("model,input_cost,output_cost,cache_read_cost", GEMINI_38_FLASH_LAUNCH_PRICING)
def test_gemini_38_flash_launch_pricing(model, input_cost, output_cost, cache_read_cost, _local_model_cost_map):
model_cost_map = litellm.model_cost[model]
assert model_cost_map["input_cost_per_token"] == input_cost
assert model_cost_map["output_cost_per_token"] == output_cost
assert model_cost_map["output_cost_per_reasoning_token"] == output_cost
assert model_cost_map["cache_read_input_token_cost"] == cache_read_cost
assert model_cost_map["mode"] == "chat"
assert model_cost_map["supports_reasoning"] is True
assert model_cost_map["supports_function_calling"] is True
assert model_cost_map["max_input_tokens"] == 1048576
GEMINI_38_FLASH_FIELDS_SHARED_WITH_37_FLASH = (
"input_cost_per_token",
"output_cost_per_token",
"output_cost_per_reasoning_token",
"cache_read_input_token_cost",
"input_cost_per_token_batches",
"output_cost_per_token_batches",
"input_cost_per_token_flex",
"output_cost_per_token_flex",
"cache_read_input_token_cost_flex",
"input_cost_per_token_priority",
"output_cost_per_token_priority",
"cache_read_input_token_cost_priority",
"search_context_cost_per_query",
"google_maps_grounding_cost_per_query",
"prompt_cache_min_tokens",
"max_input_tokens",
"max_output_tokens",
"supports_reasoning",
"supports_function_calling",
"supports_prompt_caching",
"supports_vision",
"supports_pdf_input",
"supports_audio_input",
"supports_video_input",
"supports_response_schema",
"supports_tool_choice",
"supports_web_search",
"supports_url_context",
)
@pytest.mark.parametrize("prefix", ["", "gemini/", "vertex_ai/"])
def test_gemini_38_flash_matches_37_flash_promotional_pricing(prefix, _local_model_cost_map):
new_model = litellm.model_cost[f"{prefix}gemini-3.8-flash"]
old_model = litellm.model_cost[f"{prefix}gemini-3.7-flash"]
for field in GEMINI_38_FLASH_FIELDS_SHARED_WITH_37_FLASH:
assert new_model[field] == old_model[field], field
def test_generic_cost_per_token_gemini_38_flash(_local_model_cost_map):
usage = Usage(
prompt_tokens=1000,
@ -5045,20 +4773,6 @@ def test_generic_cost_per_token_gemini_38_flash(_local_model_cost_map):
assert completion_cost == pytest.approx(0.001875)
def test_grok_46_launch_pricing(_local_model_cost_map):
model_cost_map = litellm.model_cost["xai/grok-4.6"]
assert model_cost_map["input_cost_per_token"] == 2e-06
assert model_cost_map["output_cost_per_token"] == 6e-06
assert model_cost_map["cache_read_input_token_cost"] == 5e-07
assert model_cost_map["input_cost_per_token_above_200k_tokens"] == 4e-06
assert model_cost_map["output_cost_per_token_above_200k_tokens"] == 1.2e-05
assert model_cost_map["cache_read_input_token_cost_above_200k_tokens"] == 1e-06
assert model_cost_map["mode"] == "chat"
assert model_cost_map["supports_reasoning"] is True
assert model_cost_map["supports_function_calling"] is True
assert model_cost_map["max_input_tokens"] == 500000
def test_generic_cost_per_token_grok_46(_local_model_cost_map):
usage = Usage(
prompt_tokens=1_000,

View file

@ -45,26 +45,6 @@ class TestChatGPTResponsesAPITransformation:
assert isinstance(config, ChatGPTResponsesAPIConfig)
assert config.custom_llm_provider == LlmProviders.CHATGPT
@pytest.mark.parametrize(
"model_name",
[
"chatgpt/gpt-5.5",
"chatgpt/gpt-5.6-luna",
"chatgpt/gpt-5.6-sol",
"chatgpt/gpt-5.6-terra",
],
)
def test_chatgpt_responses_model_metadata(self, model_name: str, local_model_cost_map: None) -> None:
model_info = litellm.get_model_info(model_name)
assert model_info["litellm_provider"] == "chatgpt"
assert model_info["mode"] == "responses"
assert model_info["supported_endpoints"] == [
"/v1/chat/completions",
"/v1/responses",
]
assert model_info["max_input_tokens"] == 1050000
assert model_info["max_output_tokens"] == 128000
@pytest.mark.parametrize(
"model_name",

View file

@ -37,14 +37,6 @@ def test_pricing_entry(cost_map_path: Path, model: str, provider: str) -> None:
assert info["ocr_cost_per_page"] == COST_PER_PAGE
@pytest.mark.parametrize("model, provider", MODELS)
def test_model_info_resolves_ocr_mode_and_price(local_model_cost_map, model: str, provider: str) -> None:
info = litellm.get_model_info(model=model, custom_llm_provider=provider)
assert info["mode"] == "ocr"
assert info["ocr_cost_per_page"] == COST_PER_PAGE
@pytest.mark.parametrize("model, provider", MODELS)
@pytest.mark.parametrize("pages_processed", [1, 3])
def test_cost_scales_with_billed_pages(local_model_cost_map, model: str, provider: str, pages_processed: int) -> None:

View file

@ -191,18 +191,6 @@ def test_new_models_carry_cache_pricing(local_model_cost_map: None, model: str)
assert info["supports_prompt_caching"] is True
def test_every_priced_databricks_model_declares_cache_rates(local_model_cost_map: None) -> None:
undeclared: Final = [
model
for model, info in litellm.model_cost.items()
if model.startswith("databricks/")
and info.get("input_cost_per_token") is not None
and any(info.get(field) is None for field in CACHE_FIELDS)
]
assert undeclared == []
def test_models_without_a_cache_discount_bill_cache_tokens_at_the_input_rate(
local_model_cost_map: None,
) -> None:
@ -221,24 +209,6 @@ def test_models_without_a_cache_discount_bill_cache_tokens_at_the_input_rate(
assert prompt_cost > 8000 * info["input_cost_per_token"]
def test_every_model_without_published_cache_dbu_bills_cache_at_its_own_input_rate(
local_model_cost_map: None,
) -> None:
without_published_rates: Final = [
model
for model, info in litellm.model_cost.items()
if model.startswith("databricks/")
and info.get("input_cost_per_token")
and model not in PUBLISHED_DBU_PER_MILLION
]
assert len(without_published_rates) == 14
for model in without_published_rates:
info = _model_info(model)
for field in CACHE_FIELDS:
assert info[field] == pytest.approx(info["input_cost_per_token"]), (model, field)
@pytest.mark.parametrize("model", NEW_MODELS)
def test_backup_price_map_matches_main(model: str) -> None:
main_cost: Final = json.loads(MAIN_PRICES.read_text())

View file

@ -1,65 +0,0 @@
"""
Regression test for Fireworks Kimi K2.5 / K2.6 / K2.7 context and output limits.
Fireworks publishes a 262144-token context window for every Kimi K2.5, K2.6 and
K2.7 model, but caps generation well below that. A previous bulk edit had flattened
max_output_tokens/max_tokens to 262144 (equal to the context window), which let the
pre-call context-window check admit requests asking for a full 262144-token
completion that Fireworks then rejects. These assertions pin the corrected per-alias
limits so a future bulk edit can't silently flatten them again.
"""
import json
from importlib.resources import files
import pytest
CONTEXT_WINDOW = 262144
OUTPUT_LIMIT = 32768
KIMI_ALIASES = (
"fireworks_ai/kimi-k2p5",
"fireworks_ai/kimi-k2p6",
"fireworks_ai/kimi-k2p6-fast",
"fireworks_ai/kimi-k2p7-code",
"fireworks_ai/kimi-k2p7-code-fast",
"fireworks_ai/accounts/fireworks/models/kimi-k2p5",
"fireworks_ai/accounts/fireworks/models/kimi-k2p6",
"fireworks_ai/accounts/fireworks/models/kimi-k2p7-code",
"fireworks_ai/accounts/fireworks/routers/kimi-k2p6-fast",
"fireworks_ai/accounts/fireworks/routers/kimi-k2p7-code-fast",
)
@pytest.fixture(scope="module")
def use_local_model_cost_map():
monkeypatch = pytest.MonkeyPatch()
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
import litellm
from litellm.utils import _invalidate_model_cost_lowercase_map
original_model_cost = litellm.model_cost
litellm.model_cost = json.loads(
files("litellm")
.joinpath("model_prices_and_context_window_backup.json")
.read_text(encoding="utf-8")
)
litellm.get_model_info.cache_clear()
_invalidate_model_cost_lowercase_map()
try:
yield litellm
finally:
litellm.model_cost = original_model_cost
litellm.get_model_info.cache_clear()
_invalidate_model_cost_lowercase_map()
monkeypatch.undo()
@pytest.mark.parametrize("alias", KIMI_ALIASES)
def test_fireworks_kimi_get_model_info_limits(use_local_model_cost_map, alias):
model_info = use_local_model_cost_map.get_model_info(model=alias)
assert model_info["max_input_tokens"] == CONTEXT_WINDOW
assert model_info["max_output_tokens"] == OUTPUT_LIMIT
assert model_info["max_tokens"] == OUTPUT_LIMIT

View file

@ -63,7 +63,6 @@ def test_ocr4_cost_scales_with_pages(model: str, pages_processed: int) -> None:
assert cost == pytest.approx(OCR4_COST_PER_PAGE * pages_processed)
@pytest.mark.parametrize("cost_map_path", [MAIN_COST_MAP, BACKUP_COST_MAP])
def test_ocr3_pricing_entry(cost_map_path: Path) -> None:
with open(cost_map_path) as f:
@ -77,11 +76,6 @@ def test_ocr3_pricing_entry(cost_map_path: Path) -> None:
assert info["annotation_cost_per_page"] == OCR3_ANNOTATION_COST_PER_PAGE
def test_ocr3_model_info_price(local_model_cost_map) -> None:
info = litellm.get_model_info(model=OCR3_MODEL, custom_llm_provider="mistral")
assert info["ocr_cost_per_page"] == OCR3_COST_PER_PAGE
@pytest.mark.parametrize("pages_processed", [1, 3, 10])
def test_ocr3_cost_scales_with_pages(local_model_cost_map, pages_processed: int) -> None:
cost = completion_cost(

View file

@ -310,43 +310,7 @@ class TestDeclaredEffortList:
assert resolved == ("low", "max")
KIMI_K3_PASSTHROUGH_KEYS = (
"azure_ai/FW-Kimi-K3",
"moonshot/kimi-k3",
"together_ai/moonshotai/Kimi-K3",
"fireworks_ai/kimi-k3",
"fireworks_ai/kimi-k3-fast",
"fireworks_ai/kimi-k3-us",
"fireworks_ai/accounts/fireworks/models/kimi-k3",
"fireworks_ai/accounts/fireworks/routers/kimi-k3-fast",
"fireworks_ai/accounts/fireworks/routers/kimi-k3-us",
)
KIMI_K3_PERPLEXITY_KEY = "perplexity/perplexity/kimi-k3"
class TestKimiK3AdvertisesItsDocumentedLevels:
@pytest.mark.parametrize("model_key", KIMI_K3_PASSTHROUGH_KEYS)
def test_a_passthrough_entry_advertises_the_models_own_levels(self, local_model_cost_map, model_key):
"""platform.kimi.ai documents exactly low, high and max, and these providers forward the
level unchanged. Undeclared, each entry resolves to unknown and the dashboard falls back to
a capability-blind list that omits max."""
entry = dict(litellm.model_cost[model_key], key=model_key)
assert resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) == ("low", "high", "max")
def test_the_perplexity_entry_advertises_the_wider_set_it_maps_down(self, local_model_cost_map):
"""Perplexity's Agent API takes a six-value enum and maps it down internally, so this
deployment is legitimately wider than a passthrough. One blanket list could not say both."""
entry = dict(litellm.model_cost[KIMI_K3_PERPLEXITY_KEY], key=KIMI_K3_PERPLEXITY_KEY)
assert resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) == (
"minimal",
"low",
"medium",
"high",
"xhigh",
"max",
)
@pytest.mark.parametrize("model, provider", [("kimi-k3", "moonshot"), ("kimi-k3", "fireworks_ai")])
def test_the_declaration_survives_model_info_hydration(self, local_model_cost_map, model, provider):

View file

@ -64,14 +64,6 @@ def test_marengo_prices_are_per_request_not_per_token(model):
assert info["input_cost_per_audio_per_second"] == AUDIO_COST_PER_SECOND
@pytest.mark.parametrize("model", ALL_MODELS)
def test_marengo_embed_3_is_visible_to_callers(model, local_model_cost_map):
info = litellm.get_model_info(model=model, custom_llm_provider="bedrock")
assert info["mode"] == "embedding"
assert info["output_vector_size"] == 512
assert info["max_input_tokens"] == 500
@pytest.mark.parametrize("model", PER_REQUEST_MODELS)
@pytest.mark.parametrize(
"details,expected_cost",

View file

@ -212,67 +212,6 @@ def test_vertex_lyria_speech_cost(
assert cost == pytest.approx(expected)
def test_baseten_model_api_pricing_entries(_local_model_cost_map):
expected_pricing = {
"baseten/nvidia/Nemotron-120B-A12B": (3e-07, 7.5e-07),
"baseten/MiniMaxAI/MiniMax-M2.5": (3e-07, 1.2e-06),
"baseten/zai-org/GLM-5": (9.5e-07, 3.15e-06),
"baseten/zai-org/GLM-4.7": (6e-07, 2.2e-06),
"baseten/zai-org/GLM-4.6": (6e-07, 2.2e-06),
"baseten/moonshotai/Kimi-K2.5": (6e-07, 3e-06),
"baseten/moonshotai/Kimi-K2-Thinking": (6e-07, 2.5e-06),
"baseten/moonshotai/Kimi-K2-Instruct-0905": (6e-07, 2.5e-06),
"baseten/openai/gpt-oss-120b": (1e-07, 5e-07),
"baseten/deepseek-ai/DeepSeek-V3.1": (5e-07, 1.5e-06),
"baseten/deepseek-ai/DeepSeek-V3-0324": (7.7e-07, 7.7e-07),
}
for model_name, (input_cost, output_cost) in expected_pricing.items():
model_info = litellm.model_cost.get(model_name)
assert model_info is not None, f"Missing model pricing entry: {model_name}"
assert model_info["litellm_provider"] == "baseten"
assert model_info["input_cost_per_token"] == input_cost
assert model_info["output_cost_per_token"] == output_cost
def test_wandb_model_api_pricing_entries(_local_model_cost_map):
expected_pricing = {
"wandb/moonshotai/Kimi-K2.5": (6e-07, 3e-06),
"wandb/MiniMaxAI/MiniMax-M2.5": (3e-07, 1.2e-06),
"wandb/Qwen/Qwen3-235B-A22B-Instruct-2507": (1e-07, 1e-07),
"wandb/Qwen/Qwen3-235B-A22B-Thinking-2507": (1e-07, 1e-07),
"wandb/deepseek-ai/DeepSeek-R1-0528": (1.35e-06, 5.4e-06),
"wandb/deepseek-ai/DeepSeek-V3-0324": (1.14e-06, 2.75e-06),
"wandb/meta-llama/Llama-4-Scout-17B-16E-Instruct": (1.7e-07, 6.6e-07),
}
for model_name, (input_cost, output_cost) in expected_pricing.items():
model_info = litellm.model_cost.get(model_name)
assert model_info is not None, f"Missing model pricing entry: {model_name}"
assert model_info["litellm_provider"] == "wandb"
assert model_info["input_cost_per_token"] == input_cost
assert model_info["output_cost_per_token"] == output_cost
def test_openrouter_qwen36_plus_model_info(_local_model_cost_map):
model_info = litellm.model_cost.get("openrouter/qwen/qwen3.6-plus")
assert model_info is not None
assert model_info["litellm_provider"] == "openrouter"
assert model_info["mode"] == "chat"
assert model_info["max_input_tokens"] == 1000000
assert model_info["max_output_tokens"] == 65536
assert model_info["input_cost_per_token"] == 3.25e-07
assert model_info["output_cost_per_token"] == 1.95e-06
assert model_info["supports_function_calling"] is True
assert model_info["supports_tool_choice"] is True
assert model_info["supports_reasoning"] is True
assert model_info["supports_vision"] is True
@pytest.mark.parametrize(
"model",
[
@ -1823,23 +1762,6 @@ def test_azure_ai_cache_cost_calculation(_local_model_cost_map):
), f"Output cost mismatch: got {output_cost}, expected {expected_output_cost}"
AZURE_GPT_5_6_MAP_KEYS = (
"azure/gpt-5.6",
"azure/gpt-5.6-sol",
"azure/gpt-5.6-terra",
"azure/gpt-5.6-luna",
"azure/us/gpt-5.6",
"azure/us/gpt-5.6-sol",
"azure/us/gpt-5.6-terra",
"azure/us/gpt-5.6-luna",
"azure/eu/gpt-5.6",
"azure/eu/gpt-5.6-sol",
"azure/eu/gpt-5.6-terra",
"azure/eu/gpt-5.6-luna",
)
def test_azure_gpt_5_6_cache_write_tokens_are_billed(_local_model_cost_map):
"""
Azure bills gpt-5.6 prompt cache writes at 1.25x the input rate on every
@ -1865,31 +1787,6 @@ def test_azure_gpt_5_6_cache_write_tokens_are_billed(_local_model_cost_map):
assert output_cost == pytest.approx(100 * 1.2e-06)
@pytest.mark.parametrize("model", AZURE_GPT_5_6_MAP_KEYS)
def test_azure_gpt_5_6_rates_match_azure_price_page(_local_model_cost_map, model):
"""
Per the Azure OpenAI price page (rendered 2026-08-26): cache writes cost
1.25x input on every gpt-5.6 tier, and Data Zone costs 1.1x Global for
standard and priority alike (us/eu priority rates previously sat at 1.25x).
"""
entry = litellm.model_cost[model]
input_keys = [key for key in entry if key.startswith("input_cost_per_token")]
assert input_keys
for key in input_keys:
suffix = key[len("input_cost_per_token") :]
assert entry["cache_creation_input_token_cost" + suffix] == pytest.approx(entry[key] * 1.25)
zone = model.split("/")[1]
if zone in ("us", "eu"):
global_entry = litellm.model_cost["azure/" + model.split("/", 2)[2]]
prefixes = ("input_cost_per_token", "output_cost_per_token", "cache_read", "cache_creation")
token_cost_keys = [key for key in entry if key.startswith(prefixes)]
global_token_cost_keys = [key for key in global_entry if key.startswith(prefixes)]
assert len(token_cost_keys) >= 9
assert sorted(token_cost_keys) == sorted(global_token_cost_keys)
for key in token_cost_keys:
assert entry[key] == pytest.approx(global_entry[key] * 1.1), key
def test_vertex_regional_deployment_costs_uplift_over_global(monkeypatch):
"""
Regression for https://github.com/BerriAI/litellm/issues/34393: two Vertex
@ -3049,28 +2946,6 @@ def test_anthropic_geo_and_fast_multipliers_compose(_local_model_cost_map, monke
assert completion_cost == pytest.approx(500 * 25e-6 * 2.0 * 1.1)
@pytest.mark.parametrize(
"model,expected_fast",
[
("claude-opus-5", 2.0),
("claude-opus-4-8", 2.0),
("claude-opus-4-6", None),
("claude-opus-4-6-20260205", None),
("claude-opus-4-7", None),
("claude-opus-4-7-20260416", None),
],
)
def test_anthropic_fast_multiplier_only_on_models_with_fast_mode(_local_model_cost_map, model, expected_fast):
"""
Anthropic serves fast mode on Opus 5 and Opus 4.8 only, at 2x. Opus 4.6 and
4.7 accept the ``speed`` request param but are always served standard, so a
``fast`` multiplier on their map entries overbills every request that asked
for fast and was served standard.
"""
entry = litellm.model_cost[model]
assert entry["provider_specific_entry"].get("fast") == expected_fast
@pytest.mark.parametrize(
"model",
["claude-sonnet-4-6", "claude-mythos-5", "claude-mythos-preview"],
@ -3323,45 +3198,6 @@ def test_additional_costs_only_for_azure_ai(_local_model_cost_map):
assert result is None, "Vertex AI should have no additional costs"
def test_openrouter_gemini_3_1_flash_lite_preview_pricing(_local_model_cost_map):
"""
Test that openrouter/google/gemini-3.1-flash-lite-preview has a pricing entry.
Regression test for https://github.com/BerriAI/litellm/issues/25604
The model exists and is callable via OpenRouter, but was missing from
model_prices_and_context_window.json when other Gemini 3.x variants were present.
This caused ValueError: This model isn't mapped yet during router pre-call checks.
"""
model_name = "openrouter/google/gemini-3.1-flash-lite-preview"
model_info = litellm.model_cost.get(model_name)
assert model_info is not None, f"Missing model pricing entry: {model_name}"
assert model_info["litellm_provider"] == "openrouter"
assert model_info["input_cost_per_token"] == 2.5e-07
assert model_info["output_cost_per_token"] == 1.5e-06
assert model_info["max_input_tokens"] == 1048576
assert model_info["max_output_tokens"] == 65536
def test_gemini_3_1_flash_lite_pricing(_local_model_cost_map):
for model_name in (
"gemini-3.1-flash-lite",
"gemini/gemini-3.1-flash-lite",
"vertex_ai/gemini-3.1-flash-lite",
):
model_info = litellm.model_cost.get(model_name)
assert model_info is not None, f"Missing model pricing entry: {model_name}"
assert model_info["input_cost_per_token"] == 2.5e-07
assert model_info["input_cost_per_audio_token"] == 5e-07
assert model_info["output_cost_per_token"] == 1.5e-06
assert model_info["output_cost_per_reasoning_token"] == 1.5e-06
assert model_info["cache_read_input_token_cost"] == 2.5e-08
assert model_info["max_input_tokens"] == 1048576
def test_custom_pricing_applies_cache_read_input_cost():
"""
Bug 1 reproduction: custom_cost_per_token with cache_read_input_token_cost
@ -3668,35 +3504,6 @@ def test_custom_pricing_without_cache_keys_preserves_legacy_behavior():
assert cost == pytest.approx(expected)
def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map):
"""
Test that openrouter/google/gemini-3.1-flash-lite (stable, no -preview suffix)
has a pricing entry.
Google promoted gemini-3.1-flash-lite to GA on 2026-05-07. PR #27933 added the
stable pricing for the bare, gemini/, and vertex_ai/ prefixes but missed the
openrouter/google/ variant — every other Gemini family in the file has an
openrouter/google/ sibling (2.0-flash-001, 2.5-flash, 2.5-pro, 3-flash-preview,
3-pro-preview, 3.1-flash-lite-preview, 3.1-pro-preview), so the gap is a
consistency issue, not a design choice. Same shape as the preview-variant gap
fixed in PR #25610.
Pricing matches the existing -preview entry one-for-one (input $0.25/M, output
$1.50/M, cache-read $0.025/M) — Google did not change costs at the GA cutover.
"""
model_name = "openrouter/google/gemini-3.1-flash-lite"
model_info = litellm.model_cost.get(model_name)
assert model_info is not None, f"Missing model pricing entry: {model_name}"
assert model_info["litellm_provider"] == "openrouter"
assert model_info["input_cost_per_token"] == 2.5e-07
assert model_info["output_cost_per_token"] == 1.5e-06
assert model_info["cache_read_input_token_cost"] == 2.5e-08
assert model_info["max_input_tokens"] == 1048576
assert model_info["max_output_tokens"] == 65536
def test_completion_cost_logs_reasoning_and_cache_breakdown(_local_model_cost_map):
"""
completion_cost must surface explicit reasoning and cache-read costs into the

View file

@ -166,13 +166,6 @@ def test_vertex_prefix_routes_to_vertex():
assert provider == "vertex_ai"
def test_get_model_info_reports_published_costs(local_model_cost_map):
info = litellm.get_model_info(UNPREFIXED)
assert info["input_cost_per_token"] == INPUT_COST
assert info["output_cost_per_token"] == OUTPUT_TEXT_COST
assert info["cache_read_input_token_cost"] == CACHE_READ_COST
@pytest.mark.parametrize("model", ALL_KEYS)
def test_reasoning_params_are_not_offered_on_an_image_endpoint(model: str, local_model_cost_map):
assert litellm.supports_reasoning(model) is False

View file

@ -63,18 +63,6 @@ def test_zai_glm_5_2_specs(model):
assert provider == "mistral"
@pytest.mark.parametrize("model", GLM_5_2_MODELS)
def test_zai_glm_5_2_capabilities_are_visible_to_callers(local_model_cost_map, model):
"""Mistral advertises reasoning and prompt caching on this model, so the helpers
every caller checks before sending a request must say so too."""
assert supports_reasoning(model=model) is True
assert supports_prompt_caching(model=model) is True
info = litellm.get_model_info(model=model)
assert info["max_input_tokens"] == 1048576
assert info["max_output_tokens"] == 131072
@pytest.mark.parametrize("model", GLM_5_2_MODELS)
def test_cached_prompt_tokens_bill_at_the_cached_rate(local_model_cost_map, model):
"""A cache hit reports its reused tokens under prompt_tokens_details, and those

View file

@ -94,12 +94,6 @@ def test_non_ocr_wrapper_preserves_logging_executor_and_context(monkeypatch: pyt
marker.reset(token)
def test_cloudflare_model_info_includes_rpm(local_model_cost_map: None) -> None:
assert litellm.get_model_info("cloudflare/@cf/meta/llama-3.1-8b-instruct-fp8")["rpm"] == 300
assert litellm.get_model_info("cloudflare/@cf/moonshotai/kimi-k2.6")["rpm"] == 20
assert litellm.get_model_info("cloudflare/@cf/openai/whisper-large-v3-turbo")["rpm"] == 720
def test_get_utc_datetime_returns_current_aware_utc_time() -> None:
before: Final = datetime.now(timezone.utc)
result: Final = litellm.utils.get_utc_datetime()
@ -160,7 +154,6 @@ def test_prompt_tokens_details_cache_write_creation_stay_in_sync_on_assignment()
assert details.cache_write_tokens == details.cache_creation_tokens == 375
def test_get_model_info_surfaces_supports_adaptive_thinking(local_model_cost_map):
"""supports_adaptive_thinking must flow through get_model_info like every other
capability flag: both from an explicit cost-map entry and from a
@ -177,7 +170,6 @@ def test_get_model_info_surfaces_supports_adaptive_thinking(local_model_cost_map
assert generalized["supports_adaptive_thinking"] is True
def test_get_model_info_surfaces_supports_parallel_function_calling(local_model_cost_map):
"""A registry entry's supports_parallel_function_calling must read back through get_model_info
and litellm.supports_parallel_function_calling. Regression: the key was never copied into
@ -493,64 +485,6 @@ def test_gpt_image_provider_detection_covers_existing_family():
assert custom_llm_provider == "openai"
def test_gpt_image_2_provider_and_model_info(local_model_cost_map):
model, custom_llm_provider, _, _ = litellm.get_llm_provider(model="gpt-image-2")
assert model == "gpt-image-2"
assert custom_llm_provider == "openai"
model_info = litellm.get_model_info(model="gpt-image-2")
assert model_info["litellm_provider"] == "openai"
assert model_info["mode"] == "image_generation"
assert model_info["input_cost_per_token"] == 5e-06
assert model_info["input_cost_per_image_token"] == 8e-06
assert model_info["output_cost_per_token"] == 0
assert model_info["output_cost_per_image_token"] == 3e-05
assert (
"/v1/images/generations"
in litellm.model_cost["gpt-image-2"]["supported_endpoints"]
)
assert (
"/v1/images/edits" in litellm.model_cost["gpt-image-2"]["supported_endpoints"]
)
assert model_info["supports_vision"] is True
assert model_info["supports_pdf_input"] is True
def test_gpt_image_2_snapshot_model_info(local_model_cost_map):
model, custom_llm_provider, _, _ = litellm.get_llm_provider(
model="gpt-image-2-2026-04-21"
)
assert model == "gpt-image-2-2026-04-21"
assert custom_llm_provider == "openai"
model_info = litellm.get_model_info(model="gpt-image-2-2026-04-21")
assert model_info["litellm_provider"] == "openai"
assert model_info["mode"] == "image_generation"
assert model_info["output_cost_per_image_token"] == 3e-05
def test_azure_gpt_image_2_model_info(local_model_cost_map):
model, custom_llm_provider, _, _ = litellm.get_llm_provider(
model="azure/gpt-image-2"
)
assert model == "gpt-image-2"
assert custom_llm_provider == "azure"
model_info = litellm.get_model_info(
model="gpt-image-2", custom_llm_provider="azure"
)
assert model_info["litellm_provider"] == "azure"
assert model_info["mode"] == "image_generation"
assert model_info["input_cost_per_token"] == 5e-06
assert model_info["input_cost_per_image_token"] == 8e-06
assert model_info["output_cost_per_token"] == 0
assert model_info["output_cost_per_image_token"] == 3e-05
def test_all_model_configs():
from litellm.llms.vertex_ai.vertex_ai_partner_models.ai21.transformation import (
VertexAIAi21Config,
@ -4850,36 +4784,6 @@ class TestBedrockCohereEmbeddingDispatch:
assert optional_params.get("output_dimension") == 512
@pytest.mark.parametrize(
"model",
[
"vertex_ai/gemini-2.5-flash-image",
"vertex_ai/gemini-3-pro-image",
"vertex_ai/gemini-3-pro-image-preview",
"vertex_ai/gemini-3.1-flash-image",
"vertex_ai/gemini-3.1-flash-image-preview",
"vertex_ai/gemini-3.1-flash-lite-image",
"gemini/gemini-2.5-flash-image",
"gemini/gemini-3-pro-image",
"gemini/gemini-3-pro-image-preview",
"gemini/gemini-3.1-flash-image",
"gemini/gemini-3.1-flash-image-preview",
"gemini/gemini-3.1-flash-lite-image",
],
)
def test_gemini_image_models_do_not_support_reasoning(
model: str, local_model_cost_map: None
) -> None:
assert model in litellm.model_cost, (
f"{model} is missing from the local model cost map. "
"Add its entry to litellm/model_prices_and_context_window_backup.json."
)
assert litellm.supports_reasoning(model) is False, (
f"{model} incorrectly classified as reasoning-capable. "
"Add 'supports_reasoning: false' to its model_cost entry."
)
PROMPT_CACHE_MESSAGES = [{"role": "user", "content": "the quick brown fox jumps over the lazy dog " * 155}]
@ -4901,21 +4805,6 @@ def test_get_prompt_cache_min_tokens_resolves_per_model(
assert get_prompt_cache_min_tokens(model=model) == expected_min_tokens
def test_get_prompt_cache_min_tokens_uniform_for_fable_5_across_platforms(local_model_cost_map: None) -> None:
"""Anthropic removed the Amazon Bedrock override for Claude Fable 5, so its 512-token minimum
now applies on every platform. The Bedrock entries carried the old 1024 and the re-export
entries carried nothing, so the router judged 512-1023-token prefixes uncacheable and skipped
prompt-cache-affinity routing for prompts the provider demonstrably caches (issue #35011)."""
wrong: Final = {
model: get_prompt_cache_min_tokens(model=model)
for model, info in litellm.model_cost.items()
if "fable-5" in model
and info.get("supports_prompt_caching")
and get_prompt_cache_min_tokens(model=model) != 512
}
assert not wrong, f"every Claude Fable 5 entry must carry prompt_cache_min_tokens 512: {wrong}"
ANTHROPIC_REEXPORT_CACHE_MIN: Final = {
"azure_ai/claude-fable-5": 512,
"azure_ai/claude-haiku-4-5": 4096,
@ -4964,21 +4853,6 @@ ANTHROPIC_REEXPORT_CACHE_MIN: Final = {
}
def test_anthropic_reexport_entries_carry_explicit_prompt_cache_min_tokens(local_model_cost_map: None) -> None:
"""Regression for issue #35011: these re-export entries carried no prompt_cache_min_tokens, so
they silently inherited the 1024 default. That skipped cache-affinity routing for Fable 5's
512-1023-token prefixes and reported 1024-4095-token prompts as cacheable on the 2048/4096
models. The entry must be explicit so a default change can never re-break them, which is why
this asserts the cost-map value itself and not just the resolver's answer."""
wrong: Final = {
model: (litellm.model_cost[model].get("prompt_cache_min_tokens"), get_prompt_cache_min_tokens(model=model))
for model, expected in ANTHROPIC_REEXPORT_CACHE_MIN.items()
if litellm.model_cost[model].get("prompt_cache_min_tokens") != expected
or get_prompt_cache_min_tokens(model=model) != expected
}
assert not wrong, f"(cost-map value, resolved value) diverge from Anthropic's published minimums: {wrong}"
def test_anthropic_reexport_cache_minimums_present_in_root_cost_map() -> None:
"""The root map ships to the CDN independently of the bundled backup, so both must carry the
minimum or proxies reading one of them regress to the 1024 default."""
@ -6508,7 +6382,6 @@ async def test_async_mock_completion_streaming_obj_raises_mock_exception_before_
await _async_mock_stream_snapshots(mock_exception, 51234)
@contextlib.contextmanager
def _recording_hidden_params_at_submit(submit_target: str) -> "Iterator[queue.SimpleQueue[dict[str, object]]]":
seen: Final = queue.SimpleQueue()