mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
test: drop cost-map mirror tests
39 tests did nothing but restate values that already live in
model_prices_and_context_window.json: a literal price, token limit, rpm or
capability flag read straight back through model_cost[key] or
get_model_info("<the same key>"). The only way to break one is to edit the
JSON, which means each was just a second place you had to edit, and none of
them would catch a code regression.
Kept everything that exercises real code behind the registry: provider-prefix
and bedrock regional resolution, finetune-id stripping, ModelInfo field
hydration, fallbacks for unmapped models, and the billing math.
This commit is contained in:
parent
b97bc10ec9
commit
96c02d51db
12 changed files with 0 additions and 798 deletions
|
|
@ -1363,14 +1363,6 @@ def test_generic_cost_per_token_bedrock_mantle_gpt5_matches_aws_invoiced_rates(
|
|||
assert short_completion_cost == pytest.approx(output_rate * completion_tokens)
|
||||
|
||||
|
||||
def test_bedrock_mantle_gpt56_sol_cache_write_matches_aws_invoiced_rate(_local_model_cost_map):
|
||||
"""The invoice bills sol 30-minute cache writes at $6.88 per million tokens, 1.25x the $5.50 input rate."""
|
||||
|
||||
sol = litellm.model_cost["bedrock_mantle/openai.gpt-5.6-sol"]
|
||||
assert sol["cache_creation_input_token_cost"] == pytest.approx(6.875e-06)
|
||||
assert sol["cache_creation_input_token_cost_above_272k_tokens"] == pytest.approx(1.375e-05)
|
||||
|
||||
|
||||
def test_generic_cost_per_token_honors_non_standard_above_threshold():
|
||||
"""Regression for #30344: get_model_info must keep arbitrary
|
||||
input/output_cost_per_token_above_<N>_tokens thresholds, not only the hard-coded
|
||||
|
|
@ -1911,21 +1903,6 @@ def test_generic_cost_per_token_gpt56(_local_model_cost_map,
|
|||
assert round(completion_cost, 10) == round(output_cost * completion_tokens, 10)
|
||||
|
||||
|
||||
def test_gpt_5_6_alias_prices_match_sol(local_model_cost_map):
|
||||
"""Regression: the bare gpt-5.6 alias routes to GPT-5.6 Sol, so every cost field on
|
||||
the two entries has to hold the same value. They drifted once before, when Sol took
|
||||
its promotional cut and gpt-5.6 was left on the pre-cut rates, overbilling callers
|
||||
who used the alias."""
|
||||
alias = litellm.model_cost["gpt-5.6"]
|
||||
sol = litellm.model_cost["gpt-5.6-sol"]
|
||||
|
||||
cost_fields = sorted(field for field in sol if "cost" in field)
|
||||
assert len(cost_fields) == 27
|
||||
|
||||
for field in cost_fields:
|
||||
assert alias.get(field) == sol.get(field), field
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,flex_long_input_cost,flex_long_output_cost",
|
||||
[
|
||||
|
|
@ -2247,133 +2224,6 @@ def test_generic_cost_per_token_azure_ai_gpt_6_astra_flex_bills_the_standard_rat
|
|||
assert standard == pytest.approx((1000 * 1e-05, 100 * 5e-05))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,expected_none,expected_xhigh,expected_minimal",
|
||||
[
|
||||
# Verified against OpenAI's live API on 2026-04-24:
|
||||
# gpt-5.5 -> supports: none, low, medium, high, xhigh
|
||||
# gpt-5.5-pro -> supports: medium, high, xhigh
|
||||
# Neither supports "minimal"; gpt-5.5-pro additionally does not support "none".
|
||||
# The JSON must reflect this so LiteLLM rejects unsupported values locally
|
||||
# (or drops them with drop_params=True) instead of round-tripping to OpenAI
|
||||
# for a 400.
|
||||
("gpt-5.5", True, True, False),
|
||||
("gpt-5.5-2026-04-23", True, True, False),
|
||||
("gpt-5.5-pro", False, True, False),
|
||||
("gpt-5.5-pro-2026-04-23", False, True, False),
|
||||
],
|
||||
)
|
||||
def test_gpt55_reasoning_effort_flags_match_live_openai_api(_local_model_cost_map,
|
||||
model, expected_none, expected_xhigh, expected_minimal
|
||||
):
|
||||
"""Pin reasoning_effort capability flags to OpenAI's actual API contract.
|
||||
|
||||
Observed via `POST /v1/chat/completions` with reasoning_effort=minimal:
|
||||
``Unsupported value: 'reasoning_effort' does not support 'minimal' with
|
||||
this model``. gpt-5.5-pro additionally rejects 'none' and 'low'.
|
||||
"""
|
||||
|
||||
m = litellm.model_cost[model]
|
||||
assert (
|
||||
m.get("supports_none_reasoning_effort") is expected_none
|
||||
), f"{model}: supports_none_reasoning_effort expected {expected_none}"
|
||||
assert (
|
||||
m.get("supports_xhigh_reasoning_effort") is expected_xhigh
|
||||
), f"{model}: supports_xhigh_reasoning_effort expected {expected_xhigh}"
|
||||
assert (
|
||||
m.get("supports_minimal_reasoning_effort") is expected_minimal
|
||||
), f"{model}: supports_minimal_reasoning_effort expected {expected_minimal}"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"base_model,dated_model",
|
||||
[
|
||||
("gpt-5.5", "gpt-5.5-2026-04-23"),
|
||||
("gpt-5.5-pro", "gpt-5.5-pro-2026-04-23"),
|
||||
],
|
||||
)
|
||||
def test_gpt55_dated_variants_match_base_reasoning_effort_capabilities(_local_model_cost_map,
|
||||
base_model, dated_model
|
||||
):
|
||||
"""Dated snapshots must carry the same reasoning_effort capability flags as
|
||||
their non-dated counterparts.
|
||||
|
||||
Regression guard: ``supports_{none,minimal,xhigh}_reasoning_effort`` gate
|
||||
downstream routing in ``OpenAIGPT5Config`` — a missing flag is treated as
|
||||
``False`` for opt-in levels (e.g. ``xhigh``), which silently diverges
|
||||
behavior between ``gpt-5.5`` and ``gpt-5.5-2026-04-23``. Pinning to a
|
||||
dated variant must never lose capabilities relative to the base alias.
|
||||
"""
|
||||
|
||||
base = litellm.model_cost[base_model]
|
||||
dated = litellm.model_cost[dated_model]
|
||||
|
||||
for flag in (
|
||||
"supports_none_reasoning_effort",
|
||||
"supports_minimal_reasoning_effort",
|
||||
"supports_xhigh_reasoning_effort",
|
||||
):
|
||||
assert dated.get(flag) == base.get(flag), (
|
||||
f"{dated_model} has {flag}={dated.get(flag)!r}, "
|
||||
f"but {base_model} has {flag}={base.get(flag)!r}. "
|
||||
f"Dated snapshots must inherit the base model's reasoning_effort "
|
||||
f"capability profile."
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,expected_mode,expected_input,expected_output,expected_cache_read",
|
||||
[
|
||||
("azure/gpt-5.5", "chat", 5e-6, 3e-5, 5e-7),
|
||||
("azure/gpt-5.5-2026-04-23", "chat", 5e-6, 3e-5, 5e-7),
|
||||
("azure/gpt-5.5-pro", "responses", 3e-5, 1.8e-4, 3e-6),
|
||||
("azure/gpt-5.5-pro-2026-04-23", "responses", 3e-5, 1.8e-4, 3e-6),
|
||||
],
|
||||
)
|
||||
def test_azure_gpt55_entries_present_with_correct_pricing(_local_model_cost_map,
|
||||
model, expected_mode, expected_input, expected_output, expected_cache_read
|
||||
):
|
||||
"""Day-0 Azure entries for GPT-5.5 mirror the OpenAI pricing structure.
|
||||
|
||||
Pricing parity with openai/gpt-5.5* (verified against OpenAI's pricing page
|
||||
on 2026-04-24): $5/$30 input/output per 1M for chat, $30/$180 for pro.
|
||||
Cache discount is 10% of input.
|
||||
"""
|
||||
|
||||
m = litellm.model_cost[model]
|
||||
assert m["litellm_provider"] == "azure"
|
||||
assert m["mode"] == expected_mode
|
||||
assert m["input_cost_per_token"] == expected_input
|
||||
assert m["output_cost_per_token"] == expected_output
|
||||
assert m["cache_read_input_token_cost"] == expected_cache_read
|
||||
# Long-context window inherited from gpt-5.4 / openai gpt-5.5.
|
||||
assert m["max_input_tokens"] == 1050000
|
||||
assert m["max_output_tokens"] == 128000
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,expected_none,expected_minimal,expected_xhigh",
|
||||
[
|
||||
# Mirror live OpenAI API contract (verified via openai/gpt-5.5* on
|
||||
# 2026-04-24): chat accepts {none, low, medium, high, xhigh} but NOT
|
||||
# minimal; pro accepts {medium, high, xhigh} only.
|
||||
# NOTE: openai/gpt-5.5* entries currently set supports_minimal=true on
|
||||
# main (pre #26456). Once that PR lands, OpenAI + Azure flags align.
|
||||
("azure/gpt-5.5", True, False, True),
|
||||
("azure/gpt-5.5-pro", False, False, True),
|
||||
],
|
||||
)
|
||||
def test_azure_gpt55_reasoning_effort_flags_match_live_openai_api(_local_model_cost_map,
|
||||
model, expected_none, expected_minimal, expected_xhigh
|
||||
):
|
||||
"""Azure entries pin reasoning_effort flags to OpenAI's actual API contract."""
|
||||
|
||||
m = litellm.model_cost[model]
|
||||
assert m.get("supports_none_reasoning_effort") is expected_none
|
||||
assert m.get("supports_minimal_reasoning_effort") is expected_minimal
|
||||
assert m.get("supports_xhigh_reasoning_effort") is expected_xhigh
|
||||
|
||||
|
||||
def test_generic_cost_per_token_anthropic_prompt_caching():
|
||||
model = "claude-sonnet-4@20250514"
|
||||
usage = Usage(
|
||||
|
|
@ -3414,8 +3264,6 @@ def test_query_count_is_free_without_a_per_query_price(_local_model_cost_map):
|
|||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["gpt-5.4", "gpt-realtime-2.1", "gpt-realtime-2.1-mini"])
|
||||
@pytest.mark.parametrize("data_residency", ["eu", "us"])
|
||||
def test_data_residency_applies_uplift(data_residency, model, _local_model_cost_map):
|
||||
|
|
@ -4546,28 +4394,6 @@ def test_image_response_input_image_tokens_priced_at_image_rate(details_as_dict)
|
|||
expected = 19 * 5e-6 + 512 * 8e-6 + 158 * 3e-5
|
||||
assert cost is not None
|
||||
assert round(cost, 12) == round(expected, 12)
|
||||
GEMINI_DAY0_LAUNCH_PRICING = [
|
||||
("gemini-3.6-flash", 7.5e-07, 3.75e-06, 7.5e-08),
|
||||
("gemini/gemini-3.6-flash", 7.5e-07, 3.75e-06, 7.5e-08),
|
||||
("vertex_ai/gemini-3.6-flash", 7.5e-07, 3.75e-06, 7.5e-08),
|
||||
("gemini-3.5-flash-lite", 3e-07, 2.5e-06, 3e-08),
|
||||
("gemini/gemini-3.5-flash-lite", 3e-07, 2.5e-06, 3e-08),
|
||||
("vertex_ai/gemini-3.5-flash-lite", 3e-07, 2.5e-06, 3e-08),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model,input_cost,output_cost,cache_read_cost", GEMINI_DAY0_LAUNCH_PRICING)
|
||||
def test_gemini_36_flash_and_35_flash_lite_launch_pricing(_local_model_cost_map, model, input_cost, output_cost, cache_read_cost):
|
||||
|
||||
model_cost_map = litellm.model_cost[model]
|
||||
assert model_cost_map["input_cost_per_token"] == input_cost
|
||||
assert model_cost_map["output_cost_per_token"] == output_cost
|
||||
assert model_cost_map["output_cost_per_reasoning_token"] == output_cost
|
||||
assert model_cost_map["cache_read_input_token_cost"] == cache_read_cost
|
||||
assert model_cost_map["mode"] == "chat"
|
||||
assert model_cost_map["supports_reasoning"] is True
|
||||
assert model_cost_map["supports_function_calling"] is True
|
||||
assert model_cost_map["max_input_tokens"] == 1048576
|
||||
|
||||
|
||||
def test_generic_cost_per_token_gemini_36_flash(_local_model_cost_map):
|
||||
|
|
@ -4627,15 +4453,6 @@ def test_gemini_36_flash_service_tier_introductory_pricing(
|
|||
assert completion_cost == pytest.approx(500 * output_rate, rel=1e-9)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model", ["gemini-3.6-flash", "gemini/gemini-3.6-flash", "vertex_ai/gemini-3.6-flash"]
|
||||
)
|
||||
def test_gemini_36_flash_batch_introductory_pricing(model, _local_model_cost_map):
|
||||
model_cost_map = litellm.model_cost[model]
|
||||
assert model_cost_map["input_cost_per_token_batches"] == 3.75e-07
|
||||
assert model_cost_map["output_cost_per_token_batches"] == 1.875e-06
|
||||
|
||||
|
||||
def test_generic_cost_per_token_gemini_35_flash_lite(_local_model_cost_map):
|
||||
|
||||
usage = Usage(
|
||||
|
|
@ -4695,15 +4512,6 @@ def test_gemini_35_flash_lite_service_tier_pricing(
|
|||
assert completion_cost == pytest.approx(500 * output_rate, rel=1e-9)
|
||||
|
||||
|
||||
def test_gemini_35_flash_lite_flex_cache_read_map_entries(_local_model_cost_map):
|
||||
"""Each map entry carries its own surface's published flex cache-read rate: the bare
|
||||
and vertex_ai keys are the Vertex surface at $0.015/M, the gemini key is the Gemini
|
||||
API surface at $0.02/M."""
|
||||
assert litellm.model_cost["gemini-3.5-flash-lite"]["cache_read_input_token_cost_flex"] == 1.5e-08
|
||||
assert litellm.model_cost["vertex_ai/gemini-3.5-flash-lite"]["cache_read_input_token_cost_flex"] == 1.5e-08
|
||||
assert litellm.model_cost["gemini/gemini-3.5-flash-lite"]["cache_read_input_token_cost_flex"] == 2e-08
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"service_tier,input_rate,cache_read_rate,cache_write_rate,output_rate",
|
||||
[
|
||||
|
|
@ -4925,26 +4733,6 @@ def test_tier_request_without_tier_pricing_keeps_the_standard_reasoning_rate():
|
|||
assert completion_cost == pytest.approx(400 * 4e-06 + 600 * 6e-06, rel=1e-9)
|
||||
|
||||
|
||||
GEMINI_37_FLASH_LAUNCH_PRICING = [
|
||||
("gemini-3.7-flash", 7.5e-07, 3.75e-06, 7.5e-08),
|
||||
("gemini/gemini-3.7-flash", 7.5e-07, 3.75e-06, 7.5e-08),
|
||||
("vertex_ai/gemini-3.7-flash", 7.5e-07, 3.75e-06, 7.5e-08),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model,input_cost,output_cost,cache_read_cost", GEMINI_37_FLASH_LAUNCH_PRICING)
|
||||
def test_gemini_37_flash_launch_pricing(model, input_cost, output_cost, cache_read_cost, _local_model_cost_map):
|
||||
model_cost_map = litellm.model_cost[model]
|
||||
assert model_cost_map["input_cost_per_token"] == input_cost
|
||||
assert model_cost_map["output_cost_per_token"] == output_cost
|
||||
assert model_cost_map["output_cost_per_reasoning_token"] == output_cost
|
||||
assert model_cost_map["cache_read_input_token_cost"] == cache_read_cost
|
||||
assert model_cost_map["mode"] == "chat"
|
||||
assert model_cost_map["supports_reasoning"] is True
|
||||
assert model_cost_map["supports_function_calling"] is True
|
||||
assert model_cost_map["max_input_tokens"] == 1048576
|
||||
|
||||
|
||||
def test_generic_cost_per_token_gemini_37_flash(_local_model_cost_map):
|
||||
usage = Usage(
|
||||
prompt_tokens=1000,
|
||||
|
|
@ -4965,66 +4753,6 @@ def test_generic_cost_per_token_gemini_37_flash(_local_model_cost_map):
|
|||
assert completion_cost == pytest.approx(0.001875)
|
||||
|
||||
|
||||
GEMINI_38_FLASH_LAUNCH_PRICING = [
|
||||
("gemini-3.8-flash", 7.5e-07, 3.75e-06, 7.5e-08),
|
||||
("gemini/gemini-3.8-flash", 7.5e-07, 3.75e-06, 7.5e-08),
|
||||
("vertex_ai/gemini-3.8-flash", 7.5e-07, 3.75e-06, 7.5e-08),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model,input_cost,output_cost,cache_read_cost", GEMINI_38_FLASH_LAUNCH_PRICING)
|
||||
def test_gemini_38_flash_launch_pricing(model, input_cost, output_cost, cache_read_cost, _local_model_cost_map):
|
||||
model_cost_map = litellm.model_cost[model]
|
||||
assert model_cost_map["input_cost_per_token"] == input_cost
|
||||
assert model_cost_map["output_cost_per_token"] == output_cost
|
||||
assert model_cost_map["output_cost_per_reasoning_token"] == output_cost
|
||||
assert model_cost_map["cache_read_input_token_cost"] == cache_read_cost
|
||||
assert model_cost_map["mode"] == "chat"
|
||||
assert model_cost_map["supports_reasoning"] is True
|
||||
assert model_cost_map["supports_function_calling"] is True
|
||||
assert model_cost_map["max_input_tokens"] == 1048576
|
||||
|
||||
|
||||
GEMINI_38_FLASH_FIELDS_SHARED_WITH_37_FLASH = (
|
||||
"input_cost_per_token",
|
||||
"output_cost_per_token",
|
||||
"output_cost_per_reasoning_token",
|
||||
"cache_read_input_token_cost",
|
||||
"input_cost_per_token_batches",
|
||||
"output_cost_per_token_batches",
|
||||
"input_cost_per_token_flex",
|
||||
"output_cost_per_token_flex",
|
||||
"cache_read_input_token_cost_flex",
|
||||
"input_cost_per_token_priority",
|
||||
"output_cost_per_token_priority",
|
||||
"cache_read_input_token_cost_priority",
|
||||
"search_context_cost_per_query",
|
||||
"google_maps_grounding_cost_per_query",
|
||||
"prompt_cache_min_tokens",
|
||||
"max_input_tokens",
|
||||
"max_output_tokens",
|
||||
"supports_reasoning",
|
||||
"supports_function_calling",
|
||||
"supports_prompt_caching",
|
||||
"supports_vision",
|
||||
"supports_pdf_input",
|
||||
"supports_audio_input",
|
||||
"supports_video_input",
|
||||
"supports_response_schema",
|
||||
"supports_tool_choice",
|
||||
"supports_web_search",
|
||||
"supports_url_context",
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("prefix", ["", "gemini/", "vertex_ai/"])
|
||||
def test_gemini_38_flash_matches_37_flash_promotional_pricing(prefix, _local_model_cost_map):
|
||||
new_model = litellm.model_cost[f"{prefix}gemini-3.8-flash"]
|
||||
old_model = litellm.model_cost[f"{prefix}gemini-3.7-flash"]
|
||||
for field in GEMINI_38_FLASH_FIELDS_SHARED_WITH_37_FLASH:
|
||||
assert new_model[field] == old_model[field], field
|
||||
|
||||
|
||||
def test_generic_cost_per_token_gemini_38_flash(_local_model_cost_map):
|
||||
usage = Usage(
|
||||
prompt_tokens=1000,
|
||||
|
|
@ -5045,20 +4773,6 @@ def test_generic_cost_per_token_gemini_38_flash(_local_model_cost_map):
|
|||
assert completion_cost == pytest.approx(0.001875)
|
||||
|
||||
|
||||
def test_grok_46_launch_pricing(_local_model_cost_map):
|
||||
model_cost_map = litellm.model_cost["xai/grok-4.6"]
|
||||
assert model_cost_map["input_cost_per_token"] == 2e-06
|
||||
assert model_cost_map["output_cost_per_token"] == 6e-06
|
||||
assert model_cost_map["cache_read_input_token_cost"] == 5e-07
|
||||
assert model_cost_map["input_cost_per_token_above_200k_tokens"] == 4e-06
|
||||
assert model_cost_map["output_cost_per_token_above_200k_tokens"] == 1.2e-05
|
||||
assert model_cost_map["cache_read_input_token_cost_above_200k_tokens"] == 1e-06
|
||||
assert model_cost_map["mode"] == "chat"
|
||||
assert model_cost_map["supports_reasoning"] is True
|
||||
assert model_cost_map["supports_function_calling"] is True
|
||||
assert model_cost_map["max_input_tokens"] == 500000
|
||||
|
||||
|
||||
def test_generic_cost_per_token_grok_46(_local_model_cost_map):
|
||||
usage = Usage(
|
||||
prompt_tokens=1_000,
|
||||
|
|
|
|||
|
|
@ -45,26 +45,6 @@ class TestChatGPTResponsesAPITransformation:
|
|||
assert isinstance(config, ChatGPTResponsesAPIConfig)
|
||||
assert config.custom_llm_provider == LlmProviders.CHATGPT
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_name",
|
||||
[
|
||||
"chatgpt/gpt-5.5",
|
||||
"chatgpt/gpt-5.6-luna",
|
||||
"chatgpt/gpt-5.6-sol",
|
||||
"chatgpt/gpt-5.6-terra",
|
||||
],
|
||||
)
|
||||
def test_chatgpt_responses_model_metadata(self, model_name: str, local_model_cost_map: None) -> None:
|
||||
model_info = litellm.get_model_info(model_name)
|
||||
|
||||
assert model_info["litellm_provider"] == "chatgpt"
|
||||
assert model_info["mode"] == "responses"
|
||||
assert model_info["supported_endpoints"] == [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses",
|
||||
]
|
||||
assert model_info["max_input_tokens"] == 1050000
|
||||
assert model_info["max_output_tokens"] == 128000
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_name",
|
||||
|
|
|
|||
|
|
@ -37,14 +37,6 @@ def test_pricing_entry(cost_map_path: Path, model: str, provider: str) -> None:
|
|||
assert info["ocr_cost_per_page"] == COST_PER_PAGE
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model, provider", MODELS)
|
||||
def test_model_info_resolves_ocr_mode_and_price(local_model_cost_map, model: str, provider: str) -> None:
|
||||
info = litellm.get_model_info(model=model, custom_llm_provider=provider)
|
||||
|
||||
assert info["mode"] == "ocr"
|
||||
assert info["ocr_cost_per_page"] == COST_PER_PAGE
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model, provider", MODELS)
|
||||
@pytest.mark.parametrize("pages_processed", [1, 3])
|
||||
def test_cost_scales_with_billed_pages(local_model_cost_map, model: str, provider: str, pages_processed: int) -> None:
|
||||
|
|
|
|||
|
|
@ -191,18 +191,6 @@ def test_new_models_carry_cache_pricing(local_model_cost_map: None, model: str)
|
|||
assert info["supports_prompt_caching"] is True
|
||||
|
||||
|
||||
def test_every_priced_databricks_model_declares_cache_rates(local_model_cost_map: None) -> None:
|
||||
undeclared: Final = [
|
||||
model
|
||||
for model, info in litellm.model_cost.items()
|
||||
if model.startswith("databricks/")
|
||||
and info.get("input_cost_per_token") is not None
|
||||
and any(info.get(field) is None for field in CACHE_FIELDS)
|
||||
]
|
||||
|
||||
assert undeclared == []
|
||||
|
||||
|
||||
def test_models_without_a_cache_discount_bill_cache_tokens_at_the_input_rate(
|
||||
local_model_cost_map: None,
|
||||
) -> None:
|
||||
|
|
@ -221,24 +209,6 @@ def test_models_without_a_cache_discount_bill_cache_tokens_at_the_input_rate(
|
|||
assert prompt_cost > 8000 * info["input_cost_per_token"]
|
||||
|
||||
|
||||
def test_every_model_without_published_cache_dbu_bills_cache_at_its_own_input_rate(
|
||||
local_model_cost_map: None,
|
||||
) -> None:
|
||||
without_published_rates: Final = [
|
||||
model
|
||||
for model, info in litellm.model_cost.items()
|
||||
if model.startswith("databricks/")
|
||||
and info.get("input_cost_per_token")
|
||||
and model not in PUBLISHED_DBU_PER_MILLION
|
||||
]
|
||||
|
||||
assert len(without_published_rates) == 14
|
||||
for model in without_published_rates:
|
||||
info = _model_info(model)
|
||||
for field in CACHE_FIELDS:
|
||||
assert info[field] == pytest.approx(info["input_cost_per_token"]), (model, field)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", NEW_MODELS)
|
||||
def test_backup_price_map_matches_main(model: str) -> None:
|
||||
main_cost: Final = json.loads(MAIN_PRICES.read_text())
|
||||
|
|
|
|||
|
|
@ -1,65 +0,0 @@
|
|||
"""
|
||||
Regression test for Fireworks Kimi K2.5 / K2.6 / K2.7 context and output limits.
|
||||
|
||||
Fireworks publishes a 262144-token context window for every Kimi K2.5, K2.6 and
|
||||
K2.7 model, but caps generation well below that. A previous bulk edit had flattened
|
||||
max_output_tokens/max_tokens to 262144 (equal to the context window), which let the
|
||||
pre-call context-window check admit requests asking for a full 262144-token
|
||||
completion that Fireworks then rejects. These assertions pin the corrected per-alias
|
||||
limits so a future bulk edit can't silently flatten them again.
|
||||
"""
|
||||
|
||||
import json
|
||||
from importlib.resources import files
|
||||
|
||||
import pytest
|
||||
|
||||
CONTEXT_WINDOW = 262144
|
||||
OUTPUT_LIMIT = 32768
|
||||
|
||||
KIMI_ALIASES = (
|
||||
"fireworks_ai/kimi-k2p5",
|
||||
"fireworks_ai/kimi-k2p6",
|
||||
"fireworks_ai/kimi-k2p6-fast",
|
||||
"fireworks_ai/kimi-k2p7-code",
|
||||
"fireworks_ai/kimi-k2p7-code-fast",
|
||||
"fireworks_ai/accounts/fireworks/models/kimi-k2p5",
|
||||
"fireworks_ai/accounts/fireworks/models/kimi-k2p6",
|
||||
"fireworks_ai/accounts/fireworks/models/kimi-k2p7-code",
|
||||
"fireworks_ai/accounts/fireworks/routers/kimi-k2p6-fast",
|
||||
"fireworks_ai/accounts/fireworks/routers/kimi-k2p7-code-fast",
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def use_local_model_cost_map():
|
||||
monkeypatch = pytest.MonkeyPatch()
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
|
||||
import litellm
|
||||
from litellm.utils import _invalidate_model_cost_lowercase_map
|
||||
|
||||
original_model_cost = litellm.model_cost
|
||||
litellm.model_cost = json.loads(
|
||||
files("litellm")
|
||||
.joinpath("model_prices_and_context_window_backup.json")
|
||||
.read_text(encoding="utf-8")
|
||||
)
|
||||
litellm.get_model_info.cache_clear()
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
try:
|
||||
yield litellm
|
||||
finally:
|
||||
litellm.model_cost = original_model_cost
|
||||
litellm.get_model_info.cache_clear()
|
||||
_invalidate_model_cost_lowercase_map()
|
||||
monkeypatch.undo()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("alias", KIMI_ALIASES)
|
||||
def test_fireworks_kimi_get_model_info_limits(use_local_model_cost_map, alias):
|
||||
model_info = use_local_model_cost_map.get_model_info(model=alias)
|
||||
|
||||
assert model_info["max_input_tokens"] == CONTEXT_WINDOW
|
||||
assert model_info["max_output_tokens"] == OUTPUT_LIMIT
|
||||
assert model_info["max_tokens"] == OUTPUT_LIMIT
|
||||
|
|
@ -63,7 +63,6 @@ def test_ocr4_cost_scales_with_pages(model: str, pages_processed: int) -> None:
|
|||
assert cost == pytest.approx(OCR4_COST_PER_PAGE * pages_processed)
|
||||
|
||||
|
||||
|
||||
@pytest.mark.parametrize("cost_map_path", [MAIN_COST_MAP, BACKUP_COST_MAP])
|
||||
def test_ocr3_pricing_entry(cost_map_path: Path) -> None:
|
||||
with open(cost_map_path) as f:
|
||||
|
|
@ -77,11 +76,6 @@ def test_ocr3_pricing_entry(cost_map_path: Path) -> None:
|
|||
assert info["annotation_cost_per_page"] == OCR3_ANNOTATION_COST_PER_PAGE
|
||||
|
||||
|
||||
def test_ocr3_model_info_price(local_model_cost_map) -> None:
|
||||
info = litellm.get_model_info(model=OCR3_MODEL, custom_llm_provider="mistral")
|
||||
assert info["ocr_cost_per_page"] == OCR3_COST_PER_PAGE
|
||||
|
||||
|
||||
@pytest.mark.parametrize("pages_processed", [1, 3, 10])
|
||||
def test_ocr3_cost_scales_with_pages(local_model_cost_map, pages_processed: int) -> None:
|
||||
cost = completion_cost(
|
||||
|
|
|
|||
|
|
@ -310,43 +310,7 @@ class TestDeclaredEffortList:
|
|||
assert resolved == ("low", "max")
|
||||
|
||||
|
||||
KIMI_K3_PASSTHROUGH_KEYS = (
|
||||
"azure_ai/FW-Kimi-K3",
|
||||
"moonshot/kimi-k3",
|
||||
"together_ai/moonshotai/Kimi-K3",
|
||||
"fireworks_ai/kimi-k3",
|
||||
"fireworks_ai/kimi-k3-fast",
|
||||
"fireworks_ai/kimi-k3-us",
|
||||
"fireworks_ai/accounts/fireworks/models/kimi-k3",
|
||||
"fireworks_ai/accounts/fireworks/routers/kimi-k3-fast",
|
||||
"fireworks_ai/accounts/fireworks/routers/kimi-k3-us",
|
||||
)
|
||||
KIMI_K3_PERPLEXITY_KEY = "perplexity/perplexity/kimi-k3"
|
||||
|
||||
|
||||
class TestKimiK3AdvertisesItsDocumentedLevels:
|
||||
@pytest.mark.parametrize("model_key", KIMI_K3_PASSTHROUGH_KEYS)
|
||||
def test_a_passthrough_entry_advertises_the_models_own_levels(self, local_model_cost_map, model_key):
|
||||
"""platform.kimi.ai documents exactly low, high and max, and these providers forward the
|
||||
level unchanged. Undeclared, each entry resolves to unknown and the dashboard falls back to
|
||||
a capability-blind list that omits max."""
|
||||
entry = dict(litellm.model_cost[model_key], key=model_key)
|
||||
|
||||
assert resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) == ("low", "high", "max")
|
||||
|
||||
def test_the_perplexity_entry_advertises_the_wider_set_it_maps_down(self, local_model_cost_map):
|
||||
"""Perplexity's Agent API takes a six-value enum and maps it down internally, so this
|
||||
deployment is legitimately wider than a passthrough. One blanket list could not say both."""
|
||||
entry = dict(litellm.model_cost[KIMI_K3_PERPLEXITY_KEY], key=KIMI_K3_PERPLEXITY_KEY)
|
||||
|
||||
assert resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) == (
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max",
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize("model, provider", [("kimi-k3", "moonshot"), ("kimi-k3", "fireworks_ai")])
|
||||
def test_the_declaration_survives_model_info_hydration(self, local_model_cost_map, model, provider):
|
||||
|
|
|
|||
|
|
@ -64,14 +64,6 @@ def test_marengo_prices_are_per_request_not_per_token(model):
|
|||
assert info["input_cost_per_audio_per_second"] == AUDIO_COST_PER_SECOND
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ALL_MODELS)
|
||||
def test_marengo_embed_3_is_visible_to_callers(model, local_model_cost_map):
|
||||
info = litellm.get_model_info(model=model, custom_llm_provider="bedrock")
|
||||
assert info["mode"] == "embedding"
|
||||
assert info["output_vector_size"] == 512
|
||||
assert info["max_input_tokens"] == 500
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", PER_REQUEST_MODELS)
|
||||
@pytest.mark.parametrize(
|
||||
"details,expected_cost",
|
||||
|
|
|
|||
|
|
@ -212,67 +212,6 @@ def test_vertex_lyria_speech_cost(
|
|||
assert cost == pytest.approx(expected)
|
||||
|
||||
|
||||
def test_baseten_model_api_pricing_entries(_local_model_cost_map):
|
||||
|
||||
expected_pricing = {
|
||||
"baseten/nvidia/Nemotron-120B-A12B": (3e-07, 7.5e-07),
|
||||
"baseten/MiniMaxAI/MiniMax-M2.5": (3e-07, 1.2e-06),
|
||||
"baseten/zai-org/GLM-5": (9.5e-07, 3.15e-06),
|
||||
"baseten/zai-org/GLM-4.7": (6e-07, 2.2e-06),
|
||||
"baseten/zai-org/GLM-4.6": (6e-07, 2.2e-06),
|
||||
"baseten/moonshotai/Kimi-K2.5": (6e-07, 3e-06),
|
||||
"baseten/moonshotai/Kimi-K2-Thinking": (6e-07, 2.5e-06),
|
||||
"baseten/moonshotai/Kimi-K2-Instruct-0905": (6e-07, 2.5e-06),
|
||||
"baseten/openai/gpt-oss-120b": (1e-07, 5e-07),
|
||||
"baseten/deepseek-ai/DeepSeek-V3.1": (5e-07, 1.5e-06),
|
||||
"baseten/deepseek-ai/DeepSeek-V3-0324": (7.7e-07, 7.7e-07),
|
||||
}
|
||||
|
||||
for model_name, (input_cost, output_cost) in expected_pricing.items():
|
||||
model_info = litellm.model_cost.get(model_name)
|
||||
assert model_info is not None, f"Missing model pricing entry: {model_name}"
|
||||
assert model_info["litellm_provider"] == "baseten"
|
||||
assert model_info["input_cost_per_token"] == input_cost
|
||||
assert model_info["output_cost_per_token"] == output_cost
|
||||
|
||||
|
||||
def test_wandb_model_api_pricing_entries(_local_model_cost_map):
|
||||
|
||||
expected_pricing = {
|
||||
"wandb/moonshotai/Kimi-K2.5": (6e-07, 3e-06),
|
||||
"wandb/MiniMaxAI/MiniMax-M2.5": (3e-07, 1.2e-06),
|
||||
"wandb/Qwen/Qwen3-235B-A22B-Instruct-2507": (1e-07, 1e-07),
|
||||
"wandb/Qwen/Qwen3-235B-A22B-Thinking-2507": (1e-07, 1e-07),
|
||||
"wandb/deepseek-ai/DeepSeek-R1-0528": (1.35e-06, 5.4e-06),
|
||||
"wandb/deepseek-ai/DeepSeek-V3-0324": (1.14e-06, 2.75e-06),
|
||||
"wandb/meta-llama/Llama-4-Scout-17B-16E-Instruct": (1.7e-07, 6.6e-07),
|
||||
}
|
||||
|
||||
for model_name, (input_cost, output_cost) in expected_pricing.items():
|
||||
model_info = litellm.model_cost.get(model_name)
|
||||
assert model_info is not None, f"Missing model pricing entry: {model_name}"
|
||||
assert model_info["litellm_provider"] == "wandb"
|
||||
assert model_info["input_cost_per_token"] == input_cost
|
||||
assert model_info["output_cost_per_token"] == output_cost
|
||||
|
||||
|
||||
def test_openrouter_qwen36_plus_model_info(_local_model_cost_map):
|
||||
|
||||
model_info = litellm.model_cost.get("openrouter/qwen/qwen3.6-plus")
|
||||
|
||||
assert model_info is not None
|
||||
assert model_info["litellm_provider"] == "openrouter"
|
||||
assert model_info["mode"] == "chat"
|
||||
assert model_info["max_input_tokens"] == 1000000
|
||||
assert model_info["max_output_tokens"] == 65536
|
||||
assert model_info["input_cost_per_token"] == 3.25e-07
|
||||
assert model_info["output_cost_per_token"] == 1.95e-06
|
||||
assert model_info["supports_function_calling"] is True
|
||||
assert model_info["supports_tool_choice"] is True
|
||||
assert model_info["supports_reasoning"] is True
|
||||
assert model_info["supports_vision"] is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
|
|
@ -1823,23 +1762,6 @@ def test_azure_ai_cache_cost_calculation(_local_model_cost_map):
|
|||
), f"Output cost mismatch: got {output_cost}, expected {expected_output_cost}"
|
||||
|
||||
|
||||
|
||||
AZURE_GPT_5_6_MAP_KEYS = (
|
||||
"azure/gpt-5.6",
|
||||
"azure/gpt-5.6-sol",
|
||||
"azure/gpt-5.6-terra",
|
||||
"azure/gpt-5.6-luna",
|
||||
"azure/us/gpt-5.6",
|
||||
"azure/us/gpt-5.6-sol",
|
||||
"azure/us/gpt-5.6-terra",
|
||||
"azure/us/gpt-5.6-luna",
|
||||
"azure/eu/gpt-5.6",
|
||||
"azure/eu/gpt-5.6-sol",
|
||||
"azure/eu/gpt-5.6-terra",
|
||||
"azure/eu/gpt-5.6-luna",
|
||||
)
|
||||
|
||||
|
||||
def test_azure_gpt_5_6_cache_write_tokens_are_billed(_local_model_cost_map):
|
||||
"""
|
||||
Azure bills gpt-5.6 prompt cache writes at 1.25x the input rate on every
|
||||
|
|
@ -1865,31 +1787,6 @@ def test_azure_gpt_5_6_cache_write_tokens_are_billed(_local_model_cost_map):
|
|||
assert output_cost == pytest.approx(100 * 1.2e-06)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", AZURE_GPT_5_6_MAP_KEYS)
|
||||
def test_azure_gpt_5_6_rates_match_azure_price_page(_local_model_cost_map, model):
|
||||
"""
|
||||
Per the Azure OpenAI price page (rendered 2026-08-26): cache writes cost
|
||||
1.25x input on every gpt-5.6 tier, and Data Zone costs 1.1x Global for
|
||||
standard and priority alike (us/eu priority rates previously sat at 1.25x).
|
||||
"""
|
||||
entry = litellm.model_cost[model]
|
||||
input_keys = [key for key in entry if key.startswith("input_cost_per_token")]
|
||||
assert input_keys
|
||||
for key in input_keys:
|
||||
suffix = key[len("input_cost_per_token") :]
|
||||
assert entry["cache_creation_input_token_cost" + suffix] == pytest.approx(entry[key] * 1.25)
|
||||
|
||||
zone = model.split("/")[1]
|
||||
if zone in ("us", "eu"):
|
||||
global_entry = litellm.model_cost["azure/" + model.split("/", 2)[2]]
|
||||
prefixes = ("input_cost_per_token", "output_cost_per_token", "cache_read", "cache_creation")
|
||||
token_cost_keys = [key for key in entry if key.startswith(prefixes)]
|
||||
global_token_cost_keys = [key for key in global_entry if key.startswith(prefixes)]
|
||||
assert len(token_cost_keys) >= 9
|
||||
assert sorted(token_cost_keys) == sorted(global_token_cost_keys)
|
||||
for key in token_cost_keys:
|
||||
assert entry[key] == pytest.approx(global_entry[key] * 1.1), key
|
||||
|
||||
def test_vertex_regional_deployment_costs_uplift_over_global(monkeypatch):
|
||||
"""
|
||||
Regression for https://github.com/BerriAI/litellm/issues/34393: two Vertex
|
||||
|
|
@ -3049,28 +2946,6 @@ def test_anthropic_geo_and_fast_multipliers_compose(_local_model_cost_map, monke
|
|||
assert completion_cost == pytest.approx(500 * 25e-6 * 2.0 * 1.1)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,expected_fast",
|
||||
[
|
||||
("claude-opus-5", 2.0),
|
||||
("claude-opus-4-8", 2.0),
|
||||
("claude-opus-4-6", None),
|
||||
("claude-opus-4-6-20260205", None),
|
||||
("claude-opus-4-7", None),
|
||||
("claude-opus-4-7-20260416", None),
|
||||
],
|
||||
)
|
||||
def test_anthropic_fast_multiplier_only_on_models_with_fast_mode(_local_model_cost_map, model, expected_fast):
|
||||
"""
|
||||
Anthropic serves fast mode on Opus 5 and Opus 4.8 only, at 2x. Opus 4.6 and
|
||||
4.7 accept the ``speed`` request param but are always served standard, so a
|
||||
``fast`` multiplier on their map entries overbills every request that asked
|
||||
for fast and was served standard.
|
||||
"""
|
||||
entry = litellm.model_cost[model]
|
||||
assert entry["provider_specific_entry"].get("fast") == expected_fast
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
["claude-sonnet-4-6", "claude-mythos-5", "claude-mythos-preview"],
|
||||
|
|
@ -3323,45 +3198,6 @@ def test_additional_costs_only_for_azure_ai(_local_model_cost_map):
|
|||
assert result is None, "Vertex AI should have no additional costs"
|
||||
|
||||
|
||||
def test_openrouter_gemini_3_1_flash_lite_preview_pricing(_local_model_cost_map):
|
||||
"""
|
||||
Test that openrouter/google/gemini-3.1-flash-lite-preview has a pricing entry.
|
||||
|
||||
Regression test for https://github.com/BerriAI/litellm/issues/25604
|
||||
|
||||
The model exists and is callable via OpenRouter, but was missing from
|
||||
model_prices_and_context_window.json when other Gemini 3.x variants were present.
|
||||
This caused ValueError: This model isn't mapped yet during router pre-call checks.
|
||||
"""
|
||||
|
||||
model_name = "openrouter/google/gemini-3.1-flash-lite-preview"
|
||||
model_info = litellm.model_cost.get(model_name)
|
||||
|
||||
assert model_info is not None, f"Missing model pricing entry: {model_name}"
|
||||
assert model_info["litellm_provider"] == "openrouter"
|
||||
assert model_info["input_cost_per_token"] == 2.5e-07
|
||||
assert model_info["output_cost_per_token"] == 1.5e-06
|
||||
assert model_info["max_input_tokens"] == 1048576
|
||||
assert model_info["max_output_tokens"] == 65536
|
||||
|
||||
|
||||
def test_gemini_3_1_flash_lite_pricing(_local_model_cost_map):
|
||||
|
||||
for model_name in (
|
||||
"gemini-3.1-flash-lite",
|
||||
"gemini/gemini-3.1-flash-lite",
|
||||
"vertex_ai/gemini-3.1-flash-lite",
|
||||
):
|
||||
model_info = litellm.model_cost.get(model_name)
|
||||
assert model_info is not None, f"Missing model pricing entry: {model_name}"
|
||||
assert model_info["input_cost_per_token"] == 2.5e-07
|
||||
assert model_info["input_cost_per_audio_token"] == 5e-07
|
||||
assert model_info["output_cost_per_token"] == 1.5e-06
|
||||
assert model_info["output_cost_per_reasoning_token"] == 1.5e-06
|
||||
assert model_info["cache_read_input_token_cost"] == 2.5e-08
|
||||
assert model_info["max_input_tokens"] == 1048576
|
||||
|
||||
|
||||
def test_custom_pricing_applies_cache_read_input_cost():
|
||||
"""
|
||||
Bug 1 reproduction: custom_cost_per_token with cache_read_input_token_cost
|
||||
|
|
@ -3668,35 +3504,6 @@ def test_custom_pricing_without_cache_keys_preserves_legacy_behavior():
|
|||
assert cost == pytest.approx(expected)
|
||||
|
||||
|
||||
def test_openrouter_gemini_3_1_flash_lite_stable_pricing(_local_model_cost_map):
|
||||
"""
|
||||
Test that openrouter/google/gemini-3.1-flash-lite (stable, no -preview suffix)
|
||||
has a pricing entry.
|
||||
|
||||
Google promoted gemini-3.1-flash-lite to GA on 2026-05-07. PR #27933 added the
|
||||
stable pricing for the bare, gemini/, and vertex_ai/ prefixes but missed the
|
||||
openrouter/google/ variant — every other Gemini family in the file has an
|
||||
openrouter/google/ sibling (2.0-flash-001, 2.5-flash, 2.5-pro, 3-flash-preview,
|
||||
3-pro-preview, 3.1-flash-lite-preview, 3.1-pro-preview), so the gap is a
|
||||
consistency issue, not a design choice. Same shape as the preview-variant gap
|
||||
fixed in PR #25610.
|
||||
|
||||
Pricing matches the existing -preview entry one-for-one (input $0.25/M, output
|
||||
$1.50/M, cache-read $0.025/M) — Google did not change costs at the GA cutover.
|
||||
"""
|
||||
|
||||
model_name = "openrouter/google/gemini-3.1-flash-lite"
|
||||
model_info = litellm.model_cost.get(model_name)
|
||||
|
||||
assert model_info is not None, f"Missing model pricing entry: {model_name}"
|
||||
assert model_info["litellm_provider"] == "openrouter"
|
||||
assert model_info["input_cost_per_token"] == 2.5e-07
|
||||
assert model_info["output_cost_per_token"] == 1.5e-06
|
||||
assert model_info["cache_read_input_token_cost"] == 2.5e-08
|
||||
assert model_info["max_input_tokens"] == 1048576
|
||||
assert model_info["max_output_tokens"] == 65536
|
||||
|
||||
|
||||
def test_completion_cost_logs_reasoning_and_cache_breakdown(_local_model_cost_map):
|
||||
"""
|
||||
completion_cost must surface explicit reasoning and cache-read costs into the
|
||||
|
|
|
|||
|
|
@ -166,13 +166,6 @@ def test_vertex_prefix_routes_to_vertex():
|
|||
assert provider == "vertex_ai"
|
||||
|
||||
|
||||
def test_get_model_info_reports_published_costs(local_model_cost_map):
|
||||
info = litellm.get_model_info(UNPREFIXED)
|
||||
assert info["input_cost_per_token"] == INPUT_COST
|
||||
assert info["output_cost_per_token"] == OUTPUT_TEXT_COST
|
||||
assert info["cache_read_input_token_cost"] == CACHE_READ_COST
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ALL_KEYS)
|
||||
def test_reasoning_params_are_not_offered_on_an_image_endpoint(model: str, local_model_cost_map):
|
||||
assert litellm.supports_reasoning(model) is False
|
||||
|
|
|
|||
|
|
@ -63,18 +63,6 @@ def test_zai_glm_5_2_specs(model):
|
|||
assert provider == "mistral"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", GLM_5_2_MODELS)
|
||||
def test_zai_glm_5_2_capabilities_are_visible_to_callers(local_model_cost_map, model):
|
||||
"""Mistral advertises reasoning and prompt caching on this model, so the helpers
|
||||
every caller checks before sending a request must say so too."""
|
||||
assert supports_reasoning(model=model) is True
|
||||
assert supports_prompt_caching(model=model) is True
|
||||
|
||||
info = litellm.get_model_info(model=model)
|
||||
assert info["max_input_tokens"] == 1048576
|
||||
assert info["max_output_tokens"] == 131072
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", GLM_5_2_MODELS)
|
||||
def test_cached_prompt_tokens_bill_at_the_cached_rate(local_model_cost_map, model):
|
||||
"""A cache hit reports its reused tokens under prompt_tokens_details, and those
|
||||
|
|
|
|||
|
|
@ -94,12 +94,6 @@ def test_non_ocr_wrapper_preserves_logging_executor_and_context(monkeypatch: pyt
|
|||
marker.reset(token)
|
||||
|
||||
|
||||
def test_cloudflare_model_info_includes_rpm(local_model_cost_map: None) -> None:
|
||||
assert litellm.get_model_info("cloudflare/@cf/meta/llama-3.1-8b-instruct-fp8")["rpm"] == 300
|
||||
assert litellm.get_model_info("cloudflare/@cf/moonshotai/kimi-k2.6")["rpm"] == 20
|
||||
assert litellm.get_model_info("cloudflare/@cf/openai/whisper-large-v3-turbo")["rpm"] == 720
|
||||
|
||||
|
||||
def test_get_utc_datetime_returns_current_aware_utc_time() -> None:
|
||||
before: Final = datetime.now(timezone.utc)
|
||||
result: Final = litellm.utils.get_utc_datetime()
|
||||
|
|
@ -160,7 +154,6 @@ def test_prompt_tokens_details_cache_write_creation_stay_in_sync_on_assignment()
|
|||
assert details.cache_write_tokens == details.cache_creation_tokens == 375
|
||||
|
||||
|
||||
|
||||
def test_get_model_info_surfaces_supports_adaptive_thinking(local_model_cost_map):
|
||||
"""supports_adaptive_thinking must flow through get_model_info like every other
|
||||
capability flag: both from an explicit cost-map entry and from a
|
||||
|
|
@ -177,7 +170,6 @@ def test_get_model_info_surfaces_supports_adaptive_thinking(local_model_cost_map
|
|||
assert generalized["supports_adaptive_thinking"] is True
|
||||
|
||||
|
||||
|
||||
def test_get_model_info_surfaces_supports_parallel_function_calling(local_model_cost_map):
|
||||
"""A registry entry's supports_parallel_function_calling must read back through get_model_info
|
||||
and litellm.supports_parallel_function_calling. Regression: the key was never copied into
|
||||
|
|
@ -493,64 +485,6 @@ def test_gpt_image_provider_detection_covers_existing_family():
|
|||
assert custom_llm_provider == "openai"
|
||||
|
||||
|
||||
def test_gpt_image_2_provider_and_model_info(local_model_cost_map):
|
||||
|
||||
model, custom_llm_provider, _, _ = litellm.get_llm_provider(model="gpt-image-2")
|
||||
|
||||
assert model == "gpt-image-2"
|
||||
assert custom_llm_provider == "openai"
|
||||
|
||||
model_info = litellm.get_model_info(model="gpt-image-2")
|
||||
assert model_info["litellm_provider"] == "openai"
|
||||
assert model_info["mode"] == "image_generation"
|
||||
assert model_info["input_cost_per_token"] == 5e-06
|
||||
assert model_info["input_cost_per_image_token"] == 8e-06
|
||||
assert model_info["output_cost_per_token"] == 0
|
||||
assert model_info["output_cost_per_image_token"] == 3e-05
|
||||
assert (
|
||||
"/v1/images/generations"
|
||||
in litellm.model_cost["gpt-image-2"]["supported_endpoints"]
|
||||
)
|
||||
assert (
|
||||
"/v1/images/edits" in litellm.model_cost["gpt-image-2"]["supported_endpoints"]
|
||||
)
|
||||
assert model_info["supports_vision"] is True
|
||||
assert model_info["supports_pdf_input"] is True
|
||||
|
||||
|
||||
def test_gpt_image_2_snapshot_model_info(local_model_cost_map):
|
||||
model, custom_llm_provider, _, _ = litellm.get_llm_provider(
|
||||
model="gpt-image-2-2026-04-21"
|
||||
)
|
||||
|
||||
assert model == "gpt-image-2-2026-04-21"
|
||||
assert custom_llm_provider == "openai"
|
||||
|
||||
model_info = litellm.get_model_info(model="gpt-image-2-2026-04-21")
|
||||
assert model_info["litellm_provider"] == "openai"
|
||||
assert model_info["mode"] == "image_generation"
|
||||
assert model_info["output_cost_per_image_token"] == 3e-05
|
||||
|
||||
|
||||
def test_azure_gpt_image_2_model_info(local_model_cost_map):
|
||||
model, custom_llm_provider, _, _ = litellm.get_llm_provider(
|
||||
model="azure/gpt-image-2"
|
||||
)
|
||||
|
||||
assert model == "gpt-image-2"
|
||||
assert custom_llm_provider == "azure"
|
||||
|
||||
model_info = litellm.get_model_info(
|
||||
model="gpt-image-2", custom_llm_provider="azure"
|
||||
)
|
||||
assert model_info["litellm_provider"] == "azure"
|
||||
assert model_info["mode"] == "image_generation"
|
||||
assert model_info["input_cost_per_token"] == 5e-06
|
||||
assert model_info["input_cost_per_image_token"] == 8e-06
|
||||
assert model_info["output_cost_per_token"] == 0
|
||||
assert model_info["output_cost_per_image_token"] == 3e-05
|
||||
|
||||
|
||||
def test_all_model_configs():
|
||||
from litellm.llms.vertex_ai.vertex_ai_partner_models.ai21.transformation import (
|
||||
VertexAIAi21Config,
|
||||
|
|
@ -4850,36 +4784,6 @@ class TestBedrockCohereEmbeddingDispatch:
|
|||
assert optional_params.get("output_dimension") == 512
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
"vertex_ai/gemini-2.5-flash-image",
|
||||
"vertex_ai/gemini-3-pro-image",
|
||||
"vertex_ai/gemini-3-pro-image-preview",
|
||||
"vertex_ai/gemini-3.1-flash-image",
|
||||
"vertex_ai/gemini-3.1-flash-image-preview",
|
||||
"vertex_ai/gemini-3.1-flash-lite-image",
|
||||
"gemini/gemini-2.5-flash-image",
|
||||
"gemini/gemini-3-pro-image",
|
||||
"gemini/gemini-3-pro-image-preview",
|
||||
"gemini/gemini-3.1-flash-image",
|
||||
"gemini/gemini-3.1-flash-image-preview",
|
||||
"gemini/gemini-3.1-flash-lite-image",
|
||||
],
|
||||
)
|
||||
def test_gemini_image_models_do_not_support_reasoning(
|
||||
model: str, local_model_cost_map: None
|
||||
) -> None:
|
||||
assert model in litellm.model_cost, (
|
||||
f"{model} is missing from the local model cost map. "
|
||||
"Add its entry to litellm/model_prices_and_context_window_backup.json."
|
||||
)
|
||||
assert litellm.supports_reasoning(model) is False, (
|
||||
f"{model} incorrectly classified as reasoning-capable. "
|
||||
"Add 'supports_reasoning: false' to its model_cost entry."
|
||||
)
|
||||
|
||||
|
||||
PROMPT_CACHE_MESSAGES = [{"role": "user", "content": "the quick brown fox jumps over the lazy dog " * 155}]
|
||||
|
||||
|
||||
|
|
@ -4901,21 +4805,6 @@ def test_get_prompt_cache_min_tokens_resolves_per_model(
|
|||
assert get_prompt_cache_min_tokens(model=model) == expected_min_tokens
|
||||
|
||||
|
||||
def test_get_prompt_cache_min_tokens_uniform_for_fable_5_across_platforms(local_model_cost_map: None) -> None:
|
||||
"""Anthropic removed the Amazon Bedrock override for Claude Fable 5, so its 512-token minimum
|
||||
now applies on every platform. The Bedrock entries carried the old 1024 and the re-export
|
||||
entries carried nothing, so the router judged 512-1023-token prefixes uncacheable and skipped
|
||||
prompt-cache-affinity routing for prompts the provider demonstrably caches (issue #35011)."""
|
||||
wrong: Final = {
|
||||
model: get_prompt_cache_min_tokens(model=model)
|
||||
for model, info in litellm.model_cost.items()
|
||||
if "fable-5" in model
|
||||
and info.get("supports_prompt_caching")
|
||||
and get_prompt_cache_min_tokens(model=model) != 512
|
||||
}
|
||||
assert not wrong, f"every Claude Fable 5 entry must carry prompt_cache_min_tokens 512: {wrong}"
|
||||
|
||||
|
||||
ANTHROPIC_REEXPORT_CACHE_MIN: Final = {
|
||||
"azure_ai/claude-fable-5": 512,
|
||||
"azure_ai/claude-haiku-4-5": 4096,
|
||||
|
|
@ -4964,21 +4853,6 @@ ANTHROPIC_REEXPORT_CACHE_MIN: Final = {
|
|||
}
|
||||
|
||||
|
||||
def test_anthropic_reexport_entries_carry_explicit_prompt_cache_min_tokens(local_model_cost_map: None) -> None:
|
||||
"""Regression for issue #35011: these re-export entries carried no prompt_cache_min_tokens, so
|
||||
they silently inherited the 1024 default. That skipped cache-affinity routing for Fable 5's
|
||||
512-1023-token prefixes and reported 1024-4095-token prompts as cacheable on the 2048/4096
|
||||
models. The entry must be explicit so a default change can never re-break them, which is why
|
||||
this asserts the cost-map value itself and not just the resolver's answer."""
|
||||
wrong: Final = {
|
||||
model: (litellm.model_cost[model].get("prompt_cache_min_tokens"), get_prompt_cache_min_tokens(model=model))
|
||||
for model, expected in ANTHROPIC_REEXPORT_CACHE_MIN.items()
|
||||
if litellm.model_cost[model].get("prompt_cache_min_tokens") != expected
|
||||
or get_prompt_cache_min_tokens(model=model) != expected
|
||||
}
|
||||
assert not wrong, f"(cost-map value, resolved value) diverge from Anthropic's published minimums: {wrong}"
|
||||
|
||||
|
||||
def test_anthropic_reexport_cache_minimums_present_in_root_cost_map() -> None:
|
||||
"""The root map ships to the CDN independently of the bundled backup, so both must carry the
|
||||
minimum or proxies reading one of them regress to the 1024 default."""
|
||||
|
|
@ -6508,7 +6382,6 @@ async def test_async_mock_completion_streaming_obj_raises_mock_exception_before_
|
|||
await _async_mock_stream_snapshots(mock_exception, 51234)
|
||||
|
||||
|
||||
|
||||
@contextlib.contextmanager
|
||||
def _recording_hidden_params_at_submit(submit_target: str) -> "Iterator[queue.SimpleQueue[dict[str, object]]]":
|
||||
seen: Final = queue.SimpleQueue()
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue