test: keep behavior tests that read the cost map for a later fixture rewrite

Fifty six of the deleted tests turn out to assert the output of litellm code rather than the catalog lookup itself, things like map_openai_params, get_supported_openai_params, should_fake_stream, transform_request bodies, cost_per_token arithmetic, get_llm_provider routing, and provider config dispatch. They only happen to read shipped entries as inputs, so they belong in the later rewrite that injects a local model_cost, not in this deletion

Each one is restored verbatim from origin/main along with the fixtures, helpers, constants and imports it needs, and tests/test_litellm/test_sambanova_model_metadata.py is restored wholesale

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
kerry 2026-09-18 04:52:06 +00:00
parent d2ac51893b
commit 7975987107
36 changed files with 1605 additions and 0 deletions

View file

@ -41,6 +41,36 @@ class TestAzureOpenAIO3Mini(BaseOSeriesModelsTest, BaseLLMChatTest):
"""Temporary override. o1 prompt caching is not working."""
pass
def test_override_fake_stream(self):
"""Test that native streaming is not supported for o1."""
router = litellm.Router(
model_list=[
{
"model_name": "azure/o1-preview",
"litellm_params": {
"model": "azure/o1-preview",
"api_key": "my-fake-o1-key",
"api_base": "https://openai-gpt-4-test-v-1.openai.azure.com",
},
"model_info": {
"supports_native_streaming": True,
},
}
]
)
## check model info
model_info = litellm.get_model_info(
model="azure/o1-preview", custom_llm_provider="azure"
)
assert model_info["supports_native_streaming"] is True
fake_stream = litellm.AzureOpenAIO1Config().should_fake_stream(
model="azure/o1-preview", stream=True
)
assert fake_stream is False
class TestAzureOpenAIO3(BaseOSeriesModelsTest):
def get_base_completion_call_args(self):

View file

@ -1586,11 +1586,40 @@ class TestEnableAnthropicPromptCaching:
points = self._points(model="us.anthropic.claude-sonnet-4-5-20250929-v1:0", provider="bedrock")
assert [p["index"] for p in points] == [None, -1]
@pytest.mark.parametrize("model, provider", [("gpt-4o", "openai"), ("gemini-2.0-flash", "gemini")])
def test_non_anthropic_providers_never_injected(self, monkeypatch, model, provider):
"""These report supports_prompt_caching=True but never consume cache_control markers."""
from litellm.utils import supports_prompt_caching
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
assert supports_prompt_caching(model=model, custom_llm_provider=provider) is True
assert self._points(model=model, provider=provider) == []
def test_databricks_claude_not_injected_despite_caching_support(self, monkeypatch, local_model_cost_map):
from litellm.utils import supports_prompt_caching
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
model = "databricks/databricks-claude-sonnet-4-5"
assert supports_prompt_caching(model=model, custom_llm_provider="databricks") is True
assert self._points(model=model, provider="databricks") == []
def test_model_without_caching_support_not_injected(self, monkeypatch):
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
assert self._points(model="anthropic.claude-3-5-sonnet-20240620-v1:0", provider="bedrock") == []
@pytest.mark.parametrize("model", ["us.xai.grok-4.6", "global.xai.grok-4.6"])
def test_bedrock_grok_not_injected(self, monkeypatch, local_model_cost_map, model):
"""Bedrock supports only implicit prompt caching for Grok: explicit cachePoint
breakpoints make it reject the whole request ("You invoked an unsupported model
or your request did not allow prompt caching"), so supports_prompt_caching stays
false, while implicit cache hits still bill at the cache-read rate."""
from litellm.utils import supports_prompt_caching
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
assert supports_prompt_caching(model=model, custom_llm_provider="bedrock") is False
assert self._points(model=model, provider="bedrock") == []
entry = litellm.model_cost[model]
assert 0 < entry["cache_read_input_token_cost"] < entry["input_cost_per_token"]
def test_stands_down_when_client_sent_cache_control(self, monkeypatch):
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)

View file

@ -361,6 +361,58 @@ def test_completion_cost_includes_web_search_without_standard_built_in_tools_par
), f"completion_cost ({cost}) should include web search cost ({web_search_cost})"
@pytest.mark.parametrize(
"model",
[
"vertex_ai/gemini-3.1-flash-lite", # resolves directly via get_model_info
"gemini/gemini-3.1-flash-lite", # provider-prefixed, resolves via model_cost fallback
],
)
def test_gemini_3x_web_search_billed_per_query(model, local_model_cost_map):
"""
Gemini 3.x bills web search per individual query (web_search_billing_unit == "per_query"),
so N searches cost N * $0.014.
Regression for the bug where the billing unit was dropped between the pricing JSON and the
cost calculator: the field was missing from the ModelInfoBase TypedDict and from the
ModelInfoBase(...) constructor in _get_model_info_helper, so get_model_info returned it as
None and cost_per_web_search_request fell back to the per_prompt clamp, collapsing N queries
to a single charge. The "gemini/..." case additionally covers response_cost_calculator
resolving a provider-prefixed model name that get_model_info cannot map under vertex_ai.
"""
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
web_search_requests = 2
model_info = litellm.get_model_info(model)
assert model_info["web_search_billing_unit"] == "per_query"
per_query_cost = model_info["search_context_cost_per_query"][
"search_context_size_medium"
]
expected_cost = per_query_cost * web_search_requests
usage = Usage(
prompt_tokens=11,
completion_tokens=100,
total_tokens=111,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=11, web_search_requests=web_search_requests
),
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model=model,
usage=usage,
response_object=None,
custom_llm_provider="vertex_ai",
standard_built_in_tools_params=None,
)
assert cost == pytest.approx(expected_cost), (
f"Expected {web_search_requests} x ${per_query_cost} = ${expected_cost} "
f"per_query search fee, got ${cost}"
)
def test_gemini_combined_search_and_maps_costs_are_additive(local_model_cost_map):
"""A prompt grounded with both Google Search and Google Maps pays both fees."""
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
@ -388,6 +440,86 @@ def test_gemini_combined_search_and_maps_costs_are_additive(local_model_cost_map
assert cost == pytest.approx(search_rate * 2 + maps_rate)
def test_gemini_2x_web_search_still_billed_per_prompt(local_model_cost_map):
"""
Gemini 2.x bills web search per grounded prompt: multiple internal queries are one flat
$0.035 fee. Guards the per_prompt clamp against the per_query plumbing, which makes
web_search_billing_unit always present on the resolved ModelInfo (None for 2.x), so the
clamp must treat a None billing unit as per_prompt rather than skipping the clamp.
"""
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
model = "vertex_ai/gemini-2.5-flash"
model_info = litellm.get_model_info(model)
assert not model_info.get("web_search_billing_unit")
expected_cost = model_info["search_context_cost_per_query"][
"search_context_size_medium"
]
usage = Usage(
prompt_tokens=11,
completion_tokens=100,
total_tokens=111,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=11, web_search_requests=2
),
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model=model,
usage=usage,
response_object=None,
custom_llm_provider="vertex_ai",
standard_built_in_tools_params=None,
)
assert cost == pytest.approx(expected_cost), (
f"Expected flat ${expected_cost} per_prompt search fee (2 queries clamped to 1), "
f"got ${cost}"
)
def test_web_search_provider_prefix_fallback_does_not_misprice_non_gemini_model(
local_model_cost_map,
):
"""
Regression for the provider-prefix fallback in _handle_web_search_cost. When the initial
get_model_info lookup fails for a "/"-containing model, the retry re-resolves model_info from
the prefix and must adopt that prefix's provider for routing. Otherwise an unrelated model
(here OpenRouter, which carries no web search pricing) is re-resolved but still routed through
the request's vertex_ai Gemini calculator, which charges its $0.035 per_prompt default for a
model that should cost nothing for web search.
"""
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
model = "openrouter/google/gemini-3.1-flash-lite"
model_info = litellm.get_model_info(model)
assert model_info["litellm_provider"] == "openrouter"
assert not model_info.get("search_context_cost_per_query")
usage = Usage(
prompt_tokens=11,
completion_tokens=100,
total_tokens=111,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=11, web_search_requests=2
),
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model=model,
usage=usage,
response_object=None,
custom_llm_provider="vertex_ai",
standard_built_in_tools_params=None,
)
assert cost == 0.0, (
"A non-Gemini provider-prefixed model with no web search pricing must not be charged "
f"the vertex_ai per_prompt default via the prefix fallback, got ${cost}"
)
def _openai_responses_with_web_search_calls(model, num_calls):
from openai.types.responses.response_function_web_search import (
ActionSearch,

View file

@ -3037,6 +3037,28 @@ def test_add_cache_point_tool_block_passes_ttl_for_claude_4_5(monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", old_env)
def test_add_cache_point_tool_block_stands_down_for_model_without_prompt_caching(monkeypatch):
"""A tool carrying cache_control must not become a cachePoint for a Bedrock model
whose cost-map entry lacks prompt caching support, since Bedrock rejects the whole
request. An unmapped id keeps emitting so ARN deployments do not lose caching."""
from litellm.litellm_core_utils.prompt_templates.factory import (
add_cache_point_tool_block,
)
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
tool = {"cache_control": {"type": "ephemeral"}}
assert add_cache_point_tool_block(tool, model="nvidia.nemotron-super-3-120b") is None
assert add_cache_point_tool_block(tool, model="us.nvidia.nemotron-super-3-120b") is None
assert add_cache_point_tool_block(
tool, model="arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/abc123"
) == {"cachePoint": {"type": "default"}}
assert add_cache_point_tool_block(tool, model="us.anthropic.claude-sonnet-4-5-20250929-v1:0") == {
"cachePoint": {"type": "default"}
}
def test_bedrock_tools_pt_passes_ttl_for_claude_4_5(monkeypatch):
"""
End-to-end: _bedrock_tools_pt should produce cachePoint blocks with ttl

View file

@ -802,6 +802,10 @@ def test_shipped_rules_flag_unmapped_wandb_ids_as_reasoning(shipped_cost_map):
assert litellm.supports_reasoning(model="zai-org/GLM-6-Turbo", custom_llm_provider="wandb") is True
def test_shipped_wandb_rule_does_not_fill_missing_mapped_entries(shipped_cost_map):
assert match_fill_missing_generalizations("wandb/meta-llama/Llama-3.1-8B-Instruct", "wandb") is None
def test_shipped_wandb_rule_is_anchored_to_the_wandb_namespace(shipped_cost_map):
"""``^wandb/`` is anchored, so it cannot leak onto another provider's ids."""
assert match_capability_generalizations("wandb/some-new-model") == {"supports_reasoning": True}
@ -900,6 +904,25 @@ def test_shipped_openai_reasoning_rule_matches_only_openai(shipped_cost_map):
assert match_fill_missing_generalizations("gpt-5.4", "openrouter") is None
def test_shipped_claude_thinking_rules_backfill_only_anthropic(shipped_cost_map):
model = "perplexity/anthropic/claude-sonnet-4-6"
assert model in litellm.model_cost
raw_entry = litellm.model_cost[model]
assert "supports_adaptive_thinking" not in raw_entry
assert "max_input_tokens" not in raw_entry
info = litellm.get_model_info(model="anthropic/claude-sonnet-4-6", custom_llm_provider="perplexity")
assert info.get("supports_adaptive_thinking") is None
assert info.get("supports_legacy_thinking") is None
assert info.get("max_input_tokens") is None
assert match_fill_missing_generalizations("claude-sonnet-4-6", "anthropic") == {
"supports_adaptive_thinking": True,
"supports_legacy_thinking": True,
"supports_tool_search": True,
}
assert match_fill_missing_generalizations("claude-sonnet-4-6", "perplexity") is None
@pytest.mark.parametrize(
"model,provider,tool_search",
[
@ -924,3 +947,25 @@ def test_shipped_tool_search_rule_version_boundaries(shipped_cost_map, model, pr
assert info.get("supports_tool_search") is tool_search, model
def test_shipped_tool_search_rule_fills_mapped_claude_entries_without_flag(shipped_cost_map):
"""A mapped Claude 4.5+ entry with no supports_tool_search key gets it from the rule
on Anthropic direct, Vertex and Bedrock, a mapped pre-4.5 entry stays without one,
and Azure Foundry and reseller copies of the same model are not touched."""
for key, model, provider in (
("claude-opus-4-7", "claude-opus-4-7", "anthropic"),
("vertex_ai/claude-opus-5", "claude-opus-5", "vertex_ai"),
):
assert "supports_tool_search" not in litellm.model_cost[key]
assert litellm.get_model_info(model, custom_llm_provider=provider)["supports_tool_search"] is True
assert "supports_tool_search" not in litellm.model_cost["claude-opus-4-1"]
opus_4_1_info = litellm.get_model_info("claude-opus-4-1", custom_llm_provider="anthropic")
assert opus_4_1_info.get("supports_tool_search") is None
assert "supports_tool_search" not in litellm.model_cost["azure_ai/claude-opus-5"]
azure_opus_5_info = litellm.get_model_info("claude-opus-5", custom_llm_provider="azure_ai")
assert azure_opus_5_info.get("supports_tool_search") is None
assert match_fill_missing_generalizations("claude-opus-5", "bedrock")["supports_tool_search"] is True
assert "supports_tool_search" not in match_fill_missing_generalizations("claude-opus-5", "azure_ai")
assert match_fill_missing_generalizations("claude-opus-5", "perplexity") is None

View file

@ -396,6 +396,47 @@ class TestGetRouterDeploymentModelInfo:
assert logging_obj.get_router_deployment_model_info() is None
def test_a_published_batch_rate_never_displaces_a_declared_standard_rate(self) -> None:
"""Ownership is per token direction, not per field.
Filling the batch field from the published entry let that rate win, so a
deployment configuring only its standard rate had batches billed at the
published batch price instead of half the rate it configured.
"""
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
model = "ft:gpt-3.5-turbo"
published = litellm.get_model_info(model=model)
assert published["input_cost_per_token_batches"] is not None
deployment_id = "deploy-standard-input-only-1"
litellm.model_cost[deployment_id] = {
"id": deployment_id,
"input_cost_per_token": 1e-06,
"litellm_provider": "openai",
"mode": "chat",
}
obj = LiteLLMLoggingObj(
model=model,
messages=[],
stream=False,
call_type="aretrieve_batch",
start_time=time.time(),
litellm_call_id="direction-ownership",
function_id="f",
)
obj.litellm_params = {"litellm_metadata": {"model_info": {"id": deployment_id}}, "model": model}
obj.model_call_details["model"] = model
try:
info = obj.get_router_deployment_model_info()
assert info is not None
assert info["input_cost_per_token"] == 1e-06
assert info["input_cost_per_token_batches"] is None
assert info["output_cost_per_token"] == published["output_cost_per_token"]
assert info["output_cost_per_token_batches"] == published["output_cost_per_token_batches"]
finally:
litellm.model_cost.pop(deployment_id, None)
def test_merging_does_not_mutate_the_cached_model_info(self) -> None:
"""The published-rate merge must not write into get_model_info's lru-cached dict.

View file

@ -1120,6 +1120,45 @@ def test_stream_chunk_builder_tolerates_trailing_chunk_without_choices():
assert response.choices[0].message.content == "Hello world"
def test_anthropic_speed_and_geo_survive_stream_assembly():
"""Anthropic prices fast mode and non-global regions with a multiplier read off
``usage.speed`` / ``usage.inference_geo``. Dropping them while reassembling a stream
bills streamed fast-mode calls at the standard rate."""
from litellm.llms.anthropic.cost_calculation import cost_per_token
def _usage(**extra):
usage = Usage(completion_tokens=100, prompt_tokens=1000, total_tokens=1100)
for key, value in extra.items():
setattr(usage, key, value)
return usage
def _chunk(usage):
return ModelResponseStream(
id="chatcmpl-1",
created=1745513206,
model="claude-opus-4-8",
choices=[StreamingChoices(finish_reason="stop", index=0, delta=Delta(content="Hi"))],
usage=usage,
)
fast_chunk = _chunk(_usage(speed="fast", inference_geo="global"))
fast_usage = ChunkProcessor(chunks=[fast_chunk]).calculate_usage(
chunks=[fast_chunk], model="claude-opus-4-8", completion_output="Hi"
)
standard_chunk = _chunk(_usage(inference_geo="global"))
standard_usage = ChunkProcessor(chunks=[standard_chunk]).calculate_usage(
chunks=[standard_chunk], model="claude-opus-4-8", completion_output="Hi"
)
assert fast_usage.speed == "fast"
assert fast_usage.inference_geo == "global"
assert getattr(standard_usage, "speed", None) is None
fast_cost = sum(cost_per_token(model="claude-opus-4-8", usage=fast_usage))
standard_cost = sum(cost_per_token(model="claude-opus-4-8", usage=standard_usage))
assert fast_cost == pytest.approx(standard_cost * 2.0)
def test_prompt_tokens_details_survive_later_usage_chunk_without_details():
"""Regression for #34801: a trailing usage chunk that omits
`prompt_tokens_details` must not wipe the OpenAI cache-read/cache-write split,

View file

@ -2464,6 +2464,42 @@ def test_get_max_tokens_for_model_none():
assert max_tokens == 4096
def test_get_config_with_model_uses_dynamic_max_tokens():
"""
Test that get_config returns dynamic max_tokens based on model.
Fixes: https://github.com/BerriAI/litellm/issues/8835
"""
def _mock_get_max_tokens(model):
"""Return expected max_output_tokens for each model."""
model_map = {
"claude-3-sonnet-20240229": 4096,
"claude-3-5-sonnet-20241022": 8192,
"claude-3-7-sonnet-20250219": 64000,
}
result = model_map.get(model)
if result is None:
raise Exception(f"Model {model} not found")
return result
with patch(
"litellm.llms.anthropic.chat.transformation.get_max_tokens",
side_effect=_mock_get_max_tokens,
):
# Claude 3 model should get 4096
config_claude3 = AnthropicConfig.get_config(model="claude-3-sonnet-20240229")
assert config_claude3["max_tokens"] == 4096
# Claude 3.5 model should get 8192
config_claude35 = AnthropicConfig.get_config(model="claude-3-5-sonnet-20241022")
assert config_claude35["max_tokens"] == 8192
# Claude 3.7 model should get 64000 (64K default, 128K requires beta header)
config_claude37 = AnthropicConfig.get_config(model="claude-3-7-sonnet-20250219")
assert config_claude37["max_tokens"] == 64000
def test_get_config_without_model_uses_fallback():
"""
Test that get_config without model parameter uses 4096 fallback.
@ -3166,6 +3202,27 @@ def test_max_effort_accepted_for_opus_47():
assert result["output_config"]["effort"] == "max"
def test_effort_beta_header_not_injected_for_46_models():
"""
Test that is_effort_used returns False for Claude 4.6 models.
Claude 4.6 models use output_config as a stable API feature —
no beta header should be injected.
"""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
model_info = AnthropicModelInfo()
for model in ["claude-opus-4-6-20250514", "claude-sonnet-4-6-20260219"]:
# Even with output_config present, should return False for 4.6 models
result = model_info.is_effort_used(
optional_params={"output_config": {"effort": "high"}},
model=model,
custom_llm_provider="anthropic",
)
assert result is False, f"is_effort_used should return False for {model}"
@pytest.mark.parametrize(
"model",
[
@ -3261,6 +3318,23 @@ def test_reasoning_effort_minimal_floors_at_anthropic_provider_minimum():
assert result["thinking"]["budget_tokens"] >= 1024
def test_effort_beta_header_still_injected_for_older_models():
"""
Test that is_effort_used still returns True for pre-4.6 models
when output_config is present.
"""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
model_info = AnthropicModelInfo()
result = model_info.is_effort_used(
optional_params={"output_config": {"effort": "low"}},
model="claude-opus-4-5-20251101",
custom_llm_provider="anthropic",
)
assert result is True
def test_code_execution_tool_results_extraction():
"""
Test that code execution tool results (bash_code_execution_tool_result,
@ -4075,6 +4149,48 @@ def test_fast_mode_usage_calculation():
assert usage.speed == "fast"
def test_fast_mode_cost_calculation():
"""
Test that fast mode applies the 'fast' multiplier from provider_specific_entry
on top of the base model cost (1.1x for claude-opus-4-6).
"""
from litellm.llms.anthropic.cost_calculation import cost_per_token
from litellm.types.utils import Usage
base_prompt = 0.005
base_completion = 0.025
with (
patch(
"litellm.llms.anthropic.cost_calculation.generic_cost_per_token"
) as mock_cost,
patch("litellm.get_model_info") as mock_info,
):
mock_cost.return_value = (base_prompt, base_completion)
mock_info.return_value = {"provider_specific_entry": {"fast": 1.1, "us": 1.1}}
usage_fast = Usage(
prompt_tokens=1000,
completion_tokens=1000,
speed="fast",
)
prompt_cost, completion_cost = cost_per_token(
model="claude-opus-4-6",
usage=usage_fast,
)
# generic_cost_per_token called with the plain base model name
mock_cost.assert_called_once()
assert mock_cost.call_args[1]["model"] == "claude-opus-4-6"
assert mock_cost.call_args[1]["custom_llm_provider"] == "anthropic"
# 1.1x multiplier applied
assert abs(prompt_cost - base_prompt * 1.1) < 1e-10
assert abs(completion_cost - base_completion * 1.1) < 1e-10
def test_fast_mode_with_inference_geo():
"""
Test that fast mode + inference_geo both apply their multipliers from
@ -5929,6 +6045,35 @@ def test_sampling_params_forwarded_on_models_that_accept_them(model):
assert result["top_p"] == 0.9
def test_sampling_param_gating_driven_by_model_map_flag(monkeypatch):
"""The drop/raise decision must come from ``supports_sampling_params`` in
the model map, not just name matching: a flagged entry gates a model whose
name says nothing, and an explicit ``true`` overrides the name fallback."""
monkeypatch.setitem(
litellm.model_cost, "claude-zeta-9", {"supports_sampling_params": False}
)
monkeypatch.setitem(
litellm.model_cost, "claude-fable-5-test", {"supports_sampling_params": True}
)
config = AnthropicConfig()
flagged_off = config.map_openai_params(
non_default_params={"top_p": 0.9},
optional_params={},
model="claude-zeta-9",
drop_params=True,
)
assert "top_p" not in flagged_off
flagged_on = config.map_openai_params(
non_default_params={"top_p": 0.9},
optional_params={},
model="claude-fable-5-test",
drop_params=True,
)
assert flagged_on["top_p"] == 0.9
def test_top_k_dropped_at_transform_for_models_that_removed_it():
"""``top_k`` is a provider-specific kwarg that bypasses
``map_openai_params``, so it must be stripped at the transform_request

View file

@ -180,6 +180,32 @@ def test_a_gpt_5_name_without_a_foundry_row_keeps_reading_its_own_entry(
assert optional_params["logprobs"] is True
def test_azure_ai_grok_stop_parameter_handling():
"""
Test that Grok models properly handle stop parameter filtering in Azure AI Studio.
"""
config = AzureAIStudioConfig()
# Test Grok model detection
assert config._supports_stop_reason("grok-4-fast") is False
assert config._supports_stop_reason("grok-4.3") is False
assert config._supports_stop_reason("grok-4") is False
assert config._supports_stop_reason("grok-3-mini") is False
assert config._supports_stop_reason("grok-code-fast") is False
assert config._supports_stop_reason("gpt-4") is True
# Test supported parameters for Grok models
for model in ("grok-4-fast", "grok-4.3"):
grok_params = config.get_supported_openai_params(model)
assert (
"stop" not in grok_params
), "Grok models should not support stop parameter"
# Test supported parameters for non-Grok models
gpt_params = config.get_supported_openai_params("gpt-4")
assert "stop" in gpt_params, "GPT models should support stop parameter"
def test_azure_model_router_response_shows_actual_model():
"""
Test that Azure Model Router returns the actual model used in the response,

View file

@ -317,6 +317,46 @@ class TestProviderConfigManagerAzureAnthropicMessages:
assert config is None
def test_messages_thinking_shape_follows_exact_azure_entry_flag(local_model_cost_map, monkeypatch):
"""The Azure messages config must probe capabilities under ``azure_ai`` so an
operator setting ``supports_adaptive_thinking: false`` on the exact
``azure_ai/claude-opus-4-8`` entry beats the unmodified ``anthropic`` entry.
With the inherited ``"anthropic"`` provider default the flip was ignored and
the transform kept emitting ``thinking.type='adaptive'``."""
import litellm
config = AzureAnthropicMessagesConfig()
def transform():
return config.transform_anthropic_messages_request(
model="claude-opus-4-8",
messages=[{"role": "user", "content": "Hello"}],
anthropic_messages_optional_request_params={
"max_tokens": 4096,
"reasoning_effort": "medium",
},
litellm_params=GenericLiteLLMParams(),
headers={},
)
result = transform()
assert result.get("thinking") == {"type": "adaptive", "display": "summarized"}
assert result.get("output_config") == {"effort": "medium"}
monkeypatch.setitem(
litellm.model_cost["azure_ai/claude-opus-4-8"], "supports_adaptive_thinking", False
)
litellm.get_model_info.cache_clear()
assert litellm.model_cost["claude-opus-4-8"]["supports_adaptive_thinking"] is True
flipped = transform()
thinking = flipped.get("thinking")
assert isinstance(thinking, dict)
assert thinking.get("type") == "enabled"
assert isinstance(thinking.get("budget_tokens"), int)
assert "output_config" not in flipped
def _azure_transform(model, messages, system=None):
config = AzureAnthropicMessagesConfig()
params = {"max_tokens": 256}

View file

@ -194,6 +194,32 @@ def test_transform_usage_reads_invoke_model_count_suffixed_cache_keys(
assert openai_usage.total_tokens == 12270
def test_bedrock_invoke_nova_cache_read_billed_at_discounted_rate(monkeypatch):
"""Nova cache reads are billed at the entry's discounted cache read rate; without a
``cache_read_input_token_cost`` entry the cached tokens were billed at nothing."""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
usage = ConverseTokenUsageBlock(
**{
"inputTokens": 5,
"outputTokens": 3,
"totalTokens": 12270,
"cacheReadInputTokenCount": 12262,
"cacheWriteInputTokenCount": 0,
}
)
openai_usage = AmazonConverseConfig().transform_usage(usage)
model = "bedrock/invoke/us.amazon.nova-pro-v1:0"
prompt_cost, completion_cost = litellm.cost_calculator.cost_per_token(model=model, usage_object=openai_usage)
model_info = litellm.get_model_info(model=model)
assert 0 < model_info["cache_read_input_token_cost"] < model_info["input_cost_per_token"]
assert prompt_cost == pytest.approx(
5 * model_info["input_cost_per_token"] + 12262 * model_info["cache_read_input_token_cost"]
)
assert prompt_cost > 5 * model_info["input_cost_per_token"]
assert completion_cost == pytest.approx(3 * model_info["output_cost_per_token"])
def test_transform_usage_with_reasoning_content():
"""Test that completion_tokens_details correctly tracks reasoning vs text tokens."""
usage = ConverseTokenUsageBlock(
@ -5444,6 +5470,87 @@ def test_cache_control_injection_tool_config_drops_ttl_for_unsupported_model():
assert tools[-1] == {"cachePoint": {"type": "default"}}
@pytest.mark.parametrize(
("model", "expects_cache_points"),
[
pytest.param("nvidia.nemotron-super-3-120b", False, id="mapped-model-without-prompt-caching"),
pytest.param("us.nvidia.nemotron-super-3-120b", False, id="regional-prefix-resolves-through-base-model"),
pytest.param(
"us.anthropic.claude-3-5-sonnet-20240620-v1:0", False, id="claude-named-but-not-caching-on-bedrock"
),
pytest.param("us.anthropic.claude-sonnet-4-5-20250929-v1:0", True, id="mapped-model-with-prompt-caching"),
pytest.param(
"arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/abc123",
True,
id="unmapped-arn-keeps-emitting",
),
pytest.param("global.openai.gpt-6-astra", False, id="openai-family-implicit-caching-only"),
pytest.param("openai.gpt-oss-120b-1:0", False, id="openai-gpt-oss"),
pytest.param("us.openai.gpt-99-unmapped", False, id="unmapped-openai-family-still-suppressed"),
],
)
def test_cache_points_emitted_only_for_models_that_support_prompt_caching(model, expects_cache_points, monkeypatch):
"""Bedrock rejects cachePoint blocks for models without prompt caching support
("You invoked an unsupported model or your request did not allow prompt caching"),
and clients like Claude Code attach cache_control to every request, so a map-known
model without the capability must not receive them. Unmapped ids (application
inference profile ARNs, models newer than the map) keep emitting so existing
caching setups never silently degrade."""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
body = AmazonConverseConfig().transform_request(
model=model,
messages=[
{"role": "system", "content": [{"type": "text", "text": "sys", "cache_control": {"type": "ephemeral"}}]},
{"role": "user", "content": [{"type": "text", "text": "hi", "cache_control": {"type": "ephemeral"}}]},
],
optional_params={},
litellm_params={},
headers={},
)
assert ("cachePoint" in json.dumps(body)) is expects_cache_points
assert body["system"][0]["text"] == "sys"
assert body["messages"][0]["content"][0]["text"] == "hi"
def test_tool_config_cachepoint_not_placed_or_credited_for_model_without_prompt_caching(monkeypatch):
"""The tool_config injection point must stand down with the rest of the cachePoint
emission when the model cannot cache, and spend attribution must not credit the
gateway for a breakpoint that was never placed."""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
bucket: dict = {"user_api_key": "sk-test"}
data = AmazonConverseConfig()._transform_request_helper(
model="nvidia.nemotron-super-3-120b",
system_content_blocks=[],
optional_params={
"tools": [
{
"type": "function",
"function": {
"name": "get_weather",
"description": "Get weather",
"parameters": {
"type": "object",
"properties": {"location": {"type": "string"}},
"required": ["location"],
},
},
}
],
"cache_control_injection_points": [{"location": "tool_config"}],
},
messages=[{"role": "user", "content": "hi"}],
litellm_params={"metadata": bucket, "litellm_metadata": None, "model_info": {"id": "dep-bedrock"}},
)
assert "cachePoint" not in json.dumps(data.get("toolConfig", {}))
assert "litellm_gateway_injected_cache" not in bucket
def test_translate_response_format_json_schema_still_injects_tool():
"""
response_format with an explicit json_schema should still use the

View file

@ -2900,6 +2900,49 @@ def test_bedrock_messages_tool_search_follows_claude_tool_search_rule(local_mode
assert cfg._supports_tool_search_on_bedrock(model) is expected
def test_bedrock_messages_thinking_shape_follows_exact_bedrock_entry_flag(
local_model_cost_map, monkeypatch
):
"""The outbound thinking payload must follow the exact Bedrock cost-map entry.
Before threading the caller's provider through the capability probes, the probe
was pinned to ``"anthropic"``: the exact ``global.anthropic.claude-opus-4-8``
entry was rejected by the provider match and the anthropic-scoped fallback rule
forced ``thinking.type='adaptive'`` even with ``supports_adaptive_thinking``
explicitly set to ``false`` on the entry."""
import litellm
from litellm.types.router import GenericLiteLLMParams
model = "global.anthropic.claude-opus-4-8"
cfg = AmazonAnthropicClaudeMessagesConfig()
def transform():
return cfg.transform_anthropic_messages_request(
model=model,
messages=[{"role": "user", "content": [{"type": "text", "text": "Hello"}]}],
anthropic_messages_optional_request_params={
"max_tokens": 4096,
"reasoning_effort": "medium",
},
litellm_params=GenericLiteLLMParams(),
headers={},
)
result = transform()
assert result.get("thinking") == {"type": "adaptive", "display": "summarized"}
assert result.get("output_config") == {"effort": "medium"}
monkeypatch.setitem(litellm.model_cost[model], "supports_adaptive_thinking", False)
litellm.get_model_info.cache_clear()
flipped = transform()
thinking = flipped.get("thinking")
assert isinstance(thinking, dict)
assert thinking.get("type") == "enabled"
assert isinstance(thinking.get("budget_tokens"), int)
assert "output_config" not in flipped
@pytest.mark.parametrize(
"search_results, expected_evidence",
[

View file

@ -1296,6 +1296,37 @@ class TestMantleBaseSegment:
the /openai/v1 base, everything else on /v1. An unmapped model defaults to /v1.
"""
@pytest.mark.parametrize(
"model,model_cost,expected",
[
(
"openai.gpt-5.5",
{"bedrock_mantle/openai.gpt-5.5": {"use_openai_responses_path": True}},
"openai/v1",
),
(
"google.gemma-4-31b",
{
"bedrock_mantle/google.gemma-4-31b": {
"use_openai_responses_path": True
}
},
"openai/v1",
),
(
"openai.gpt-oss-120b",
{"bedrock_mantle/openai.gpt-oss-120b": {}},
"v1",
),
("openai.gpt-oss-120b", {}, "v1"),
(None, {}, "v1"),
],
)
def test_base_segment(self, model, model_cost, expected):
from litellm.llms.bedrock_mantle.common_utils import mantle_base_segment
assert mantle_base_segment(model, model_cost) == expected
class TestMantleSupportsResponses:
"""The capability helper is data-driven (supported_endpoints / mode), with no

View file

@ -442,6 +442,29 @@ class TestDashscopeCostCalculator:
assert math.isclose(completion_cost, 200 * 1.6e-06, rel_tol=1e-10)
def test_dashscope_model_zero_reasoning_rate_bills_reasoning_free(self):
"""
Regression: a model declaring an explicit zero reasoning rate had it treated as
missing, billing reasoning tokens at the plain output rate instead of free.
"""
litellm.model_cost["dashscope/qwen-zero-reasoning-test"] = {
"litellm_provider": "dashscope",
"mode": "chat",
"input_cost_per_token": 4e-07,
"output_cost_per_token": 1.6e-06,
"output_cost_per_reasoning_token": 0,
}
usage = Usage(
prompt_tokens=500,
completion_tokens=200,
completion_tokens_details=CompletionTokensDetailsWrapper(reasoning_tokens=150),
)
_, completion_cost = dashscope_cost_per_token(
model="qwen-zero-reasoning-test", usage=usage
)
assert math.isclose(completion_cost, 50 * 1.6e-06, rel_tol=1e-10)
def test_dashscope_tier_zero_reasoning_rate_bills_reasoning_free(self):
"""

View file

@ -337,6 +337,39 @@ def test_get_supported_openai_params_preserves_generic_reasoning_fallback():
assert "reasoning_effort" in supported_params
def test_get_supported_openai_params_parallel_tool_calls_without_tool_choice(
monkeypatch,
):
"""Test that parallel_tool_calls is gated on tools, not tool_choice."""
config = FireworksAIConfig()
model = "fireworks_ai/test-tools-without-tool-choice"
monkeypatch.setitem(
litellm.model_cost,
model,
{
"supports_function_calling": True,
"supports_tool_choice": False,
},
)
supported_params = config.get_supported_openai_params(model)
assert "tools" in supported_params
assert "parallel_tool_calls" in supported_params
assert "tool_choice" not in supported_params
def test_get_provider_info_omits_false_supports_reasoning(monkeypatch):
"""Test that Fireworks only overrides supports_reasoning for supported models."""
config = FireworksAIConfig()
model = "fireworks_ai/test-reasoning-false"
monkeypatch.setitem(litellm.model_cost, model, {"supports_reasoning": False})
info = config.get_provider_info(model)
assert "supports_reasoning" not in info
@pytest.mark.parametrize(
"api_base, expected_url_prefix",
[
@ -426,6 +459,14 @@ def test_transform_messages_helper_removes_provider_specific_fields():
assert "provider_specific_fields" not in msg
def test_unmapped_model_fallback_function_calling():
"""Test that a model not in model_cost still defaults to supporting function calling for Fireworks."""
config = FireworksAIConfig()
model = "fireworks_ai/unmapped-future-model"
info = config.get_provider_info(model)
assert info["supports_function_calling"] is True
def test_transform_messages_helper_strips_thinking_blocks_but_keeps_reasoning_content():
"""Fireworks rejects thinking_blocks but requires reasoning_content to be replayed for reasoning_history."""
config = FireworksAIConfig()
@ -1050,6 +1091,26 @@ def test_transform_messages_helper_no_transform_inline():
assert "#transform=inline" not in block["image_url"]
def test_get_provider_info_vision_from_model_cost(monkeypatch):
config = FireworksAIConfig()
vision_model = "fireworks_ai/test-vision-from-cost"
monkeypatch.setitem(
litellm.model_cost,
vision_model,
{"supports_vision": True, "supports_pdf_input": True},
)
info = config.get_provider_info(vision_model)
assert info["supports_vision"] is True
assert info["supports_pdf_input"] is True
no_vision_model = "fireworks_ai/test-no-vision-from-cost"
monkeypatch.setitem(litellm.model_cost, no_vision_model, {})
info_no_vision = config.get_provider_info(no_vision_model)
assert info_no_vision.get("supports_vision") is not True
assert "supports_pdf_input" not in info_no_vision
def test_reasoning_effort_boolean_true_to_medium():
config = FireworksAIConfig()
result = config.map_openai_params(

View file

@ -122,6 +122,20 @@ def test_off_peak_window_bills_cached_tokens_at_the_off_peak_input_rate_without_
assert math.isclose(peak_prompt_cost, 1000 * STANDARD_INPUT_COST, rel_tol=1e-10)
def test_off_peak_defaults_to_the_current_time():
"""The proxy's cost dispatch passes no clock, so an all-day window has to apply on the
default current time."""
_register_off_peak_model(
{"hours_utc": "00:00-00:00", "input_cost_per_token": 1e-08, "output_cost_per_token": 2e-08}
)
usage = _usage(prompt_tokens=1000, cached_tokens=0, completion_tokens=200)
prompt_cost, completion_cost = cost_per_token(model=OFF_PEAK_MODEL, usage=usage)
assert math.isclose(prompt_cost, 1000 * 1e-08, rel_tol=1e-10)
assert math.isclose(completion_cost, 200 * 2e-08, rel_tol=1e-10)
COMPONENT_MODEL = "accounts/fireworks/models/cost-components-test"
COMPONENT_INPUT_COST = 1e-06
COMPONENT_OUTPUT_COST = 2e-06

View file

@ -2226,6 +2226,18 @@ class TestReasoningFollowsModelSupport:
)
assert mapped["reasoning"] == reasoning
def test_an_explicit_supports_reasoning_false_beats_the_bundled_floor(self, local_model_cost_map, monkeypatch):
overridden = {
name: ({**entry, "supports_reasoning": False} if name == "o3" else entry)
for name, entry in litellm.model_cost.items()
}
monkeypatch.setattr(litellm, "model_cost", overridden)
mapped = OpenAIResponsesAPIConfig().map_openai_params(
response_api_optional_params={"reasoning": {"effort": "medium"}},
model="o3",
drop_params=True,
)
assert "reasoning" not in mapped
def test_azure_deployments_keep_reasoning_even_on_a_non_reasoning_model_name(self, local_model_cost_map):
mapped = AzureOpenAIResponsesAPIConfig().map_openai_params(

View file

@ -1314,6 +1314,19 @@ def test_gpt5_6_forwards_reasoning_effort_max_for_the_responses_bridge(config: O
assert params["reasoning_effort"] == "max"
@pytest.mark.parametrize("model", ["gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"])
def test_gpt5_6_never_advertises_reasoning_effort_max(model: str):
"""/v1/chat/completions answers max with "Unsupported value: 'reasoning_effort' does not support
'max' with this model. Supported values are: 'none', 'low', 'medium', 'high', and 'xhigh'", so no
gpt-5.6 entry asserts supports_max_reasoning_effort and the advertised set stops at xhigh."""
from litellm.router_utils.reasoning_effort_capability import resolve_supported_reasoning_efforts
resolved = resolve_supported_reasoning_efforts(litellm.get_model_info(model), deployment_is_mapped=True)
assert resolved is not None
assert "max" not in resolved
assert "xhigh" in resolved
def test_gpt5_6_keeps_reasoning_effort_max_on_the_responses_api(
responses_config: OpenAIResponsesAPIConfig,
):

View file

@ -79,6 +79,19 @@ class TestTensormeshProviderConfig:
matching the text_completion flag in provider_endpoints_support.json."""
assert "tensormesh" in litellm.openai_text_completion_compatible_providers
def test_tensormesh_responses_api_enabled(self):
"""Tensormesh declares /v1/responses in supported_endpoints, so litellm
resolves a responses config for it."""
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
from litellm.utils import ProviderConfigManager
assert JSONProviderRegistry.supports_responses_api("tensormesh") is True
config = ProviderConfigManager.get_provider_responses_api_config(
provider="tensormesh",
model="tensormesh/openai/gpt-oss-120b",
)
assert config is not None
assert config.custom_llm_provider == "tensormesh"
def test_tensormesh_router_config(self):
"""Test that tensormesh can be used in Router configuration"""

View file

@ -204,6 +204,18 @@ class TestPerplexityCostCalculator:
assert math.isclose(prompt_cost, (1000 * 1e-07) + (100 * 2e-06), rel_tol=1e-10)
assert math.isclose(completion_cost, (150 * 2e-07) + (50 * 3e-06) + 0.005, rel_tol=1e-10)
def test_off_peak_defaults_to_the_current_time(self):
"""The proxy's cost dispatch passes no clock, so an all-day window has to apply on the
default current time."""
self._register_off_peak_model(
{"hours_utc": "00:00-00:00", "input_cost_per_token": 1e-07, "output_cost_per_token": 2e-07}
)
usage = Usage(prompt_tokens=1000, completion_tokens=200, total_tokens=1200)
prompt_cost, completion_cost = perplexity_cost_per_token(model=self.OFF_PEAK_MODEL, usage=usage)
assert math.isclose(prompt_cost, 1000 * 1e-07, rel_tol=1e-10)
assert math.isclose(completion_cost, 200 * 2e-07, rel_tol=1e-10)
def test_provider_stated_cost_still_wins_inside_an_off_peak_window(self):
"""A response that carries Perplexity's own metered cost bills that cost whatever the

View file

@ -640,6 +640,58 @@ def test_get_vertex_url_global_region(stream, expected_endpoint_suffix):
assert url == expected_url
@pytest.mark.parametrize(
"model_cost_entry, vertex_region, expected_region",
[
# Model with supported_regions=["global"], no user region -> use "global"
({"supported_regions": ["global"]}, None, "global"),
# Model with supported_regions=["global"], user passes unsupported region -> override to "global"
({"supported_regions": ["global"]}, "us-central1", "global"),
# Model with supported_regions=["global"], user passes unsupported region -> override to "global"
({"supported_regions": ["global"]}, "europe-west1", "global"),
# Model with supported_regions=["us-west2"], no user region -> use "us-west2"
({"supported_regions": ["us-west2"]}, None, "us-west2"),
# Model with supported_regions=["us-west2", "us-central1"], user passes supported region -> respect it
(
{"supported_regions": ["us-west2", "us-central1"]},
"us-central1",
"us-central1",
),
# Model with supported_regions=["us-west2", "us-central1"], user passes unsupported region -> override
(
{"supported_regions": ["us-west2", "us-central1"]},
"europe-west1",
"us-west2",
),
# No model_cost entry, no user region -> default us-central1
({}, None, "us-central1"),
# No model_cost entry, user specifies region -> use specified region
({}, "europe-west1", "europe-west1"),
# No model_cost entry, user specifies region -> use specified region
({}, "us-east1", "us-east1"),
],
)
def test_get_vertex_region_global_only_model(
model_cost_entry, vertex_region, expected_region
):
"""Test get_vertex_region resolves region from model_cost supported_regions"""
import litellm
from litellm.llms.vertex_ai.vertex_llm_base import VertexBase
vertex_base = VertexBase()
with patch.dict(
litellm.model_cost,
{"vertex_ai/test-model": model_cost_entry},
clear=False,
):
result = vertex_base.get_vertex_region(
vertex_region=vertex_region, model="test-model"
)
assert result == expected_region
def test_vertex_filter_format_uri():
import json

View file

@ -181,6 +181,58 @@ class TestVertexAILyriaTextToSpeechConfig:
assert isinstance(config, VertexAILyriaTextToSpeechConfig)
@pytest.mark.parametrize(
("model", "vertex_ai_audio_api", "supported_audio_formats", "expected_url"),
[
(
"future-lyria-predict",
"lyria_predict",
["wav"],
"https://us-central1-aiplatform.googleapis.com/v1/projects/music-project/locations/"
"us-central1/publishers/google/models/future-lyria-predict:predict",
),
(
"future-music-interactions",
"lyria_interactions",
["mp3", "wav"],
"https://aiplatform.googleapis.com/v1beta1/projects/music-project/locations/global/interactions",
),
],
)
def test_dispatches_from_model_metadata(
self,
monkeypatch,
model,
vertex_ai_audio_api,
supported_audio_formats,
expected_url,
):
monkeypatch.setitem(
litellm.model_cost,
f"vertex_ai/{model}",
{
"vertex_ai_audio_api": vertex_ai_audio_api,
"supported_audio_formats": supported_audio_formats,
},
)
config = ProviderConfigManager.get_provider_text_to_speech_config(
model=model,
provider=LlmProviders.VERTEX_AI,
)
assert isinstance(config, VertexAILyriaTextToSpeechConfig)
assert (
config.get_complete_url(
model=model,
api_base=None,
litellm_params={
"vertex_project": "music-project",
"vertex_location": "us-central1",
},
)
== expected_url
)
def test_vertex_chirp_does_not_select_lyria_config(self):
config = ProviderConfigManager.get_provider_text_to_speech_config(

View file

@ -32,6 +32,49 @@ def test_get_supported_params_thinking():
assert "thinking" in params
def test_vertex_ai_anthropic_web_search_header_in_completion():
"""Test that web search tool adds the required beta header for Vertex AI completion requests"""
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
# Create the config instance
model_info = AnthropicModelInfo()
# Test the header generation directly
tools = [{"type": "web_search_20250305", "name": "web_search", "max_uses": 5}]
# Check if web search tool is detected
web_search_detected = model_info.is_web_search_tool_used(tools=tools)
assert web_search_detected is True, "Web search tool should be detected"
# Generate headers with is_vertex_request=True
headers = model_info.get_anthropic_headers(
api_key="test-key",
web_search_tool_used=web_search_detected,
is_vertex_request=True,
)
# Assert that the anthropic-beta header with web-search is present
assert "anthropic-beta" in headers, "anthropic-beta header should be present"
assert (
headers["anthropic-beta"] == "web-search-2025-03-05"
), f"anthropic-beta should be 'web-search-2025-03-05', got: {headers['anthropic-beta']}"
# Test that header is NOT added for non-Vertex requests
headers_non_vertex = model_info.get_anthropic_headers(
api_key="test-key",
web_search_tool_used=web_search_detected,
is_vertex_request=False,
)
# For non-Vertex (Anthropic-hosted), the web search header should NOT be in anthropic-beta
# because Anthropic doesn't require it
assert (
"anthropic-beta" not in headers_non_vertex
or "web-search" not in headers_non_vertex.get("anthropic-beta", "")
), "anthropic-beta with web-search should not be present for non-Vertex requests"
def test_vertex_ai_anthropic_context_management_compact_beta_header():
"""Test that context_management with compact adds the correct beta header for Vertex AI"""
config = VertexAIAnthropicConfig()

View file

@ -13,6 +13,7 @@ import httpx
import pytest
import litellm
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
from litellm.llms.vertex_ai.videos.transformation import (
VertexAIVideoConfig,
_convert_image_to_vertex_format,
@ -122,6 +123,25 @@ class TestVertexAIVideoConfig:
)
def test_veo_31_lite_provider_routing_from_local_model_map(
self, monkeypatch: pytest.MonkeyPatch
):
model_cost = _load_model_cost_map(BACKUP_MODEL_COST_PATH)
vertex_video_models = {
model_name.removeprefix("vertex_ai/")
for model_name, info in model_cost.items()
if info.get("litellm_provider") == "vertex_ai-video-models"
}
monkeypatch.setattr(litellm, "vertex_ai_video_models", vertex_video_models)
model, custom_llm_provider, _, _ = get_llm_provider(
model="veo-3.1-lite-generate-001"
)
assert model == "veo-3.1-lite-generate-001"
assert custom_llm_provider == "vertex_ai"
def test_transform_video_create_request(self):
"""Test transformation of video creation request."""
prompt = "A cat playing with a ball of yarn"

View file

@ -75,6 +75,21 @@ def wandb_request_mock(respx_mock: respx.MockRouter) -> respx.Route:
class TestWandbConfig:
"""Test class for WandB Inference functionality"""
@pytest.mark.parametrize("model", WANDB_REASONING_MODELS)
def test_map_openai_params_preserves_reasoning_effort(self, wandb_test_config, model: str):
assert litellm.model_cost[f"wandb/{model}"].get("supports_reasoning") is True
supported_params = litellm.get_supported_openai_params(model=f"wandb/{model}")
assert supported_params is not None
assert "reasoning_effort" in supported_params
result = WandbConfig().map_openai_params(
non_default_params={"reasoning_effort": "medium", "max_completion_tokens": 64},
optional_params={},
model=model,
drop_params=True,
)
assert result == {"reasoning_effort": "medium", "max_tokens": 64}
def test_default_api_base(self):
"""Test that default API base is used when none is provided"""
@ -228,6 +243,52 @@ class TestWandbConfig:
assert request_body["max_tokens"] == 64
assert "max_completion_tokens" not in request_body
@pytest.mark.respx(assert_all_called=False)
@pytest.mark.parametrize("drop_params", [True, False])
@pytest.mark.parametrize(
"model,explicit_false",
[
("meta-llama/Llama-3.1-8B-Instruct", False),
("openai/gpt-oss-20b", True),
],
)
def test_wandb_completion_without_reasoning_support(
self,
wandb_test_config,
wandb_request_mock: respx.Route,
respx_mock: respx.MockRouter,
monkeypatch: pytest.MonkeyPatch,
model: str,
explicit_false: bool,
drop_params: bool,
):
with monkeypatch.context() as context:
if explicit_false:
context.setitem(litellm.model_cost[f"wandb/{model}"], "supports_reasoning", False)
kwargs = {
"model": f"wandb/{model}",
"messages": [{"role": "user", "content": "Hello"}],
"api_key": "fake-wandb-key",
"api_base": "https://api.inference.wandb.ai/v1",
"reasoning_effort": "medium",
"drop_params": drop_params,
}
if not drop_params:
with pytest.raises(litellm.UnsupportedParamsError, match="reasoning_effort"):
completion(**kwargs)
assert len(respx_mock.calls) == 0
return
completion(**kwargs)
assert wandb_request_mock.call_count == 1
request_body = json.loads(wandb_request_mock.calls[0].request.content)
assert request_body["model"] == model
assert "reasoning_effort" not in request_body
supported_params = litellm.get_supported_openai_params(model=f"wandb/{model}")
assert supported_params is not None
assert "reasoning_effort" not in supported_params
@pytest.mark.respx()
def test_wandb_completion_keeps_reasoning_effort_for_an_unregistered_model(

View file

@ -7,6 +7,8 @@ from __future__ import annotations
import json
from pathlib import Path
import pytest
REPO_ROOT = Path(__file__).parents[4]
PRICES_PATH = REPO_ROOT / "model_prices_and_context_window.json"
BACKUP_PRICES_PATH = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
@ -21,6 +23,12 @@ RESPONSES_ONLY_MODELS = (
MAP_PATHS = (PRICES_PATH, BACKUP_PRICES_PATH)
@pytest.fixture(scope="module", params=[p.name for p in MAP_PATHS])
def cost_map(request: pytest.FixtureRequest) -> dict:
path = next(p for p in MAP_PATHS if p.name == request.param)
return json.loads(path.read_text(encoding="utf-8"))
def test_both_cost_maps_agree_on_xai_entries():
prices = json.loads(PRICES_PATH.read_text(encoding="utf-8"))
backup = json.loads(BACKUP_PRICES_PATH.read_text(encoding="utf-8"))

View file

@ -7,6 +7,7 @@ from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_toke
from litellm.llms.anthropic.cost_calculation import cost_per_token as anthropic_cost_per_token
from litellm.proxy.spend_tracking.savings import (
_baseline_usage,
_resolve_model,
compute_autorouter_savings,
compute_savings_spend,
marks_gateway_injection,
@ -757,6 +758,84 @@ def test_a_switch_onto_a_partly_cached_model_still_pays_for_the_write():
assert reported < if_treated_as_same_model / 10, "a mostly-cold switch must not be priced as a continuation"
def test_a_baseline_that_prices_caching_implicitly_still_pays_for_its_prompt():
"""OpenAI, Azure and Gemini entries carry no `cache_creation_input_token_cost`,
because those providers cache implicitly and charge nothing to write. Leaving this
request's written tokens in the creation bucket priced them at the 0.0 the cost
resolver falls back to, so the baseline carried a 20k prompt for free and a first
turn that saved money reported a loss. Those tokens are plain input on such a model.
"""
first_turn = _usage(fresh=0, cached=0, written=20_000, out=1_000)
reported = compute_autorouter_savings(
baseline_model="gpt-5",
selected_model="claude-haiku-4-5",
selected_provider="anthropic",
usage=first_turn,
conversation_continuing=False,
)
gpt5 = litellm.get_model_info("gpt-5", "openai")
assert gpt5.get("cache_creation_input_token_cost") is None, "pick a baseline with no cache-write rate"
haiku = litellm.get_model_info("claude-haiku-4-5", "anthropic")
baseline_pays_input = 20_000 * gpt5["input_cost_per_token"] + 1_000 * gpt5["output_cost_per_token"]
actually_paid = 20_000 * haiku["cache_creation_input_token_cost"] + 1_000 * haiku["output_cost_per_token"]
assert reported == pytest.approx(baseline_pays_input - actually_paid)
assert reported > 0, "routing a cold first turn onto a cheaper model is a saving, not a loss"
def _priced_chat_model_without_cache_read_rate() -> tuple[str, str, str]:
"""A chat model the bundled map prices per token for input and output but not for cache
reads, derived from the map itself: a hardcoded pick goes stale the moment the registry
prices that model's cache reads, which is exactly how this test's premise last broke.
Candidates go through the savings module's own resolver, so the pick is one the code
under test can actually price."""
for key in sorted(litellm.model_cost):
entry = litellm.model_cost[key]
provider = entry.get("litellm_provider")
if not isinstance(provider, str) or not key.startswith(f"{provider}/"):
continue
if entry.get("mode") != "chat" or entry.get("cache_read_input_token_cost") is not None:
continue
if not entry.get("input_cost_per_token") or not entry.get("output_cost_per_token"):
continue
if _resolve_model(key, None) is None:
continue
priced = compute_autorouter_savings(
baseline_model=key,
selected_model="claude-haiku-4-5",
selected_provider="anthropic",
usage=_usage(fresh=1_000, cached=0, written=0, out=100),
conversation_continuing=True,
)
if priced == 0.0:
continue
return key, key.removeprefix(f"{provider}/"), provider
raise AssertionError("the bundled map has no per-token chat model without a cache-read rate")
def test_a_baseline_with_no_cache_read_rate_is_charged_its_input_rate():
"""The same hole on the other bucket. A baseline whose entry has no
`cache_read_input_token_cost` reads for 0.0, so a continuing turn priced the whole
prompt at nothing and every switch away from it reported a loss.
"""
baseline_key, baseline_name, baseline_provider = _priced_chat_model_without_cache_read_rate()
continuing = _usage(fresh=0, cached=0, written=20_000, out=1_000)
reported = compute_autorouter_savings(
baseline_model=baseline_key,
selected_model="claude-haiku-4-5",
selected_provider="anthropic",
usage=continuing,
conversation_continuing=True,
)
baseline = litellm.get_model_info(baseline_name, baseline_provider)
assert baseline.get("cache_read_input_token_cost") is None, "pick a baseline with no cache-read rate"
haiku = litellm.get_model_info("claude-haiku-4-5", "anthropic")
baseline_pays_input = 20_000 * baseline["input_cost_per_token"] + 1_000 * baseline["output_cost_per_token"]
actually_paid = 20_000 * haiku["cache_creation_input_token_cost"] + 1_000 * haiku["output_cost_per_token"]
assert reported == pytest.approx(baseline_pays_input - actually_paid)
def _breakdown(input_cost: float, output_cost: float = 0.0, **extra: object) -> dict:
"""A `cost_breakdown` as the cost calculator records it on the spend log."""
return {"input_cost": input_cost, "output_cost": output_cost, **extra}
@ -796,6 +875,51 @@ def test_the_served_arm_is_read_from_the_record_not_repriced():
assert reported == pytest.approx(public - (negotiated_input + negotiated_output))
@pytest.mark.parametrize(
"basis, expected_multiplier",
[
pytest.param({"service_tier": "priority"}, 2.5, id="priority tier uplifts the baseline"),
pytest.param({"data_residency": "eu"}, 1.1, id="eu residency uplifts the baseline"),
pytest.param({}, 1.0, id="no basis recorded prices at standard"),
pytest.param(None, 1.0, id="row predating the field prices at standard"),
pytest.param({"service_tier": True, "data_residency": 17}, 1.0, id="a non-string basis is dropped"),
],
)
def test_the_baseline_is_priced_on_the_basis_the_request_was_billed_at(basis, expected_multiplier):
"""A request billed at a priority tier, or through a regional host, would have been
billed the same way on the single model an operator ran instead of the router, so the
counterfactual carries that basis too. Dropping it prices the two arms from different
books; neither multiplier cancels out of the difference, because both are per-model.
The served model has no tiered rates and no uplift of its own, so only the baseline
can move: a fix that forwards the basis to the served arm alone leaves these numbers
unchanged. The non-string case guards the JSON round trip, where `.lower()` inside
the pricer would raise and be swallowed into a silent $0.00 for the whole row.
"""
gpt = litellm.get_model_info("gpt-5.5", "openai")
haiku = litellm.get_model_info("claude-haiku-4-5", "anthropic")
assert gpt.get("input_cost_per_token_priority") == pytest.approx(2.5 * gpt["input_cost_per_token"])
assert gpt.get("output_cost_per_token_priority") == pytest.approx(2.5 * gpt["output_cost_per_token"])
assert gpt.get("regional_processing_uplift_multiplier_eu") == 1.1
assert haiku.get("input_cost_per_token_priority") is None, "served model must not move with the basis"
assert haiku.get("regional_processing_uplift_multiplier_eu") is None
usage = _usage(fresh=20_000, cached=0, written=0, out=1_000)
served = 20_000 * haiku["input_cost_per_token"] + 1_000 * haiku["output_cost_per_token"]
reported = compute_autorouter_savings(
baseline_model="openai/gpt-5.5",
selected_model="claude-haiku-4-5",
selected_provider="anthropic",
usage=usage,
conversation_continuing=False,
cost_breakdown=None if basis is None else _breakdown(served, **basis),
)
baseline = 20_000 * gpt["input_cost_per_token"] + 1_000 * gpt["output_cost_per_token"]
assert reported == pytest.approx(expected_multiplier * baseline - served)
def test_the_baseline_is_priced_on_the_vertex_location_the_request_was_billed_at(monkeypatch):
"""A request served from a regional Vertex endpoint was billed with the
regional-endpoint uplift, so the counterfactual single-model operator would

View file

@ -2151,6 +2151,32 @@ async def test_proxy_only_error_5xx_keeps_traceback_and_runs_sync_callbacks(monk
assert "test_proxy_utils" in captured["async_traceback"]
def test_create_model_info_response_resolves_mode_through_deployment_model():
"""`mode` is derived from the same lookup, so an aliased embedding deployment
currently reports no mode at all; it must report `embedding`."""
from litellm import Router
saved_model_cost = dict(litellm.model_cost)
try:
router = Router(
model_list=[
{
"model_name": "my-embeddings",
"litellm_params": {"model": "openai/text-embedding-3-small"},
}
]
)
response = create_model_info_response(
model_id="my-embeddings", provider="openai", llm_router=router
)
finally:
litellm.model_cost.clear()
litellm.model_cost.update(saved_model_cost)
assert response["mode"] == "embedding"
@pytest.mark.parametrize(
"key_metadata, team_metadata, expected_to_run",
[

View file

@ -105,6 +105,20 @@ def test_build_jev_request_includes_system_prompt_and_criteria() -> None:
assert request.questions["tier"].criteria == criteria
def test_jev_classifier_cost_uses_registry_pricing(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setitem(
litellm.model_cost,
"typesafe/jev-1.13.0",
{"input_cost_per_token": 0.0001, "output_cost_per_token": 0.0002},
)
response: Final = JevSystemOneResponse(
model="jev-1.13.0",
answers={"tier": _answer()},
usage=JevUsage(input_tokens=3, output_tokens=4),
)
assert jev_classifier_cost(response, "jev-latest") == pytest.approx(0.0011)
def test_jev_classifier_cost_is_none_without_registry_pricing() -> None:
assert "typesafe/jev-unpriced" not in litellm.model_cost
response: Final = JevSystemOneResponse(

View file

@ -325,7 +325,28 @@ KIMI_K3_PERPLEXITY_KEY = "perplexity/perplexity/kimi-k3"
class TestKimiK3AdvertisesItsDocumentedLevels:
@pytest.mark.parametrize("model_key", KIMI_K3_PASSTHROUGH_KEYS)
def test_a_passthrough_entry_advertises_the_models_own_levels(self, local_model_cost_map, model_key):
"""platform.kimi.ai documents exactly low, high and max, and these providers forward the
level unchanged. Undeclared, each entry resolves to unknown and the dashboard falls back to
a capability-blind list that omits max."""
entry = dict(litellm.model_cost[model_key], key=model_key)
assert resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) == ("low", "high", "max")
def test_the_perplexity_entry_advertises_the_wider_set_it_maps_down(self, local_model_cost_map):
"""Perplexity's Agent API takes a six-value enum and maps it down internally, so this
deployment is legitimately wider than a passthrough. One blanket list could not say both."""
entry = dict(litellm.model_cost[KIMI_K3_PERPLEXITY_KEY], key=KIMI_K3_PERPLEXITY_KEY)
assert resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) == (
"minimal",
"low",
"medium",
"high",
"xhigh",
"max",
)
@pytest.mark.parametrize("model, provider", [("kimi-k3", "moonshot"), ("kimi-k3", "fireworks_ai")])
def test_the_declaration_survives_model_info_hydration(self, local_model_cost_map, model, provider):
@ -338,6 +359,19 @@ class TestKimiK3AdvertisesItsDocumentedLevels:
assert model_info["reasoning_effort_levels"] == ["low", "high", "max"]
assert resolve_supported_reasoning_efforts(model_info, deployment_is_mapped=True) == ("low", "high", "max")
def test_a_kimi_k3_deployment_now_narrows_a_mixed_group(self, local_model_cost_map):
"""kimi used to contribute unknown, which never narrows, so the group advertised whatever
its other deployments agreed on."""
kimi = resolve_supported_reasoning_efforts(
dict(litellm.model_cost["fireworks_ai/kimi-k3"], key="fireworks_ai/kimi-k3"),
deployment_is_mapped=True,
)
assert intersect_supported_reasoning_efforts(("none", "minimal", "low", "medium", "high", "xhigh"), kimi) == (
"low",
"high",
)
class TestGpt6AstraAdvertisesItsDocumentedLevels:
def test_the_entry_advertises_low_through_max_without_none(self, local_model_cost_map):

View file

@ -1,8 +1,11 @@
from pathlib import Path
from typing import Final
import pytest
from pydantic import TypeAdapter
from litellm import cost_per_token, get_model_info
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
REPO_ROOT: Final = Path(__file__).parents[2]
MODEL: Final = "azure_ai/grok-4.6"
@ -13,6 +16,27 @@ def _cost_map_entry(path: Path) -> dict[str, object]:
return COST_MAP_ADAPTER.validate_json(path.read_bytes())[MODEL]
@pytest.mark.usefixtures("local_model_cost_map")
def test_azure_ai_grok_4_6_is_priced_and_routed() -> None:
routed_model, provider, _, _ = get_llm_provider(model=MODEL)
assert (routed_model, provider) == ("grok-4.6", "azure_ai")
info = get_model_info(model=routed_model, custom_llm_provider=provider)
assert info["litellm_provider"] == "azure_ai"
assert info["mode"] == "chat"
assert info["supports_function_calling"] is True
assert info["supports_prompt_caching"] is True
assert info["supports_reasoning"] is True
assert info["supports_response_schema"] is True
assert info["supports_tool_choice"] is True
assert info["supports_vision"] is True
assert info["supports_web_search"] is True
prompt_cost, completion_cost = cost_per_token(model=MODEL, prompt_tokens=1_000_000, completion_tokens=1_000_000)
assert prompt_cost > 0
assert completion_cost > 0
def test_azure_ai_grok_4_6_entry_source_and_backup_match() -> None:
main_entry = _cost_map_entry(REPO_ROOT / "model_prices_and_context_window.json")
backup_entry = _cost_map_entry(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json")

View file

@ -427,6 +427,74 @@ def test_transcription_usage_cost_returns_zero_for_unknown_type():
assert _transcription_usage_cost({}, {}) == 0.0
def test_get_transcription_model_falls_back_to_session_model(monkeypatch):
"""session.model is used when transcription-specific model fields are absent."""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
from litellm.cost_calculator import _get_transcription_model_name_from_results
results: OpenAIRealtimeStreamList = [
{"type": "session.created", "session": {"model": "gpt-realtime-whisper"}},
]
assert _get_transcription_model_name_from_results(results) == "gpt-realtime-whisper"
from litellm import Router
router = Router(
model_list=[
{
"model_name": "prod/claude-3-5-sonnet-20240620",
"litellm_params": {
"model": "anthropic/claude-sonnet-4-5-20250929",
"api_key": "test_api_key",
},
"model_info": {
"id": "my-unique-model-id",
"input_cost_per_token": 0.000006,
"output_cost_per_token": 0.00003,
"cache_creation_input_token_cost": 0.0000075,
"cache_read_input_token_cost": 0.0000006,
},
},
{
"model_name": "claude-3-5-sonnet-20240620",
"litellm_params": {
"model": "anthropic/claude-sonnet-4-5-20250929",
"api_key": "test_api_key",
},
"model_info": {
"input_cost_per_token": 100,
"output_cost_per_token": 200,
},
},
]
)
result = router.completion(
model="claude-3-5-sonnet-20240620",
messages=[{"role": "user", "content": "Hello, world!"}],
mock_response=True,
)
result_2 = router.completion(
model="prod/claude-3-5-sonnet-20240620",
messages=[{"role": "user", "content": "Hello, world!"}],
mock_response=True,
)
assert result._hidden_params["response_cost"] > result_2._hidden_params["response_cost"]
model_info = router.get_deployment_model_info(
model_id="my-unique-model-id", model_name="anthropic/claude-sonnet-4-5-20250929"
)
assert model_info is not None
assert model_info["input_cost_per_token"] == 0.000006
assert model_info["output_cost_per_token"] == 0.00003
assert model_info["cache_creation_input_token_cost"] == 0.0000075
assert model_info["cache_read_input_token_cost"] == 0.0000006
def test_custom_pricing_cost_calc_uses_router_model_id_from_litellm_metadata():
"""When custom pricing is in litellm_metadata.model_info,
use_custom_pricing_for_model should return True and
@ -2270,6 +2338,42 @@ def test_anthropic_geo_multiplier_applies_to_cache_tokens(_local_model_cost_map,
assert geo_completion_cost == pytest.approx(base_completion_cost * 1.1)
def test_anthropic_geo_and_fast_multipliers_compose(_local_model_cost_map, monkeypatch):
"""
Anthropic's fast-mode pricing doubles every token type, cache reads and
writes included, and the regional uplift stacks on top, so a fast +
regional row prices as ``(non_cache + cache) * fast * geo``.
"""
from litellm.llms.anthropic.cost_calculation import (
cost_per_token as anthropic_cost_per_token,
)
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
model = "claude-test-geo-fast-cache-model"
_register_anthropic_geo_cache_model(model)
usage = Usage(
prompt_tokens=10_000,
completion_tokens=500,
total_tokens=10_500,
prompt_tokens_details=PromptTokensDetailsWrapper(
cached_tokens=2_000,
cache_creation_tokens=6_000,
),
)
usage.inference_geo = "us"
usage.speed = "fast"
prompt_cost, completion_cost = anthropic_cost_per_token(model=model, usage=usage)
cache_cost = 2_000 * 0.5e-6 + 6_000 * 6.25e-6
non_cache_cost = 2_000 * 5e-6
assert prompt_cost == pytest.approx((non_cache_cost + cache_cost) * 2.0 * 1.1)
assert completion_cost == pytest.approx(500 * 25e-6 * 2.0 * 1.1)
@pytest.mark.parametrize(
"model",
["claude-sonnet-4-6", "claude-mythos-5", "claude-mythos-preview"],
@ -2806,6 +2910,60 @@ def test_custom_pricing_without_cache_keys_preserves_legacy_behavior():
assert cost == pytest.approx(expected)
def test_completion_cost_logs_the_rates_it_billed_at(monkeypatch):
"""A caller reporting the cost lines beside their per-token rates reads both off this one call.
completion_cost infers the provider, and xai's inclusive tier thresholds put a request sitting
exactly on 200k at the tier rate, which a lookup made without that inferred provider would miss.
"""
from datetime import datetime
from litellm.litellm_core_utils.litellm_logging import Logging
monkeypatch.setitem(
litellm.model_cost,
"xai/tiered-model",
{
"input_cost_per_token": 3e-6,
"output_cost_per_token": 15e-6,
"cache_read_input_token_cost": 3e-7,
"input_cost_per_token_above_200k_tokens": 6e-6,
"output_cost_per_token_above_200k_tokens": 3e-5,
"cache_read_input_token_cost_above_200k_tokens": 6e-7,
"litellm_provider": "xai",
"mode": "chat",
},
)
logging_obj = Logging(
model="xai/tiered-model",
messages=[{"role": "user", "content": "Hello"}],
stream=False,
call_type="completion",
start_time=datetime.now(),
litellm_call_id="billed-rates",
function_id="f",
)
usage = Usage(
prompt_tokens=200_000,
completion_tokens=1_000,
total_tokens=201_000,
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=100_000),
)
litellm.completion_cost(
completion_response=ModelResponse(model="xai/tiered-model", usage=usage),
model="xai/tiered-model",
custom_llm_provider=None,
litellm_logging_obj=logging_obj,
)
rates = logging_obj.billed_token_rates
assert rates is not None
assert rates.input_cost_per_token == pytest.approx(6e-6)
assert rates.cache_read_input_token_cost == pytest.approx(6e-7)
assert logging_obj.cost_breakdown["cache_read_cost"] == pytest.approx(100_000 * rates.cache_read_input_token_cost)
assert logging_obj.cost_breakdown["output_cost"] == pytest.approx(1_000 * rates.output_cost_per_token)
def test_completion_cost_logs_cache_and_reasoning_breakdown_for_custom_pricing():
"""
A custom-priced deployment bills cache tokens at its custom cache rates, but the
@ -3065,6 +3223,35 @@ def test_completion_cost_bills_interactions_google_search_per_query():
assert cost > 3 * per_query_cost
def test_completion_cost_bills_interactions_video_output_at_video_rate():
from litellm.types.interactions import InteractionsAPIResponse
model_info = litellm.get_model_info(model="gemini-omni-flash-preview", custom_llm_provider="gemini")
video_tokens = 5792 * 8
response = InteractionsAPIResponse(
id="interactions/video123",
model="gemini-omni-flash-preview",
status="completed",
steps=[],
usage={
"total_tokens": 10 + video_tokens,
"total_input_tokens": 10,
"input_tokens_by_modality": [{"modality": "text", "tokens": 10}],
"total_cached_tokens": 0,
"total_output_tokens": video_tokens,
"output_tokens_by_modality": [{"modality": "video", "tokens": video_tokens}],
"total_tool_use_tokens": 0,
"total_thought_tokens": 0,
},
)
cost = completion_cost(completion_response=response, custom_llm_provider="gemini")
expected = 10 * model_info["input_cost_per_token"] + video_tokens * model_info["output_cost_per_video_token"]
assert model_info["output_cost_per_video_token"] != model_info["output_cost_per_token"]
assert cost == pytest.approx(expected)
@pytest.mark.parametrize("video_count", [2, 3])
def test_completion_cost_multiplies_video_cost_by_generated_video_count(video_count: int) -> None:
"""Regression for LIT-6896: a Veo request for N samples generates N videos and must be billed N times."""

View file

@ -3,6 +3,8 @@ from pathlib import Path
import pytest
import litellm
REPO_ROOT = Path(__file__).parents[2]
MAIN_PATH = REPO_ROOT / "model_prices_and_context_window.json"
BACKUP_PATH = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
@ -19,6 +21,17 @@ def _load(path):
return json.load(f)
@pytest.fixture
def local_model_cost_map(monkeypatch):
"""Force get_model_info to resolve against the in-repo cost map instead of the
remote one fetched at import time, which still carries the pre-merge pricing."""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
litellm.get_model_info.cache_clear()
yield
litellm.get_model_info.cache_clear()
@pytest.mark.parametrize("model", GLM_5_2_MODELS)
def test_backup_matches_main(model):
"""Ensure the bundled (backup) cost map stays in sync with the canonical file."""

View file

@ -0,0 +1,25 @@
import json
from pathlib import Path
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
def test_sambanova_minimax_m27_model_info():
model = "sambanova/MiniMax-M2.7"
json_path = Path(__file__).parents[2] / "model_prices_and_context_window.json"
with open(json_path) as f:
model_cost = json.load(f)
info = model_cost.get(model)
assert info is not None, f"{model} not found in model_prices_and_context_window.json"
assert info["litellm_provider"] == "sambanova"
assert info["mode"] == "chat"
assert info["input_cost_per_token"] > 0
assert info["output_cost_per_token"] > 0
assert info["supports_function_calling"] is True
assert info["supports_reasoning"] is True
assert info["supports_tool_choice"] is True
routed_model, provider, _, _ = get_llm_provider(model=model)
assert routed_model == "MiniMax-M2.7"
assert provider == "sambanova"

View file

@ -3378,6 +3378,38 @@ _FIREWORKS_ROUTER_SHORT_FORMS = [
]
@pytest.fixture
def fireworks_short_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> Iterator[None]:
monkeypatch.setattr(
litellm,
"model_cost",
{
"fireworks_ai/accounts/fireworks/models/glm-5p3": {
"input_cost_per_token": 1e-6,
"output_cost_per_token": 2e-6,
"litellm_provider": "fireworks_ai",
"mode": "chat",
"max_tokens": 100,
},
"fireworks_ai/accounts/fireworks/routers/glm-5p3-fast": {
"input_cost_per_token": 2.1e-6,
"output_cost_per_token": 6.6e-6,
"litellm_provider": "fireworks_ai",
"mode": "chat",
},
"fireworks_ai/nomic-ai/nomic-embed-text-v1.5": {
"input_cost_per_token": 8e-9,
"output_cost_per_token": 0.0,
"litellm_provider": "fireworks_ai",
"mode": "embedding",
},
},
)
litellm.get_model_info.cache_clear()
yield
litellm.get_model_info.cache_clear()
class TestBedrockBaseModelLabelKeepsTools:
"""Regression for #29618: a Bedrock deployment whose ``base_model`` is a friendly
label must not silently drop ``tools``/``tool_choice`` under ``drop_params``."""

View file

@ -3,6 +3,8 @@ from typing import Final
import pytest
import litellm
from litellm import get_model_info
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
from litellm.utils import supports_prompt_caching
MODEL: Final = "vertex_ai/xai/grok-4.6"
@ -24,3 +26,13 @@ def test_grok_models_with_cache_read_price_advertise_prompt_caching() -> None:
)
@pytest.mark.usefixtures("local_model_cost_map")
def test_vertex_ai_grok_4_6_supports_prompt_caching_via_get_model_info() -> None:
routed_model, provider, _, _ = get_llm_provider(model=MODEL)
assert (routed_model, provider) == ("xai/grok-4.6", "vertex_ai")
info = get_model_info(model=routed_model, custom_llm_provider=provider)
assert info["litellm_provider"] == "vertex_ai"
assert info.get("supports_prompt_caching") is True
assert supports_prompt_caching(model=MODEL) is True