mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-07 02:59:05 +00:00
test: keep behavior tests that read the cost map for a later fixture rewrite
Fifty six of the deleted tests turn out to assert the output of litellm code rather than the catalog lookup itself, things like map_openai_params, get_supported_openai_params, should_fake_stream, transform_request bodies, cost_per_token arithmetic, get_llm_provider routing, and provider config dispatch. They only happen to read shipped entries as inputs, so they belong in the later rewrite that injects a local model_cost, not in this deletion Each one is restored verbatim from origin/main along with the fixtures, helpers, constants and imports it needs, and tests/test_litellm/test_sambanova_model_metadata.py is restored wholesale Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
d2ac51893b
commit
7975987107
36 changed files with 1605 additions and 0 deletions
|
|
@ -41,6 +41,36 @@ class TestAzureOpenAIO3Mini(BaseOSeriesModelsTest, BaseLLMChatTest):
|
|||
"""Temporary override. o1 prompt caching is not working."""
|
||||
pass
|
||||
|
||||
def test_override_fake_stream(self):
|
||||
"""Test that native streaming is not supported for o1."""
|
||||
router = litellm.Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "azure/o1-preview",
|
||||
"litellm_params": {
|
||||
"model": "azure/o1-preview",
|
||||
"api_key": "my-fake-o1-key",
|
||||
"api_base": "https://openai-gpt-4-test-v-1.openai.azure.com",
|
||||
},
|
||||
"model_info": {
|
||||
"supports_native_streaming": True,
|
||||
},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
## check model info
|
||||
|
||||
model_info = litellm.get_model_info(
|
||||
model="azure/o1-preview", custom_llm_provider="azure"
|
||||
)
|
||||
assert model_info["supports_native_streaming"] is True
|
||||
|
||||
fake_stream = litellm.AzureOpenAIO1Config().should_fake_stream(
|
||||
model="azure/o1-preview", stream=True
|
||||
)
|
||||
assert fake_stream is False
|
||||
|
||||
|
||||
class TestAzureOpenAIO3(BaseOSeriesModelsTest):
|
||||
def get_base_completion_call_args(self):
|
||||
|
|
|
|||
|
|
@ -1586,11 +1586,40 @@ class TestEnableAnthropicPromptCaching:
|
|||
points = self._points(model="us.anthropic.claude-sonnet-4-5-20250929-v1:0", provider="bedrock")
|
||||
assert [p["index"] for p in points] == [None, -1]
|
||||
|
||||
@pytest.mark.parametrize("model, provider", [("gpt-4o", "openai"), ("gemini-2.0-flash", "gemini")])
|
||||
def test_non_anthropic_providers_never_injected(self, monkeypatch, model, provider):
|
||||
"""These report supports_prompt_caching=True but never consume cache_control markers."""
|
||||
from litellm.utils import supports_prompt_caching
|
||||
|
||||
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
|
||||
assert supports_prompt_caching(model=model, custom_llm_provider=provider) is True
|
||||
assert self._points(model=model, provider=provider) == []
|
||||
|
||||
def test_databricks_claude_not_injected_despite_caching_support(self, monkeypatch, local_model_cost_map):
|
||||
from litellm.utils import supports_prompt_caching
|
||||
|
||||
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
|
||||
model = "databricks/databricks-claude-sonnet-4-5"
|
||||
assert supports_prompt_caching(model=model, custom_llm_provider="databricks") is True
|
||||
assert self._points(model=model, provider="databricks") == []
|
||||
|
||||
def test_model_without_caching_support_not_injected(self, monkeypatch):
|
||||
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
|
||||
assert self._points(model="anthropic.claude-3-5-sonnet-20240620-v1:0", provider="bedrock") == []
|
||||
|
||||
@pytest.mark.parametrize("model", ["us.xai.grok-4.6", "global.xai.grok-4.6"])
|
||||
def test_bedrock_grok_not_injected(self, monkeypatch, local_model_cost_map, model):
|
||||
"""Bedrock supports only implicit prompt caching for Grok: explicit cachePoint
|
||||
breakpoints make it reject the whole request ("You invoked an unsupported model
|
||||
or your request did not allow prompt caching"), so supports_prompt_caching stays
|
||||
false, while implicit cache hits still bill at the cache-read rate."""
|
||||
from litellm.utils import supports_prompt_caching
|
||||
|
||||
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
|
||||
assert supports_prompt_caching(model=model, custom_llm_provider="bedrock") is False
|
||||
assert self._points(model=model, provider="bedrock") == []
|
||||
entry = litellm.model_cost[model]
|
||||
assert 0 < entry["cache_read_input_token_cost"] < entry["input_cost_per_token"]
|
||||
|
||||
def test_stands_down_when_client_sent_cache_control(self, monkeypatch):
|
||||
monkeypatch.setattr(litellm, "enable_anthropic_prompt_caching", True)
|
||||
|
|
|
|||
|
|
@ -361,6 +361,58 @@ def test_completion_cost_includes_web_search_without_standard_built_in_tools_par
|
|||
), f"completion_cost ({cost}) should include web search cost ({web_search_cost})"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
"vertex_ai/gemini-3.1-flash-lite", # resolves directly via get_model_info
|
||||
"gemini/gemini-3.1-flash-lite", # provider-prefixed, resolves via model_cost fallback
|
||||
],
|
||||
)
|
||||
def test_gemini_3x_web_search_billed_per_query(model, local_model_cost_map):
|
||||
"""
|
||||
Gemini 3.x bills web search per individual query (web_search_billing_unit == "per_query"),
|
||||
so N searches cost N * $0.014.
|
||||
|
||||
Regression for the bug where the billing unit was dropped between the pricing JSON and the
|
||||
cost calculator: the field was missing from the ModelInfoBase TypedDict and from the
|
||||
ModelInfoBase(...) constructor in _get_model_info_helper, so get_model_info returned it as
|
||||
None and cost_per_web_search_request fell back to the per_prompt clamp, collapsing N queries
|
||||
to a single charge. The "gemini/..." case additionally covers response_cost_calculator
|
||||
resolving a provider-prefixed model name that get_model_info cannot map under vertex_ai.
|
||||
"""
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
||||
web_search_requests = 2
|
||||
model_info = litellm.get_model_info(model)
|
||||
assert model_info["web_search_billing_unit"] == "per_query"
|
||||
per_query_cost = model_info["search_context_cost_per_query"][
|
||||
"search_context_size_medium"
|
||||
]
|
||||
expected_cost = per_query_cost * web_search_requests
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=11,
|
||||
completion_tokens=100,
|
||||
total_tokens=111,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=11, web_search_requests=web_search_requests
|
||||
),
|
||||
)
|
||||
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
model=model,
|
||||
usage=usage,
|
||||
response_object=None,
|
||||
custom_llm_provider="vertex_ai",
|
||||
standard_built_in_tools_params=None,
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(expected_cost), (
|
||||
f"Expected {web_search_requests} x ${per_query_cost} = ${expected_cost} "
|
||||
f"per_query search fee, got ${cost}"
|
||||
)
|
||||
|
||||
|
||||
def test_gemini_combined_search_and_maps_costs_are_additive(local_model_cost_map):
|
||||
"""A prompt grounded with both Google Search and Google Maps pays both fees."""
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
|
@ -388,6 +440,86 @@ def test_gemini_combined_search_and_maps_costs_are_additive(local_model_cost_map
|
|||
assert cost == pytest.approx(search_rate * 2 + maps_rate)
|
||||
|
||||
|
||||
def test_gemini_2x_web_search_still_billed_per_prompt(local_model_cost_map):
|
||||
"""
|
||||
Gemini 2.x bills web search per grounded prompt: multiple internal queries are one flat
|
||||
$0.035 fee. Guards the per_prompt clamp against the per_query plumbing, which makes
|
||||
web_search_billing_unit always present on the resolved ModelInfo (None for 2.x), so the
|
||||
clamp must treat a None billing unit as per_prompt rather than skipping the clamp.
|
||||
"""
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
||||
model = "vertex_ai/gemini-2.5-flash"
|
||||
model_info = litellm.get_model_info(model)
|
||||
assert not model_info.get("web_search_billing_unit")
|
||||
expected_cost = model_info["search_context_cost_per_query"][
|
||||
"search_context_size_medium"
|
||||
]
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=11,
|
||||
completion_tokens=100,
|
||||
total_tokens=111,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=11, web_search_requests=2
|
||||
),
|
||||
)
|
||||
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
model=model,
|
||||
usage=usage,
|
||||
response_object=None,
|
||||
custom_llm_provider="vertex_ai",
|
||||
standard_built_in_tools_params=None,
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(expected_cost), (
|
||||
f"Expected flat ${expected_cost} per_prompt search fee (2 queries clamped to 1), "
|
||||
f"got ${cost}"
|
||||
)
|
||||
|
||||
|
||||
def test_web_search_provider_prefix_fallback_does_not_misprice_non_gemini_model(
|
||||
local_model_cost_map,
|
||||
):
|
||||
"""
|
||||
Regression for the provider-prefix fallback in _handle_web_search_cost. When the initial
|
||||
get_model_info lookup fails for a "/"-containing model, the retry re-resolves model_info from
|
||||
the prefix and must adopt that prefix's provider for routing. Otherwise an unrelated model
|
||||
(here OpenRouter, which carries no web search pricing) is re-resolved but still routed through
|
||||
the request's vertex_ai Gemini calculator, which charges its $0.035 per_prompt default for a
|
||||
model that should cost nothing for web search.
|
||||
"""
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
||||
model = "openrouter/google/gemini-3.1-flash-lite"
|
||||
model_info = litellm.get_model_info(model)
|
||||
assert model_info["litellm_provider"] == "openrouter"
|
||||
assert not model_info.get("search_context_cost_per_query")
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=11,
|
||||
completion_tokens=100,
|
||||
total_tokens=111,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=11, web_search_requests=2
|
||||
),
|
||||
)
|
||||
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
model=model,
|
||||
usage=usage,
|
||||
response_object=None,
|
||||
custom_llm_provider="vertex_ai",
|
||||
standard_built_in_tools_params=None,
|
||||
)
|
||||
|
||||
assert cost == 0.0, (
|
||||
"A non-Gemini provider-prefixed model with no web search pricing must not be charged "
|
||||
f"the vertex_ai per_prompt default via the prefix fallback, got ${cost}"
|
||||
)
|
||||
|
||||
|
||||
def _openai_responses_with_web_search_calls(model, num_calls):
|
||||
from openai.types.responses.response_function_web_search import (
|
||||
ActionSearch,
|
||||
|
|
|
|||
|
|
@ -3037,6 +3037,28 @@ def test_add_cache_point_tool_block_passes_ttl_for_claude_4_5(monkeypatch):
|
|||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", old_env)
|
||||
|
||||
|
||||
def test_add_cache_point_tool_block_stands_down_for_model_without_prompt_caching(monkeypatch):
|
||||
"""A tool carrying cache_control must not become a cachePoint for a Bedrock model
|
||||
whose cost-map entry lacks prompt caching support, since Bedrock rejects the whole
|
||||
request. An unmapped id keeps emitting so ARN deployments do not lose caching."""
|
||||
from litellm.litellm_core_utils.prompt_templates.factory import (
|
||||
add_cache_point_tool_block,
|
||||
)
|
||||
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
tool = {"cache_control": {"type": "ephemeral"}}
|
||||
|
||||
assert add_cache_point_tool_block(tool, model="nvidia.nemotron-super-3-120b") is None
|
||||
assert add_cache_point_tool_block(tool, model="us.nvidia.nemotron-super-3-120b") is None
|
||||
assert add_cache_point_tool_block(
|
||||
tool, model="arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/abc123"
|
||||
) == {"cachePoint": {"type": "default"}}
|
||||
assert add_cache_point_tool_block(tool, model="us.anthropic.claude-sonnet-4-5-20250929-v1:0") == {
|
||||
"cachePoint": {"type": "default"}
|
||||
}
|
||||
|
||||
|
||||
def test_bedrock_tools_pt_passes_ttl_for_claude_4_5(monkeypatch):
|
||||
"""
|
||||
End-to-end: _bedrock_tools_pt should produce cachePoint blocks with ttl
|
||||
|
|
|
|||
|
|
@ -802,6 +802,10 @@ def test_shipped_rules_flag_unmapped_wandb_ids_as_reasoning(shipped_cost_map):
|
|||
assert litellm.supports_reasoning(model="zai-org/GLM-6-Turbo", custom_llm_provider="wandb") is True
|
||||
|
||||
|
||||
def test_shipped_wandb_rule_does_not_fill_missing_mapped_entries(shipped_cost_map):
|
||||
assert match_fill_missing_generalizations("wandb/meta-llama/Llama-3.1-8B-Instruct", "wandb") is None
|
||||
|
||||
|
||||
def test_shipped_wandb_rule_is_anchored_to_the_wandb_namespace(shipped_cost_map):
|
||||
"""``^wandb/`` is anchored, so it cannot leak onto another provider's ids."""
|
||||
assert match_capability_generalizations("wandb/some-new-model") == {"supports_reasoning": True}
|
||||
|
|
@ -900,6 +904,25 @@ def test_shipped_openai_reasoning_rule_matches_only_openai(shipped_cost_map):
|
|||
assert match_fill_missing_generalizations("gpt-5.4", "openrouter") is None
|
||||
|
||||
|
||||
def test_shipped_claude_thinking_rules_backfill_only_anthropic(shipped_cost_map):
|
||||
model = "perplexity/anthropic/claude-sonnet-4-6"
|
||||
assert model in litellm.model_cost
|
||||
raw_entry = litellm.model_cost[model]
|
||||
assert "supports_adaptive_thinking" not in raw_entry
|
||||
assert "max_input_tokens" not in raw_entry
|
||||
|
||||
info = litellm.get_model_info(model="anthropic/claude-sonnet-4-6", custom_llm_provider="perplexity")
|
||||
assert info.get("supports_adaptive_thinking") is None
|
||||
assert info.get("supports_legacy_thinking") is None
|
||||
assert info.get("max_input_tokens") is None
|
||||
assert match_fill_missing_generalizations("claude-sonnet-4-6", "anthropic") == {
|
||||
"supports_adaptive_thinking": True,
|
||||
"supports_legacy_thinking": True,
|
||||
"supports_tool_search": True,
|
||||
}
|
||||
assert match_fill_missing_generalizations("claude-sonnet-4-6", "perplexity") is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,provider,tool_search",
|
||||
[
|
||||
|
|
@ -924,3 +947,25 @@ def test_shipped_tool_search_rule_version_boundaries(shipped_cost_map, model, pr
|
|||
assert info.get("supports_tool_search") is tool_search, model
|
||||
|
||||
|
||||
def test_shipped_tool_search_rule_fills_mapped_claude_entries_without_flag(shipped_cost_map):
|
||||
"""A mapped Claude 4.5+ entry with no supports_tool_search key gets it from the rule
|
||||
on Anthropic direct, Vertex and Bedrock, a mapped pre-4.5 entry stays without one,
|
||||
and Azure Foundry and reseller copies of the same model are not touched."""
|
||||
for key, model, provider in (
|
||||
("claude-opus-4-7", "claude-opus-4-7", "anthropic"),
|
||||
("vertex_ai/claude-opus-5", "claude-opus-5", "vertex_ai"),
|
||||
):
|
||||
assert "supports_tool_search" not in litellm.model_cost[key]
|
||||
assert litellm.get_model_info(model, custom_llm_provider=provider)["supports_tool_search"] is True
|
||||
|
||||
assert "supports_tool_search" not in litellm.model_cost["claude-opus-4-1"]
|
||||
opus_4_1_info = litellm.get_model_info("claude-opus-4-1", custom_llm_provider="anthropic")
|
||||
assert opus_4_1_info.get("supports_tool_search") is None
|
||||
|
||||
assert "supports_tool_search" not in litellm.model_cost["azure_ai/claude-opus-5"]
|
||||
azure_opus_5_info = litellm.get_model_info("claude-opus-5", custom_llm_provider="azure_ai")
|
||||
assert azure_opus_5_info.get("supports_tool_search") is None
|
||||
|
||||
assert match_fill_missing_generalizations("claude-opus-5", "bedrock")["supports_tool_search"] is True
|
||||
assert "supports_tool_search" not in match_fill_missing_generalizations("claude-opus-5", "azure_ai")
|
||||
assert match_fill_missing_generalizations("claude-opus-5", "perplexity") is None
|
||||
|
|
|
|||
|
|
@ -396,6 +396,47 @@ class TestGetRouterDeploymentModelInfo:
|
|||
assert logging_obj.get_router_deployment_model_info() is None
|
||||
|
||||
|
||||
def test_a_published_batch_rate_never_displaces_a_declared_standard_rate(self) -> None:
|
||||
"""Ownership is per token direction, not per field.
|
||||
|
||||
Filling the batch field from the published entry let that rate win, so a
|
||||
deployment configuring only its standard rate had batches billed at the
|
||||
published batch price instead of half the rate it configured.
|
||||
"""
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
||||
model = "ft:gpt-3.5-turbo"
|
||||
published = litellm.get_model_info(model=model)
|
||||
assert published["input_cost_per_token_batches"] is not None
|
||||
|
||||
deployment_id = "deploy-standard-input-only-1"
|
||||
litellm.model_cost[deployment_id] = {
|
||||
"id": deployment_id,
|
||||
"input_cost_per_token": 1e-06,
|
||||
"litellm_provider": "openai",
|
||||
"mode": "chat",
|
||||
}
|
||||
obj = LiteLLMLoggingObj(
|
||||
model=model,
|
||||
messages=[],
|
||||
stream=False,
|
||||
call_type="aretrieve_batch",
|
||||
start_time=time.time(),
|
||||
litellm_call_id="direction-ownership",
|
||||
function_id="f",
|
||||
)
|
||||
obj.litellm_params = {"litellm_metadata": {"model_info": {"id": deployment_id}}, "model": model}
|
||||
obj.model_call_details["model"] = model
|
||||
try:
|
||||
info = obj.get_router_deployment_model_info()
|
||||
assert info is not None
|
||||
assert info["input_cost_per_token"] == 1e-06
|
||||
assert info["input_cost_per_token_batches"] is None
|
||||
assert info["output_cost_per_token"] == published["output_cost_per_token"]
|
||||
assert info["output_cost_per_token_batches"] == published["output_cost_per_token_batches"]
|
||||
finally:
|
||||
litellm.model_cost.pop(deployment_id, None)
|
||||
|
||||
def test_merging_does_not_mutate_the_cached_model_info(self) -> None:
|
||||
"""The published-rate merge must not write into get_model_info's lru-cached dict.
|
||||
|
||||
|
|
|
|||
|
|
@ -1120,6 +1120,45 @@ def test_stream_chunk_builder_tolerates_trailing_chunk_without_choices():
|
|||
assert response.choices[0].message.content == "Hello world"
|
||||
|
||||
|
||||
def test_anthropic_speed_and_geo_survive_stream_assembly():
|
||||
"""Anthropic prices fast mode and non-global regions with a multiplier read off
|
||||
``usage.speed`` / ``usage.inference_geo``. Dropping them while reassembling a stream
|
||||
bills streamed fast-mode calls at the standard rate."""
|
||||
from litellm.llms.anthropic.cost_calculation import cost_per_token
|
||||
|
||||
def _usage(**extra):
|
||||
usage = Usage(completion_tokens=100, prompt_tokens=1000, total_tokens=1100)
|
||||
for key, value in extra.items():
|
||||
setattr(usage, key, value)
|
||||
return usage
|
||||
|
||||
def _chunk(usage):
|
||||
return ModelResponseStream(
|
||||
id="chatcmpl-1",
|
||||
created=1745513206,
|
||||
model="claude-opus-4-8",
|
||||
choices=[StreamingChoices(finish_reason="stop", index=0, delta=Delta(content="Hi"))],
|
||||
usage=usage,
|
||||
)
|
||||
|
||||
fast_chunk = _chunk(_usage(speed="fast", inference_geo="global"))
|
||||
fast_usage = ChunkProcessor(chunks=[fast_chunk]).calculate_usage(
|
||||
chunks=[fast_chunk], model="claude-opus-4-8", completion_output="Hi"
|
||||
)
|
||||
standard_chunk = _chunk(_usage(inference_geo="global"))
|
||||
standard_usage = ChunkProcessor(chunks=[standard_chunk]).calculate_usage(
|
||||
chunks=[standard_chunk], model="claude-opus-4-8", completion_output="Hi"
|
||||
)
|
||||
|
||||
assert fast_usage.speed == "fast"
|
||||
assert fast_usage.inference_geo == "global"
|
||||
assert getattr(standard_usage, "speed", None) is None
|
||||
|
||||
fast_cost = sum(cost_per_token(model="claude-opus-4-8", usage=fast_usage))
|
||||
standard_cost = sum(cost_per_token(model="claude-opus-4-8", usage=standard_usage))
|
||||
assert fast_cost == pytest.approx(standard_cost * 2.0)
|
||||
|
||||
|
||||
def test_prompt_tokens_details_survive_later_usage_chunk_without_details():
|
||||
"""Regression for #34801: a trailing usage chunk that omits
|
||||
`prompt_tokens_details` must not wipe the OpenAI cache-read/cache-write split,
|
||||
|
|
|
|||
|
|
@ -2464,6 +2464,42 @@ def test_get_max_tokens_for_model_none():
|
|||
assert max_tokens == 4096
|
||||
|
||||
|
||||
def test_get_config_with_model_uses_dynamic_max_tokens():
|
||||
"""
|
||||
Test that get_config returns dynamic max_tokens based on model.
|
||||
|
||||
Fixes: https://github.com/BerriAI/litellm/issues/8835
|
||||
"""
|
||||
|
||||
def _mock_get_max_tokens(model):
|
||||
"""Return expected max_output_tokens for each model."""
|
||||
model_map = {
|
||||
"claude-3-sonnet-20240229": 4096,
|
||||
"claude-3-5-sonnet-20241022": 8192,
|
||||
"claude-3-7-sonnet-20250219": 64000,
|
||||
}
|
||||
result = model_map.get(model)
|
||||
if result is None:
|
||||
raise Exception(f"Model {model} not found")
|
||||
return result
|
||||
|
||||
with patch(
|
||||
"litellm.llms.anthropic.chat.transformation.get_max_tokens",
|
||||
side_effect=_mock_get_max_tokens,
|
||||
):
|
||||
# Claude 3 model should get 4096
|
||||
config_claude3 = AnthropicConfig.get_config(model="claude-3-sonnet-20240229")
|
||||
assert config_claude3["max_tokens"] == 4096
|
||||
|
||||
# Claude 3.5 model should get 8192
|
||||
config_claude35 = AnthropicConfig.get_config(model="claude-3-5-sonnet-20241022")
|
||||
assert config_claude35["max_tokens"] == 8192
|
||||
|
||||
# Claude 3.7 model should get 64000 (64K default, 128K requires beta header)
|
||||
config_claude37 = AnthropicConfig.get_config(model="claude-3-7-sonnet-20250219")
|
||||
assert config_claude37["max_tokens"] == 64000
|
||||
|
||||
|
||||
def test_get_config_without_model_uses_fallback():
|
||||
"""
|
||||
Test that get_config without model parameter uses 4096 fallback.
|
||||
|
|
@ -3166,6 +3202,27 @@ def test_max_effort_accepted_for_opus_47():
|
|||
assert result["output_config"]["effort"] == "max"
|
||||
|
||||
|
||||
def test_effort_beta_header_not_injected_for_46_models():
|
||||
"""
|
||||
Test that is_effort_used returns False for Claude 4.6 models.
|
||||
|
||||
Claude 4.6 models use output_config as a stable API feature —
|
||||
no beta header should be injected.
|
||||
"""
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
|
||||
model_info = AnthropicModelInfo()
|
||||
|
||||
for model in ["claude-opus-4-6-20250514", "claude-sonnet-4-6-20260219"]:
|
||||
# Even with output_config present, should return False for 4.6 models
|
||||
result = model_info.is_effort_used(
|
||||
optional_params={"output_config": {"effort": "high"}},
|
||||
model=model,
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result is False, f"is_effort_used should return False for {model}"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
[
|
||||
|
|
@ -3261,6 +3318,23 @@ def test_reasoning_effort_minimal_floors_at_anthropic_provider_minimum():
|
|||
assert result["thinking"]["budget_tokens"] >= 1024
|
||||
|
||||
|
||||
def test_effort_beta_header_still_injected_for_older_models():
|
||||
"""
|
||||
Test that is_effort_used still returns True for pre-4.6 models
|
||||
when output_config is present.
|
||||
"""
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
|
||||
model_info = AnthropicModelInfo()
|
||||
|
||||
result = model_info.is_effort_used(
|
||||
optional_params={"output_config": {"effort": "low"}},
|
||||
model="claude-opus-4-5-20251101",
|
||||
custom_llm_provider="anthropic",
|
||||
)
|
||||
assert result is True
|
||||
|
||||
|
||||
def test_code_execution_tool_results_extraction():
|
||||
"""
|
||||
Test that code execution tool results (bash_code_execution_tool_result,
|
||||
|
|
@ -4075,6 +4149,48 @@ def test_fast_mode_usage_calculation():
|
|||
assert usage.speed == "fast"
|
||||
|
||||
|
||||
def test_fast_mode_cost_calculation():
|
||||
"""
|
||||
Test that fast mode applies the 'fast' multiplier from provider_specific_entry
|
||||
on top of the base model cost (1.1x for claude-opus-4-6).
|
||||
"""
|
||||
|
||||
from litellm.llms.anthropic.cost_calculation import cost_per_token
|
||||
from litellm.types.utils import Usage
|
||||
|
||||
base_prompt = 0.005
|
||||
base_completion = 0.025
|
||||
|
||||
with (
|
||||
patch(
|
||||
"litellm.llms.anthropic.cost_calculation.generic_cost_per_token"
|
||||
) as mock_cost,
|
||||
patch("litellm.get_model_info") as mock_info,
|
||||
):
|
||||
mock_cost.return_value = (base_prompt, base_completion)
|
||||
mock_info.return_value = {"provider_specific_entry": {"fast": 1.1, "us": 1.1}}
|
||||
|
||||
usage_fast = Usage(
|
||||
prompt_tokens=1000,
|
||||
completion_tokens=1000,
|
||||
speed="fast",
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model="claude-opus-4-6",
|
||||
usage=usage_fast,
|
||||
)
|
||||
|
||||
# generic_cost_per_token called with the plain base model name
|
||||
mock_cost.assert_called_once()
|
||||
assert mock_cost.call_args[1]["model"] == "claude-opus-4-6"
|
||||
assert mock_cost.call_args[1]["custom_llm_provider"] == "anthropic"
|
||||
|
||||
# 1.1x multiplier applied
|
||||
assert abs(prompt_cost - base_prompt * 1.1) < 1e-10
|
||||
assert abs(completion_cost - base_completion * 1.1) < 1e-10
|
||||
|
||||
|
||||
def test_fast_mode_with_inference_geo():
|
||||
"""
|
||||
Test that fast mode + inference_geo both apply their multipliers from
|
||||
|
|
@ -5929,6 +6045,35 @@ def test_sampling_params_forwarded_on_models_that_accept_them(model):
|
|||
assert result["top_p"] == 0.9
|
||||
|
||||
|
||||
def test_sampling_param_gating_driven_by_model_map_flag(monkeypatch):
|
||||
"""The drop/raise decision must come from ``supports_sampling_params`` in
|
||||
the model map, not just name matching: a flagged entry gates a model whose
|
||||
name says nothing, and an explicit ``true`` overrides the name fallback."""
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost, "claude-zeta-9", {"supports_sampling_params": False}
|
||||
)
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost, "claude-fable-5-test", {"supports_sampling_params": True}
|
||||
)
|
||||
config = AnthropicConfig()
|
||||
|
||||
flagged_off = config.map_openai_params(
|
||||
non_default_params={"top_p": 0.9},
|
||||
optional_params={},
|
||||
model="claude-zeta-9",
|
||||
drop_params=True,
|
||||
)
|
||||
assert "top_p" not in flagged_off
|
||||
|
||||
flagged_on = config.map_openai_params(
|
||||
non_default_params={"top_p": 0.9},
|
||||
optional_params={},
|
||||
model="claude-fable-5-test",
|
||||
drop_params=True,
|
||||
)
|
||||
assert flagged_on["top_p"] == 0.9
|
||||
|
||||
|
||||
def test_top_k_dropped_at_transform_for_models_that_removed_it():
|
||||
"""``top_k`` is a provider-specific kwarg that bypasses
|
||||
``map_openai_params``, so it must be stripped at the transform_request
|
||||
|
|
|
|||
|
|
@ -180,6 +180,32 @@ def test_a_gpt_5_name_without_a_foundry_row_keeps_reading_its_own_entry(
|
|||
assert optional_params["logprobs"] is True
|
||||
|
||||
|
||||
def test_azure_ai_grok_stop_parameter_handling():
|
||||
"""
|
||||
Test that Grok models properly handle stop parameter filtering in Azure AI Studio.
|
||||
"""
|
||||
config = AzureAIStudioConfig()
|
||||
|
||||
# Test Grok model detection
|
||||
assert config._supports_stop_reason("grok-4-fast") is False
|
||||
assert config._supports_stop_reason("grok-4.3") is False
|
||||
assert config._supports_stop_reason("grok-4") is False
|
||||
assert config._supports_stop_reason("grok-3-mini") is False
|
||||
assert config._supports_stop_reason("grok-code-fast") is False
|
||||
assert config._supports_stop_reason("gpt-4") is True
|
||||
|
||||
# Test supported parameters for Grok models
|
||||
for model in ("grok-4-fast", "grok-4.3"):
|
||||
grok_params = config.get_supported_openai_params(model)
|
||||
assert (
|
||||
"stop" not in grok_params
|
||||
), "Grok models should not support stop parameter"
|
||||
|
||||
# Test supported parameters for non-Grok models
|
||||
gpt_params = config.get_supported_openai_params("gpt-4")
|
||||
assert "stop" in gpt_params, "GPT models should support stop parameter"
|
||||
|
||||
|
||||
def test_azure_model_router_response_shows_actual_model():
|
||||
"""
|
||||
Test that Azure Model Router returns the actual model used in the response,
|
||||
|
|
|
|||
|
|
@ -317,6 +317,46 @@ class TestProviderConfigManagerAzureAnthropicMessages:
|
|||
assert config is None
|
||||
|
||||
|
||||
def test_messages_thinking_shape_follows_exact_azure_entry_flag(local_model_cost_map, monkeypatch):
|
||||
"""The Azure messages config must probe capabilities under ``azure_ai`` so an
|
||||
operator setting ``supports_adaptive_thinking: false`` on the exact
|
||||
``azure_ai/claude-opus-4-8`` entry beats the unmodified ``anthropic`` entry.
|
||||
With the inherited ``"anthropic"`` provider default the flip was ignored and
|
||||
the transform kept emitting ``thinking.type='adaptive'``."""
|
||||
import litellm
|
||||
|
||||
config = AzureAnthropicMessagesConfig()
|
||||
|
||||
def transform():
|
||||
return config.transform_anthropic_messages_request(
|
||||
model="claude-opus-4-8",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
anthropic_messages_optional_request_params={
|
||||
"max_tokens": 4096,
|
||||
"reasoning_effort": "medium",
|
||||
},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
result = transform()
|
||||
assert result.get("thinking") == {"type": "adaptive", "display": "summarized"}
|
||||
assert result.get("output_config") == {"effort": "medium"}
|
||||
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost["azure_ai/claude-opus-4-8"], "supports_adaptive_thinking", False
|
||||
)
|
||||
litellm.get_model_info.cache_clear()
|
||||
assert litellm.model_cost["claude-opus-4-8"]["supports_adaptive_thinking"] is True
|
||||
|
||||
flipped = transform()
|
||||
thinking = flipped.get("thinking")
|
||||
assert isinstance(thinking, dict)
|
||||
assert thinking.get("type") == "enabled"
|
||||
assert isinstance(thinking.get("budget_tokens"), int)
|
||||
assert "output_config" not in flipped
|
||||
|
||||
|
||||
def _azure_transform(model, messages, system=None):
|
||||
config = AzureAnthropicMessagesConfig()
|
||||
params = {"max_tokens": 256}
|
||||
|
|
|
|||
|
|
@ -194,6 +194,32 @@ def test_transform_usage_reads_invoke_model_count_suffixed_cache_keys(
|
|||
assert openai_usage.total_tokens == 12270
|
||||
|
||||
|
||||
def test_bedrock_invoke_nova_cache_read_billed_at_discounted_rate(monkeypatch):
|
||||
"""Nova cache reads are billed at the entry's discounted cache read rate; without a
|
||||
``cache_read_input_token_cost`` entry the cached tokens were billed at nothing."""
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
usage = ConverseTokenUsageBlock(
|
||||
**{
|
||||
"inputTokens": 5,
|
||||
"outputTokens": 3,
|
||||
"totalTokens": 12270,
|
||||
"cacheReadInputTokenCount": 12262,
|
||||
"cacheWriteInputTokenCount": 0,
|
||||
}
|
||||
)
|
||||
openai_usage = AmazonConverseConfig().transform_usage(usage)
|
||||
model = "bedrock/invoke/us.amazon.nova-pro-v1:0"
|
||||
prompt_cost, completion_cost = litellm.cost_calculator.cost_per_token(model=model, usage_object=openai_usage)
|
||||
model_info = litellm.get_model_info(model=model)
|
||||
assert 0 < model_info["cache_read_input_token_cost"] < model_info["input_cost_per_token"]
|
||||
assert prompt_cost == pytest.approx(
|
||||
5 * model_info["input_cost_per_token"] + 12262 * model_info["cache_read_input_token_cost"]
|
||||
)
|
||||
assert prompt_cost > 5 * model_info["input_cost_per_token"]
|
||||
assert completion_cost == pytest.approx(3 * model_info["output_cost_per_token"])
|
||||
|
||||
|
||||
def test_transform_usage_with_reasoning_content():
|
||||
"""Test that completion_tokens_details correctly tracks reasoning vs text tokens."""
|
||||
usage = ConverseTokenUsageBlock(
|
||||
|
|
@ -5444,6 +5470,87 @@ def test_cache_control_injection_tool_config_drops_ttl_for_unsupported_model():
|
|||
assert tools[-1] == {"cachePoint": {"type": "default"}}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("model", "expects_cache_points"),
|
||||
[
|
||||
pytest.param("nvidia.nemotron-super-3-120b", False, id="mapped-model-without-prompt-caching"),
|
||||
pytest.param("us.nvidia.nemotron-super-3-120b", False, id="regional-prefix-resolves-through-base-model"),
|
||||
pytest.param(
|
||||
"us.anthropic.claude-3-5-sonnet-20240620-v1:0", False, id="claude-named-but-not-caching-on-bedrock"
|
||||
),
|
||||
pytest.param("us.anthropic.claude-sonnet-4-5-20250929-v1:0", True, id="mapped-model-with-prompt-caching"),
|
||||
pytest.param(
|
||||
"arn:aws:bedrock:us-east-1:123456789012:application-inference-profile/abc123",
|
||||
True,
|
||||
id="unmapped-arn-keeps-emitting",
|
||||
),
|
||||
pytest.param("global.openai.gpt-6-astra", False, id="openai-family-implicit-caching-only"),
|
||||
pytest.param("openai.gpt-oss-120b-1:0", False, id="openai-gpt-oss"),
|
||||
pytest.param("us.openai.gpt-99-unmapped", False, id="unmapped-openai-family-still-suppressed"),
|
||||
],
|
||||
)
|
||||
def test_cache_points_emitted_only_for_models_that_support_prompt_caching(model, expects_cache_points, monkeypatch):
|
||||
"""Bedrock rejects cachePoint blocks for models without prompt caching support
|
||||
("You invoked an unsupported model or your request did not allow prompt caching"),
|
||||
and clients like Claude Code attach cache_control to every request, so a map-known
|
||||
model without the capability must not receive them. Unmapped ids (application
|
||||
inference profile ARNs, models newer than the map) keep emitting so existing
|
||||
caching setups never silently degrade."""
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
body = AmazonConverseConfig().transform_request(
|
||||
model=model,
|
||||
messages=[
|
||||
{"role": "system", "content": [{"type": "text", "text": "sys", "cache_control": {"type": "ephemeral"}}]},
|
||||
{"role": "user", "content": [{"type": "text", "text": "hi", "cache_control": {"type": "ephemeral"}}]},
|
||||
],
|
||||
optional_params={},
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert ("cachePoint" in json.dumps(body)) is expects_cache_points
|
||||
assert body["system"][0]["text"] == "sys"
|
||||
assert body["messages"][0]["content"][0]["text"] == "hi"
|
||||
|
||||
|
||||
def test_tool_config_cachepoint_not_placed_or_credited_for_model_without_prompt_caching(monkeypatch):
|
||||
"""The tool_config injection point must stand down with the rest of the cachePoint
|
||||
emission when the model cannot cache, and spend attribution must not credit the
|
||||
gateway for a breakpoint that was never placed."""
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
bucket: dict = {"user_api_key": "sk-test"}
|
||||
data = AmazonConverseConfig()._transform_request_helper(
|
||||
model="nvidia.nemotron-super-3-120b",
|
||||
system_content_blocks=[],
|
||||
optional_params={
|
||||
"tools": [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "get_weather",
|
||||
"description": "Get weather",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {"location": {"type": "string"}},
|
||||
"required": ["location"],
|
||||
},
|
||||
},
|
||||
}
|
||||
],
|
||||
"cache_control_injection_points": [{"location": "tool_config"}],
|
||||
},
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
litellm_params={"metadata": bucket, "litellm_metadata": None, "model_info": {"id": "dep-bedrock"}},
|
||||
)
|
||||
|
||||
assert "cachePoint" not in json.dumps(data.get("toolConfig", {}))
|
||||
assert "litellm_gateway_injected_cache" not in bucket
|
||||
|
||||
|
||||
def test_translate_response_format_json_schema_still_injects_tool():
|
||||
"""
|
||||
response_format with an explicit json_schema should still use the
|
||||
|
|
|
|||
|
|
@ -2900,6 +2900,49 @@ def test_bedrock_messages_tool_search_follows_claude_tool_search_rule(local_mode
|
|||
assert cfg._supports_tool_search_on_bedrock(model) is expected
|
||||
|
||||
|
||||
def test_bedrock_messages_thinking_shape_follows_exact_bedrock_entry_flag(
|
||||
local_model_cost_map, monkeypatch
|
||||
):
|
||||
"""The outbound thinking payload must follow the exact Bedrock cost-map entry.
|
||||
Before threading the caller's provider through the capability probes, the probe
|
||||
was pinned to ``"anthropic"``: the exact ``global.anthropic.claude-opus-4-8``
|
||||
entry was rejected by the provider match and the anthropic-scoped fallback rule
|
||||
forced ``thinking.type='adaptive'`` even with ``supports_adaptive_thinking``
|
||||
explicitly set to ``false`` on the entry."""
|
||||
import litellm
|
||||
|
||||
from litellm.types.router import GenericLiteLLMParams
|
||||
|
||||
model = "global.anthropic.claude-opus-4-8"
|
||||
cfg = AmazonAnthropicClaudeMessagesConfig()
|
||||
|
||||
def transform():
|
||||
return cfg.transform_anthropic_messages_request(
|
||||
model=model,
|
||||
messages=[{"role": "user", "content": [{"type": "text", "text": "Hello"}]}],
|
||||
anthropic_messages_optional_request_params={
|
||||
"max_tokens": 4096,
|
||||
"reasoning_effort": "medium",
|
||||
},
|
||||
litellm_params=GenericLiteLLMParams(),
|
||||
headers={},
|
||||
)
|
||||
|
||||
result = transform()
|
||||
assert result.get("thinking") == {"type": "adaptive", "display": "summarized"}
|
||||
assert result.get("output_config") == {"effort": "medium"}
|
||||
|
||||
monkeypatch.setitem(litellm.model_cost[model], "supports_adaptive_thinking", False)
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
flipped = transform()
|
||||
thinking = flipped.get("thinking")
|
||||
assert isinstance(thinking, dict)
|
||||
assert thinking.get("type") == "enabled"
|
||||
assert isinstance(thinking.get("budget_tokens"), int)
|
||||
assert "output_config" not in flipped
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"search_results, expected_evidence",
|
||||
[
|
||||
|
|
|
|||
|
|
@ -1296,6 +1296,37 @@ class TestMantleBaseSegment:
|
|||
the /openai/v1 base, everything else on /v1. An unmapped model defaults to /v1.
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,model_cost,expected",
|
||||
[
|
||||
(
|
||||
"openai.gpt-5.5",
|
||||
{"bedrock_mantle/openai.gpt-5.5": {"use_openai_responses_path": True}},
|
||||
"openai/v1",
|
||||
),
|
||||
(
|
||||
"google.gemma-4-31b",
|
||||
{
|
||||
"bedrock_mantle/google.gemma-4-31b": {
|
||||
"use_openai_responses_path": True
|
||||
}
|
||||
},
|
||||
"openai/v1",
|
||||
),
|
||||
(
|
||||
"openai.gpt-oss-120b",
|
||||
{"bedrock_mantle/openai.gpt-oss-120b": {}},
|
||||
"v1",
|
||||
),
|
||||
("openai.gpt-oss-120b", {}, "v1"),
|
||||
(None, {}, "v1"),
|
||||
],
|
||||
)
|
||||
def test_base_segment(self, model, model_cost, expected):
|
||||
from litellm.llms.bedrock_mantle.common_utils import mantle_base_segment
|
||||
|
||||
assert mantle_base_segment(model, model_cost) == expected
|
||||
|
||||
|
||||
class TestMantleSupportsResponses:
|
||||
"""The capability helper is data-driven (supported_endpoints / mode), with no
|
||||
|
|
|
|||
|
|
@ -442,6 +442,29 @@ class TestDashscopeCostCalculator:
|
|||
|
||||
assert math.isclose(completion_cost, 200 * 1.6e-06, rel_tol=1e-10)
|
||||
|
||||
def test_dashscope_model_zero_reasoning_rate_bills_reasoning_free(self):
|
||||
"""
|
||||
Regression: a model declaring an explicit zero reasoning rate had it treated as
|
||||
missing, billing reasoning tokens at the plain output rate instead of free.
|
||||
"""
|
||||
litellm.model_cost["dashscope/qwen-zero-reasoning-test"] = {
|
||||
"litellm_provider": "dashscope",
|
||||
"mode": "chat",
|
||||
"input_cost_per_token": 4e-07,
|
||||
"output_cost_per_token": 1.6e-06,
|
||||
"output_cost_per_reasoning_token": 0,
|
||||
}
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=500,
|
||||
completion_tokens=200,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(reasoning_tokens=150),
|
||||
)
|
||||
_, completion_cost = dashscope_cost_per_token(
|
||||
model="qwen-zero-reasoning-test", usage=usage
|
||||
)
|
||||
|
||||
assert math.isclose(completion_cost, 50 * 1.6e-06, rel_tol=1e-10)
|
||||
|
||||
def test_dashscope_tier_zero_reasoning_rate_bills_reasoning_free(self):
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -337,6 +337,39 @@ def test_get_supported_openai_params_preserves_generic_reasoning_fallback():
|
|||
assert "reasoning_effort" in supported_params
|
||||
|
||||
|
||||
def test_get_supported_openai_params_parallel_tool_calls_without_tool_choice(
|
||||
monkeypatch,
|
||||
):
|
||||
"""Test that parallel_tool_calls is gated on tools, not tool_choice."""
|
||||
config = FireworksAIConfig()
|
||||
model = "fireworks_ai/test-tools-without-tool-choice"
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost,
|
||||
model,
|
||||
{
|
||||
"supports_function_calling": True,
|
||||
"supports_tool_choice": False,
|
||||
},
|
||||
)
|
||||
|
||||
supported_params = config.get_supported_openai_params(model)
|
||||
|
||||
assert "tools" in supported_params
|
||||
assert "parallel_tool_calls" in supported_params
|
||||
assert "tool_choice" not in supported_params
|
||||
|
||||
|
||||
def test_get_provider_info_omits_false_supports_reasoning(monkeypatch):
|
||||
"""Test that Fireworks only overrides supports_reasoning for supported models."""
|
||||
config = FireworksAIConfig()
|
||||
model = "fireworks_ai/test-reasoning-false"
|
||||
monkeypatch.setitem(litellm.model_cost, model, {"supports_reasoning": False})
|
||||
|
||||
info = config.get_provider_info(model)
|
||||
|
||||
assert "supports_reasoning" not in info
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"api_base, expected_url_prefix",
|
||||
[
|
||||
|
|
@ -426,6 +459,14 @@ def test_transform_messages_helper_removes_provider_specific_fields():
|
|||
assert "provider_specific_fields" not in msg
|
||||
|
||||
|
||||
def test_unmapped_model_fallback_function_calling():
|
||||
"""Test that a model not in model_cost still defaults to supporting function calling for Fireworks."""
|
||||
config = FireworksAIConfig()
|
||||
model = "fireworks_ai/unmapped-future-model"
|
||||
info = config.get_provider_info(model)
|
||||
assert info["supports_function_calling"] is True
|
||||
|
||||
|
||||
def test_transform_messages_helper_strips_thinking_blocks_but_keeps_reasoning_content():
|
||||
"""Fireworks rejects thinking_blocks but requires reasoning_content to be replayed for reasoning_history."""
|
||||
config = FireworksAIConfig()
|
||||
|
|
@ -1050,6 +1091,26 @@ def test_transform_messages_helper_no_transform_inline():
|
|||
assert "#transform=inline" not in block["image_url"]
|
||||
|
||||
|
||||
def test_get_provider_info_vision_from_model_cost(monkeypatch):
|
||||
config = FireworksAIConfig()
|
||||
|
||||
vision_model = "fireworks_ai/test-vision-from-cost"
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost,
|
||||
vision_model,
|
||||
{"supports_vision": True, "supports_pdf_input": True},
|
||||
)
|
||||
info = config.get_provider_info(vision_model)
|
||||
assert info["supports_vision"] is True
|
||||
assert info["supports_pdf_input"] is True
|
||||
|
||||
no_vision_model = "fireworks_ai/test-no-vision-from-cost"
|
||||
monkeypatch.setitem(litellm.model_cost, no_vision_model, {})
|
||||
info_no_vision = config.get_provider_info(no_vision_model)
|
||||
assert info_no_vision.get("supports_vision") is not True
|
||||
assert "supports_pdf_input" not in info_no_vision
|
||||
|
||||
|
||||
def test_reasoning_effort_boolean_true_to_medium():
|
||||
config = FireworksAIConfig()
|
||||
result = config.map_openai_params(
|
||||
|
|
|
|||
|
|
@ -122,6 +122,20 @@ def test_off_peak_window_bills_cached_tokens_at_the_off_peak_input_rate_without_
|
|||
assert math.isclose(peak_prompt_cost, 1000 * STANDARD_INPUT_COST, rel_tol=1e-10)
|
||||
|
||||
|
||||
def test_off_peak_defaults_to_the_current_time():
|
||||
"""The proxy's cost dispatch passes no clock, so an all-day window has to apply on the
|
||||
default current time."""
|
||||
_register_off_peak_model(
|
||||
{"hours_utc": "00:00-00:00", "input_cost_per_token": 1e-08, "output_cost_per_token": 2e-08}
|
||||
)
|
||||
usage = _usage(prompt_tokens=1000, cached_tokens=0, completion_tokens=200)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model=OFF_PEAK_MODEL, usage=usage)
|
||||
|
||||
assert math.isclose(prompt_cost, 1000 * 1e-08, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, 200 * 2e-08, rel_tol=1e-10)
|
||||
|
||||
|
||||
COMPONENT_MODEL = "accounts/fireworks/models/cost-components-test"
|
||||
COMPONENT_INPUT_COST = 1e-06
|
||||
COMPONENT_OUTPUT_COST = 2e-06
|
||||
|
|
|
|||
|
|
@ -2226,6 +2226,18 @@ class TestReasoningFollowsModelSupport:
|
|||
)
|
||||
assert mapped["reasoning"] == reasoning
|
||||
|
||||
def test_an_explicit_supports_reasoning_false_beats_the_bundled_floor(self, local_model_cost_map, monkeypatch):
|
||||
overridden = {
|
||||
name: ({**entry, "supports_reasoning": False} if name == "o3" else entry)
|
||||
for name, entry in litellm.model_cost.items()
|
||||
}
|
||||
monkeypatch.setattr(litellm, "model_cost", overridden)
|
||||
mapped = OpenAIResponsesAPIConfig().map_openai_params(
|
||||
response_api_optional_params={"reasoning": {"effort": "medium"}},
|
||||
model="o3",
|
||||
drop_params=True,
|
||||
)
|
||||
assert "reasoning" not in mapped
|
||||
|
||||
def test_azure_deployments_keep_reasoning_even_on_a_non_reasoning_model_name(self, local_model_cost_map):
|
||||
mapped = AzureOpenAIResponsesAPIConfig().map_openai_params(
|
||||
|
|
|
|||
|
|
@ -1314,6 +1314,19 @@ def test_gpt5_6_forwards_reasoning_effort_max_for_the_responses_bridge(config: O
|
|||
assert params["reasoning_effort"] == "max"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", ["gpt-5.6", "gpt-5.6-sol", "gpt-5.6-terra", "gpt-5.6-luna"])
|
||||
def test_gpt5_6_never_advertises_reasoning_effort_max(model: str):
|
||||
"""/v1/chat/completions answers max with "Unsupported value: 'reasoning_effort' does not support
|
||||
'max' with this model. Supported values are: 'none', 'low', 'medium', 'high', and 'xhigh'", so no
|
||||
gpt-5.6 entry asserts supports_max_reasoning_effort and the advertised set stops at xhigh."""
|
||||
from litellm.router_utils.reasoning_effort_capability import resolve_supported_reasoning_efforts
|
||||
|
||||
resolved = resolve_supported_reasoning_efforts(litellm.get_model_info(model), deployment_is_mapped=True)
|
||||
assert resolved is not None
|
||||
assert "max" not in resolved
|
||||
assert "xhigh" in resolved
|
||||
|
||||
|
||||
def test_gpt5_6_keeps_reasoning_effort_max_on_the_responses_api(
|
||||
responses_config: OpenAIResponsesAPIConfig,
|
||||
):
|
||||
|
|
|
|||
|
|
@ -79,6 +79,19 @@ class TestTensormeshProviderConfig:
|
|||
matching the text_completion flag in provider_endpoints_support.json."""
|
||||
assert "tensormesh" in litellm.openai_text_completion_compatible_providers
|
||||
|
||||
def test_tensormesh_responses_api_enabled(self):
|
||||
"""Tensormesh declares /v1/responses in supported_endpoints, so litellm
|
||||
resolves a responses config for it."""
|
||||
from litellm.llms.openai_like.json_loader import JSONProviderRegistry
|
||||
from litellm.utils import ProviderConfigManager
|
||||
|
||||
assert JSONProviderRegistry.supports_responses_api("tensormesh") is True
|
||||
config = ProviderConfigManager.get_provider_responses_api_config(
|
||||
provider="tensormesh",
|
||||
model="tensormesh/openai/gpt-oss-120b",
|
||||
)
|
||||
assert config is not None
|
||||
assert config.custom_llm_provider == "tensormesh"
|
||||
|
||||
def test_tensormesh_router_config(self):
|
||||
"""Test that tensormesh can be used in Router configuration"""
|
||||
|
|
|
|||
|
|
@ -204,6 +204,18 @@ class TestPerplexityCostCalculator:
|
|||
assert math.isclose(prompt_cost, (1000 * 1e-07) + (100 * 2e-06), rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, (150 * 2e-07) + (50 * 3e-06) + 0.005, rel_tol=1e-10)
|
||||
|
||||
def test_off_peak_defaults_to_the_current_time(self):
|
||||
"""The proxy's cost dispatch passes no clock, so an all-day window has to apply on the
|
||||
default current time."""
|
||||
self._register_off_peak_model(
|
||||
{"hours_utc": "00:00-00:00", "input_cost_per_token": 1e-07, "output_cost_per_token": 2e-07}
|
||||
)
|
||||
usage = Usage(prompt_tokens=1000, completion_tokens=200, total_tokens=1200)
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(model=self.OFF_PEAK_MODEL, usage=usage)
|
||||
|
||||
assert math.isclose(prompt_cost, 1000 * 1e-07, rel_tol=1e-10)
|
||||
assert math.isclose(completion_cost, 200 * 2e-07, rel_tol=1e-10)
|
||||
|
||||
def test_provider_stated_cost_still_wins_inside_an_off_peak_window(self):
|
||||
"""A response that carries Perplexity's own metered cost bills that cost whatever the
|
||||
|
|
|
|||
|
|
@ -640,6 +640,58 @@ def test_get_vertex_url_global_region(stream, expected_endpoint_suffix):
|
|||
assert url == expected_url
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_cost_entry, vertex_region, expected_region",
|
||||
[
|
||||
# Model with supported_regions=["global"], no user region -> use "global"
|
||||
({"supported_regions": ["global"]}, None, "global"),
|
||||
# Model with supported_regions=["global"], user passes unsupported region -> override to "global"
|
||||
({"supported_regions": ["global"]}, "us-central1", "global"),
|
||||
# Model with supported_regions=["global"], user passes unsupported region -> override to "global"
|
||||
({"supported_regions": ["global"]}, "europe-west1", "global"),
|
||||
# Model with supported_regions=["us-west2"], no user region -> use "us-west2"
|
||||
({"supported_regions": ["us-west2"]}, None, "us-west2"),
|
||||
# Model with supported_regions=["us-west2", "us-central1"], user passes supported region -> respect it
|
||||
(
|
||||
{"supported_regions": ["us-west2", "us-central1"]},
|
||||
"us-central1",
|
||||
"us-central1",
|
||||
),
|
||||
# Model with supported_regions=["us-west2", "us-central1"], user passes unsupported region -> override
|
||||
(
|
||||
{"supported_regions": ["us-west2", "us-central1"]},
|
||||
"europe-west1",
|
||||
"us-west2",
|
||||
),
|
||||
# No model_cost entry, no user region -> default us-central1
|
||||
({}, None, "us-central1"),
|
||||
# No model_cost entry, user specifies region -> use specified region
|
||||
({}, "europe-west1", "europe-west1"),
|
||||
# No model_cost entry, user specifies region -> use specified region
|
||||
({}, "us-east1", "us-east1"),
|
||||
],
|
||||
)
|
||||
def test_get_vertex_region_global_only_model(
|
||||
model_cost_entry, vertex_region, expected_region
|
||||
):
|
||||
"""Test get_vertex_region resolves region from model_cost supported_regions"""
|
||||
import litellm
|
||||
from litellm.llms.vertex_ai.vertex_llm_base import VertexBase
|
||||
|
||||
vertex_base = VertexBase()
|
||||
|
||||
with patch.dict(
|
||||
litellm.model_cost,
|
||||
{"vertex_ai/test-model": model_cost_entry},
|
||||
clear=False,
|
||||
):
|
||||
result = vertex_base.get_vertex_region(
|
||||
vertex_region=vertex_region, model="test-model"
|
||||
)
|
||||
|
||||
assert result == expected_region
|
||||
|
||||
|
||||
def test_vertex_filter_format_uri():
|
||||
import json
|
||||
|
||||
|
|
|
|||
|
|
@ -181,6 +181,58 @@ class TestVertexAILyriaTextToSpeechConfig:
|
|||
|
||||
assert isinstance(config, VertexAILyriaTextToSpeechConfig)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("model", "vertex_ai_audio_api", "supported_audio_formats", "expected_url"),
|
||||
[
|
||||
(
|
||||
"future-lyria-predict",
|
||||
"lyria_predict",
|
||||
["wav"],
|
||||
"https://us-central1-aiplatform.googleapis.com/v1/projects/music-project/locations/"
|
||||
"us-central1/publishers/google/models/future-lyria-predict:predict",
|
||||
),
|
||||
(
|
||||
"future-music-interactions",
|
||||
"lyria_interactions",
|
||||
["mp3", "wav"],
|
||||
"https://aiplatform.googleapis.com/v1beta1/projects/music-project/locations/global/interactions",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_dispatches_from_model_metadata(
|
||||
self,
|
||||
monkeypatch,
|
||||
model,
|
||||
vertex_ai_audio_api,
|
||||
supported_audio_formats,
|
||||
expected_url,
|
||||
):
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost,
|
||||
f"vertex_ai/{model}",
|
||||
{
|
||||
"vertex_ai_audio_api": vertex_ai_audio_api,
|
||||
"supported_audio_formats": supported_audio_formats,
|
||||
},
|
||||
)
|
||||
|
||||
config = ProviderConfigManager.get_provider_text_to_speech_config(
|
||||
model=model,
|
||||
provider=LlmProviders.VERTEX_AI,
|
||||
)
|
||||
|
||||
assert isinstance(config, VertexAILyriaTextToSpeechConfig)
|
||||
assert (
|
||||
config.get_complete_url(
|
||||
model=model,
|
||||
api_base=None,
|
||||
litellm_params={
|
||||
"vertex_project": "music-project",
|
||||
"vertex_location": "us-central1",
|
||||
},
|
||||
)
|
||||
== expected_url
|
||||
)
|
||||
|
||||
def test_vertex_chirp_does_not_select_lyria_config(self):
|
||||
config = ProviderConfigManager.get_provider_text_to_speech_config(
|
||||
|
|
|
|||
|
|
@ -32,6 +32,49 @@ def test_get_supported_params_thinking():
|
|||
assert "thinking" in params
|
||||
|
||||
|
||||
def test_vertex_ai_anthropic_web_search_header_in_completion():
|
||||
"""Test that web search tool adds the required beta header for Vertex AI completion requests"""
|
||||
|
||||
from litellm.llms.anthropic.common_utils import AnthropicModelInfo
|
||||
|
||||
# Create the config instance
|
||||
model_info = AnthropicModelInfo()
|
||||
|
||||
# Test the header generation directly
|
||||
tools = [{"type": "web_search_20250305", "name": "web_search", "max_uses": 5}]
|
||||
|
||||
# Check if web search tool is detected
|
||||
web_search_detected = model_info.is_web_search_tool_used(tools=tools)
|
||||
assert web_search_detected is True, "Web search tool should be detected"
|
||||
|
||||
# Generate headers with is_vertex_request=True
|
||||
headers = model_info.get_anthropic_headers(
|
||||
api_key="test-key",
|
||||
web_search_tool_used=web_search_detected,
|
||||
is_vertex_request=True,
|
||||
)
|
||||
|
||||
# Assert that the anthropic-beta header with web-search is present
|
||||
assert "anthropic-beta" in headers, "anthropic-beta header should be present"
|
||||
assert (
|
||||
headers["anthropic-beta"] == "web-search-2025-03-05"
|
||||
), f"anthropic-beta should be 'web-search-2025-03-05', got: {headers['anthropic-beta']}"
|
||||
|
||||
# Test that header is NOT added for non-Vertex requests
|
||||
headers_non_vertex = model_info.get_anthropic_headers(
|
||||
api_key="test-key",
|
||||
web_search_tool_used=web_search_detected,
|
||||
is_vertex_request=False,
|
||||
)
|
||||
|
||||
# For non-Vertex (Anthropic-hosted), the web search header should NOT be in anthropic-beta
|
||||
# because Anthropic doesn't require it
|
||||
assert (
|
||||
"anthropic-beta" not in headers_non_vertex
|
||||
or "web-search" not in headers_non_vertex.get("anthropic-beta", "")
|
||||
), "anthropic-beta with web-search should not be present for non-Vertex requests"
|
||||
|
||||
|
||||
def test_vertex_ai_anthropic_context_management_compact_beta_header():
|
||||
"""Test that context_management with compact adds the correct beta header for Vertex AI"""
|
||||
config = VertexAIAnthropicConfig()
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ import httpx
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
|
||||
from litellm.llms.vertex_ai.videos.transformation import (
|
||||
VertexAIVideoConfig,
|
||||
_convert_image_to_vertex_format,
|
||||
|
|
@ -122,6 +123,25 @@ class TestVertexAIVideoConfig:
|
|||
)
|
||||
|
||||
|
||||
def test_veo_31_lite_provider_routing_from_local_model_map(
|
||||
self, monkeypatch: pytest.MonkeyPatch
|
||||
):
|
||||
model_cost = _load_model_cost_map(BACKUP_MODEL_COST_PATH)
|
||||
vertex_video_models = {
|
||||
model_name.removeprefix("vertex_ai/")
|
||||
for model_name, info in model_cost.items()
|
||||
if info.get("litellm_provider") == "vertex_ai-video-models"
|
||||
}
|
||||
monkeypatch.setattr(litellm, "vertex_ai_video_models", vertex_video_models)
|
||||
|
||||
model, custom_llm_provider, _, _ = get_llm_provider(
|
||||
model="veo-3.1-lite-generate-001"
|
||||
)
|
||||
|
||||
assert model == "veo-3.1-lite-generate-001"
|
||||
assert custom_llm_provider == "vertex_ai"
|
||||
|
||||
|
||||
def test_transform_video_create_request(self):
|
||||
"""Test transformation of video creation request."""
|
||||
prompt = "A cat playing with a ball of yarn"
|
||||
|
|
|
|||
|
|
@ -75,6 +75,21 @@ def wandb_request_mock(respx_mock: respx.MockRouter) -> respx.Route:
|
|||
class TestWandbConfig:
|
||||
"""Test class for WandB Inference functionality"""
|
||||
|
||||
@pytest.mark.parametrize("model", WANDB_REASONING_MODELS)
|
||||
def test_map_openai_params_preserves_reasoning_effort(self, wandb_test_config, model: str):
|
||||
assert litellm.model_cost[f"wandb/{model}"].get("supports_reasoning") is True
|
||||
supported_params = litellm.get_supported_openai_params(model=f"wandb/{model}")
|
||||
assert supported_params is not None
|
||||
assert "reasoning_effort" in supported_params
|
||||
|
||||
result = WandbConfig().map_openai_params(
|
||||
non_default_params={"reasoning_effort": "medium", "max_completion_tokens": 64},
|
||||
optional_params={},
|
||||
model=model,
|
||||
drop_params=True,
|
||||
)
|
||||
|
||||
assert result == {"reasoning_effort": "medium", "max_tokens": 64}
|
||||
|
||||
def test_default_api_base(self):
|
||||
"""Test that default API base is used when none is provided"""
|
||||
|
|
@ -228,6 +243,52 @@ class TestWandbConfig:
|
|||
assert request_body["max_tokens"] == 64
|
||||
assert "max_completion_tokens" not in request_body
|
||||
|
||||
@pytest.mark.respx(assert_all_called=False)
|
||||
@pytest.mark.parametrize("drop_params", [True, False])
|
||||
@pytest.mark.parametrize(
|
||||
"model,explicit_false",
|
||||
[
|
||||
("meta-llama/Llama-3.1-8B-Instruct", False),
|
||||
("openai/gpt-oss-20b", True),
|
||||
],
|
||||
)
|
||||
def test_wandb_completion_without_reasoning_support(
|
||||
self,
|
||||
wandb_test_config,
|
||||
wandb_request_mock: respx.Route,
|
||||
respx_mock: respx.MockRouter,
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
model: str,
|
||||
explicit_false: bool,
|
||||
drop_params: bool,
|
||||
):
|
||||
with monkeypatch.context() as context:
|
||||
if explicit_false:
|
||||
context.setitem(litellm.model_cost[f"wandb/{model}"], "supports_reasoning", False)
|
||||
|
||||
kwargs = {
|
||||
"model": f"wandb/{model}",
|
||||
"messages": [{"role": "user", "content": "Hello"}],
|
||||
"api_key": "fake-wandb-key",
|
||||
"api_base": "https://api.inference.wandb.ai/v1",
|
||||
"reasoning_effort": "medium",
|
||||
"drop_params": drop_params,
|
||||
}
|
||||
if not drop_params:
|
||||
with pytest.raises(litellm.UnsupportedParamsError, match="reasoning_effort"):
|
||||
completion(**kwargs)
|
||||
assert len(respx_mock.calls) == 0
|
||||
return
|
||||
|
||||
completion(**kwargs)
|
||||
assert wandb_request_mock.call_count == 1
|
||||
request_body = json.loads(wandb_request_mock.calls[0].request.content)
|
||||
assert request_body["model"] == model
|
||||
assert "reasoning_effort" not in request_body
|
||||
|
||||
supported_params = litellm.get_supported_openai_params(model=f"wandb/{model}")
|
||||
assert supported_params is not None
|
||||
assert "reasoning_effort" not in supported_params
|
||||
|
||||
@pytest.mark.respx()
|
||||
def test_wandb_completion_keeps_reasoning_effort_for_an_unregistered_model(
|
||||
|
|
|
|||
|
|
@ -7,6 +7,8 @@ from __future__ import annotations
|
|||
import json
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[4]
|
||||
PRICES_PATH = REPO_ROOT / "model_prices_and_context_window.json"
|
||||
BACKUP_PRICES_PATH = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
|
||||
|
|
@ -21,6 +23,12 @@ RESPONSES_ONLY_MODELS = (
|
|||
MAP_PATHS = (PRICES_PATH, BACKUP_PRICES_PATH)
|
||||
|
||||
|
||||
@pytest.fixture(scope="module", params=[p.name for p in MAP_PATHS])
|
||||
def cost_map(request: pytest.FixtureRequest) -> dict:
|
||||
path = next(p for p in MAP_PATHS if p.name == request.param)
|
||||
return json.loads(path.read_text(encoding="utf-8"))
|
||||
|
||||
|
||||
def test_both_cost_maps_agree_on_xai_entries():
|
||||
prices = json.loads(PRICES_PATH.read_text(encoding="utf-8"))
|
||||
backup = json.loads(BACKUP_PRICES_PATH.read_text(encoding="utf-8"))
|
||||
|
|
|
|||
|
|
@ -7,6 +7,7 @@ from litellm.litellm_core_utils.llm_cost_calc.utils import generic_cost_per_toke
|
|||
from litellm.llms.anthropic.cost_calculation import cost_per_token as anthropic_cost_per_token
|
||||
from litellm.proxy.spend_tracking.savings import (
|
||||
_baseline_usage,
|
||||
_resolve_model,
|
||||
compute_autorouter_savings,
|
||||
compute_savings_spend,
|
||||
marks_gateway_injection,
|
||||
|
|
@ -757,6 +758,84 @@ def test_a_switch_onto_a_partly_cached_model_still_pays_for_the_write():
|
|||
assert reported < if_treated_as_same_model / 10, "a mostly-cold switch must not be priced as a continuation"
|
||||
|
||||
|
||||
def test_a_baseline_that_prices_caching_implicitly_still_pays_for_its_prompt():
|
||||
"""OpenAI, Azure and Gemini entries carry no `cache_creation_input_token_cost`,
|
||||
because those providers cache implicitly and charge nothing to write. Leaving this
|
||||
request's written tokens in the creation bucket priced them at the 0.0 the cost
|
||||
resolver falls back to, so the baseline carried a 20k prompt for free and a first
|
||||
turn that saved money reported a loss. Those tokens are plain input on such a model.
|
||||
"""
|
||||
first_turn = _usage(fresh=0, cached=0, written=20_000, out=1_000)
|
||||
reported = compute_autorouter_savings(
|
||||
baseline_model="gpt-5",
|
||||
selected_model="claude-haiku-4-5",
|
||||
selected_provider="anthropic",
|
||||
usage=first_turn,
|
||||
conversation_continuing=False,
|
||||
)
|
||||
|
||||
gpt5 = litellm.get_model_info("gpt-5", "openai")
|
||||
assert gpt5.get("cache_creation_input_token_cost") is None, "pick a baseline with no cache-write rate"
|
||||
haiku = litellm.get_model_info("claude-haiku-4-5", "anthropic")
|
||||
baseline_pays_input = 20_000 * gpt5["input_cost_per_token"] + 1_000 * gpt5["output_cost_per_token"]
|
||||
actually_paid = 20_000 * haiku["cache_creation_input_token_cost"] + 1_000 * haiku["output_cost_per_token"]
|
||||
assert reported == pytest.approx(baseline_pays_input - actually_paid)
|
||||
assert reported > 0, "routing a cold first turn onto a cheaper model is a saving, not a loss"
|
||||
|
||||
|
||||
def _priced_chat_model_without_cache_read_rate() -> tuple[str, str, str]:
|
||||
"""A chat model the bundled map prices per token for input and output but not for cache
|
||||
reads, derived from the map itself: a hardcoded pick goes stale the moment the registry
|
||||
prices that model's cache reads, which is exactly how this test's premise last broke.
|
||||
Candidates go through the savings module's own resolver, so the pick is one the code
|
||||
under test can actually price."""
|
||||
for key in sorted(litellm.model_cost):
|
||||
entry = litellm.model_cost[key]
|
||||
provider = entry.get("litellm_provider")
|
||||
if not isinstance(provider, str) or not key.startswith(f"{provider}/"):
|
||||
continue
|
||||
if entry.get("mode") != "chat" or entry.get("cache_read_input_token_cost") is not None:
|
||||
continue
|
||||
if not entry.get("input_cost_per_token") or not entry.get("output_cost_per_token"):
|
||||
continue
|
||||
if _resolve_model(key, None) is None:
|
||||
continue
|
||||
priced = compute_autorouter_savings(
|
||||
baseline_model=key,
|
||||
selected_model="claude-haiku-4-5",
|
||||
selected_provider="anthropic",
|
||||
usage=_usage(fresh=1_000, cached=0, written=0, out=100),
|
||||
conversation_continuing=True,
|
||||
)
|
||||
if priced == 0.0:
|
||||
continue
|
||||
return key, key.removeprefix(f"{provider}/"), provider
|
||||
raise AssertionError("the bundled map has no per-token chat model without a cache-read rate")
|
||||
|
||||
|
||||
def test_a_baseline_with_no_cache_read_rate_is_charged_its_input_rate():
|
||||
"""The same hole on the other bucket. A baseline whose entry has no
|
||||
`cache_read_input_token_cost` reads for 0.0, so a continuing turn priced the whole
|
||||
prompt at nothing and every switch away from it reported a loss.
|
||||
"""
|
||||
baseline_key, baseline_name, baseline_provider = _priced_chat_model_without_cache_read_rate()
|
||||
continuing = _usage(fresh=0, cached=0, written=20_000, out=1_000)
|
||||
reported = compute_autorouter_savings(
|
||||
baseline_model=baseline_key,
|
||||
selected_model="claude-haiku-4-5",
|
||||
selected_provider="anthropic",
|
||||
usage=continuing,
|
||||
conversation_continuing=True,
|
||||
)
|
||||
|
||||
baseline = litellm.get_model_info(baseline_name, baseline_provider)
|
||||
assert baseline.get("cache_read_input_token_cost") is None, "pick a baseline with no cache-read rate"
|
||||
haiku = litellm.get_model_info("claude-haiku-4-5", "anthropic")
|
||||
baseline_pays_input = 20_000 * baseline["input_cost_per_token"] + 1_000 * baseline["output_cost_per_token"]
|
||||
actually_paid = 20_000 * haiku["cache_creation_input_token_cost"] + 1_000 * haiku["output_cost_per_token"]
|
||||
assert reported == pytest.approx(baseline_pays_input - actually_paid)
|
||||
|
||||
|
||||
def _breakdown(input_cost: float, output_cost: float = 0.0, **extra: object) -> dict:
|
||||
"""A `cost_breakdown` as the cost calculator records it on the spend log."""
|
||||
return {"input_cost": input_cost, "output_cost": output_cost, **extra}
|
||||
|
|
@ -796,6 +875,51 @@ def test_the_served_arm_is_read_from_the_record_not_repriced():
|
|||
assert reported == pytest.approx(public - (negotiated_input + negotiated_output))
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"basis, expected_multiplier",
|
||||
[
|
||||
pytest.param({"service_tier": "priority"}, 2.5, id="priority tier uplifts the baseline"),
|
||||
pytest.param({"data_residency": "eu"}, 1.1, id="eu residency uplifts the baseline"),
|
||||
pytest.param({}, 1.0, id="no basis recorded prices at standard"),
|
||||
pytest.param(None, 1.0, id="row predating the field prices at standard"),
|
||||
pytest.param({"service_tier": True, "data_residency": 17}, 1.0, id="a non-string basis is dropped"),
|
||||
],
|
||||
)
|
||||
def test_the_baseline_is_priced_on_the_basis_the_request_was_billed_at(basis, expected_multiplier):
|
||||
"""A request billed at a priority tier, or through a regional host, would have been
|
||||
billed the same way on the single model an operator ran instead of the router, so the
|
||||
counterfactual carries that basis too. Dropping it prices the two arms from different
|
||||
books; neither multiplier cancels out of the difference, because both are per-model.
|
||||
|
||||
The served model has no tiered rates and no uplift of its own, so only the baseline
|
||||
can move: a fix that forwards the basis to the served arm alone leaves these numbers
|
||||
unchanged. The non-string case guards the JSON round trip, where `.lower()` inside
|
||||
the pricer would raise and be swallowed into a silent $0.00 for the whole row.
|
||||
"""
|
||||
gpt = litellm.get_model_info("gpt-5.5", "openai")
|
||||
haiku = litellm.get_model_info("claude-haiku-4-5", "anthropic")
|
||||
assert gpt.get("input_cost_per_token_priority") == pytest.approx(2.5 * gpt["input_cost_per_token"])
|
||||
assert gpt.get("output_cost_per_token_priority") == pytest.approx(2.5 * gpt["output_cost_per_token"])
|
||||
assert gpt.get("regional_processing_uplift_multiplier_eu") == 1.1
|
||||
assert haiku.get("input_cost_per_token_priority") is None, "served model must not move with the basis"
|
||||
assert haiku.get("regional_processing_uplift_multiplier_eu") is None
|
||||
|
||||
usage = _usage(fresh=20_000, cached=0, written=0, out=1_000)
|
||||
served = 20_000 * haiku["input_cost_per_token"] + 1_000 * haiku["output_cost_per_token"]
|
||||
|
||||
reported = compute_autorouter_savings(
|
||||
baseline_model="openai/gpt-5.5",
|
||||
selected_model="claude-haiku-4-5",
|
||||
selected_provider="anthropic",
|
||||
usage=usage,
|
||||
conversation_continuing=False,
|
||||
cost_breakdown=None if basis is None else _breakdown(served, **basis),
|
||||
)
|
||||
|
||||
baseline = 20_000 * gpt["input_cost_per_token"] + 1_000 * gpt["output_cost_per_token"]
|
||||
assert reported == pytest.approx(expected_multiplier * baseline - served)
|
||||
|
||||
|
||||
def test_the_baseline_is_priced_on_the_vertex_location_the_request_was_billed_at(monkeypatch):
|
||||
"""A request served from a regional Vertex endpoint was billed with the
|
||||
regional-endpoint uplift, so the counterfactual single-model operator would
|
||||
|
|
|
|||
|
|
@ -2151,6 +2151,32 @@ async def test_proxy_only_error_5xx_keeps_traceback_and_runs_sync_callbacks(monk
|
|||
assert "test_proxy_utils" in captured["async_traceback"]
|
||||
|
||||
|
||||
def test_create_model_info_response_resolves_mode_through_deployment_model():
|
||||
"""`mode` is derived from the same lookup, so an aliased embedding deployment
|
||||
currently reports no mode at all; it must report `embedding`."""
|
||||
from litellm import Router
|
||||
|
||||
saved_model_cost = dict(litellm.model_cost)
|
||||
try:
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "my-embeddings",
|
||||
"litellm_params": {"model": "openai/text-embedding-3-small"},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
response = create_model_info_response(
|
||||
model_id="my-embeddings", provider="openai", llm_router=router
|
||||
)
|
||||
finally:
|
||||
litellm.model_cost.clear()
|
||||
litellm.model_cost.update(saved_model_cost)
|
||||
|
||||
assert response["mode"] == "embedding"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"key_metadata, team_metadata, expected_to_run",
|
||||
[
|
||||
|
|
|
|||
|
|
@ -105,6 +105,20 @@ def test_build_jev_request_includes_system_prompt_and_criteria() -> None:
|
|||
assert request.questions["tier"].criteria == criteria
|
||||
|
||||
|
||||
def test_jev_classifier_cost_uses_registry_pricing(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost,
|
||||
"typesafe/jev-1.13.0",
|
||||
{"input_cost_per_token": 0.0001, "output_cost_per_token": 0.0002},
|
||||
)
|
||||
response: Final = JevSystemOneResponse(
|
||||
model="jev-1.13.0",
|
||||
answers={"tier": _answer()},
|
||||
usage=JevUsage(input_tokens=3, output_tokens=4),
|
||||
)
|
||||
assert jev_classifier_cost(response, "jev-latest") == pytest.approx(0.0011)
|
||||
|
||||
|
||||
def test_jev_classifier_cost_is_none_without_registry_pricing() -> None:
|
||||
assert "typesafe/jev-unpriced" not in litellm.model_cost
|
||||
response: Final = JevSystemOneResponse(
|
||||
|
|
|
|||
|
|
@ -325,7 +325,28 @@ KIMI_K3_PERPLEXITY_KEY = "perplexity/perplexity/kimi-k3"
|
|||
|
||||
|
||||
class TestKimiK3AdvertisesItsDocumentedLevels:
|
||||
@pytest.mark.parametrize("model_key", KIMI_K3_PASSTHROUGH_KEYS)
|
||||
def test_a_passthrough_entry_advertises_the_models_own_levels(self, local_model_cost_map, model_key):
|
||||
"""platform.kimi.ai documents exactly low, high and max, and these providers forward the
|
||||
level unchanged. Undeclared, each entry resolves to unknown and the dashboard falls back to
|
||||
a capability-blind list that omits max."""
|
||||
entry = dict(litellm.model_cost[model_key], key=model_key)
|
||||
|
||||
assert resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) == ("low", "high", "max")
|
||||
|
||||
def test_the_perplexity_entry_advertises_the_wider_set_it_maps_down(self, local_model_cost_map):
|
||||
"""Perplexity's Agent API takes a six-value enum and maps it down internally, so this
|
||||
deployment is legitimately wider than a passthrough. One blanket list could not say both."""
|
||||
entry = dict(litellm.model_cost[KIMI_K3_PERPLEXITY_KEY], key=KIMI_K3_PERPLEXITY_KEY)
|
||||
|
||||
assert resolve_supported_reasoning_efforts(entry, deployment_is_mapped=True) == (
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max",
|
||||
)
|
||||
|
||||
@pytest.mark.parametrize("model, provider", [("kimi-k3", "moonshot"), ("kimi-k3", "fireworks_ai")])
|
||||
def test_the_declaration_survives_model_info_hydration(self, local_model_cost_map, model, provider):
|
||||
|
|
@ -338,6 +359,19 @@ class TestKimiK3AdvertisesItsDocumentedLevels:
|
|||
assert model_info["reasoning_effort_levels"] == ["low", "high", "max"]
|
||||
assert resolve_supported_reasoning_efforts(model_info, deployment_is_mapped=True) == ("low", "high", "max")
|
||||
|
||||
def test_a_kimi_k3_deployment_now_narrows_a_mixed_group(self, local_model_cost_map):
|
||||
"""kimi used to contribute unknown, which never narrows, so the group advertised whatever
|
||||
its other deployments agreed on."""
|
||||
kimi = resolve_supported_reasoning_efforts(
|
||||
dict(litellm.model_cost["fireworks_ai/kimi-k3"], key="fireworks_ai/kimi-k3"),
|
||||
deployment_is_mapped=True,
|
||||
)
|
||||
|
||||
assert intersect_supported_reasoning_efforts(("none", "minimal", "low", "medium", "high", "xhigh"), kimi) == (
|
||||
"low",
|
||||
"high",
|
||||
)
|
||||
|
||||
|
||||
class TestGpt6AstraAdvertisesItsDocumentedLevels:
|
||||
def test_the_entry_advertises_low_through_max_without_none(self, local_model_cost_map):
|
||||
|
|
|
|||
|
|
@ -1,8 +1,11 @@
|
|||
from pathlib import Path
|
||||
from typing import Final
|
||||
|
||||
import pytest
|
||||
from pydantic import TypeAdapter
|
||||
|
||||
from litellm import cost_per_token, get_model_info
|
||||
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
|
||||
|
||||
REPO_ROOT: Final = Path(__file__).parents[2]
|
||||
MODEL: Final = "azure_ai/grok-4.6"
|
||||
|
|
@ -13,6 +16,27 @@ def _cost_map_entry(path: Path) -> dict[str, object]:
|
|||
return COST_MAP_ADAPTER.validate_json(path.read_bytes())[MODEL]
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map")
|
||||
def test_azure_ai_grok_4_6_is_priced_and_routed() -> None:
|
||||
routed_model, provider, _, _ = get_llm_provider(model=MODEL)
|
||||
assert (routed_model, provider) == ("grok-4.6", "azure_ai")
|
||||
|
||||
info = get_model_info(model=routed_model, custom_llm_provider=provider)
|
||||
assert info["litellm_provider"] == "azure_ai"
|
||||
assert info["mode"] == "chat"
|
||||
assert info["supports_function_calling"] is True
|
||||
assert info["supports_prompt_caching"] is True
|
||||
assert info["supports_reasoning"] is True
|
||||
assert info["supports_response_schema"] is True
|
||||
assert info["supports_tool_choice"] is True
|
||||
assert info["supports_vision"] is True
|
||||
assert info["supports_web_search"] is True
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model=MODEL, prompt_tokens=1_000_000, completion_tokens=1_000_000)
|
||||
assert prompt_cost > 0
|
||||
assert completion_cost > 0
|
||||
|
||||
|
||||
def test_azure_ai_grok_4_6_entry_source_and_backup_match() -> None:
|
||||
main_entry = _cost_map_entry(REPO_ROOT / "model_prices_and_context_window.json")
|
||||
backup_entry = _cost_map_entry(REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json")
|
||||
|
|
|
|||
|
|
@ -427,6 +427,74 @@ def test_transcription_usage_cost_returns_zero_for_unknown_type():
|
|||
assert _transcription_usage_cost({}, {}) == 0.0
|
||||
|
||||
|
||||
def test_get_transcription_model_falls_back_to_session_model(monkeypatch):
|
||||
"""session.model is used when transcription-specific model fields are absent."""
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
from litellm.cost_calculator import _get_transcription_model_name_from_results
|
||||
|
||||
results: OpenAIRealtimeStreamList = [
|
||||
{"type": "session.created", "session": {"model": "gpt-realtime-whisper"}},
|
||||
]
|
||||
assert _get_transcription_model_name_from_results(results) == "gpt-realtime-whisper"
|
||||
|
||||
from litellm import Router
|
||||
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "prod/claude-3-5-sonnet-20240620",
|
||||
"litellm_params": {
|
||||
"model": "anthropic/claude-sonnet-4-5-20250929",
|
||||
"api_key": "test_api_key",
|
||||
},
|
||||
"model_info": {
|
||||
"id": "my-unique-model-id",
|
||||
"input_cost_per_token": 0.000006,
|
||||
"output_cost_per_token": 0.00003,
|
||||
"cache_creation_input_token_cost": 0.0000075,
|
||||
"cache_read_input_token_cost": 0.0000006,
|
||||
},
|
||||
},
|
||||
{
|
||||
"model_name": "claude-3-5-sonnet-20240620",
|
||||
"litellm_params": {
|
||||
"model": "anthropic/claude-sonnet-4-5-20250929",
|
||||
"api_key": "test_api_key",
|
||||
},
|
||||
"model_info": {
|
||||
"input_cost_per_token": 100,
|
||||
"output_cost_per_token": 200,
|
||||
},
|
||||
},
|
||||
]
|
||||
)
|
||||
|
||||
result = router.completion(
|
||||
model="claude-3-5-sonnet-20240620",
|
||||
messages=[{"role": "user", "content": "Hello, world!"}],
|
||||
mock_response=True,
|
||||
)
|
||||
|
||||
result_2 = router.completion(
|
||||
model="prod/claude-3-5-sonnet-20240620",
|
||||
messages=[{"role": "user", "content": "Hello, world!"}],
|
||||
mock_response=True,
|
||||
)
|
||||
|
||||
assert result._hidden_params["response_cost"] > result_2._hidden_params["response_cost"]
|
||||
|
||||
model_info = router.get_deployment_model_info(
|
||||
model_id="my-unique-model-id", model_name="anthropic/claude-sonnet-4-5-20250929"
|
||||
)
|
||||
assert model_info is not None
|
||||
assert model_info["input_cost_per_token"] == 0.000006
|
||||
assert model_info["output_cost_per_token"] == 0.00003
|
||||
assert model_info["cache_creation_input_token_cost"] == 0.0000075
|
||||
assert model_info["cache_read_input_token_cost"] == 0.0000006
|
||||
|
||||
|
||||
def test_custom_pricing_cost_calc_uses_router_model_id_from_litellm_metadata():
|
||||
"""When custom pricing is in litellm_metadata.model_info,
|
||||
use_custom_pricing_for_model should return True and
|
||||
|
|
@ -2270,6 +2338,42 @@ def test_anthropic_geo_multiplier_applies_to_cache_tokens(_local_model_cost_map,
|
|||
assert geo_completion_cost == pytest.approx(base_completion_cost * 1.1)
|
||||
|
||||
|
||||
def test_anthropic_geo_and_fast_multipliers_compose(_local_model_cost_map, monkeypatch):
|
||||
"""
|
||||
Anthropic's fast-mode pricing doubles every token type, cache reads and
|
||||
writes included, and the regional uplift stacks on top, so a fast +
|
||||
regional row prices as ``(non_cache + cache) * fast * geo``.
|
||||
"""
|
||||
from litellm.llms.anthropic.cost_calculation import (
|
||||
cost_per_token as anthropic_cost_per_token,
|
||||
)
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
|
||||
model = "claude-test-geo-fast-cache-model"
|
||||
_register_anthropic_geo_cache_model(model)
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=10_000,
|
||||
completion_tokens=500,
|
||||
total_tokens=10_500,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
cached_tokens=2_000,
|
||||
cache_creation_tokens=6_000,
|
||||
),
|
||||
)
|
||||
usage.inference_geo = "us"
|
||||
usage.speed = "fast"
|
||||
|
||||
prompt_cost, completion_cost = anthropic_cost_per_token(model=model, usage=usage)
|
||||
|
||||
cache_cost = 2_000 * 0.5e-6 + 6_000 * 6.25e-6
|
||||
non_cache_cost = 2_000 * 5e-6
|
||||
assert prompt_cost == pytest.approx((non_cache_cost + cache_cost) * 2.0 * 1.1)
|
||||
assert completion_cost == pytest.approx(500 * 25e-6 * 2.0 * 1.1)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
["claude-sonnet-4-6", "claude-mythos-5", "claude-mythos-preview"],
|
||||
|
|
@ -2806,6 +2910,60 @@ def test_custom_pricing_without_cache_keys_preserves_legacy_behavior():
|
|||
assert cost == pytest.approx(expected)
|
||||
|
||||
|
||||
def test_completion_cost_logs_the_rates_it_billed_at(monkeypatch):
|
||||
"""A caller reporting the cost lines beside their per-token rates reads both off this one call.
|
||||
completion_cost infers the provider, and xai's inclusive tier thresholds put a request sitting
|
||||
exactly on 200k at the tier rate, which a lookup made without that inferred provider would miss.
|
||||
"""
|
||||
from datetime import datetime
|
||||
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging
|
||||
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost,
|
||||
"xai/tiered-model",
|
||||
{
|
||||
"input_cost_per_token": 3e-6,
|
||||
"output_cost_per_token": 15e-6,
|
||||
"cache_read_input_token_cost": 3e-7,
|
||||
"input_cost_per_token_above_200k_tokens": 6e-6,
|
||||
"output_cost_per_token_above_200k_tokens": 3e-5,
|
||||
"cache_read_input_token_cost_above_200k_tokens": 6e-7,
|
||||
"litellm_provider": "xai",
|
||||
"mode": "chat",
|
||||
},
|
||||
)
|
||||
logging_obj = Logging(
|
||||
model="xai/tiered-model",
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
stream=False,
|
||||
call_type="completion",
|
||||
start_time=datetime.now(),
|
||||
litellm_call_id="billed-rates",
|
||||
function_id="f",
|
||||
)
|
||||
usage = Usage(
|
||||
prompt_tokens=200_000,
|
||||
completion_tokens=1_000,
|
||||
total_tokens=201_000,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=100_000),
|
||||
)
|
||||
|
||||
litellm.completion_cost(
|
||||
completion_response=ModelResponse(model="xai/tiered-model", usage=usage),
|
||||
model="xai/tiered-model",
|
||||
custom_llm_provider=None,
|
||||
litellm_logging_obj=logging_obj,
|
||||
)
|
||||
|
||||
rates = logging_obj.billed_token_rates
|
||||
assert rates is not None
|
||||
assert rates.input_cost_per_token == pytest.approx(6e-6)
|
||||
assert rates.cache_read_input_token_cost == pytest.approx(6e-7)
|
||||
assert logging_obj.cost_breakdown["cache_read_cost"] == pytest.approx(100_000 * rates.cache_read_input_token_cost)
|
||||
assert logging_obj.cost_breakdown["output_cost"] == pytest.approx(1_000 * rates.output_cost_per_token)
|
||||
|
||||
|
||||
def test_completion_cost_logs_cache_and_reasoning_breakdown_for_custom_pricing():
|
||||
"""
|
||||
A custom-priced deployment bills cache tokens at its custom cache rates, but the
|
||||
|
|
@ -3065,6 +3223,35 @@ def test_completion_cost_bills_interactions_google_search_per_query():
|
|||
assert cost > 3 * per_query_cost
|
||||
|
||||
|
||||
def test_completion_cost_bills_interactions_video_output_at_video_rate():
|
||||
from litellm.types.interactions import InteractionsAPIResponse
|
||||
|
||||
model_info = litellm.get_model_info(model="gemini-omni-flash-preview", custom_llm_provider="gemini")
|
||||
video_tokens = 5792 * 8
|
||||
response = InteractionsAPIResponse(
|
||||
id="interactions/video123",
|
||||
model="gemini-omni-flash-preview",
|
||||
status="completed",
|
||||
steps=[],
|
||||
usage={
|
||||
"total_tokens": 10 + video_tokens,
|
||||
"total_input_tokens": 10,
|
||||
"input_tokens_by_modality": [{"modality": "text", "tokens": 10}],
|
||||
"total_cached_tokens": 0,
|
||||
"total_output_tokens": video_tokens,
|
||||
"output_tokens_by_modality": [{"modality": "video", "tokens": video_tokens}],
|
||||
"total_tool_use_tokens": 0,
|
||||
"total_thought_tokens": 0,
|
||||
},
|
||||
)
|
||||
|
||||
cost = completion_cost(completion_response=response, custom_llm_provider="gemini")
|
||||
|
||||
expected = 10 * model_info["input_cost_per_token"] + video_tokens * model_info["output_cost_per_video_token"]
|
||||
assert model_info["output_cost_per_video_token"] != model_info["output_cost_per_token"]
|
||||
assert cost == pytest.approx(expected)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("video_count", [2, 3])
|
||||
def test_completion_cost_multiplies_video_cost_by_generated_video_count(video_count: int) -> None:
|
||||
"""Regression for LIT-6896: a Veo request for N samples generates N videos and must be billed N times."""
|
||||
|
|
|
|||
|
|
@ -3,6 +3,8 @@ from pathlib import Path
|
|||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
|
||||
REPO_ROOT = Path(__file__).parents[2]
|
||||
MAIN_PATH = REPO_ROOT / "model_prices_and_context_window.json"
|
||||
BACKUP_PATH = REPO_ROOT / "litellm" / "model_prices_and_context_window_backup.json"
|
||||
|
|
@ -19,6 +21,17 @@ def _load(path):
|
|||
return json.load(f)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def local_model_cost_map(monkeypatch):
|
||||
"""Force get_model_info to resolve against the in-repo cost map instead of the
|
||||
remote one fetched at import time, which still carries the pre-merge pricing."""
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
litellm.get_model_info.cache_clear()
|
||||
yield
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", GLM_5_2_MODELS)
|
||||
def test_backup_matches_main(model):
|
||||
"""Ensure the bundled (backup) cost map stays in sync with the canonical file."""
|
||||
|
|
|
|||
25
tests/test_litellm/test_sambanova_model_metadata.py
Normal file
25
tests/test_litellm/test_sambanova_model_metadata.py
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
|
||||
|
||||
|
||||
def test_sambanova_minimax_m27_model_info():
|
||||
model = "sambanova/MiniMax-M2.7"
|
||||
json_path = Path(__file__).parents[2] / "model_prices_and_context_window.json"
|
||||
with open(json_path) as f:
|
||||
model_cost = json.load(f)
|
||||
|
||||
info = model_cost.get(model)
|
||||
assert info is not None, f"{model} not found in model_prices_and_context_window.json"
|
||||
assert info["litellm_provider"] == "sambanova"
|
||||
assert info["mode"] == "chat"
|
||||
assert info["input_cost_per_token"] > 0
|
||||
assert info["output_cost_per_token"] > 0
|
||||
assert info["supports_function_calling"] is True
|
||||
assert info["supports_reasoning"] is True
|
||||
assert info["supports_tool_choice"] is True
|
||||
|
||||
routed_model, provider, _, _ = get_llm_provider(model=model)
|
||||
assert routed_model == "MiniMax-M2.7"
|
||||
assert provider == "sambanova"
|
||||
|
|
@ -3378,6 +3378,38 @@ _FIREWORKS_ROUTER_SHORT_FORMS = [
|
|||
]
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def fireworks_short_model_cost_map(monkeypatch: pytest.MonkeyPatch) -> Iterator[None]:
|
||||
monkeypatch.setattr(
|
||||
litellm,
|
||||
"model_cost",
|
||||
{
|
||||
"fireworks_ai/accounts/fireworks/models/glm-5p3": {
|
||||
"input_cost_per_token": 1e-6,
|
||||
"output_cost_per_token": 2e-6,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"mode": "chat",
|
||||
"max_tokens": 100,
|
||||
},
|
||||
"fireworks_ai/accounts/fireworks/routers/glm-5p3-fast": {
|
||||
"input_cost_per_token": 2.1e-6,
|
||||
"output_cost_per_token": 6.6e-6,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"mode": "chat",
|
||||
},
|
||||
"fireworks_ai/nomic-ai/nomic-embed-text-v1.5": {
|
||||
"input_cost_per_token": 8e-9,
|
||||
"output_cost_per_token": 0.0,
|
||||
"litellm_provider": "fireworks_ai",
|
||||
"mode": "embedding",
|
||||
},
|
||||
},
|
||||
)
|
||||
litellm.get_model_info.cache_clear()
|
||||
yield
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
|
||||
class TestBedrockBaseModelLabelKeepsTools:
|
||||
"""Regression for #29618: a Bedrock deployment whose ``base_model`` is a friendly
|
||||
label must not silently drop ``tools``/``tool_choice`` under ``drop_params``."""
|
||||
|
|
|
|||
|
|
@ -3,6 +3,8 @@ from typing import Final
|
|||
import pytest
|
||||
|
||||
import litellm
|
||||
from litellm import get_model_info
|
||||
from litellm.litellm_core_utils.get_llm_provider_logic import get_llm_provider
|
||||
from litellm.utils import supports_prompt_caching
|
||||
|
||||
MODEL: Final = "vertex_ai/xai/grok-4.6"
|
||||
|
|
@ -24,3 +26,13 @@ def test_grok_models_with_cache_read_price_advertise_prompt_caching() -> None:
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map")
|
||||
def test_vertex_ai_grok_4_6_supports_prompt_caching_via_get_model_info() -> None:
|
||||
routed_model, provider, _, _ = get_llm_provider(model=MODEL)
|
||||
assert (routed_model, provider) == ("xai/grok-4.6", "vertex_ai")
|
||||
|
||||
info = get_model_info(model=routed_model, custom_llm_provider=provider)
|
||||
assert info["litellm_provider"] == "vertex_ai"
|
||||
assert info.get("supports_prompt_caching") is True
|
||||
|
||||
assert supports_prompt_caching(model=MODEL) is True
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue