Merge pull request #41443 from BerriAI/litellm_remove_brittle_price_pinning_tests

test: delete unit-test assertions that pin cost-map prices, limits and deprecation dates
This commit is contained in:
kerry-berri 2026-09-17 17:52:42 -07:00 • committed by GitHub
commit f51f01fb54
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
48 changed files with 0 additions and 3036 deletions

View file

@ -153,23 +153,12 @@ def test_custom_pricing_as_completion_cost_param():
assert round(cost, 5) == round(expected_cost, 5)
def test_get_gpt3_tokens():
max_tokens = get_max_tokens("gpt-3.5-turbo")
print(max_tokens)
assert max_tokens == 4096
# print(results)
# test_get_gpt3_tokens()
def test_get_gemini_tokens():
# # 🦄🦄🦄🦄🦄🦄🦄🦄
max_tokens = get_max_tokens("gemini/gemini-1.5-flash")
assert max_tokens == 8192
print(max_tokens)
# test_get_palm_tokens()
@ -273,36 +262,6 @@ def test_cost_azure_gpt_35():
# test_cost_azure_gpt_35()
def test_cost_azure_embedding():
try:
import asyncio
litellm.set_verbose = True
async def _test():
response = await litellm.aembedding(
model="azure/text-embedding-ada-002",
input=["good morning from litellm", "gm"],
)
print(response)
return response
response = asyncio.run(_test())
cost = litellm.completion_cost(completion_response=response)
print("Cost", cost)
expected_cost = float("7e-07")
assert cost == expected_cost
except Exception as e:
pytest.fail(
f"Cost Calc failed for azure/gpt-3.5-turbo. Expected {expected_cost}, Calculated cost {cost}"
)
# test_cost_azure_embedding()
@ -639,56 +598,6 @@ def test_vertex_ai_medlm_completion_cost():
assert predictive_cost > 0
def test_vertex_ai_claude_completion_cost():
from litellm import Choices, Message, ModelResponse
from litellm.utils import Usage
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
litellm.set_verbose = True
input_tokens = litellm.token_counter(
model="vertex_ai/claude-3-sonnet@20240229",
messages=[{"role": "user", "content": "Hey, how's it going?"}],
)
print(f"input_tokens: {input_tokens}")
output_tokens = litellm.token_counter(
model="vertex_ai/claude-3-sonnet@20240229",
text="It's all going well",
count_response_tokens=True,
)
print(f"output_tokens: {output_tokens}")
response = ModelResponse(
id="chatcmpl-e41836bb-bb8b-4df2-8e70-8f3e160155ac",
choices=[
Choices(
finish_reason=None,
index=0,
message=Message(
content="It's all going well",
role="assistant",
),
)
],
created=1700775391,
model="claude-3-sonnet",
object="chat.completion",
system_fingerprint=None,
usage=Usage(
prompt_tokens=input_tokens,
completion_tokens=output_tokens,
total_tokens=input_tokens + output_tokens,
),
)
cost = litellm.completion_cost(
model="vertex_ai/claude-3-sonnet",
completion_response=response,
messages=[{"role": "user", "content": "Hey, how's it going?"}],
)
predicted_cost = input_tokens * 0.000003 + 0.000015 * output_tokens
assert cost == predicted_cost
def test_vertex_ai_embedding_completion_cost(caplog):
"""
Relevant issue - https://github.com/BerriAI/litellm/issues/4630
@ -1212,105 +1121,6 @@ def test_completion_cost_fireworks_ai(model):
assert cost > 0
def test_cost_azure_openai_prompt_caching():
from litellm.utils import Choices, Message, ModelResponse, Usage
from litellm.types.utils import (
PromptTokensDetailsWrapper,
CompletionTokensDetailsWrapper,
)
from litellm import get_model_info
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "azure/o1-mini"
## LLM API CALL ## (MORE EXPENSIVE)
response_1 = ModelResponse(
id="chatcmpl-3f427194-0840-4d08-b571-56bfe38a5424",
choices=[
Choices(
finish_reason="length",
index=0,
message=Message(
content="Hello! I'm doing well, thank you for",
role="assistant",
tool_calls=None,
function_call=None,
),
)
],
created=1725036547,
model=model,
object="chat.completion",
system_fingerprint=None,
usage=Usage(
completion_tokens=10,
prompt_tokens=14,
total_tokens=24,
completion_tokens_details=CompletionTokensDetailsWrapper(
reasoning_tokens=2
),
),
)
## PROMPT CACHE HIT ## (LESS EXPENSIVE)
response_2 = ModelResponse(
id="chatcmpl-3f427194-0840-4d08-b571-56bfe38a5424",
choices=[
Choices(
finish_reason="length",
index=0,
message=Message(
content="Hello! I'm doing well, thank you for",
role="assistant",
tool_calls=None,
function_call=None,
),
)
],
created=1725036547,
model=model,
object="chat.completion",
system_fingerprint=None,
usage=Usage(
completion_tokens=10,
prompt_tokens=0,
total_tokens=10,
prompt_tokens_details=PromptTokensDetailsWrapper(
cached_tokens=14,
),
completion_tokens_details=CompletionTokensDetailsWrapper(
reasoning_tokens=2
),
),
)
cost_1 = completion_cost(model=model, completion_response=response_1)
cost_2 = completion_cost(model=model, completion_response=response_2)
assert cost_1 > cost_2
model_info = get_model_info(model=model, custom_llm_provider="azure")
usage = response_2.usage
_expected_cost2 = (
(usage.prompt_tokens - usage.prompt_tokens_details.cached_tokens)
* model_info["input_cost_per_token"]
+ (usage.completion_tokens * model_info["output_cost_per_token"])
+ (
usage.prompt_tokens_details.cached_tokens
* model_info["cache_read_input_token_cost"]
)
)
print("_expected_cost2", _expected_cost2)
print("cost_2", cost_2)
assert (
abs(cost_2 - _expected_cost2) < 1e-5
) # Allow for small floating-point differences
def test_completion_cost_vertex_llama3():
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")

View file

@ -1670,8 +1670,6 @@ async def test_handle_completed_bedrock_batch_prices_from_deployment_model(monke
)
assert (result.usage.prompt_tokens, result.usage.completion_tokens, result.usage.total_tokens) == (1800, 1000, 2800)
# 3e-06 / 1.5e-05 on-demand, halved for batch.
assert result.cost == pytest.approx(1800 * 3e-06 / 2 + 1000 * 1.5e-05 / 2)
# The response model alone cannot price a bedrock batch: this is the $0 bug.
zero_result = await bu._handle_completed_batch(

View file

@ -377,8 +377,6 @@ class TestOpenAIContainerTransformation:
in container._hidden_params["additional_headers"]
)
# Verify the cost matches expected value for OpenAI code interpreter (1 session)
# OpenAI charges $0.03 per code interpreter session
expected_cost = StandardBuiltInToolCostTracking.get_cost_for_code_interpreter(
sessions=1, provider="openai"
)
@ -387,4 +385,3 @@ class TestOpenAIContainerTransformation:
]
assert actual_cost == expected_cost
assert actual_cost == 0.03 # OpenAI code interpreter costs $0.03 per session

View file

@ -90,15 +90,6 @@ class TestAzureAssistantCostTracking:
)
assert cost == 0.0, "Should return 0 for zero sessions"
def test_openai_code_interpreter_free(self):
"""Test OpenAI code interpreter cost from model cost map."""
cost = StandardBuiltInToolCostTracking.get_cost_for_code_interpreter(
sessions=5,
provider="openai",
)
assert (
cost == 0.15
), "OpenAI code interpreter should return 0.15 based on current implementation"
@pytest.mark.parametrize(
"input_tokens,output_tokens,expected_cost",
@ -222,14 +213,3 @@ class TestAzureAssistantCostTracking:
)
assert StandardBuiltInToolCostTracking.get_cost_for_vector_store(None) == 0.0
def test_constants_loaded_correctly(self):
"""Test that Azure pricing constants are loaded with expected values."""
assert AZURE_FILE_SEARCH_COST_PER_GB_PER_DAY == 0.1
# Code interpreter cost is now in model cost map
azure_container_info = litellm.model_cost.get("azure/container", {})
assert azure_container_info.get("code_interpreter_cost_per_session") == 0.03
assert AZURE_COMPUTER_USE_INPUT_COST_PER_1K_TOKENS == 3.0
assert AZURE_COMPUTER_USE_OUTPUT_COST_PER_1K_TOKENS == 12.0
assert AZURE_VECTOR_STORE_COST_PER_GB_PER_DAY == 0.1

View file

@ -1685,35 +1685,6 @@ def test_azure_gpt55_reasoning_effort_flags_match_live_openai_api(
assert m.get("supports_xhigh_reasoning_effort") is expected_xhigh
def test_generic_cost_per_token_anthropic_prompt_caching_with_cache_creation():
model = "claude-haiku-4-5-20251001"
usage = Usage(
completion_tokens=90,
prompt_tokens=28436,
total_tokens=28526,
completion_tokens_details=CompletionTokensDetailsWrapper(
accepted_prediction_tokens=None,
audio_tokens=None,
reasoning_tokens=0,
rejected_prediction_tokens=None,
text_tokens=None,
),
prompt_tokens_details=None,
cache_creation_input_tokens=2000,
)
custom_llm_provider = "anthropic"
prompt_cost, completion_cost = generic_cost_per_token(
model=model,
usage=usage,
custom_llm_provider=custom_llm_provider,
)
print(f"prompt_cost: {prompt_cost}")
assert round(prompt_cost, 3) == 0.029
def test_string_cost_values():
"""Test that cost values defined as strings are properly converted to floats."""
from unittest.mock import patch
@ -2350,140 +2321,6 @@ def test_gemini_image_generation_cost_falls_back_to_flat_image_pricing(_local_mo
assert round(cost, 10) == round(expected_cost, 10)
def test_bedrock_anthropic_prompt_caching():
"""Test Bedrock Anthropic models with prompt caching return correct costs."""
model = "us.anthropic.claude-sonnet-4-5-20250929-v1:0"
usage = Usage(
prompt_tokens=52123,
completion_tokens=497,
total_tokens=52620,
cache_creation_input_tokens=7183,
cache_read_input_tokens=22465,
)
custom_llm_provider = "bedrock"
prompt_cost, completion_cost = generic_cost_per_token(
model=model,
usage=usage,
custom_llm_provider=custom_llm_provider,
)
assert prompt_cost >= 0
assert completion_cost >= 0
assert round(prompt_cost, 3) == 0.111
assert round(completion_cost, 5) == 0.00820
def test_reasoning_tokens_without_text_tokens_gpt5_nano():
"""
Test fix for GitHub issue #18599:
https://github.com/BerriAI/litellm/issues/18599
When OpenAI models (gpt-5-nano, o1, o3) return reasoning_tokens but don't provide
text_tokens, LiteLLM should calculate text_tokens as:
text_tokens = completion_tokens - reasoning_tokens - audio_tokens - image_tokens
This ensures ALL completion tokens are billed, not just reasoning tokens.
"""
model = "gpt-5-nano"
custom_llm_provider = "openai"
# Simulate OpenAI gpt-5-nano response where text_tokens is NOT provided
# completion_tokens: 977 total
# reasoning_tokens: 768
# text_tokens: should be calculated as 977 - 768 = 209
usage = Usage(
prompt_tokens=17,
completion_tokens=977,
total_tokens=994,
completion_tokens_details=CompletionTokensDetailsWrapper(
reasoning_tokens=768,
audio_tokens=0,
# text_tokens NOT provided - this is the key part of the bug
),
)
prompt_cost, completion_cost = generic_cost_per_token(
model=model,
usage=usage,
custom_llm_provider=custom_llm_provider,
)
# gpt-5-nano pricing: $0.05/1M input, $0.40/1M output
expected_prompt_cost = 17 * 0.05 / 1_000_000
expected_completion_cost = 977 * 0.40 / 1_000_000 # ALL tokens, not just reasoning
assert abs(prompt_cost - expected_prompt_cost) < 1e-10, (
f"Prompt cost incorrect: {prompt_cost} vs {expected_prompt_cost}"
)
assert abs(completion_cost - expected_completion_cost) < 1e-10, (
f"Completion cost incorrect: {completion_cost} vs {expected_completion_cost}"
)
# Verify it's NOT using only reasoning_tokens (the bug)
wrong_cost = 768 * 0.40 / 1_000_000 # Only reasoning tokens
assert abs(completion_cost - wrong_cost) > 1e-6, (
"Bug detected: Cost calculation is using only reasoning_tokens instead of all completion_tokens!"
)
def test_image_count_prevents_text_tokens_fallback(_local_model_cost_map):
"""
Test that the text_tokens fallback in generic_cost_per_token does not
override text_tokens=0 when image_count > 0.
Regression test for: Bedrock image embedding double-charging bug.
When image_count > 0, text_tokens=0 is intentional (image-only request),
not "text_tokens not set by provider."
"""
# Simulate Nova image-only embedding: prompt_tokens estimated from
# embedding dimensions (768 for 3072-dim), image_count=1
usage = Usage(
prompt_tokens=768,
completion_tokens=0,
total_tokens=768,
prompt_tokens_details=PromptTokensDetailsWrapper(
image_count=1,
),
)
prompt_cost, completion_cost = generic_cost_per_token(
model="amazon.nova-2-multimodal-embeddings-v1:0",
usage=usage,
custom_llm_provider="bedrock",
)
# Cost should be 1 * input_cost_per_image ($6e-05) = $0.00006
# NOT 768 * input_cost_per_token ($1.35e-07) + $0.00006 = $0.000164
expected_image_cost = 1 * 6e-05
assert prompt_cost == expected_image_cost, (
f"Expected prompt_cost={expected_image_cost} (image-only), "
f"got {prompt_cost}. text_tokens fallback may be double-charging."
)
assert completion_cost == 0.0
def test_query_count_bills_input_cost_per_query(_local_model_cost_map):
usage = Usage(
prompt_tokens=0,
completion_tokens=0,
total_tokens=0,
prompt_tokens_details=PromptTokensDetailsWrapper(query_count=3, image_count=1),
)
prompt_cost, completion_cost = generic_cost_per_token(
model="us.twelvelabs.marengo-embed-3-0-v1:0",
usage=usage,
custom_llm_provider="bedrock",
)
assert prompt_cost == pytest.approx(3 * 7e-05 + 1e-04)
assert completion_cost == 0.0
def test_query_count_is_free_without_a_per_query_price(_local_model_cost_map):
usage = Usage(
prompt_tokens=0,
@ -2692,36 +2529,6 @@ def test_vertex_uplift_invalid_multiplier_defaults_to_one():
)
def test_priority_service_tier_above_threshold_uses_priority_tier_rates_for_cached_tokens(
_local_model_cost_map,
):
"""Regression: for a model that publishes both service_tier and above_threshold rate
variants, a priority request over the threshold must bill cached tokens at
cache_read_input_token_cost_above_200k_tokens_priority (and analogously for
input/output above-threshold), not the standard above-threshold rate."""
usage = Usage(
prompt_tokens=250_000,
completion_tokens=1_000,
total_tokens=251_000,
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=200_000, text_tokens=50_000),
completion_tokens_details=CompletionTokensDetailsWrapper(text_tokens=1_000),
)
prompt_cost, completion_cost = generic_cost_per_token(
model="gemini-3-pro-preview",
usage=usage,
custom_llm_provider="gemini",
service_tier="priority",
)
# gemini-3-pro-preview priority + above_200k rates from the pricing JSON:
# input 7.2e-6, output 3.24e-5, cache_read 7.2e-7
expected_prompt = 50_000 * 7.2e-6 + 200_000 * 7.2e-7
expected_completion = 1_000 * 3.24e-5
assert prompt_cost == pytest.approx(expected_prompt, rel=1e-9)
assert completion_cost == pytest.approx(expected_completion, rel=1e-9)
def test_service_tier_suffixes_constant_in_sync_with_enum():
from litellm.litellm_core_utils.llm_cost_calc.utils import _SERVICE_TIER_SUFFIXES
from litellm.types.utils import ServiceTier
@ -3614,28 +3421,6 @@ def test_gemini_38_flash_matches_37_flash_promotional_pricing(prefix, _local_mod
assert new_model[field] == old_model[field], field
@pytest.mark.parametrize(
("model", "provider", "image_token_rate"),
[
("gpt-realtime-2.1", "openai", 5e-06),
("gpt-realtime-2.1-mini", "openai", 8e-07),
("azure/gpt-realtime-2.1", "azure", 5e-06),
("azure/gpt-realtime-2.1-mini", "azure", 8e-07),
],
)
def test_realtime_image_tokens_priced_per_token(model, provider, image_token_rate, _local_model_cost_map):
"""Realtime image input is billed per 1M image tokens, not per image."""
usage = Usage(
prompt_tokens=1_100,
completion_tokens=0,
total_tokens=1_100,
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=100, image_tokens=1_000),
)
prompt_cost, _ = generic_cost_per_token(model=model, usage=usage, custom_llm_provider=provider)
text_rate = litellm.model_cost[model]["input_cost_per_token"]
assert prompt_cost == pytest.approx(100 * text_rate + 1_000 * image_token_rate)
@pytest.mark.parametrize(
("response_quality", "requested_quality", "expected_cost"),
[
@ -3830,28 +3615,6 @@ def test_cached_audio_tokens_fall_back_to_cache_read_input_token_cost() -> None:
assert prompt_cost == pytest.approx(expected)
def test_cache_read_breakdown_splits_cached_audio_at_the_audio_cache_rate(_local_model_cost_map: None) -> None:
usage = Usage(
prompt_tokens=4863,
completion_tokens=1087,
total_tokens=5950,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=1693,
audio_tokens=3170,
cached_tokens=2816,
cached_tokens_details={"text_tokens": 896, "audio_tokens": 1920},
),
)
breakdown = get_token_type_cost_breakdown(model="gpt-realtime-2.1-mini", custom_llm_provider="openai", usage=usage)
prompt_cost, _ = generic_cost_per_token(model="gpt-realtime-2.1-mini", usage=usage, custom_llm_provider="openai")
assert breakdown.cache_read_cost == pytest.approx(896 * 6e-8 + 1920 * 3e-7)
assert breakdown.rates is not None
assert breakdown.rates.cache_read_input_audio_token_cost == pytest.approx(3e-7)
assert prompt_cost == pytest.approx((1693 - 896) * 6e-7 + (3170 - 1920) * 1e-5 + breakdown.cache_read_cost)
def test_generic_cost_per_token_bills_cache_creation_at_the_input_rate_without_a_write_price():
"""Azure and OpenAI publish no cache-write price and bill cache writes as ordinary input.
A deployment priced with only input, output, and cache-read rates must bill the creation

View file

@ -309,102 +309,6 @@ def test_get_cost_for_gemini_web_search(model):
assert cost > 0.0
@pytest.mark.parametrize(
"model,custom_llm_provider",
[
("vertex_ai/gemini-2.5-flash", "vertex_ai"),
("gemini-2.5-flash", "vertex_ai"),
],
)
def test_get_cost_for_vertex_ai_gemini_web_search(model, custom_llm_provider):
"""
Test that Vertex AI Gemini web search costs are tracked when passing
a ModelResponse with usage.prompt_tokens_details.web_search_requests.
This tests the fix for: https://github.com/BerriAI/litellm/issues/XXXXX
The issue: When a ModelResponse is passed, the detection logic only checks
for url_citation annotations, not usage.prompt_tokens_details.web_search_requests.
This causes Vertex AI grounding costs to not be tracked.
"""
from litellm.types.utils import Choices, Message, PromptTokensDetailsWrapper, Usage
# Create a realistic ModelResponse like what Vertex AI returns
response = ModelResponse(
id="test-id",
choices=[
Choices(
finish_reason="stop",
index=0,
message=Message(
content="Test response with grounding", role="assistant"
),
)
],
created=1234567890,
model=model,
object="chat.completion",
system_fingerprint=None,
)
# Add usage with web_search_requests (how Vertex AI indicates grounding was used)
usage = Usage(
prompt_tokens=11,
completion_tokens=100,
total_tokens=111,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=11, web_search_requests=1 # This should trigger grounding cost
),
)
response.usage = usage
# Calculate cost - should include grounding cost
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model=model,
usage=usage,
response_object=response, # Pass the ModelResponse
custom_llm_provider=custom_llm_provider,
standard_built_in_tools_params=None,
)
# Vertex AI charges $0.035 per grounded request
assert cost == 0.035, f"Expected $0.035 grounding cost, got ${cost}"
def test_azure_assistant_features_integrated_cost_tracking(monkeypatch):
"""
Test integrated cost tracking for Azure assistant features.
"""
# Force use of local model cost map for CI/CD consistency
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
litellm.model_cost = litellm.get_model_cost_map(url="")
model = "azure/gpt-4o"
# Test with multiple Azure assistant features
standard_built_in_tools_params = StandardBuiltInToolsParams(
vector_store_usage={"storage_gb": 1.0, "days": 10},
computer_use_usage={"input_tokens": 1000, "output_tokens": 500},
code_interpreter_sessions=2,
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model=model,
response_object=None,
usage=None,
custom_llm_provider="azure",
standard_built_in_tools_params=standard_built_in_tools_params,
)
# Should calculate costs for:
# - Vector store: 1.0 * 10 * 0.1 = $1.00
# - Computer use: (1000/1000 * 3.0) + (500/1000 * 12.0) = $9.00
# - Code interpreter: 2 * 0.03 = $0.06
# Total: $10.06
expected_cost = 1.0 + 9.0 + 0.06
assert abs(cost - expected_cost) < 0.01, f"Expected ~{expected_cost}, got {cost}"
def test_completion_cost_includes_web_search_without_standard_built_in_tools_params():
"""
Test that completion_cost includes web search cost even when
@ -510,68 +414,6 @@ def test_gemini_3x_web_search_billed_per_query(model, local_model_cost_map):
)
@pytest.mark.parametrize(
"model,custom_llm_provider",
[
("gemini/gemini-2.5-flash", "gemini"),
("vertex_ai/gemini-2.5-flash", "vertex_ai"),
],
)
def test_gemini_2x_maps_grounding_billed_at_maps_rate(model, custom_llm_provider, local_model_cost_map):
"""
Grounding with Google Maps is its own SKU: a Maps-only grounded prompt on Gemini 2.x bills the
$0.025 Maps per-prompt fee, not the $0.035 Google Search fee it was previously conflated with,
and not $0 as on Vertex AI where webSearchQueries is never populated for Maps.
Regression for https://github.com/BerriAI/litellm/issues/35906
"""
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
model_info = litellm.get_model_info(model)
expected_cost = model_info["google_maps_grounding_cost_per_query"]
assert expected_cost == pytest.approx(0.025)
usage = Usage(
prompt_tokens=15,
completion_tokens=100,
total_tokens=115,
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=15, google_maps_grounding_requests=1),
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model=model,
usage=usage,
response_object=None,
custom_llm_provider=custom_llm_provider,
standard_built_in_tools_params=None,
)
assert cost == pytest.approx(expected_cost)
def test_gemini_3x_maps_grounding_billed_per_query(local_model_cost_map):
"""Gemini 3.x bills Maps grounding per executed query: N queries cost N * $0.014."""
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
model = "vertex_ai/gemini-3.5-flash"
model_info = litellm.get_model_info(model)
assert model_info["web_search_billing_unit"] == "per_query"
expected_cost = model_info["google_maps_grounding_cost_per_query"] * 2
usage = Usage(
prompt_tokens=15,
completion_tokens=100,
total_tokens=115,
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=15, google_maps_grounding_requests=2),
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model=model,
usage=usage,
response_object=None,
custom_llm_provider="vertex_ai",
standard_built_in_tools_params=None,
)
assert cost == pytest.approx(expected_cost)
assert cost == pytest.approx(0.028)
def test_gemini_combined_search_and_maps_costs_are_additive(local_model_cost_map):
"""A prompt grounded with both Google Search and Google Maps pays both fees."""
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
@ -708,35 +550,6 @@ def _openai_responses_with_web_search_calls(model, num_calls):
)
def test_openai_responses_web_search_priced_per_call(local_model_cost_map):
"""
Regression for LIT-5013 bug 1: OpenAI reasoning models (gpt-5 family, o-series, deep-research)
carry supports_web_search but had no search_context_cost_per_query, so get_cost_for_web_search_request
(no openai branch) returned None and the default fallback billed web search as $0. gpt-5-nano now
prices at $0.01 per call, and two web_search_call items in the Responses output must bill 2 x $0.01.
"""
from litellm.types.utils import Usage
model = "gpt-5-nano"
per_call = litellm.get_model_info(model)["search_context_cost_per_query"][
"search_context_size_medium"
]
assert per_call == 0.01
response = _openai_responses_with_web_search_calls(model, num_calls=2)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model=model,
response_object=response,
usage=Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15),
custom_llm_provider="openai",
standard_built_in_tools_params=None,
)
assert cost == pytest.approx(2 * per_call), (
f"gpt-5-nano web search must bill 2 x ${per_call}, got ${cost}"
)
def test_openai_responses_web_search_multiplied_by_call_count(local_model_cost_map):
"""
Regression for LIT-5013 bug 2: web_search_call detection was binary, so a Responses output with
@ -808,88 +621,6 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map):
)
def test_dated_search_preview_entries_carry_search_pricing(local_model_cost_map):
"""
Regression for the live QA finding: OpenAI resolves gpt-4o-search-preview requests to the
dated id gpt-4o-search-preview-2025-03-11, whose cost map entry lacked
search_context_cost_per_query, so the default chat path silently billed the $0.035 search
fee as $0. Dated entries must price identically to their undated siblings.
"""
from litellm.types.utils import Usage
for dated, undated in (
("gpt-4o-search-preview-2025-03-11", "gpt-4o-search-preview"),
("gpt-4o-mini-search-preview-2025-03-11", "gpt-4o-mini-search-preview"),
):
assert (
litellm.get_model_info(dated)["search_context_cost_per_query"]
== litellm.get_model_info(undated)["search_context_cost_per_query"]
)
response = ModelResponse(
model="gpt-4o-search-preview-2025-03-11",
choices=[
{
"index": 0,
"finish_reason": "stop",
"message": {
"role": "assistant",
"content": "headlines",
"annotations": [
{
"type": "url_citation",
"url_citation": {
"url": "https://example.com",
"title": "t",
"start_index": 0,
"end_index": 1,
},
}
],
},
}
],
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model="gpt-4o-search-preview-2025-03-11",
response_object=response,
usage=Usage(prompt_tokens=14, completion_tokens=825, total_tokens=839),
custom_llm_provider="openai",
standard_built_in_tools_params=None,
)
assert cost == pytest.approx(0.025), (
f"dated search-preview id must bill the $0.025 search fee, got ${cost}"
)
@pytest.mark.parametrize(
"web_search_options",
[
None,
WebSearchOptions(search_context_size="low"),
WebSearchOptions(search_context_size="medium"),
WebSearchOptions(search_context_size="high"),
],
)
def test_gpt_4o_mini_snapshot_bills_web_search_like_its_alias(
web_search_options: WebSearchOptions | None, local_model_cost_map: None
) -> None:
alias_info = litellm.get_model_info("gpt-4o-mini")
snapshot_info = litellm.get_model_info("gpt-4o-mini-2024-07-18")
assert not snapshot_info["supports_web_search"]
assert not alias_info["supports_web_search"]
snapshot_cost = StandardBuiltInToolCostTracking.get_cost_for_web_search(
web_search_options=web_search_options, model_info=snapshot_info
)
alias_cost = StandardBuiltInToolCostTracking.get_cost_for_web_search(
web_search_options=web_search_options, model_info=alias_info
)
assert snapshot_cost == alias_cost == 0.025
# Note: File search integration test removed due to complex annotation detection logic
# The unit tests in test_azure_assistant_cost_tracking.py provide comprehensive coverage
@ -999,81 +730,3 @@ def _web_search_cost(model: str, response: ResponsesAPIResponse, custom_llm_prov
)
@pytest.mark.parametrize("model", _BEDROCK_MANTLE_WEB_SEARCH_MODELS)
def test_bedrock_mantle_web_search_billed_per_query(local_model_cost_map, model):
"""Two Bedrock-reported web searches bill 2 x $0.012 under the prefixed and the bare model id alike."""
pricing = litellm.get_model_info(model)["search_context_cost_per_query"]
assert pricing == {
"search_context_size_low": _BEDROCK_MANTLE_WEB_SEARCH_RATE,
"search_context_size_medium": _BEDROCK_MANTLE_WEB_SEARCH_RATE,
"search_context_size_high": _BEDROCK_MANTLE_WEB_SEARCH_RATE,
}
response = _responses_with_web_search(
model,
actions=[{"type": "search", "query": "litellm"}, {"type": "search", "query": "bedrock web search"}],
tool_usage={"web_search": {"num_requests": 2}},
)
for cost_model in (model, model.split("/", 1)[1]):
cost = _web_search_cost(cost_model, response, "bedrock_mantle")
assert cost == pytest.approx(2 * _BEDROCK_MANTLE_WEB_SEARCH_RATE), (
f"{cost_model} must bill 2 x ${_BEDROCK_MANTLE_WEB_SEARCH_RATE} for 2 web searches, got ${cost}"
)
@pytest.mark.parametrize("num_requests", [1, 0])
def test_web_search_call_count_prefers_provider_reported_num_requests(local_model_cost_map, num_requests):
"""A search plus an open_page fetch bills tool_usage.web_search.num_requests, never the two items."""
model = "bedrock_mantle/openai.gpt-5.6-sol"
response = _responses_with_web_search(
model,
actions=[
{"type": "search", "query": "litellm"},
{"type": "open_page", "url": "https://docs.litellm.ai/"},
],
tool_usage={"web_search": {"num_requests": num_requests}},
)
cost = _web_search_cost(model, response, "bedrock_mantle")
assert cost == pytest.approx(num_requests * _BEDROCK_MANTLE_WEB_SEARCH_RATE), (
f"{num_requests} reported web search requests must bill {num_requests} x "
f"${_BEDROCK_MANTLE_WEB_SEARCH_RATE}, got ${cost}"
)
@pytest.mark.parametrize(
"tool_usage",
[None, {}, {"web_search": None}, {"web_search": {"num_requests": "many"}}, {"web_search": {"num_requests": -1}}],
)
def test_web_search_call_count_falls_back_to_items_without_reported_count(local_model_cost_map, tool_usage):
"""Without a usable reported count the per-call path keeps counting web_search_call items."""
model = "bedrock_mantle/openai.gpt-5.6-sol"
response = _responses_with_web_search(
model,
actions=[{"type": "search", "query": "litellm"}, {"type": "search", "query": "bedrock web search"}],
tool_usage=tool_usage,
)
cost = _web_search_cost(model, response, "bedrock_mantle")
assert cost == pytest.approx(2 * _BEDROCK_MANTLE_WEB_SEARCH_RATE), (
f"2 web_search_call items with tool_usage={tool_usage!r} must bill 2 x "
f"${_BEDROCK_MANTLE_WEB_SEARCH_RATE}, got ${cost}"
)
def test_web_search_call_count_reads_reported_count_beside_other_tool_usage_entries(local_model_cost_map):
"""OpenAI reports web_search.num_requests next to other tool entries, which must not disable the reported count."""
response = _responses_with_web_search(
"gpt-5.6",
actions=[{"type": "search", "query": "S&P 500 close"}, {"type": "open_page", "url": "https://example.com/"}],
tool_usage={
"image_gen": {"input_tokens": 0, "output_tokens": 0, "total_tokens": 0},
"web_search": {"num_requests": 1},
},
)
cost = _web_search_cost("gpt-5.6", response, "openai")
assert cost == pytest.approx(0.01), f"1 reported OpenAI web search must bill 1 x $0.01, not the 2 items, got ${cost}"

View file

@ -395,53 +395,6 @@ class TestGetRouterDeploymentModelInfo:
logging_obj.litellm_params = {"api_base": ""}
assert logging_obj.get_router_deployment_model_info() is None
@pytest.mark.parametrize(
"declared,expected_input,expected_output",
[
({"input_cost_per_token": 1e-06}, 1e-06, 1.5e-05),
({"output_cost_per_token": 5e-06}, 3e-06, 5e-06),
({"input_cost_per_token": 0.0, "output_cost_per_token": 0.0}, 0.0, 0.0),
],
ids=["input-only", "output-only", "both-zero"],
)
def test_one_sided_override_keeps_the_published_rate_for_the_other_side(
self,
declared: dict[str, float],
expected_input: float,
expected_output: float,
) -> None:
"""A deployment may configure one direction only.
Substituting its pricing wholesale billed the direction it left unset at
zero, because get_model_info fills an absent cost with 0 and that
suppressed the global fallback.
"""
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
model = "bedrock/global.anthropic.claude-sonnet-4-6"
published = litellm.get_model_info(model=model)
assert (published["input_cost_per_token"], published["output_cost_per_token"]) == (3e-06, 1.5e-05)
deployment_id = f"deploy-one-sided-{'-'.join(sorted(declared))}"
litellm.model_cost[deployment_id] = {"id": deployment_id, **declared}
obj = LiteLLMLoggingObj(
model=model,
messages=[],
stream=False,
call_type="aretrieve_batch",
start_time=time.time(),
litellm_call_id="one-sided",
function_id="f",
)
obj.litellm_params = {"litellm_metadata": {"model_info": {"id": deployment_id}}, "model": model}
obj.model_call_details["model"] = model
try:
info = obj.get_router_deployment_model_info()
assert info is not None
assert info["input_cost_per_token"] == expected_input
assert info["output_cost_per_token"] == expected_output
finally:
litellm.model_cost.pop(deployment_id, None)
def test_a_published_batch_rate_never_displaces_a_declared_standard_rate(self) -> None:
"""Ownership is per token direction, not per field.
@ -511,7 +464,6 @@ class TestGetRouterDeploymentModelInfo:
cached_before = dict(litellm.get_model_info(model=deployment_id))
info = obj.get_router_deployment_model_info()
assert info is not None
assert info["output_cost_per_token"] == 1.5e-05
assert dict(litellm.get_model_info(model=deployment_id)) == cached_before
finally:
litellm.model_cost.pop(deployment_id, None)

View file

@ -336,7 +336,6 @@ def test_streaming_preserves_anthropic_1hr_cache_creation_breakdown():
Correct cache-write cost is 50 * 6e-06 (1h) = 0.0003, not 50 * 3.75e-06 = 0.0001875.
"""
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
from litellm.llms.anthropic.cost_calculation import cost_per_token
config = AnthropicConfig()
message_start_usage = config.calculate_usage(
@ -400,13 +399,6 @@ def test_streaming_preserves_anthropic_1hr_cache_creation_breakdown():
assert usage.cache_creation_input_tokens == 50
assert usage.cache_read_input_tokens == 8728
prompt_cost, _ = cost_per_token(model="claude-sonnet-4-6", usage=usage)
# text 3*3e-06 + cache_read 8728*3e-07 + cache_write 50*6e-06 (1h rate)
expected = 3 * 3e-06 + 8728 * 3e-07 + 50 * 6e-06
assert prompt_cost == pytest.approx(expected)
# Guard against the regression: 5m-rate fallback would shave the write cost.
buggy = 3 * 3e-06 + 8728 * 3e-07 + 50 * 3.75e-06
assert prompt_cost != pytest.approx(buggy)
def test_streaming_keeps_cache_creation_breakdown_from_final_chunk():

View file

@ -130,16 +130,3 @@ def test_openai_style_unsupported_param_dropped_with_drop_params():
assert mapped == {}
def test_cost_calculator_uses_aiml_pricing_for_gpt_image_2():
"""Regression: pricing must come from the ``aiml/openai/gpt-image-2`` entry,
not the upstream OpenAI token-based entry.
"""
response = ImageResponse(
data=[
ImageObject(b64_json=None, url="https://example.com/1.png"),
ImageObject(b64_json=None, url="https://example.com/2.png"),
]
)
assert aiml_cost_calculator(
model="openai/gpt-image-2", image_response=response
) == pytest.approx(0.054 * 2)

View file

@ -2442,21 +2442,6 @@ def test_get_max_tokens_for_model_claude_35():
assert max_tokens == 8192
def test_get_max_tokens_for_model_claude_37():
"""
Test that get_max_tokens_for_model returns correct value for Claude 3.7 models.
Claude 3.7 Sonnet has max_output_tokens of 64000 by default.
128K output requires the beta header 'output-128k-2025-02-19'.
Fixes: https://github.com/BerriAI/litellm/issues/8835
"""
config = AnthropicConfig()
# Claude 3.7 Sonnet should return 64000 (64K default, 128K requires beta header)
max_tokens = config.get_max_tokens_for_model("claude-3-7-sonnet-20250219")
assert max_tokens == 64000
def test_get_max_tokens_for_model_unknown():
"""
Test that get_max_tokens_for_model returns 4096 fallback for unknown models.
@ -2631,29 +2616,6 @@ def test_transform_request_injects_dummy_tool_without_tools_param():
assert "dummy_tool" in names
def test_transform_request_uses_dynamic_max_tokens():
"""
Test that transform_request uses dynamic max_tokens based on model
when max_tokens is not explicitly provided.
Fixes: https://github.com/BerriAI/litellm/issues/8835
"""
config = AnthropicConfig()
messages = [{"role": "user", "content": "Hello"}]
# Claude 3.7 model should get 64000 as default max_tokens (from model_prices_and_context_window.json)
result = config.transform_request(
model="claude-3-7-sonnet-20250219",
messages=messages,
optional_params={}, # No max_tokens provided
litellm_params={},
headers={},
)
assert result["max_tokens"] == 64000
def test_transform_request_respects_user_max_tokens():
"""
Test that transform_request respects user-provided max_tokens
@ -2851,7 +2813,6 @@ def test_raw_adaptive_thinking_untouched_for_46_plus_model():
assert result["thinking"] == {"type": "adaptive"}
@pytest.mark.parametrize(
"model, expected",
[

View file

@ -4,7 +4,6 @@ Verifies the fix for issue #19532.
"""
import litellm
from litellm import get_model_info
from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map
@ -18,25 +17,3 @@ def reload_model_costs():
yield
@pytest.mark.parametrize(
"model,expected_cache_creation_cost,expected_cache_read_cost",
[
("claude-haiku-4-5", 1.25e-06, 1e-07),
("claude-opus-4-5", 6.25e-06, 5e-07),
("claude-opus-4-1", 1.875e-05, 1.5e-06),
("claude-sonnet-4-5", 3.75e-06, 3e-07),
],
)
def test_azure_ai_claude_cache_pricing(
model, expected_cache_creation_cost, expected_cache_read_cost
):
"""Test that Azure AI Claude models have correct cache pricing."""
model_info = get_model_info(model=model, custom_llm_provider="azure_ai")
assert model_info.get("cache_creation_input_token_cost") is not None
assert model_info.get("cache_read_input_token_cost") is not None
assert (
model_info.get("cache_creation_input_token_cost")
== expected_cache_creation_cost
)
assert model_info.get("cache_read_input_token_cost") == expected_cache_read_cost

View file

@ -26,26 +26,6 @@ def _transcription_client() -> AzureOpenAI:
)
def test_azure_ai_transcription_is_priced_at_the_azure_ai_entry():
with AUDIO_FILE.open("rb") as audio:
response = litellm.transcription(
model="azure_ai/whisper",
file=audio,
api_base="https://example.cognitiveservices.azure.com",
api_key="test-key",
api_version="2024-06-01",
client=_transcription_client(),
)
with AUDIO_FILE.open("rb") as audio:
duration = calculate_request_duration(audio)
assert duration is not None and duration > 0
assert response._hidden_params["custom_llm_provider"] == "azure_ai"
assert completion_cost(completion_response=response, call_type="transcription") == pytest.approx(
WHISPER_COST_PER_SECOND * duration
)
def test_azure_transcription_keeps_the_azure_provider():
with AUDIO_FILE.open("rb") as audio:
response = litellm.transcription(

View file

@ -158,13 +158,6 @@ class TestAzureModelRouterFlatCost:
assert prompt_cost == pytest.approx(1000 * ROUTER_FEE_PER_TOKEN, rel=1e-9)
assert completion_cost_usd == 0.0
@pytest.mark.parametrize("router_entry_name", ["model_router", "model-router"])
def test_router_entry_prices_its_own_fee(self, router_entry_name: str) -> None:
usage = Usage(prompt_tokens=1_000_000, completion_tokens=0, total_tokens=1_000_000)
prompt_cost, completion_cost_usd = cost_per_token(model=router_entry_name, usage=usage)
assert prompt_cost == pytest.approx(0.14, rel=1e-9)
assert completion_cost_usd == 0.0
def test_routed_model_is_priced_as_itself(self) -> None:
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
prompt_cost, completion_cost_usd = cost_per_token(model=ROUTED_MODEL, usage=ROUTED_USAGE)
@ -210,24 +203,6 @@ class TestAzureModelRouterFlatCost:
assert prompt_cost == pytest.approx(routed_prompt_cost + ROUTED_FEE, rel=1e-9)
assert completion_cost_usd == pytest.approx(routed_completion_cost, rel=1e-9)
def test_flat_cost_helper(self) -> None:
assert calculate_azure_model_router_flat_cost(
model="azure-model-router", prompt_tokens=10_000
) == pytest.approx(0.0014, rel=1e-9)
assert calculate_azure_model_router_flat_cost(model="gpt-5-nano", prompt_tokens=10_000) == 0.0
def test_flat_cost_reads_the_fee_from_the_deployment_named_entry(self) -> None:
litellm.register_model(
{"azure_ai/model-router": {"input_cost_per_token": 2e-07, "litellm_provider": "azure_ai", "mode": "chat"}}
)
litellm.get_model_info.cache_clear()
assert calculate_azure_model_router_flat_cost(model="model-router", prompt_tokens=1_000_000) == pytest.approx(
0.2, rel=1e-9
)
assert calculate_azure_model_router_flat_cost(
model="azure-model-router", prompt_tokens=1_000_000
) == pytest.approx(0.14, rel=1e-9)
@pytest.mark.usefixtures("local_model_cost_map")
class TestAzureModelRouterCostBreakdown:
@ -350,32 +325,3 @@ class TestAzureAIServiceTierCostCalculation:
assert flex_prompt < standard_prompt
assert flex_completion < standard_completion
def test_codestral_2501_model_info_and_cost(local_model_cost_map):
model_info = get_model_info(model="Codestral-2501", custom_llm_provider="azure_ai")
usage = Usage(prompt_tokens=1_000_000, completion_tokens=1_000_000, total_tokens=2_000_000)
prompt_cost, completion_cost = cost_per_token(model="Codestral-2501", usage=usage)
assert model_info["mode"] == "chat"
assert model_info["max_input_tokens"] == 256000
assert model_info["max_output_tokens"] == 4096
assert prompt_cost == pytest.approx(0.3)
assert completion_cost == pytest.approx(0.9)
def test_mai_thinking_1_model_info_and_cost(local_model_cost_map):
model_info = get_model_info(model="MAI-Thinking-1", custom_llm_provider="azure_ai")
usage = Usage(prompt_tokens=1_000_000, completion_tokens=1_000_000, total_tokens=2_000_000)
prompt_cost, completion_cost = cost_per_token(model="MAI-Thinking-1", usage=usage)
assert model_info["mode"] == "chat"
assert model_info["max_input_tokens"] == 256000
assert model_info["max_output_tokens"] == 64000
assert model_info["cache_read_input_token_cost"] == pytest.approx(2e-07)
assert model_info["supports_reasoning"] is True
assert model_info["supports_function_calling"] is True
assert prompt_cost == pytest.approx(2.0)
assert completion_cost == pytest.approx(8.0)

View file

@ -33,17 +33,3 @@ def use_local_model_cost_map():
monkeypatch.undo()
def test_azure_ai_kimi_k26_cost_per_token(use_local_model_cost_map):
from litellm.llms.azure_ai.cost_calculator import cost_per_token
from litellm.types.utils import Usage
usage = Usage(
prompt_tokens=1_000_000,
completion_tokens=1_000_000,
total_tokens=2_000_000,
)
prompt_cost, completion_cost = cost_per_token(model="kimi-k2.6", usage=usage)
assert prompt_cost == pytest.approx(0.95)
assert completion_cost == pytest.approx(4.0)

View file

@ -1903,7 +1903,6 @@ async def test_unified_bedrock_messages_cache_on_start_only_never_negative_cost(
custom_llm_provider="bedrock",
)
assert cost > 0
assert cost == pytest.approx(0.0093951, rel=0, abs=1e-9)
@pytest.mark.asyncio
@ -1967,13 +1966,6 @@ async def test_unified_bedrock_messages_sse_usage_and_cost_claude_sonnet_46():
assert built.usage.cache_creation_input_tokens == 10553
assert built.usage.cache_read_input_tokens == 25490
cost = completion_cost(
completion_response=built,
model="bedrock/us.anthropic.claude-sonnet-4-6",
custom_llm_provider="bedrock",
)
assert cost == pytest.approx(0.052150725, rel=0, abs=1e-9)
@pytest.mark.parametrize(
"model",

View file

@ -159,51 +159,3 @@ def test_bedrock_gpt_5_6_offers_tools_and_reasoning_effort_but_not_thinking(prof
# Cache-read prices are the `*-cache-read-input-tokens` usagetype rows of the AWS Price List API, us-east-1,
# https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-east-1/index.json on 2026-09-15
@pytest.mark.parametrize(
"model,expected_cache_read",
[
("amazon.nova-lite-v1:0", 1.5e-8),
("us.amazon.nova-lite-v1:0", 1.5e-8),
("amazon.nova-micro-v1:0", 8.75e-9),
("us.amazon.nova-micro-v1:0", 8.75e-9),
("amazon.nova-pro-v1:0", 2e-7),
("us.amazon.nova-pro-v1:0", 2e-7),
("us.amazon.nova-premier-v1:0", 6.25e-7),
],
)
def test_bedrock_nova_cache_read_prices(
model, expected_cache_read, local_model_cost_map
):
model_info = litellm.model_cost[model]
assert model_info["cache_read_input_token_cost"] == expected_cache_read
usage = Usage(
prompt_tokens=1_000,
completion_tokens=100,
total_tokens=1_100,
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=400),
)
response = _bedrock_response(model, usage)
cost = completion_cost(
completion_response=response,
model=model,
custom_llm_provider="bedrock",
)
expected_cost = (
600 * model_info["input_cost_per_token"]
+ 400 * expected_cache_read
+ 100 * model_info["output_cost_per_token"]
)
assert cost == pytest.approx(expected_cost)
uncached_usage = Usage(
prompt_tokens=1_000,
completion_tokens=100,
total_tokens=1_100,
)
uncached_cost = completion_cost(
completion_response=_bedrock_response(model, uncached_usage),
model=model,
custom_llm_provider="bedrock",
)
assert cost < uncached_cost

View file

@ -1865,38 +1865,6 @@ class TestBedrockMantleResponsesSigV4:
class TestBedrockMantleResponsesPricing:
@pytest.mark.parametrize(
"model, input_cost, output_cost",
[
("openai.gpt-5.6-sol", 5.5e-06, 3.3e-05),
("openai.gpt-5.6-terra", 2.2e-06, 1.32e-05),
("openai.gpt-5.6-luna", 2.2e-07, 1.32e-06),
],
)
def test_gpt_5_6_responses_call_cost(self, local_cost_map, model, input_cost, output_cost):
from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse
input_tokens = 100000
output_tokens = 10000
response = ResponsesAPIResponse(
id="resp-1",
created_at=1700000000,
model=model,
output=[],
usage=ResponseAPIUsage(
input_tokens=input_tokens,
output_tokens=output_tokens,
total_tokens=input_tokens + output_tokens,
),
)
cost = litellm.completion_cost(
completion_response=response,
model=f"bedrock_mantle/{model}",
custom_llm_provider="bedrock_mantle",
)
assert cost == pytest.approx(input_tokens * input_cost + output_tokens * output_cost)
def test_models_registered(self, local_cost_map):
assert "bedrock_mantle/openai.gpt-5.5" in litellm.bedrock_mantle_models

View file

@ -62,23 +62,3 @@ def test_map_openai_params_preserves_max_retries_zero_falsy() -> None:
assert "max_retries" in result and result["max_retries"] == 0, (
f"max_retries=0 (falsy) must not be silently omitted; got: {result!r}"
)
def test_qwen_3_8_27b_cost_and_tokens(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
model = "cerebras/qwen-3.8-27b"
prompt_cost, completion_cost = litellm.cost_per_token(
model=model,
prompt_tokens=1000,
completion_tokens=1000,
)
assert abs(prompt_cost - 0.00099) < 1e-9
assert abs(completion_cost - 0.00149) < 1e-9
model_info = litellm.get_model_info(model)
assert model_info["max_input_tokens"] == 65536
assert model_info["max_output_tokens"] == 32768
assert model_info["supports_vision"] is True
assert model_info["supports_reasoning"] is True
assert model_info["supports_parallel_function_calling"] is True

View file

@ -45,26 +45,6 @@ class TestChatGPTResponsesAPITransformation:
assert isinstance(config, ChatGPTResponsesAPIConfig)
assert config.custom_llm_provider == LlmProviders.CHATGPT
@pytest.mark.parametrize(
"model_name",
[
"chatgpt/gpt-5.5",
"chatgpt/gpt-5.6-luna",
"chatgpt/gpt-5.6-sol",
"chatgpt/gpt-5.6-terra",
],
)
def test_chatgpt_responses_model_metadata(self, model_name: str, local_model_cost_map: None) -> None:
model_info = litellm.get_model_info(model_name)
assert model_info["litellm_provider"] == "chatgpt"
assert model_info["mode"] == "responses"
assert model_info["supported_endpoints"] == [
"/v1/chat/completions",
"/v1/responses",
]
assert model_info["max_input_tokens"] == 1050000
assert model_info["max_output_tokens"] == 128000
@pytest.mark.parametrize(
"model_name",

View file

@ -127,24 +127,3 @@ def test_transform_image_generation_request():
) == {"prompt": "a red bicycle", "quality": "high", "num_images": 2}
@pytest.mark.parametrize(
("model", "expected_cost_for_two_images"),
[
("openai/gpt-image-2", 0.29),
("gpt-image-2", 0.29),
("openai/gpt-image-2/edit", 0.302),
],
)
def test_cost_calculator_uses_registry_price(
model, expected_cost_for_two_images, monkeypatch: pytest.MonkeyPatch
):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
litellm.get_model_info.cache_clear()
response = ImageResponse(
data=[
ImageObject(url="https://v3b.fal.media/files/b/one.png"),
ImageObject(url="https://v3b.fal.media/files/b/two.png"),
]
)
assert cost_calculator(model=model, image_response=response) == pytest.approx(expected_cost_for_two_images)

View file

@ -145,20 +145,3 @@ def test_transform_request_includes_prompt_and_mapped_params():
}
@pytest.mark.parametrize(
"model", ["fal-ai/nano-banana", "fal-ai/gemini-25-flash-image"]
)
def test_nano_banana_pricing_registered(model):
info = litellm.get_model_info(
model=model, custom_llm_provider=litellm.LlmProviders.FAL_AI.value
)
assert info["output_cost_per_image"] == 0.039
assert info["mode"] == "image_generation"
def test_cost_calculator_scales_with_image_count():
image_response = ImageResponse(
data=[ImageObject(url="https://x/1.png"), ImageObject(url="https://x/2.png")]
)
cost = cost_calculator(model="fal-ai/nano-banana", image_response=image_response)
assert cost == pytest.approx(0.078)

View file

@ -17,140 +17,3 @@ def _use_local_model_cost_map(monkeypatch):
def _image_response(num_images: int = 1) -> ImageResponse:
return ImageResponse(data=[ImageObject(url="https://example.com/img.png") for _ in range(num_images)])
def test_high_quality_1024x1024_uses_keyed_price():
cost = cost_calculator(
model="openai/gpt-image-2",
image_response=_image_response(),
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
)
assert cost == pytest.approx(0.211)
def test_alias_model_uses_keyed_price():
cost = cost_calculator(
model="gpt-image-2",
image_response=_image_response(),
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
)
assert cost == pytest.approx(0.211)
def test_provider_prefixed_model_uses_keyed_price():
cost = cost_calculator(
model="fal_ai/openai/gpt-image-2",
image_response=_image_response(),
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
)
assert cost == pytest.approx(0.211)
def test_provider_prefixed_edit_model_uses_keyed_edit_price():
cost = cost_calculator(
model="fal_ai/openai/gpt-image-2/edit",
image_response=_image_response(),
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
)
assert cost == pytest.approx(0.219)
def test_default_request_priced_at_default_size_and_quality():
cost = cost_calculator(
model="openai/gpt-image-2",
image_response=_image_response(),
optional_params={},
)
assert cost == pytest.approx(0.145)
def test_auto_quality_priced_as_high():
cost = cost_calculator(
model="openai/gpt-image-2",
image_response=_image_response(),
optional_params={"quality": "auto", "image_size": {"width": 1024, "height": 1024}},
)
assert cost == pytest.approx(0.211)
def test_low_quality_4k_uses_keyed_price():
cost = cost_calculator(
model="openai/gpt-image-2",
image_response=_image_response(),
optional_params={"quality": "low", "image_size": {"width": 3840, "height": 2160}},
)
assert cost == pytest.approx(0.012)
def test_named_fal_size_uses_keyed_price():
cost = cost_calculator(
model="openai/gpt-image-2",
image_response=_image_response(),
optional_params={"quality": "high", "image_size": "square_hd"},
)
assert cost == pytest.approx(0.211)
def test_edit_model_uses_keyed_edit_price():
cost = cost_calculator(
model="openai/gpt-image-2/edit",
image_response=_image_response(),
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
)
assert cost == pytest.approx(0.219)
def test_edit_model_without_size_falls_back_to_flat_price():
cost = cost_calculator(
model="openai/gpt-image-2/edit",
image_response=_image_response(),
optional_params={"quality": "high"},
)
assert cost == pytest.approx(0.151)
def test_missing_optional_params_falls_back_to_flat_price():
cost = cost_calculator(
model="openai/gpt-image-2",
image_response=_image_response(),
optional_params=None,
)
assert cost == pytest.approx(0.145)
def test_unlisted_size_falls_back_to_flat_price():
cost = cost_calculator(
model="openai/gpt-image-2",
image_response=_image_response(),
optional_params={"quality": "high", "image_size": {"width": 999, "height": 999}},
)
assert cost == pytest.approx(0.145)
def test_keyed_price_multiplies_per_image():
cost = cost_calculator(
model="openai/gpt-image-2",
image_response=_image_response(num_images=2),
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
)
assert cost == pytest.approx(0.422)
def test_route_image_generation_passes_optional_params_to_fal():
cost = CostCalculatorUtils.route_image_generation_cost_calculator(
model="openai/gpt-image-2",
completion_response=_image_response(),
custom_llm_provider="fal_ai",
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
)
assert cost == pytest.approx(0.211)
def test_route_image_generation_with_provider_prefixed_model_uses_keyed_price():
cost = CostCalculatorUtils.route_image_generation_cost_calculator(
model="fal_ai/openai/gpt-image-2",
completion_response=_image_response(),
custom_llm_provider="fal_ai",
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
)
assert cost == pytest.approx(0.211)

View file

@ -302,18 +302,3 @@ class TestCostRegression:
def local_cost_map(self, monkeypatch):
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
def test_registry_entries(self, local_cost_map):
batch_entry = litellm.model_cost["gemini/gemini-3.5-transcribe"]
assert batch_entry["mode"] == "audio_transcription"
assert batch_entry["input_cost_per_audio_token"] == 2e-06
assert batch_entry["input_cost_per_token"] == 2e-06
assert batch_entry["output_cost_per_token"] == 1.2e-05
assert batch_entry["supported_endpoints"] == ["/v1/audio/transcriptions"]
live_entry = litellm.model_cost["gemini/gemini-3.5-transcribe-live"]
assert live_entry["mode"] == "audio_transcription"
assert live_entry["input_cost_per_audio_token"] == 3.5e-06
assert live_entry["input_cost_per_token"] == 3.5e-06
assert live_entry["output_cost_per_token"] == 2.1e-05
assert live_entry["supported_endpoints"] == ["/v1/realtime"]

View file

@ -1856,54 +1856,6 @@ def test_map_openai_params_drops_stock_voice_case_insensitively():
assert passthrough["generationConfig"]["speechConfig"]["voiceConfig"]["prebuiltVoiceConfig"]["voiceName"] == "Kore"
def test_gemini_response_done_bills_audio_output_tokens_at_audio_rate(monkeypatch):
"""Regression for the Gemini Live AUDIO output breakdown: responseTokensDetails
must survive into response.done usage and bill at output_cost_per_audio_token,
not the text rate."""
from litellm.cost_calculator import (
RealtimeAPITokenUsageProcessor,
handle_realtime_stream_cost_calculation,
)
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
config = GeminiRealtimeConfig()
done_event = config.transform_response_done_event(
message={
"serverContent": {"turnComplete": True},
"usageMetadata": {
"promptTokenCount": 377,
"responseTokenCount": 51,
"totalTokenCount": 428,
"promptTokensDetails": [{"modality": "TEXT", "tokenCount": 377}],
"responseTokensDetails": [{"modality": "AUDIO", "tokenCount": 51}],
"thoughtsTokenCount": 37,
},
},
current_response_id="resp_lit6277",
current_conversation_id="conv_lit6277",
output_items=None,
)
usage = done_event["response"]["usage"]
assert usage["output_tokens_details"]["audio_tokens"] == 51
assert usage["output_token_details"]["audio_tokens"] == 51
results = [done_event]
combined_usage = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(
results=results,
)
assert combined_usage.completion_tokens_details is not None
assert combined_usage.completion_tokens_details.audio_tokens == 51
cost = handle_realtime_stream_cost_calculation(
results=results,
combined_usage_object=combined_usage,
custom_llm_provider="gemini",
litellm_model_name="gemini-2.5-flash-native-audio-preview-12-2025",
)
assert cost == pytest.approx(377 * 5e-07 + 51 * 1.2e-05 + 37 * 2e-06)
@pytest.fixture(autouse=False)
def patch_gemini_transcribe_live_cost_map_entry(monkeypatch):
"""Inject the gemini-3.5-transcribe-live registry entry locally.

View file

@ -21,7 +21,6 @@ WEB_SEARCH_MODELS = (
COMPOUND_MODELS = ("compound", "compound-mini", "groq/compound", "groq/compound-mini")
class TestGroqWebSearchOptions:
@pytest.mark.parametrize("model", WEB_SEARCH_MODELS + COMPOUND_MODELS)
def test_supported_on_search_capable_models(self, model: str):
@ -204,36 +203,4 @@ class TestGroqWebSearchUsageSignal:
GroqChatConfig()._add_web_search_usage(model_response=model_response)
assert getattr(model_response, "usage", None) is None
@pytest.mark.usefixtures("local_model_cost_map")
@pytest.mark.parametrize(
"executed_tools, expected_cost",
[
(EXECUTED_TOOLS_THREE_SEARCHES_TWO_OPENS, 3 * 0.005 + 2 * 0.001),
(EXECUTED_TOOLS_OPENS_ONLY, 2 * 0.001),
],
)
def test_response_billed_per_action(self, executed_tools: list, expected_cost: float):
response = _groq_completion_with_mocked_response(_searched_groq_response(executed_tools))
assert StandardBuiltInToolCostTracking.response_object_includes_web_search_call(
response_object=response, usage=response.usage
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model="groq/openai/gpt-oss-20b",
response_object=response,
usage=response.usage,
custom_llm_provider="groq",
standard_built_in_tools_params={"web_search_options": {"search_context_size": "high"}},
)
assert cost == pytest.approx(expected_cost)
class TestGroqWebSearchCost:
@pytest.mark.usefixtures("local_model_cost_map")
@pytest.mark.parametrize("model", WEB_SEARCH_MODELS)
@pytest.mark.parametrize("search_context_size", ["low", "medium", "high"])
def test_browser_search_priced_per_search(self, model: str, search_context_size: str):
cost = StandardBuiltInToolCostTracking.get_cost_for_web_search(
web_search_options={"search_context_size": search_context_size},
model_info=litellm.get_model_info(model=model, custom_llm_provider="groq"),
)
assert cost == 0.005

View file

@ -308,22 +308,3 @@ def test_inception_completion_targets_inception_endpoint():
assert response.choices[0].message.content == "hi"
def test_inception_mercury_2_5_cost_and_tokens(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
model = "inception/mercury-2.5"
prompt_cost, completion_cost = litellm.cost_per_token(
model=model,
prompt_tokens=1000,
completion_tokens=500,
)
assert abs(prompt_cost - 0.0002) < 1e-9
assert abs(completion_cost - 0.000375) < 1e-9
model_info = litellm.get_model_info(model)
assert model_info["max_input_tokens"] == 260000
assert model_info["max_output_tokens"] == 65536
assert model_info["litellm_provider"] == "inception"
assert model_info["mode"] == "chat"
assert model_info["supports_function_calling"] is True
assert model_info["supports_response_schema"] is True

View file

@ -111,28 +111,6 @@ class TestCognitionProviderIdentity:
class TestCognitionCostTracking:
@pytest.mark.parametrize(
"model, expected_prompt_cost, expected_completion_cost",
[
("cognition/swe-1.7", 0.5, 2.5),
("cognition/swe-1.7-lightning", 2.5, 12.5),
],
)
def test_cost_differs_from_openai_pricing(
self, model: str, expected_prompt_cost: float, expected_completion_cost: float
):
"""A cognition-prefixed model must never be priced off an OpenAI cost entry."""
from litellm.cost_calculator import cost_per_token
prompt_cost, completion_cost = cost_per_token(
model=model,
prompt_tokens=1_000_000,
completion_tokens=1_000_000,
custom_llm_provider="cognition",
)
assert prompt_cost == pytest.approx(expected_prompt_cost)
assert completion_cost == pytest.approx(expected_completion_cost)
def test_lightning_is_five_times_the_standard_tier(self):
standard = litellm.get_model_info(model="cognition/swe-1.7")
@ -151,51 +129,4 @@ class TestCognitionCostTracking:
assert endpoints["embeddings"] is False
class TestCognitionRouting:
@pytest.mark.asyncio
async def test_router_spend_is_attributed_to_cognition_pricing(self):
"""Routed traffic is costed off the cognition entry, not an OpenAI one."""
from litellm import Router
router = Router(
model_list=[
{
"model_name": "swe",
"litellm_params": {"model": "cognition/swe-1.7", "api_key": "sk-test"},
}
]
)
response = await router.acompletion(
model="swe",
messages=[{"role": "user", "content": "hi"}],
mock_response="hello from swe",
)
usage = response.usage
expected = usage.prompt_tokens * 5e-07 + usage.completion_tokens * 2.5e-06
assert response._hidden_params["response_cost"] == pytest.approx(expected)
@pytest.mark.asyncio
async def test_router_spend_uses_the_lightning_entry_for_lightning(self):
"""The Lightning tier is its own model, costed off its own entry."""
from litellm import Router
router = Router(
model_list=[
{
"model_name": "swe-lightning",
"litellm_params": {"model": "cognition/swe-1.7-lightning", "api_key": "sk-test"},
}
]
)
response = await router.acompletion(
model="swe-lightning",
messages=[{"role": "user", "content": "hi"}],
mock_response="hello from swe lightning",
)
usage = response.usage
expected = usage.prompt_tokens * 2.5e-06 + usage.completion_tokens * 1.25e-05
assert response._hidden_params["response_cost"] == pytest.approx(expected)

View file

@ -192,20 +192,4 @@ class TestMetaAnthropicMessages:
assert headers["anthropic-version"] == "2023-06-01"
class TestMuseSparkModelInfo:
def test_muse_spark_cost_calculation(self):
from litellm import completion_cost
from litellm.types.utils import ModelResponse, Usage
response = ModelResponse(
model="muse-spark-1.1",
usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500),
)
cost = completion_cost(
completion_response=response,
model="meta/muse-spark-1.1",
custom_llm_provider="meta",
)
expected = 1000 * 1.25e-06 + 500 * 4.25e-06
assert abs(cost - expected) < 1e-12

View file

@ -154,17 +154,3 @@ class TestTensormeshCostMap:
for model in TENSORMESH_MODELS:
assert litellm.supports_reasoning(model) is (model in reasoning_models), model
def test_cost_is_wired_and_cache_reads_are_free(self):
prompt_cost, completion_cost = litellm.cost_per_token(
model="tensormesh/openai/gpt-oss-120b",
prompt_tokens=1_000_000,
completion_tokens=1_000_000,
)
assert prompt_cost == pytest.approx(0.15)
assert completion_cost == pytest.approx(0.60)
assert (
litellm.model_cost["tensormesh/openai/gpt-oss-120b"][
"cache_read_input_token_cost"
]
== 0
)

View file

@ -431,76 +431,3 @@ class TestParallelAISearch:
assert result.snippet == ""
assert result.date is None
assert result.model_dump()["excerpts"] == ()
@pytest.mark.parametrize(
"mode,usage,max_results,expected_cost",
[
("turbo", [{"name": "sku_search", "count": 1}], None, 0.001),
("fast", [{"name": "sku_search", "count": 1}], None, 0.001),
("basic", [{"name": "sku_search", "count": 1}], None, 0.005),
("advanced", [{"name": "sku_search", "count": 1}], None, 0.005),
(
"basic",
[
{"name": "sku_search", "count": 1},
{"name": "sku_search_additional_results", "count": 2},
],
20,
0.007,
),
("basic", None, 20, 0.015),
],
)
@pytest.mark.asyncio
async def test_search_cost_uses_mode_and_provider_usage(
self, mode, usage, max_results, expected_cost, bundled_cost_map, respx_mock, httpx_transport
):
response_payload = {**MOCK_V1_RESPONSE, "usage": usage}
respx_mock.post("https://api.parallel.ai/v1/search").respond(json=response_payload)
response = await litellm.asearch(
query="AI developments",
search_provider="parallel_ai",
mode=mode,
max_results=max_results,
)
assert response._hidden_params["response_cost"] == pytest.approx(expected_cost)
@pytest.mark.asyncio
async def test_search_cost_treats_keyword_queries_as_one_request(
self, bundled_cost_map, respx_mock, httpx_transport
):
response_payload = {
**MOCK_V1_RESPONSE,
"usage": [{"name": "sku_search", "count": 1}],
}
respx_mock.post("https://api.parallel.ai/v1/search").respond(json=response_payload)
response = await litellm.asearch(
query=["AI developments", "machine learning trends"],
search_provider="parallel_ai",
mode="basic",
)
assert response._hidden_params["response_cost"] == pytest.approx(0.005)
@pytest.mark.asyncio
async def test_caller_cannot_supply_provider_usage(self, bundled_cost_map, respx_mock, httpx_transport):
"""`_parallel_ai_usage` prices the request, so a caller must not be able to set it.
The provider reports no usage here, which is the case where a caller-supplied
value would otherwise survive into the cost calculation.
"""
response_payload = {k: v for k, v in MOCK_V1_RESPONSE.items() if k != "usage"}
route = respx_mock.post("https://api.parallel.ai/v1/search").respond(json=response_payload)
response = await litellm.asearch(
query="AI developments",
search_provider="parallel_ai",
mode="basic",
_parallel_ai_usage=[{"name": "sku_search", "count": 0}],
)
assert response._hidden_params["response_cost"] == pytest.approx(0.005)
assert "_parallel_ai_usage" not in json.loads(route.calls[0].request.content)

View file

@ -140,23 +140,6 @@ class TestPerplexityCostCalculator:
assert prompt_cost == 0.0
assert completion_cost == 0.008
def test_falls_back_to_manual_calculation_when_no_cost_provided(self):
"""
Test that manual cost calculation is used when Perplexity doesn't
provide the cost object (fallback behavior).
"""
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
# No cost object - should use manual calculation
prompt_cost, completion_cost = perplexity_cost_per_token(model="sonar-deep-research", usage=usage)
# Should calculate manually: 100 * 2e-6 + 50 * 8e-6
expected_prompt = 100 * 2e-6
expected_completion = 50 * 8e-6
assert math.isclose(prompt_cost, expected_prompt, rel_tol=1e-6)
assert math.isclose(completion_cost, expected_completion, rel_tol=1e-6)
OFF_PEAK_MODEL = "sonar-off-peak-test"
OFF_PEAK_WINDOW = "14:00-00:00"
INSIDE_WINDOW = datetime(2026, 9, 3, 17, 25, tzinfo=timezone.utc)

View file

@ -150,24 +150,3 @@ class TestPerplexityIntegration:
assert hasattr(model_response.usage, "prompt_tokens_details")
assert hasattr(model_response.usage, "citation_tokens")
assert model_response.usage.prompt_tokens_details.web_search_requests == 3
@pytest.mark.parametrize("provider_name", ["perplexity", "PERPLEXITY", "Perplexity"])
def test_case_insensitive_provider_matching(self, provider_name):
"""Test that cost calculation works with different case variations of provider name."""
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
usage.citation_tokens = 10
usage.prompt_tokens_details = PromptTokensDetailsWrapper(web_search_requests=1)
# Should work regardless of case
prompt_cost, completion_cost_val = cost_per_token(
model="sonar-deep-research",
custom_llm_provider=provider_name.lower(), # Normalize to lowercase
usage_object=usage,
)
# Should calculate costs correctly
expected_prompt_cost = (100 * 2e-6) + (10 * 2e-6)
expected_completion_cost = (50 * 8e-6) + (1 * 0.005)
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
assert math.isclose(completion_cost_val, expected_completion_cost, rel_tol=1e-6)

View file

@ -1056,44 +1056,3 @@ class TestSpendTracking:
litellm.model_cost = original_model_cost
litellm.get_model_info.cache_clear()
def test_should_charge_by_audio_duration(self, monkeypatch):
import litellm
monkeypatch.setattr("time.sleep", lambda *_: None)
responses = {
"POST https://api.soniox.com/v1/transcriptions": [
_make_response({"id": "tx_1", "status": "queued"})
],
"GET https://api.soniox.com/v1/transcriptions/tx_1": [
_make_response(
{"id": "tx_1", "status": "completed", "audio_duration_ms": 600000}
),
],
"GET https://api.soniox.com/v1/transcriptions/tx_1/transcript": [
_make_response({"text": "hello world", "tokens": []}),
],
"DELETE https://api.soniox.com/v1/transcriptions/tx_1": [
_make_response({"deleted": True}),
],
}
resp = SonioxAudioTranscriptionHandler().audio_transcriptions(
audio_file=None,
optional_params={"audio_url": "https://example.com/a.wav"},
litellm_params={},
atranscription=False,
**_common_call_kwargs(_MockSyncClient(responses)),
)
assert resp._hidden_params["audio_transcription_duration"] == pytest.approx(
600.0
)
cost = litellm.completion_cost(
completion_response=resp,
model="soniox/stt-async-v4",
call_type="transcription",
)
# 10 minutes of audio billed at Soniox's ~$0.10/hour async rate.
assert cost > 0
assert cost == pytest.approx((0.10 / 3600) * 600.0, rel=1e-3)

View file

@ -22,16 +22,6 @@ def config():
class TestGetCompleteUrl:
def test_defaults_to_us_regional_host(self, config):
url = config.get_complete_url(
api_base=None,
api_key=None,
model="chirp_3",
optional_params={},
litellm_params={"vertex_project": "test-project"},
)
assert url == "https://us-speech.googleapis.com/v2/projects/test-project/locations/us/recognizers/_:recognize"
def test_uses_vertex_location_for_regional_host(self, config):
url = config.get_complete_url(
api_base=None,
@ -52,16 +42,6 @@ class TestGetCompleteUrl:
)
assert url == "https://speech.googleapis.com/v2/projects/test-project/locations/global/recognizers/_:recognize"
def test_api_base_override(self, config):
url = config.get_complete_url(
api_base="http://localhost:8080/",
api_key=None,
model="chirp_3",
optional_params={},
litellm_params={"vertex_project": "test-project"},
)
assert url == "http://localhost:8080/v2/projects/test-project/locations/us/recognizers/_:recognize"
@pytest.mark.parametrize(
"location,expected_netloc",
[
@ -317,18 +297,3 @@ class TestProviderRouting:
class TestModelCostEntry:
REPO_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../.."))
@pytest.mark.parametrize(
"cost_map_path",
[
"model_prices_and_context_window.json",
"litellm/model_prices_and_context_window_backup.json",
],
)
def test_chirp_3_registered_as_audio_transcription(self, cost_map_path):
with open(os.path.join(self.REPO_ROOT, cost_map_path)) as f:
entry = json.load(f)["vertex_ai/chirp_3"]
assert entry["mode"] == "audio_transcription"
assert entry["litellm_provider"] == "vertex_ai"
assert entry["input_cost_per_second"] == pytest.approx(0.016 / 60, rel=1e-3)
assert entry["supported_endpoints"] == ["/v1/audio/transcriptions"]

View file

@ -309,37 +309,3 @@ class TestOptionalParams:
class TestModelCostEntry:
REPO_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../.."))
@pytest.mark.parametrize(
"cost_map_path",
[
"model_prices_and_context_window.json",
"litellm/model_prices_and_context_window_backup.json",
],
)
def test_transcribe_preview_pricing(self, cost_map_path):
with open(os.path.join(self.REPO_ROOT, cost_map_path)) as f:
entry = json.load(f)["vertex_ai/gemini-3.5-transcribe-preview"]
assert entry["mode"] == "audio_transcription"
assert entry["litellm_provider"] == "vertex_ai"
assert entry["input_cost_per_audio_token"] == pytest.approx(2e-06)
assert entry["input_cost_per_token"] == pytest.approx(2e-06)
assert entry["output_cost_per_token"] == pytest.approx(1.2e-05)
assert entry["supported_endpoints"] == ["/v1/audio/transcriptions"]
@pytest.mark.parametrize(
"cost_map_path",
[
"model_prices_and_context_window.json",
"litellm/model_prices_and_context_window_backup.json",
],
)
def test_transcribe_live_preview_pricing(self, cost_map_path):
with open(os.path.join(self.REPO_ROOT, cost_map_path)) as f:
entry = json.load(f)["vertex_ai/gemini-3.5-transcribe-live-preview"]
assert entry["mode"] == "audio_transcription"
assert entry["litellm_provider"] == "vertex_ai"
assert entry["input_cost_per_audio_token"] == pytest.approx(3.5e-06)
assert entry["input_cost_per_token"] == pytest.approx(3.5e-06)
assert entry["output_cost_per_token"] == pytest.approx(2.1e-05)
assert entry["supported_endpoints"] == ["/v1/realtime"]

View file

@ -407,227 +407,4 @@ class TestProcessEmbedContentResponseUsage:
)
assert result.usage.prompt_tokens > 0
def test_file_reference_image_billed_per_image_token_rate(self):
response_json = {
"embedding": {"values": [0.1, 0.2, 0.3]},
"usageMetadata": {
"promptTokenCount": 258,
"totalTokenCount": 258,
"promptTokensDetails": [{"modality": "IMAGE", "tokenCount": 258}],
},
}
result = process_embed_content_response(
input=["files/img123"],
model_response=EmbeddingResponse(),
model=self.MODEL,
response_json=response_json,
resolved_files={
"files/img123": {
"mime_type": "image/png",
"uri": "https://example.com/img123",
}
},
)
assert result.usage.prompt_tokens_details.image_tokens == 258
assert result.usage.prompt_tokens_details.text_tokens == 0
prompt_cost, _ = generic_cost_per_token(
model=self.MODEL,
usage=result.usage,
custom_llm_provider="vertex_ai",
)
assert prompt_cost == pytest.approx(258 * 4.5e-7)
def test_file_reference_non_image_not_counted_as_image(self):
"""A files/... ref resolving to a non-image mime keeps audio token billing."""
response_json = {
"embedding": {"values": [0.1, 0.2]},
"usageMetadata": {
"promptTokenCount": 64,
"totalTokenCount": 64,
"promptTokensDetails": [{"modality": "AUDIO", "tokenCount": 64}],
},
}
result = process_embed_content_response(
input=["files/clip1"],
model_response=EmbeddingResponse(),
model=self.MODEL,
response_json=response_json,
resolved_files={
"files/clip1": {
"mime_type": "audio/mpeg",
"uri": "https://example.com/clip1",
}
},
)
assert result.usage.prompt_tokens_details.audio_tokens == 64
assert result.usage.prompt_tokens_details.image_tokens == 0
prompt_cost, _ = generic_cost_per_token(
model=self.MODEL,
usage=result.usage,
custom_llm_provider="vertex_ai",
)
assert prompt_cost == pytest.approx(64 * 6.5e-6)
def test_video_plus_audio_does_not_double_bill_text(self):
"""Video and audio responses are billed from their respective token counts."""
response_json = {
"embedding": {"values": [0.1]},
"usageMetadata": {
"promptTokenCount": 580,
"totalTokenCount": 580,
"promptTokensDetails": [
{"modality": "VIDEO", "tokenCount": 516},
{"modality": "AUDIO", "tokenCount": 64},
],
},
}
result = process_embed_content_response(
input=["gs://bucket/clip.mp4"],
model_response=EmbeddingResponse(),
model=self.MODEL,
response_json=response_json,
)
assert result.usage.prompt_tokens_details.text_tokens == 0
assert result.usage.prompt_tokens_details.video_tokens == 516
assert result.usage.prompt_tokens_details.audio_tokens == 64
prompt_cost, _ = generic_cost_per_token(
model=self.MODEL,
usage=result.usage,
custom_llm_provider="vertex_ai",
)
assert prompt_cost == pytest.approx(516 * 1.2e-5 + 64 * 6.5e-6)
def test_preview_alias_bills_audio_per_token(self):
response_json = {
"embedding": {"values": [0.1]},
"usageMetadata": {
"promptTokenCount": 64,
"totalTokenCount": 64,
"promptTokensDetails": [{"modality": "AUDIO", "tokenCount": 64}],
},
}
result = process_embed_content_response(
input="audio",
model_response=EmbeddingResponse(),
model="gemini-embedding-2-preview",
response_json=response_json,
)
prompt_cost, _ = generic_cost_per_token(
model="gemini-embedding-2-preview",
usage=result.usage,
custom_llm_provider="vertex_ai",
)
assert prompt_cost == pytest.approx(64 * 6.5e-6)
def test_image_without_modality_details_uses_image_rate(self):
response_json = {
"embedding": {"values": [0.1]},
"usageMetadata": {
"promptTokenCount": 258,
"totalTokenCount": 258,
},
}
result = process_embed_content_response(
input=IMAGE_DATA_URI,
model_response=EmbeddingResponse(),
model=self.MODEL,
response_json=response_json,
)
assert result.usage.prompt_tokens_details.image_tokens == 258
assert result.usage.prompt_tokens_details.text_tokens == 0
prompt_cost, _ = generic_cost_per_token(
model=self.MODEL,
usage=result.usage,
custom_llm_provider="vertex_ai",
)
assert prompt_cost == pytest.approx(258 * 4.5e-7)
@pytest.mark.parametrize(
"input_value,resolved_files,expected_image_tokens",
[
(GCS_URL, {}, 258),
("gs://my-bucket/clip.mp4", {}, 0),
("gs://my-bucket/unknown.bin", {}, 0),
("files/image-123", {"files/image-123": {"mime_type": "image/jpeg"}}, 258),
("files/missing", {}, 0),
("data:application/octet-stream;base64,abc", {}, 0),
([[IMAGE_DATA_URI]], {}, 258),
([], {}, 0),
],
)
def test_missing_modality_details_classifies_image_inputs(self, input_value, resolved_files, expected_image_tokens):
response_json = {
"embedding": {"values": [0.1]},
"usageMetadata": {
"promptTokenCount": 258,
"totalTokenCount": 258,
},
}
result = process_embed_content_response(
input=input_value,
model_response=EmbeddingResponse(),
model=self.MODEL,
response_json=response_json,
resolved_files=resolved_files,
)
assert result.usage.prompt_tokens_details.image_tokens == expected_image_tokens
assert result.usage.prompt_tokens_details.text_tokens == 0
prompt_cost, _ = generic_cost_per_token(
model=self.MODEL,
usage=result.usage,
custom_llm_provider="vertex_ai",
)
expected_rate = 4.5e-7 if expected_image_tokens else 2e-7
assert prompt_cost == pytest.approx(258 * expected_rate)
def test_mixed_text_and_image_without_modality_details_not_billed_as_image(self):
response_json = {
"embedding": {"values": [0.1]},
"usageMetadata": {
"promptTokenCount": 270,
"totalTokenCount": 270,
},
}
result = process_embed_content_response(
input=["a short caption", IMAGE_DATA_URI],
model_response=EmbeddingResponse(),
model=self.MODEL,
response_json=response_json,
)
assert result.usage.prompt_tokens_details.image_tokens == 0
prompt_cost, _ = generic_cost_per_token(
model=self.MODEL,
usage=result.usage,
custom_llm_provider="vertex_ai",
)
assert prompt_cost == pytest.approx(270 * 2e-7)
def test_text_without_modality_details_uses_text_rate(self):
response_json = {
"embedding": {"values": [0.1]},
"usageMetadata": {
"promptTokenCount": 12,
"totalTokenCount": 12,
},
}
result = process_embed_content_response(
input="a short caption",
model_response=EmbeddingResponse(),
model=self.MODEL,
response_json=response_json,
)
assert result.usage.prompt_tokens_details.text_tokens == 0
assert result.usage.prompt_tokens_details.image_tokens == 0
prompt_cost, _ = generic_cost_per_token(
model=self.MODEL,
usage=result.usage,
custom_llm_provider="vertex_ai",
)
assert prompt_cost == pytest.approx(12 * 2e-7)

View file

@ -238,56 +238,6 @@ def test_audio_predict_response_supports_bytes_base64_encoded(
assert logging_obj.model_call_details["response_cost"] == pytest.approx(0.06)
@pytest.mark.parametrize("runtime_entry_is_missing", (True, False))
def test_lyria_predict_cost_falls_back_to_bundled_map_when_runtime_metadata_is_incomplete(
monkeypatch: pytest.MonkeyPatch,
runtime_entry_is_missing: bool,
local_model_cost_map: None,
) -> None:
if runtime_entry_is_missing:
monkeypatch.delitem(litellm.model_cost, "vertex_ai/lyria-002")
else:
monkeypatch.setitem(
litellm.model_cost,
"vertex_ai/lyria-002",
{
key: value
for key, value in litellm.model_cost["vertex_ai/lyria-002"].items()
if key != "output_cost_per_image"
},
)
logging_obj = MagicMock()
logging_obj.model_call_details = {}
response = httpx.Response(
status_code=200,
json={
"predictions": [
{
"audioContent": "clip",
"mimeType": "audio/wav",
}
]
},
)
result = VertexPassthroughLoggingHandler.vertex_passthrough_handler(
httpx_response=response,
logging_obj=logging_obj,
url_route="/v1/projects/test/locations/us-central1/publishers/google/models/lyria-002:predict",
result=response.text,
start_time=datetime.now(),
end_time=datetime.now(),
cache_hit=False,
request_body={"instances": [{"prompt": "ambient piano"}]},
)
if runtime_entry_is_missing:
assert "vertex_ai/lyria-002" not in litellm.model_cost
assert result["kwargs"]["model"] == "lyria-002"
assert result["kwargs"]["response_cost"] == pytest.approx(0.06)
assert logging_obj.model_call_details["response_cost"] == pytest.approx(0.06)
def test_image_predict_response_is_not_billed_as_audio(
local_model_cost_map: None,
) -> None:

View file

@ -123,18 +123,6 @@ class TestVertexAIVideoConfig:
model="veo-002", api_base=None, litellm_params={}
)
def test_get_complete_url_default_location(self):
"""Test URL construction with default location."""
litellm_params = {"vertex_project": "test-project"}
url = self.config.get_complete_url(
model="veo-002", api_base=None, litellm_params=litellm_params
)
# Should default to us-central1
assert "us-central1" in url
# Should NOT include endpoint
assert not url.endswith(":predictLongRunning")
def test_veo_31_lite_provider_routing_from_local_model_map(
self, monkeypatch: pytest.MonkeyPatch
@ -154,24 +142,6 @@ class TestVertexAIVideoConfig:
assert model == "veo-3.1-lite-generate-001"
assert custom_llm_provider == "vertex_ai"
def test_veo_31_lite_cost_uses_resolution_tiers(self):
model_cost = _load_model_cost_map(BACKUP_MODEL_COST_PATH)
model_info = model_cost[VEO_31_LITE_VERTEX_MODEL]
assert video_generation_cost(
model=VEO_31_LITE_VERTEX_MODEL,
duration_seconds=10.0,
custom_llm_provider="vertex_ai",
model_info=dict(model_info),
video_resolution="720p",
) == pytest.approx(0.5)
assert video_generation_cost(
model=VEO_31_LITE_VERTEX_MODEL,
duration_seconds=10.0,
custom_llm_provider="vertex_ai",
model_info=dict(model_info),
video_resolution="1080p",
) == pytest.approx(0.8)
def test_transform_video_create_request(self):
"""Test transformation of video creation request."""

View file

@ -105,16 +105,3 @@ def test_both_cost_maps_agree_on_the_redirected_slugs():
backup = json.loads(BACKUP_PRICES_PATH.read_text(encoding="utf-8"))
for slug in (*REDIRECTED_SLUGS, *CODE_SLUGS, REDIRECT_TARGET, CODE_REDIRECT_TARGET):
assert prices[slug] == backup[slug], slug
def test_every_retired_chat_slug_is_covered(cost_map: dict):
"""The lists above must stay in step with what the registry marks retired."""
marked = {
key
for key, entry in cost_map.items()
if isinstance(entry, dict)
and entry.get("litellm_provider") == "xai"
and "deprecation_date" in entry
and entry.get("mode") == "chat"
}
assert marked == {*REDIRECTED_SLUGS, *CODE_SLUGS}

View file

@ -55,34 +55,6 @@ def test_zai_in_provider_lists():
assert "zai" in litellm.provider_list
def test_zai_glm46_cost_calculation(local_model_cost_map):
"""Test the cost calculation for glm-4.6"""
prompt_cost, completion_cost = cost_per_token(
model="zai/glm-4.6",
prompt_tokens=1000000, # 1M tokens
completion_tokens=1000000,
)
# GLM-4.6: $0.6/M input, $2.2/M output
assert math.isclose(prompt_cost, 0.6, rel_tol=1e-6)
assert math.isclose(completion_cost, 2.2, rel_tol=1e-6)
def test_glm47_cost_calculation(local_model_cost_map):
"""Test cost calculation for GLM-4.7"""
prompt_cost, completion_cost = cost_per_token(
model="zai/glm-4.7",
prompt_tokens=1000000, # 1M tokens
completion_tokens=1000000,
)
# GLM-4.7: $0.6/M input, $2.2/M output (same as GLM-4.6)
assert math.isclose(prompt_cost, 0.6, rel_tol=1e-6)
assert math.isclose(completion_cost, 2.2, rel_tol=1e-6)
@pytest.mark.asyncio
async def test_zai_completion_call(respx_mock, zai_response, monkeypatch):
"""Test completion call with zai provider using mocked response"""

View file

@ -7,31 +7,6 @@ from litellm.proxy.common_utils.prompt_cache_pricing import price_cache_tokens
from litellm.types.management_endpoints.prompt_cache_prediction import CacheTokenBuckets
@pytest.mark.parametrize(
("model", "expected"),
[("anthropic/claude-sonnet-4-5", 1.26), ("anthropic/claude-sonnet-4-6", 0.63)],
)
def test_prices_all_cache_buckets_at_total_context_tier(model: str, expected: float) -> None:
tokens: Final = CacheTokenBuckets(
uncached_input_tokens=100_000,
cache_read_input_tokens=50_000,
cache_creation_5m_input_tokens=20_000,
cache_creation_1h_input_tokens=40_000,
)
assert price_cache_tokens(model, "unconfigured-deployment", tokens) == pytest.approx(expected)
@pytest.mark.parametrize(("total", "expected"), [(200_000, 0.387), (200_001, 0.774006)])
def test_long_context_tier_starts_above_threshold(total: int, expected: float) -> None:
tokens: Final = CacheTokenBuckets(
uncached_input_tokens=total - 100_000,
cache_creation_1h_input_tokens=10_000,
cache_read_input_tokens=90_000,
)
actual: Final = price_cache_tokens("anthropic/claude-sonnet-4-5", "unconfigured-deployment", tokens)
assert actual == pytest.approx(expected)
def test_deployment_tariff_wins_without_proxy_discounts_or_margins(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(litellm, "model_cost", litellm.model_cost.copy())
litellm.Router(

View file

@ -110,54 +110,6 @@ async def _observe(
await cache.async_set_cache(_cache_key(scope, prefix.fingerprint), observation.model_dump_json(), ttl=3_600)
@pytest.mark.asyncio
@pytest.mark.parametrize(("ttl", "cold_cost"), [("5m", 0.0145), ("1h", 0.022)])
async def test_unobserved_cache_prices_cold_and_warm_bounds(ttl: str, cold_cost: float) -> None:
body: Final = _body(ttl)
arm: Final = await endpoint.predict_arm(_deployment(), body, _prefix(body), _CALLER, DualCache(), Counts())
assert arm.cache_state == "unknown"
assert arm.reason == "no_compatible_observation"
assert arm.evidence is None
assert arm.estimate is not None and arm.cold is not None and arm.warm is not None
assert arm.estimate.input_cost == pytest.approx(cold_cost)
assert arm.cold.input_cost == pytest.approx(cold_cost)
assert arm.warm.input_cost == pytest.approx(0.003)
assert arm.cold.tokens.uncached_input_tokens == 1_000
assert arm.cold.tokens.cache_read_input_tokens == 0
assert arm.cold.tokens.cache_creation_5m_input_tokens == (5_000 if ttl == "5m" else 0)
assert arm.cold.tokens.cache_creation_1h_input_tokens == (5_000 if ttl == "1h" else 0)
assert arm.warm.tokens.cache_read_input_tokens == 5_000
@pytest.mark.asyncio
@pytest.mark.parametrize(
("cached_tokens", "warm_cost", "cold_cost"), [(5_400, 0.00228, 0.0147), (4_600, 0.00372, 0.0143)]
)
@pytest.mark.parametrize("expired", [False, True])
async def test_exact_prefix_conserves_total_with_observed_count_in_all_scenarios(
cached_tokens: int, warm_cost: float, cold_cost: float, expired: bool
) -> None:
cache: Final = DualCache()
body: Final = _body()
await _observe(cache, body, cached_tokens=cached_tokens, expired=expired)
arm: Final = await endpoint.predict_arm(_deployment(), body, _prefix(body), _CALLER, cache, Counts())
assert arm.cache_state == ("stale" if expired else "warm")
assert arm.evidence is not None
assert arm.estimate is not None and arm.warm is not None and arm.cold is not None
assert arm.warm.tokens.cache_read_input_tokens == cached_tokens
assert arm.warm.tokens.cache_creation_5m_input_tokens == 0
assert arm.cold.tokens.cache_creation_5m_input_tokens == cached_tokens
assert arm.cold.tokens.cache_read_input_tokens == 0
for scenario in (arm.estimate, arm.cold, arm.warm):
assert scenario.tokens.total_tokens == 6_000
assert scenario.tokens.uncached_input_tokens == 6_000 - cached_tokens
assert arm.warm.input_cost == pytest.approx(warm_cost)
assert arm.cold.input_cost == pytest.approx(cold_cost)
assert arm.estimate.input_cost == pytest.approx(cold_cost if expired else warm_cost)
@pytest.mark.asyncio
async def test_observed_prefix_larger_than_full_request_returns_unknown() -> None:
cache: Final = DualCache()
@ -170,22 +122,6 @@ async def test_observed_prefix_larger_than_full_request_returns_unknown() -> Non
assert arm.estimate is None and arm.cold is None and arm.warm is None
@pytest.mark.asyncio
@pytest.mark.parametrize(("ttl", "expected"), [("5m", 0.0053), ("1h", 0.0068)])
async def test_append_only_prefix_reads_old_tokens_and_writes_extension(ttl: str, expected: float) -> None:
cache: Final = DualCache()
await _observe(cache, _body(ttl), cached_tokens=4_000)
body: Final = _body(ttl, extended=True)
arm: Final = await endpoint.predict_arm(_deployment(), body, _prefix(body), _CALLER, cache, Counts())
assert arm.cache_state == "partial"
assert arm.estimate is not None
assert arm.estimate.tokens.cache_read_input_tokens == 4_000
assert arm.estimate.tokens.cache_creation_5m_input_tokens == (1_000 if ttl == "5m" else 0)
assert arm.estimate.tokens.cache_creation_1h_input_tokens == (1_000 if ttl == "1h" else 0)
assert arm.estimate.input_cost == pytest.approx(expected)
@pytest.mark.asyncio
async def test_expired_observation_estimates_a_cold_rebuild() -> None:
cache: Final = DualCache()
@ -202,22 +138,6 @@ async def test_expired_observation_estimates_a_cold_rebuild() -> None:
assert arm.estimate.input_cost == arm.cold.input_cost
@pytest.mark.asyncio
async def test_below_model_minimum_prices_all_input_as_uncached() -> None:
body: Final = _body()
arm: Final = await endpoint.predict_arm(
_deployment(), body, _prefix(body), _CALLER, DualCache(), Counts(total=1_500, prefix=1_000)
)
assert arm.cache_state == "disabled"
assert arm.reason == "below_cache_minimum"
assert arm.estimate is not None
assert arm.estimate.tokens.uncached_input_tokens == 1_500
assert arm.estimate.tokens.cache_read_input_tokens == 0
assert arm.estimate.tokens.cache_creation_5m_input_tokens == 0
assert arm.estimate.input_cost == pytest.approx(0.003)
@pytest.mark.asyncio
@pytest.mark.parametrize("counts", [Counts(total=None), Counts(prefix=None), Counts(total=4_000)])
async def test_unavailable_or_inconsistent_token_counts_return_null_estimates(counts: Counts) -> None:
@ -269,20 +189,6 @@ async def test_custom_api_base_from_environment_returns_unknown_before_counting(
assert arm.estimate is None and arm.cold is None and arm.warm is None
@pytest.mark.asyncio
async def test_explicit_official_api_base_overrides_custom_environment(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setenv("ANTHROPIC_API_BASE", "https://custom.invalid")
body: Final = _body()
arm: Final = await endpoint.predict_arm(
_deployment(api_base="https://api.anthropic.com"), body, _prefix(body), _CALLER, DualCache(), Counts()
)
assert arm.cache_state == "unknown"
assert arm.reason == "no_compatible_observation"
assert arm.estimate is not None
assert arm.estimate.input_cost == pytest.approx(0.0145)
@dataclass(frozen=True)
class _ProxyLogging:
internal_usage_cache: InternalUsageCache
@ -343,38 +249,6 @@ async def _post(
)
@pytest.mark.asyncio
@pytest.mark.parametrize(
("warm_deployment", "warm_model", "expected_delta", "expected_penalty"),
[("sonnet", "claude-sonnet-5", -0.03325, 0.0), ("opus", "claude-opus-5", 0.007, 0.0115)],
)
async def test_switch_delta_accounts_for_each_deployment_cache(
monkeypatch: pytest.MonkeyPatch,
warm_deployment: str,
warm_model: str,
expected_delta: float,
expected_penalty: float,
) -> None:
cache: Final = DualCache()
body: Final = _body()
await _observe(cache, body, deployment_id=warm_deployment, model=warm_model)
app: Final = _app(monkeypatch, cache, caller=UserAPIKeyAuth(api_key=_CALLER))
response: Final = await _post(app, body)
assert response.status_code == 200, response.text
result: Final = CachePredictionResponse.model_validate(response.json())
assert result.switch_delta == pytest.approx(expected_delta)
assert result.cache_rebuild_penalty == pytest.approx(expected_penalty)
assert result.cache_guarantee is False
assert result.pricing_basis == "input_before_discounts_and_margins"
if warm_deployment == "sonnet":
assert result.switch.cache_state == "warm"
assert result.stay.cache_state == "unknown"
else:
assert result.stay.cache_state == "warm"
assert result.switch.cache_state == "unknown"
@pytest.mark.asyncio
async def test_missing_caller_identity_cannot_reuse_observations(monkeypatch: pytest.MonkeyPatch) -> None:
cache: Final = DualCache()
@ -568,53 +442,6 @@ async def test_each_count_preserves_auth_cached_request_tag_limits(
assert calls.get_nowait() == "claude-opus-5"
@pytest.mark.asyncio
async def test_provider_counter_failure_releases_parallel_capacity(monkeypatch: pytest.MonkeyPatch) -> None:
cache: Final = DualCache()
limiter: Final = _PROXY_MaxParallelRequestsHandler_v3(InternalUsageCache(cache))
caller: Final = UserAPIKeyAuth(api_key=_CALLER, max_parallel_requests=1)
async def fail_count(model: str, api_key: str, body: Mapping[str, JsonValue]) -> int | None:
raise RuntimeError("provider counter failed")
app: Final = _app(monkeypatch, cache, caller=caller, counts=fail_count, limiter=limiter)
with pytest.raises(RuntimeError, match="provider counter failed"):
await _post(app, _body())
recovered: Final = await _post(_app(monkeypatch, cache, caller=caller, limiter=limiter), _body())
assert recovered.status_code == 200, recovered.text
assert recovered.json()["switch"]["estimate"]["input_cost"] == pytest.approx(0.0145)
@pytest.mark.asyncio
async def test_cancelled_provider_counter_releases_parallel_capacity(monkeypatch: pytest.MonkeyPatch) -> None:
cache: Final = DualCache()
limiter: Final = _PROXY_MaxParallelRequestsHandler_v3(InternalUsageCache(cache))
caller: Final = UserAPIKeyAuth(api_key=_CALLER, max_parallel_requests=1)
started: Final = asyncio.Event()
release: Final = asyncio.Event()
async def wait_count(model: str, api_key: str, body: Mapping[str, JsonValue]) -> int | None:
started.set()
await release.wait()
return await Counts()(model, api_key, body)
app: Final = _app(monkeypatch, cache, caller=caller, counts=wait_count, limiter=limiter)
pending: Final = asyncio.create_task(_post(app, _body()))
try:
await asyncio.wait_for(started.wait(), timeout=5)
pending.cancel()
with pytest.raises(asyncio.CancelledError):
await pending
release.set()
recovered: Final = await asyncio.wait_for(_post(app, _body()), timeout=5)
assert recovered.status_code == 200, recovered.text
assert recovered.json()["switch"]["estimate"]["input_cost"] == pytest.approx(0.0145)
finally:
pending.cancel()
release.set()
await asyncio.gather(pending, return_exceptions=True)
async def _unexpected_count(model: str, api_key: str, body: Mapping[str, JsonValue]) -> int | None:
pytest.fail("Unsupported prediction must return before contacting the token counter")

View file

@ -2151,96 +2151,6 @@ async def test_proxy_only_error_5xx_keeps_traceback_and_runs_sync_callbacks(monk
assert "test_proxy_utils" in captured["async_traceback"]
def test_create_model_info_response_resolves_alias_to_deployment_model():
"""A public model name that is not itself a cost-map key must not be resolved through
the fallback-generalization rules: `bedrock-claude-opus-5` matches the generic
claude-family baseline (200k/64k) by substring, while the deployment it fronts really
accepts 1M/128k. Regression for the /v1/models alias resolution introduced in v1.94.0."""
from litellm import Router
saved_model_cost = dict(litellm.model_cost)
try:
router = Router(
model_list=[
{
"model_name": "bedrock-claude-opus-5",
"litellm_params": {
"custom_llm_provider": "bedrock",
"model": "bedrock/eu.anthropic.claude-opus-5",
},
"model_info": {"base_model": "eu.anthropic.claude-opus-5"},
}
]
)
response = create_model_info_response(
model_id="bedrock-claude-opus-5", provider="openai", llm_router=router
)
finally:
litellm.model_cost.clear()
litellm.model_cost.update(saved_model_cost)
assert response["max_input_tokens"] == 1000000
assert response["max_output_tokens"] == 128000
def test_create_model_info_response_keeps_exact_alias_over_generalized_deployment_model():
"""Mirror of the alias bug: when the deployment points at a custom backend name that
only matches a generalization rule, the listed name's exact cost-map entry is the
better answer and must win."""
from litellm import Router
saved_model_cost = dict(litellm.model_cost)
try:
router = Router(
model_list=[
{
"model_name": "claude-opus-5",
"litellm_params": {
"custom_llm_provider": "bedrock",
"model": "bedrock/my-claude-opus-5-provisioned",
},
}
]
)
response = create_model_info_response(
model_id="claude-opus-5", provider="openai", llm_router=router
)
finally:
litellm.model_cost.clear()
litellm.model_cost.update(saved_model_cost)
assert response["max_input_tokens"] == 1000000
def test_create_model_info_response_falls_back_to_alias_for_opaque_deployment_name():
"""An Azure deployment named after the resource rather than the model has no cost-map
entry; the listed name still does, and must keep answering."""
from litellm import Router
saved_model_cost = dict(litellm.model_cost)
try:
router = Router(
model_list=[
{
"model_name": "gpt-4o",
"litellm_params": {"model": "azure/my-gpt4o-deployment"},
}
]
)
response = create_model_info_response(
model_id="gpt-4o", provider="openai", llm_router=router
)
finally:
litellm.model_cost.clear()
litellm.model_cost.update(saved_model_cost)
assert response["max_input_tokens"] == 128000
assert response["max_output_tokens"] == 16384
def test_create_model_info_response_resolves_mode_through_deployment_model():
"""`mode` is derived from the same lookup, so an aliased embedding deployment
currently reports no mode at all; it must report `embedding`."""

View file

@ -203,164 +203,6 @@ def test_cost_calculator_with_usage(_local_model_cost_map, monkeypatch):
assert result == expected_cost, f"Got {result}, Expected {expected_cost}"
def test_transcription_cost_uses_token_pricing(_local_model_cost_map):
from litellm import completion_cost
usage = Usage(
prompt_tokens=14,
completion_tokens=45,
total_tokens=59,
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=0, audio_tokens=14),
)
response = TranscriptionResponse(text="demo text")
response.usage = usage
cost = completion_cost(
completion_response=response,
model="gpt-4o-transcribe",
custom_llm_provider="openai",
call_type="atranscription",
)
expected_cost = (14 * 2.5e-06) + (45 * 1e-05)
assert pytest.approx(cost, rel=1e-6) == expected_cost
def test_transcription_token_pricing_is_provider_aware(_local_model_cost_map):
"""Regression: the token-priced transcription path hardcoded provider openai,
so gemini transcription models raised "This model isn't mapped yet"."""
from litellm import completion_cost
usage = Usage(
prompt_tokens=200,
completion_tokens=10,
total_tokens=210,
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=1, audio_tokens=199),
)
response = TranscriptionResponse(text="demo text")
response.usage = usage
cost = completion_cost(
completion_response=response,
model="gemini/gemini-3.5-transcribe",
custom_llm_provider="gemini",
call_type="atranscription",
)
expected_cost = (199 * 2e-06) + (1 * 2e-06) + (10 * 1.2e-05)
assert pytest.approx(cost, rel=1e-6) == expected_cost
def test_transcription_cost_falls_back_to_duration(_local_model_cost_map):
from litellm import completion_cost
response = TranscriptionResponse(text="demo text")
response.duration = 10.0
cost = completion_cost(
completion_response=response,
model="whisper-1",
custom_llm_provider="openai",
call_type="atranscription",
)
expected_cost = 10.0 * 0.0001
assert pytest.approx(cost, rel=1e-6) == expected_cost
def test_vertex_chirp_3_transcription_cost_from_duration(_local_model_cost_map):
"""Regression: the chirp_3 cost map entry shipped with output_cost_per_second 0.0,
and cost_per_second prefers output_cost_per_second whenever it is not None, so
every transcription priced to $0.00 instead of using input_cost_per_second."""
from litellm import completion_cost
response = TranscriptionResponse(text="demo text")
response.duration = 18.0
cost = completion_cost(
completion_response=response,
model="vertex_ai/chirp_3",
custom_llm_provider="vertex_ai",
call_type="atranscription",
)
expected_cost = 18.0 * 0.00026667
assert cost > 0
assert pytest.approx(cost, rel=1e-6) == expected_cost
def test_handle_realtime_stream_cost_calculation():
from litellm.cost_calculator import RealtimeAPITokenUsageProcessor
# Setup test data
results: OpenAIRealtimeStreamList = [
{"type": "session.created", "session": {"model": "gpt-3.5-turbo"}},
{
"type": "response.done",
"response": {"usage": {"input_tokens": 100, "output_tokens": 50, "total_tokens": 150}},
},
{
"type": "response.done",
"response": {
"usage": {
"input_tokens": 200,
"output_tokens": 100,
"total_tokens": 300,
}
},
},
]
combined_usage_object = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(
results=results,
)
# Test with explicit model name
cost = handle_realtime_stream_cost_calculation(
results=results,
combined_usage_object=combined_usage_object,
custom_llm_provider="openai",
litellm_model_name="gpt-3.5-turbo",
)
# Calculate expected cost
# gpt-3.5-turbo costs: $0.0015/1K tokens input, $0.002/1K tokens output
expected_cost = (300 * 0.0015 / 1000) + ( # input tokens (100 + 200)
150 * 0.002 / 1000
) # output tokens (50 + 100)
assert abs(cost - expected_cost) <= 0.00075 # Allow small floating point differences
# Test with different model name in session
results[0]["session"]["model"] = "gpt-4"
cost = handle_realtime_stream_cost_calculation(
results=results,
combined_usage_object=combined_usage_object,
custom_llm_provider="openai",
litellm_model_name="gpt-3.5-turbo",
)
# Calculate expected cost using gpt-4 rates
# gpt-4 costs: $0.03/1K tokens input, $0.06/1K tokens output
expected_cost = (300 * 0.03 / 1000) + ( # input tokens
150 * 0.06 / 1000
) # output tokens
assert abs(cost - expected_cost) < 0.00076
# Test with no response.done events
results = [{"type": "session.created", "session": {"model": "gpt-3.5-turbo"}}]
combined_usage_object = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(
results=results,
)
cost = handle_realtime_stream_cost_calculation(
results=results,
combined_usage_object=combined_usage_object,
custom_llm_provider="openai",
litellm_model_name="gpt-3.5-turbo",
)
assert cost == 0.0 # No usage, no cost
def test_handle_realtime_stream_cost_calculation_stores_cost_breakdown():
"""Regression: realtime cost must populate logging_obj.cost_breakdown so the
spend logs / UI show input vs output cost (issue: cost_breakdown was None for
@ -557,101 +399,6 @@ def test_realtime_logging_object_does_not_validate_unknown_event_types():
assert len(dumped["results"]) == len(results)
def test_realtime_transcription_duration_cost(monkeypatch):
"""
gpt-realtime-whisper transcription sessions are billed by input audio duration
($0.017/min). The .completed events carry usage {type: duration, seconds: N};
cost must equal total_seconds * input_cost_per_second.
"""
from datetime import datetime
from litellm.litellm_core_utils.litellm_logging import Logging
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
from litellm.cost_calculator import RealtimeAPITokenUsageProcessor
results: OpenAIRealtimeStreamList = [
{
"type": "session.created",
"session": {
"type": "transcription",
"audio": {"input": {"transcription": {"model": "gpt-realtime-whisper"}}},
},
},
{
"type": "conversation.item.input_audio_transcription.completed",
"transcript": "hello",
"usage": {"type": "duration", "seconds": 60.0},
},
{
"type": "conversation.item.input_audio_transcription.completed",
"transcript": "world",
"usage": {"type": "duration", "seconds": 30.0},
},
]
combined = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(results=results)
logging_obj = Logging(
model="gpt-realtime-whisper",
messages=[],
stream=False,
call_type="_arealtime",
start_time=datetime.now(),
litellm_call_id="realtime-transcription-cost-breakdown-test",
function_id="realtime-transcription-cost-breakdown-test",
)
cost = handle_realtime_stream_cost_calculation(
results=results,
combined_usage_object=combined,
custom_llm_provider="openai",
litellm_model_name="gpt-realtime-whisper",
litellm_logging_obj=logging_obj,
)
# 90 seconds at $0.017/minute.
expected = 90.0 * (0.017 / 60)
assert abs(cost - expected) < 1e-9
assert cost > 0 # guards against the duration branch being dropped
assert logging_obj.cost_breakdown is not None
assert abs(logging_obj.cost_breakdown["total_cost"] - cost) < 1e-9
# The transcription cost must be attributed in the breakdown, not just folded
# into total_cost, or input_cost + output_cost + additional_costs won't sum to total_cost.
additional_costs = logging_obj.cost_breakdown.get("additional_costs")
assert additional_costs is not None
assert abs(additional_costs["transcription_cost"] - expected) < 1e-9
attributed_total = (
logging_obj.cost_breakdown["input_cost"]
+ logging_obj.cost_breakdown["output_cost"]
+ additional_costs["transcription_cost"]
)
assert abs(attributed_total - logging_obj.cost_breakdown["total_cost"]) < 1e-9
def test_realtime_transcription_duration_cost_resolves_model_from_litellm_name(
monkeypatch,
):
"""When no session event carries the ASR model, the litellm_model_name is used."""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
results: OpenAIRealtimeStreamList = [
{
"type": "conversation.item.input_audio_transcription.completed",
"usage": {"type": "duration", "seconds": 120.0},
},
]
cost = handle_realtime_stream_cost_calculation(
results=results,
combined_usage_object=Usage(),
custom_llm_provider="azure",
litellm_model_name="azure/gpt-realtime-whisper",
)
assert abs(cost - 120.0 * (0.017 / 60)) < 1e-9
def test_realtime_transcription_no_completed_events_is_zero(monkeypatch):
"""A realtime stream without transcription completed events adds no extra cost."""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
@ -673,35 +420,6 @@ def test_realtime_transcription_no_completed_events_is_zero(monkeypatch):
)
def test_realtime_transcription_token_billed_fallback(monkeypatch):
"""
Token-billed transcription models price by audio/text tokens. Verify the
fallback path multiplies audio tokens by the model's audio token cost.
"""
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
from litellm.cost_calculator import _transcription_usage_cost
# gpt-4o-transcribe: input_cost_per_audio_token = 2.5e-06, input_cost_per_token = 2.5e-06,
# output_cost_per_token = 1e-05
model_info = litellm.get_model_info(model="gpt-4o-transcribe", custom_llm_provider="openai")
usage = {
"type": "tokens",
"input_tokens": 40,
"output_tokens": 10,
"total_tokens": 50,
"input_token_details": {"audio_tokens": 30, "text_tokens": 10},
}
cost = _transcription_usage_cost(usage, model_info)
expected = (
30 * 2.5e-06 # audio tokens
+ 10 * 2.5e-06 # text tokens
+ 10 * 1e-05 # output tokens
)
assert abs(cost - expected) < 1e-12
def test_transcription_usage_cost_returns_zero_for_unknown_type():
"""An unrecognized usage type yields 0 (safe fallback, no exception)."""
from litellm.cost_calculator import _transcription_usage_cost
@ -1290,78 +1008,6 @@ def test_bedrock_cost_calculator_comparison_with_without_cache():
print(f"Cost with cache: {cost_with_cache}")
def test_gemini_25_implicit_caching_cost():
"""
Test that Gemini 2.5 models correctly calculate costs with implicit caching.
This test reproduces the issue from #11156 where cached tokens should receive
a 75% discount.
"""
from litellm import completion_cost
from litellm.types.utils import (
Choices,
Message,
ModelResponse,
PromptTokensDetailsWrapper,
Usage,
)
# Create a mock response similar to the one in the issue
litellm_model_response = ModelResponse(
id="test-response",
created=1750733889,
model="gemini/gemini-2.5-flash",
object="chat.completion",
system_fingerprint=None,
choices=[
Choices(
finish_reason="stop",
index=0,
message=Message(
content="Understood. This is a test message to check the response from the Gemini model.",
role="assistant",
tool_calls=None,
function_call=None,
),
)
],
usage=Usage(
total_tokens=15050,
prompt_tokens=15033,
completion_tokens=17,
prompt_tokens_details=PromptTokensDetailsWrapper(
audio_tokens=None,
cached_tokens=14316, # This is cachedContentTokenCount from Gemini
),
completion_tokens_details=None,
),
)
# Calculate the cost
result = completion_cost(
completion_response=litellm_model_response,
model="gemini/gemini-2.5-flash",
)
# Current pricing for gemini/gemini-2.5-flash:
# input: $0.30 / 1M tokens (3e-07 per token)
# cache_read: $0.03 / 1M tokens (3e-08 per token)
# output: $2.50 / 1M tokens (2.5e-06 per token)
# Breakdown:
# - Cached tokens: 14316 * 3e-08 = 0.00042948
# - Non-cached tokens: (15033-14316) * 3e-07 = 717 * 3e-07 = 0.00021510
# - Output tokens: 17 * 2.5e-06 = 0.00004250
# Total: 0.00042948 + 0.00021510 + 0.00004250 = 0.00068708
expected_cost = 0.00068708
# Allow for small floating point differences
assert abs(result - expected_cost) < 1e-8, f"Expected cost {expected_cost}, but got {result}"
print(f"✓ Gemini 2.5 implicit caching cost calculation is correct: ${result:.8f}")
def test_log_context_cost_calculation():
"""
Test that log context cost calculation works correctly with tiered pricing.
@ -3730,31 +3376,6 @@ def test_combine_usage_objects_sums_mirrored_cache_write_fields_once():
assert combined_pair.prompt_tokens_details.cache_creation_tokens == 100
def test_completion_cost_prices_anthropic_shaped_cache_read_tokens(_local_model_cost_map):
"""Regression: an Anthropic /v1/messages response reports cache reads as top-level
cache_read_input_tokens with input_tokens excluding them. Reading that usage as
Responses API usage dropped the cache tokens and billed the whole prompt at the
uncached input rate, overstating spend on cache hits."""
response = {
"id": "msg_1",
"type": "message",
"role": "assistant",
"model": "gpt-5.6-sol",
"stop_reason": "end_turn",
"content": [{"type": "text", "text": "1"}],
"usage": {"input_tokens": 3, "output_tokens": 5, "cache_read_input_tokens": 4014},
}
cost = litellm.completion_cost(
completion_response=response,
model="gpt-5.6-sol",
custom_llm_provider="openai",
)
assert cost == pytest.approx(3 * 4e-6 + 4014 * 4e-7 + 5 * 2e-5, rel=1e-9)
def _together_chat_response(
model: str, prompt_tokens: int, completion_tokens: int, cached_tokens: int
) -> ModelResponse:
@ -3773,60 +3394,6 @@ def _together_chat_response(
)
def test_completion_cost_prices_together_cached_tokens_at_cache_read_rate(_local_model_cost_map):
"""Regression: Together reports prompt_tokens_details.cached_tokens but no together_ai
registry entry carried cache_read_input_token_cost, so cache-hit tokens were priced at
0.0 and spend on cache-heavy workloads was understated."""
cost = completion_cost(
completion_response=_together_chat_response(
model="deepseek-ai/DeepSeek-V4-Flash-0731", prompt_tokens=7864, completion_tokens=16, cached_tokens=7863
),
custom_llm_provider="together_ai",
)
assert cost == pytest.approx(1 * 1.4e-07 + 7863 * 3e-08 + 16 * 2.8e-07, rel=1e-9)
def test_completion_cost_together_mapped_model_skips_size_bucket(_local_model_cost_map):
"""Regression: any together model whose name matches (\\d+b) was rewritten to a
together-ai-* size bucket before the registry lookup, so mapped models like
Muse-Glimmer-30B never used their per-model rates, cache fields included."""
cost = completion_cost(
completion_response=_together_chat_response(
model="meta-models/Muse-Glimmer-30B", prompt_tokens=63, completion_tokens=16, cached_tokens=0
),
custom_llm_provider="together_ai",
)
assert cost == pytest.approx(63 * 3.5e-07 + 16 * 1.5e-06, rel=1e-9)
def test_completion_cost_together_unmapped_model_still_uses_size_bucket(_local_model_cost_map):
cost = completion_cost(
completion_response=_together_chat_response(
model="qwen/Qwen2-72B-Instruct", prompt_tokens=23, completion_tokens=15, cached_tokens=0
),
custom_llm_provider="together_ai",
)
assert cost == pytest.approx((23 + 15) * 9e-07, rel=1e-9)
def test_completion_cost_together_metadata_only_model_still_uses_size_bucket(_local_model_cost_map):
assert "input_cost_per_token" not in litellm.model_cost["together_ai/togethercomputer/CodeLlama-34b-Instruct"]
cost = completion_cost(
completion_response=_together_chat_response(
model="togethercomputer/CodeLlama-34b-Instruct", prompt_tokens=23, completion_tokens=15, cached_tokens=0
),
custom_llm_provider="together_ai",
)
assert cost == pytest.approx((23 + 15) * 8e-07, rel=1e-9)
def test_select_model_name_strips_unregistered_alias_prefix(_local_model_cost_map):
"""A router-facing model_name alias containing "/" whose leading segment is NOT a
registered provider must not be double-prefixed into a non-existent cost key.
@ -4011,31 +3578,6 @@ def test_completion_cost_base_model_ignores_regional_row(_local_model_cost_map):
) == pytest.approx(1000 * flat["input_cost_per_token"])
def test_completion_cost_nonzero_for_slash_alias_model_name(_local_model_cost_map):
"""End-to-end cost through a "/"-containing alias must price above zero (#38069)."""
response = litellm.ModelResponse(
id="x",
choices=[
{
"index": 0,
"message": {"role": "assistant", "content": "hi"},
"finish_reason": "stop",
}
],
model="vertex/claude-opus-5",
)
response._hidden_params = {"custom_llm_provider": "vertex_ai"}
response.usage = litellm.Usage(prompt_tokens=100, completion_tokens=50)
cost = litellm.completion_cost(
completion_response=response,
custom_llm_provider="vertex_ai",
)
assert cost == pytest.approx(100 * 5e-6 + 50 * 2.5e-5, rel=1e-9)
def test_select_model_name_unresolvable_alias_unchanged(_local_model_cost_map):
"""An alias that resolves to no known cost key keeps the legacy double-prefixed name."""
@ -4259,52 +3801,6 @@ def test_explicit_pricing_precedes_private_provider_response_model(
assert selected == expected
def test_handle_realtime_stream_cost_calculation_bills_nested_reasoning_tokens_once(
_local_model_cost_map: None,
) -> None:
"""Realtime response.done nests reasoning_tokens inside text_tokens, so they are billed once."""
results: OpenAIRealtimeStreamList = [
{"type": "session.created", "session": {"model": "gpt-realtime-2.1-mini"}},
{
"type": "response.done",
"response": {
"usage": {
"total_tokens": 260,
"input_tokens": 237,
"output_tokens": 23,
"input_token_details": {
"text_tokens": 43,
"audio_tokens": 0,
"image_tokens": 194,
"cached_tokens": 0,
"cached_tokens_details": {"text_tokens": 0, "audio_tokens": 0, "image_tokens": 0},
},
"output_token_details": {"text_tokens": 23, "audio_tokens": 0, "reasoning_tokens": 18},
}
},
},
]
combined_usage_object = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(
results=results,
)
total_cost = handle_realtime_stream_cost_calculation(
results=results,
combined_usage_object=combined_usage_object,
custom_llm_provider="azure",
litellm_model_name="azure/gpt-realtime-2.1-mini",
)
info = litellm.get_model_info(model="azure/gpt-realtime-2.1-mini", custom_llm_provider="azure")
expected = (
43 * info["input_cost_per_token"]
+ 194 * info["input_cost_per_image_token"]
+ 23 * info["output_cost_per_token"]
)
assert total_cost == pytest.approx(expected)
assert total_cost == pytest.approx(0.0002362)
def test_collect_and_combine_realtime_usage_stores_partitioned_text_tokens() -> None:
"""The combined usage that lands in spend logs keeps reasoning out of text_tokens for every turn."""
results: OpenAIRealtimeStreamList = [

View file

@ -3409,7 +3409,6 @@ def test_a_streamed_response_bills_the_usage_the_provider_reported(local_cost_ma
cost = litellm.completion_cost(completion_response=rebuilt, model=STREAM_COST_MODEL)
assert cost == pytest.approx(_priced_at(137, 42))
assert cost == pytest.approx(0.0007625)
def test_streaming_and_not_streaming_bill_the_same_usage_the_same(local_cost_map):

View file

@ -31,13 +31,6 @@ def test_muse_spark_1_3_routes_to_meta_model_api(model: str):
assert api_base == "https://api.meta.ai/v1"
@pytest.mark.parametrize("model", (MUSE_SPARK_STANDARD, MUSE_SPARK_CONTRIBUTOR))
def test_muse_spark_1_3_web_search_cost_per_query(local_model_cost_map, model: str):
info = litellm.get_model_info(model=model)
assert StandardBuiltInToolCostTracking.get_cost_for_web_search(model_info=info) == WEB_SEARCH_COST_PER_QUERY
@pytest.mark.parametrize("model", (MUSE_SPARK_STANDARD, MUSE_SPARK_CONTRIBUTOR))
def test_muse_spark_1_3_backup_matches_main(model: str):
"""Ensure the bundled model cost map stays in sync with the canonical file."""

View file

@ -91,18 +91,3 @@ TIERED_COST_CASES = [
("gpt-5.6-luna", "priority", 8e-07, 3.6e-06),
("gpt-6-astra", "priority", 4e-05, 0.00015),
]
@pytest.mark.parametrize("model,tier,input_rate,output_rate", TIERED_COST_CASES)
def test_cost_per_token_bills_long_context_at_the_tier_rate(
model: str, tier: str, input_rate: float, output_rate: float
) -> None:
"""A prompt over 272K on flex or priority must bill at that tier's long-context rate."""
input_cost, output_cost = litellm.cost_per_token(
model=model,
prompt_tokens=LONG_CONTEXT_PROMPT_TOKENS,
completion_tokens=COMPLETION_TOKENS,
service_tier=tier,
)
assert input_cost == pytest.approx(LONG_CONTEXT_PROMPT_TOKENS * input_rate)
assert output_cost == pytest.approx(COMPLETION_TOKENS * output_rate)

View file

@ -235,37 +235,6 @@ class TestVideoGeneration:
assert response.status == "completed"
assert response.model == "sora-2"
def test_video_generation_cost_calculation(self):
"""Test video generation cost calculation."""
import json
# Try to load the local model cost map, skip if not found
cost_map_path = "model_prices_and_context_window.json"
if not os.path.exists(cost_map_path):
# Try alternative paths
alt_paths = [
os.path.join(os.path.dirname(__file__), "..", "..", cost_map_path),
os.path.join(
os.path.dirname(__file__), "..", "..", "..", cost_map_path
),
]
for path in alt_paths:
if os.path.exists(path):
cost_map_path = path
break
else:
pytest.skip("model_prices_and_context_window.json not found")
with open(cost_map_path, "r") as f:
litellm.model_cost = json.load(f)
# Test with sora-2 model
cost = default_video_cost_calculator(
model="openai/sora-2", duration_seconds=10.0, custom_llm_provider="openai"
)
# Should calculate cost based on duration (10 seconds * $0.10 per second = $1.00)
assert cost == 1.0
def test_video_generation_cost_calculation_unknown_model(self):
"""Test video generation cost calculation for unknown model."""
@ -502,96 +471,6 @@ class TestVideoGeneration:
)
assert abs(cost - 1.8) < 0.001
def test_completion_cost_video_resolution_tiers_from_cost_map(self, monkeypatch):
"""The 480p/1080p/4k tier keys resolve from the shipped runwayml cost map entries."""
from litellm.cost_calculator import completion_cost
local_map_path = os.path.join(
os.path.dirname(__file__), "..", "..", "model_prices_and_context_window.json"
)
with open(local_map_path, "r") as f:
monkeypatch.setattr(litellm, "model_cost", json.load(f))
def cost_for(model: str, resolution: str | None, duration: float) -> float:
mock_response = MagicMock()
mock_response.usage = {
"duration_seconds": duration,
**({"video_resolution": resolution} if resolution else {}),
}
type(mock_response)._hidden_params = {}
return completion_cost(
completion_response=mock_response,
model=model,
call_type="create_video",
custom_llm_provider="runwayml",
)
assert abs(cost_for("runwayml/seedance2", "4k", 8.0) - 12.0) < 0.001
assert abs(cost_for("runwayml/seedance2", "1080p", 8.0) - 3.2) < 0.001
assert abs(cost_for("runwayml/seedance2", "720p", 8.0) - 2.88) < 0.001
assert abs(cost_for("runwayml/seedance2_5", "480p", 8.0) - 1.6) < 0.001
assert abs(cost_for("runwayml/gen4.5", None, 8.0) - 0.96) < 0.001
def test_completion_cost_xai_imagine_video_720p_tier_from_cost_map(self, monkeypatch):
"""720p xAI Imagine Video requests bill the published 720p rate, not the 480p base rate."""
from litellm.cost_calculator import completion_cost
local_map_path = os.path.join(
os.path.dirname(__file__), "..", "..", "model_prices_and_context_window.json"
)
with open(local_map_path, "r") as f:
monkeypatch.setattr(litellm, "model_cost", json.load(f))
def cost_for(model: str, resolution: str, duration: float) -> float:
mock_response = MagicMock()
mock_response.usage = {"duration_seconds": duration, "video_resolution": resolution}
type(mock_response)._hidden_params = {}
return completion_cost(
completion_response=mock_response,
model=model,
call_type="create_video",
custom_llm_provider="xai",
)
assert abs(cost_for("xai/grok-imagine-video", "720p", 10.0) - 0.7) < 0.001
assert abs(cost_for("xai/grok-imagine-video-1.5", "720p", 10.0) - 1.4) < 0.001
assert abs(cost_for("xai/grok-imagine-video-1.5", "480p", 10.0) - 0.8) < 0.001
assert abs(cost_for("xai/grok-imagine-video-1.5", "1080p", 10.0) - 2.5) < 0.001
def test_completion_cost_veo_31_tiers_pin_published_rates(self, monkeypatch):
"""The gemini and vertex_ai veo 3.1 entries bill Google's published per-second tier rates."""
from litellm.cost_calculator import completion_cost
local_map_path = os.path.join(
os.path.dirname(__file__), "..", "..", "model_prices_and_context_window.json"
)
with open(local_map_path, "r") as f:
monkeypatch.setattr(litellm, "model_cost", json.load(f))
def cost_for(model: str, provider: str, resolution: str | None, duration: float) -> float:
mock_response = MagicMock()
mock_response.usage = {
"duration_seconds": duration,
**({"video_resolution": resolution} if resolution else {}),
}
type(mock_response)._hidden_params = {}
return completion_cost(
completion_response=mock_response,
model=model,
call_type="create_video",
custom_llm_provider=provider,
)
for provider in ("gemini", "vertex_ai"):
for suffix in ("generate-preview", "generate-001"):
standard = f"{provider}/veo-3.1-{suffix}"
fast = f"{provider}/veo-3.1-fast-{suffix}"
assert abs(cost_for(standard, provider, None, 8.0) - 3.2) < 1e-6
assert abs(cost_for(standard, provider, "1080p", 8.0) - 3.2) < 1e-6
assert abs(cost_for(standard, provider, "4k", 8.0) - 4.8) < 1e-6
assert abs(cost_for(fast, provider, "720p", 8.0) - 0.8) < 1e-6
assert abs(cost_for(fast, provider, "1080p", 8.0) - 0.96) < 1e-6
assert abs(cost_for(fast, provider, "4k", 8.0) - 2.4) < 1e-6
def test_video_generation_with_files(self):
"""Test video generation with file uploads."""