mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-06 02:48:13 +00:00
Merge pull request #41443 from BerriAI/litellm_remove_brittle_price_pinning_tests
test: delete unit-test assertions that pin cost-map prices, limits and deprecation dates
This commit is contained in:
commit
f51f01fb54
48 changed files with 0 additions and 3036 deletions
|
|
@ -153,23 +153,12 @@ def test_custom_pricing_as_completion_cost_param():
|
|||
assert round(cost, 5) == round(expected_cost, 5)
|
||||
|
||||
|
||||
def test_get_gpt3_tokens():
|
||||
max_tokens = get_max_tokens("gpt-3.5-turbo")
|
||||
print(max_tokens)
|
||||
assert max_tokens == 4096
|
||||
# print(results)
|
||||
|
||||
|
||||
# test_get_gpt3_tokens()
|
||||
|
||||
|
||||
def test_get_gemini_tokens():
|
||||
# # 🦄🦄🦄🦄🦄🦄🦄🦄
|
||||
max_tokens = get_max_tokens("gemini/gemini-1.5-flash")
|
||||
assert max_tokens == 8192
|
||||
print(max_tokens)
|
||||
|
||||
|
||||
# test_get_palm_tokens()
|
||||
|
||||
|
||||
|
|
@ -273,36 +262,6 @@ def test_cost_azure_gpt_35():
|
|||
# test_cost_azure_gpt_35()
|
||||
|
||||
|
||||
def test_cost_azure_embedding():
|
||||
try:
|
||||
import asyncio
|
||||
|
||||
litellm.set_verbose = True
|
||||
|
||||
async def _test():
|
||||
response = await litellm.aembedding(
|
||||
model="azure/text-embedding-ada-002",
|
||||
input=["good morning from litellm", "gm"],
|
||||
)
|
||||
|
||||
print(response)
|
||||
|
||||
return response
|
||||
|
||||
response = asyncio.run(_test())
|
||||
|
||||
cost = litellm.completion_cost(completion_response=response)
|
||||
|
||||
print("Cost", cost)
|
||||
expected_cost = float("7e-07")
|
||||
assert cost == expected_cost
|
||||
|
||||
except Exception as e:
|
||||
pytest.fail(
|
||||
f"Cost Calc failed for azure/gpt-3.5-turbo. Expected {expected_cost}, Calculated cost {cost}"
|
||||
)
|
||||
|
||||
|
||||
# test_cost_azure_embedding()
|
||||
|
||||
|
||||
|
|
@ -639,56 +598,6 @@ def test_vertex_ai_medlm_completion_cost():
|
|||
assert predictive_cost > 0
|
||||
|
||||
|
||||
def test_vertex_ai_claude_completion_cost():
|
||||
from litellm import Choices, Message, ModelResponse
|
||||
from litellm.utils import Usage
|
||||
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
||||
litellm.set_verbose = True
|
||||
input_tokens = litellm.token_counter(
|
||||
model="vertex_ai/claude-3-sonnet@20240229",
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}],
|
||||
)
|
||||
print(f"input_tokens: {input_tokens}")
|
||||
output_tokens = litellm.token_counter(
|
||||
model="vertex_ai/claude-3-sonnet@20240229",
|
||||
text="It's all going well",
|
||||
count_response_tokens=True,
|
||||
)
|
||||
print(f"output_tokens: {output_tokens}")
|
||||
response = ModelResponse(
|
||||
id="chatcmpl-e41836bb-bb8b-4df2-8e70-8f3e160155ac",
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason=None,
|
||||
index=0,
|
||||
message=Message(
|
||||
content="It's all going well",
|
||||
role="assistant",
|
||||
),
|
||||
)
|
||||
],
|
||||
created=1700775391,
|
||||
model="claude-3-sonnet",
|
||||
object="chat.completion",
|
||||
system_fingerprint=None,
|
||||
usage=Usage(
|
||||
prompt_tokens=input_tokens,
|
||||
completion_tokens=output_tokens,
|
||||
total_tokens=input_tokens + output_tokens,
|
||||
),
|
||||
)
|
||||
cost = litellm.completion_cost(
|
||||
model="vertex_ai/claude-3-sonnet",
|
||||
completion_response=response,
|
||||
messages=[{"role": "user", "content": "Hey, how's it going?"}],
|
||||
)
|
||||
predicted_cost = input_tokens * 0.000003 + 0.000015 * output_tokens
|
||||
assert cost == predicted_cost
|
||||
|
||||
|
||||
def test_vertex_ai_embedding_completion_cost(caplog):
|
||||
"""
|
||||
Relevant issue - https://github.com/BerriAI/litellm/issues/4630
|
||||
|
|
@ -1212,105 +1121,6 @@ def test_completion_cost_fireworks_ai(model):
|
|||
assert cost > 0
|
||||
|
||||
|
||||
def test_cost_azure_openai_prompt_caching():
|
||||
from litellm.utils import Choices, Message, ModelResponse, Usage
|
||||
from litellm.types.utils import (
|
||||
PromptTokensDetailsWrapper,
|
||||
CompletionTokensDetailsWrapper,
|
||||
)
|
||||
from litellm import get_model_info
|
||||
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
||||
model = "azure/o1-mini"
|
||||
|
||||
## LLM API CALL ## (MORE EXPENSIVE)
|
||||
response_1 = ModelResponse(
|
||||
id="chatcmpl-3f427194-0840-4d08-b571-56bfe38a5424",
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason="length",
|
||||
index=0,
|
||||
message=Message(
|
||||
content="Hello! I'm doing well, thank you for",
|
||||
role="assistant",
|
||||
tool_calls=None,
|
||||
function_call=None,
|
||||
),
|
||||
)
|
||||
],
|
||||
created=1725036547,
|
||||
model=model,
|
||||
object="chat.completion",
|
||||
system_fingerprint=None,
|
||||
usage=Usage(
|
||||
completion_tokens=10,
|
||||
prompt_tokens=14,
|
||||
total_tokens=24,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
reasoning_tokens=2
|
||||
),
|
||||
),
|
||||
)
|
||||
|
||||
## PROMPT CACHE HIT ## (LESS EXPENSIVE)
|
||||
response_2 = ModelResponse(
|
||||
id="chatcmpl-3f427194-0840-4d08-b571-56bfe38a5424",
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason="length",
|
||||
index=0,
|
||||
message=Message(
|
||||
content="Hello! I'm doing well, thank you for",
|
||||
role="assistant",
|
||||
tool_calls=None,
|
||||
function_call=None,
|
||||
),
|
||||
)
|
||||
],
|
||||
created=1725036547,
|
||||
model=model,
|
||||
object="chat.completion",
|
||||
system_fingerprint=None,
|
||||
usage=Usage(
|
||||
completion_tokens=10,
|
||||
prompt_tokens=0,
|
||||
total_tokens=10,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
cached_tokens=14,
|
||||
),
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
reasoning_tokens=2
|
||||
),
|
||||
),
|
||||
)
|
||||
|
||||
cost_1 = completion_cost(model=model, completion_response=response_1)
|
||||
cost_2 = completion_cost(model=model, completion_response=response_2)
|
||||
assert cost_1 > cost_2
|
||||
|
||||
model_info = get_model_info(model=model, custom_llm_provider="azure")
|
||||
usage = response_2.usage
|
||||
|
||||
_expected_cost2 = (
|
||||
(usage.prompt_tokens - usage.prompt_tokens_details.cached_tokens)
|
||||
* model_info["input_cost_per_token"]
|
||||
+ (usage.completion_tokens * model_info["output_cost_per_token"])
|
||||
+ (
|
||||
usage.prompt_tokens_details.cached_tokens
|
||||
* model_info["cache_read_input_token_cost"]
|
||||
)
|
||||
)
|
||||
|
||||
print("_expected_cost2", _expected_cost2)
|
||||
print("cost_2", cost_2)
|
||||
|
||||
assert (
|
||||
abs(cost_2 - _expected_cost2) < 1e-5
|
||||
) # Allow for small floating-point differences
|
||||
|
||||
|
||||
def test_completion_cost_vertex_llama3():
|
||||
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
|
|
|||
|
|
@ -1670,8 +1670,6 @@ async def test_handle_completed_bedrock_batch_prices_from_deployment_model(monke
|
|||
)
|
||||
|
||||
assert (result.usage.prompt_tokens, result.usage.completion_tokens, result.usage.total_tokens) == (1800, 1000, 2800)
|
||||
# 3e-06 / 1.5e-05 on-demand, halved for batch.
|
||||
assert result.cost == pytest.approx(1800 * 3e-06 / 2 + 1000 * 1.5e-05 / 2)
|
||||
|
||||
# The response model alone cannot price a bedrock batch: this is the $0 bug.
|
||||
zero_result = await bu._handle_completed_batch(
|
||||
|
|
|
|||
|
|
@ -377,8 +377,6 @@ class TestOpenAIContainerTransformation:
|
|||
in container._hidden_params["additional_headers"]
|
||||
)
|
||||
|
||||
# Verify the cost matches expected value for OpenAI code interpreter (1 session)
|
||||
# OpenAI charges $0.03 per code interpreter session
|
||||
expected_cost = StandardBuiltInToolCostTracking.get_cost_for_code_interpreter(
|
||||
sessions=1, provider="openai"
|
||||
)
|
||||
|
|
@ -387,4 +385,3 @@ class TestOpenAIContainerTransformation:
|
|||
]
|
||||
|
||||
assert actual_cost == expected_cost
|
||||
assert actual_cost == 0.03 # OpenAI code interpreter costs $0.03 per session
|
||||
|
|
|
|||
|
|
@ -90,15 +90,6 @@ class TestAzureAssistantCostTracking:
|
|||
)
|
||||
assert cost == 0.0, "Should return 0 for zero sessions"
|
||||
|
||||
def test_openai_code_interpreter_free(self):
|
||||
"""Test OpenAI code interpreter cost from model cost map."""
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_code_interpreter(
|
||||
sessions=5,
|
||||
provider="openai",
|
||||
)
|
||||
assert (
|
||||
cost == 0.15
|
||||
), "OpenAI code interpreter should return 0.15 based on current implementation"
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"input_tokens,output_tokens,expected_cost",
|
||||
|
|
@ -222,14 +213,3 @@ class TestAzureAssistantCostTracking:
|
|||
)
|
||||
assert StandardBuiltInToolCostTracking.get_cost_for_vector_store(None) == 0.0
|
||||
|
||||
def test_constants_loaded_correctly(self):
|
||||
"""Test that Azure pricing constants are loaded with expected values."""
|
||||
assert AZURE_FILE_SEARCH_COST_PER_GB_PER_DAY == 0.1
|
||||
|
||||
# Code interpreter cost is now in model cost map
|
||||
azure_container_info = litellm.model_cost.get("azure/container", {})
|
||||
assert azure_container_info.get("code_interpreter_cost_per_session") == 0.03
|
||||
|
||||
assert AZURE_COMPUTER_USE_INPUT_COST_PER_1K_TOKENS == 3.0
|
||||
assert AZURE_COMPUTER_USE_OUTPUT_COST_PER_1K_TOKENS == 12.0
|
||||
assert AZURE_VECTOR_STORE_COST_PER_GB_PER_DAY == 0.1
|
||||
|
|
|
|||
|
|
@ -1685,35 +1685,6 @@ def test_azure_gpt55_reasoning_effort_flags_match_live_openai_api(
|
|||
assert m.get("supports_xhigh_reasoning_effort") is expected_xhigh
|
||||
|
||||
|
||||
def test_generic_cost_per_token_anthropic_prompt_caching_with_cache_creation():
|
||||
model = "claude-haiku-4-5-20251001"
|
||||
usage = Usage(
|
||||
completion_tokens=90,
|
||||
prompt_tokens=28436,
|
||||
total_tokens=28526,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
accepted_prediction_tokens=None,
|
||||
audio_tokens=None,
|
||||
reasoning_tokens=0,
|
||||
rejected_prediction_tokens=None,
|
||||
text_tokens=None,
|
||||
),
|
||||
prompt_tokens_details=None,
|
||||
cache_creation_input_tokens=2000,
|
||||
)
|
||||
|
||||
custom_llm_provider = "anthropic"
|
||||
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model=model,
|
||||
usage=usage,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
print(f"prompt_cost: {prompt_cost}")
|
||||
assert round(prompt_cost, 3) == 0.029
|
||||
|
||||
|
||||
def test_string_cost_values():
|
||||
"""Test that cost values defined as strings are properly converted to floats."""
|
||||
from unittest.mock import patch
|
||||
|
|
@ -2350,140 +2321,6 @@ def test_gemini_image_generation_cost_falls_back_to_flat_image_pricing(_local_mo
|
|||
assert round(cost, 10) == round(expected_cost, 10)
|
||||
|
||||
|
||||
def test_bedrock_anthropic_prompt_caching():
|
||||
"""Test Bedrock Anthropic models with prompt caching return correct costs."""
|
||||
model = "us.anthropic.claude-sonnet-4-5-20250929-v1:0"
|
||||
usage = Usage(
|
||||
prompt_tokens=52123,
|
||||
completion_tokens=497,
|
||||
total_tokens=52620,
|
||||
cache_creation_input_tokens=7183,
|
||||
cache_read_input_tokens=22465,
|
||||
)
|
||||
|
||||
custom_llm_provider = "bedrock"
|
||||
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model=model,
|
||||
usage=usage,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
assert prompt_cost >= 0
|
||||
assert completion_cost >= 0
|
||||
assert round(prompt_cost, 3) == 0.111
|
||||
assert round(completion_cost, 5) == 0.00820
|
||||
|
||||
|
||||
def test_reasoning_tokens_without_text_tokens_gpt5_nano():
|
||||
"""
|
||||
Test fix for GitHub issue #18599:
|
||||
https://github.com/BerriAI/litellm/issues/18599
|
||||
|
||||
When OpenAI models (gpt-5-nano, o1, o3) return reasoning_tokens but don't provide
|
||||
text_tokens, LiteLLM should calculate text_tokens as:
|
||||
text_tokens = completion_tokens - reasoning_tokens - audio_tokens - image_tokens
|
||||
|
||||
This ensures ALL completion tokens are billed, not just reasoning tokens.
|
||||
"""
|
||||
model = "gpt-5-nano"
|
||||
custom_llm_provider = "openai"
|
||||
|
||||
# Simulate OpenAI gpt-5-nano response where text_tokens is NOT provided
|
||||
# completion_tokens: 977 total
|
||||
# reasoning_tokens: 768
|
||||
# text_tokens: should be calculated as 977 - 768 = 209
|
||||
usage = Usage(
|
||||
prompt_tokens=17,
|
||||
completion_tokens=977,
|
||||
total_tokens=994,
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(
|
||||
reasoning_tokens=768,
|
||||
audio_tokens=0,
|
||||
# text_tokens NOT provided - this is the key part of the bug
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model=model,
|
||||
usage=usage,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
)
|
||||
|
||||
# gpt-5-nano pricing: $0.05/1M input, $0.40/1M output
|
||||
expected_prompt_cost = 17 * 0.05 / 1_000_000
|
||||
expected_completion_cost = 977 * 0.40 / 1_000_000 # ALL tokens, not just reasoning
|
||||
|
||||
assert abs(prompt_cost - expected_prompt_cost) < 1e-10, (
|
||||
f"Prompt cost incorrect: {prompt_cost} vs {expected_prompt_cost}"
|
||||
)
|
||||
|
||||
assert abs(completion_cost - expected_completion_cost) < 1e-10, (
|
||||
f"Completion cost incorrect: {completion_cost} vs {expected_completion_cost}"
|
||||
)
|
||||
|
||||
# Verify it's NOT using only reasoning_tokens (the bug)
|
||||
wrong_cost = 768 * 0.40 / 1_000_000 # Only reasoning tokens
|
||||
assert abs(completion_cost - wrong_cost) > 1e-6, (
|
||||
"Bug detected: Cost calculation is using only reasoning_tokens instead of all completion_tokens!"
|
||||
)
|
||||
|
||||
|
||||
def test_image_count_prevents_text_tokens_fallback(_local_model_cost_map):
|
||||
"""
|
||||
Test that the text_tokens fallback in generic_cost_per_token does not
|
||||
override text_tokens=0 when image_count > 0.
|
||||
|
||||
Regression test for: Bedrock image embedding double-charging bug.
|
||||
When image_count > 0, text_tokens=0 is intentional (image-only request),
|
||||
not "text_tokens not set by provider."
|
||||
"""
|
||||
|
||||
# Simulate Nova image-only embedding: prompt_tokens estimated from
|
||||
# embedding dimensions (768 for 3072-dim), image_count=1
|
||||
usage = Usage(
|
||||
prompt_tokens=768,
|
||||
completion_tokens=0,
|
||||
total_tokens=768,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
image_count=1,
|
||||
),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model="amazon.nova-2-multimodal-embeddings-v1:0",
|
||||
usage=usage,
|
||||
custom_llm_provider="bedrock",
|
||||
)
|
||||
|
||||
# Cost should be 1 * input_cost_per_image ($6e-05) = $0.00006
|
||||
# NOT 768 * input_cost_per_token ($1.35e-07) + $0.00006 = $0.000164
|
||||
expected_image_cost = 1 * 6e-05
|
||||
assert prompt_cost == expected_image_cost, (
|
||||
f"Expected prompt_cost={expected_image_cost} (image-only), "
|
||||
f"got {prompt_cost}. text_tokens fallback may be double-charging."
|
||||
)
|
||||
assert completion_cost == 0.0
|
||||
|
||||
|
||||
def test_query_count_bills_input_cost_per_query(_local_model_cost_map):
|
||||
usage = Usage(
|
||||
prompt_tokens=0,
|
||||
completion_tokens=0,
|
||||
total_tokens=0,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(query_count=3, image_count=1),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model="us.twelvelabs.marengo-embed-3-0-v1:0",
|
||||
usage=usage,
|
||||
custom_llm_provider="bedrock",
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(3 * 7e-05 + 1e-04)
|
||||
assert completion_cost == 0.0
|
||||
|
||||
|
||||
def test_query_count_is_free_without_a_per_query_price(_local_model_cost_map):
|
||||
usage = Usage(
|
||||
prompt_tokens=0,
|
||||
|
|
@ -2692,36 +2529,6 @@ def test_vertex_uplift_invalid_multiplier_defaults_to_one():
|
|||
)
|
||||
|
||||
|
||||
def test_priority_service_tier_above_threshold_uses_priority_tier_rates_for_cached_tokens(
|
||||
_local_model_cost_map,
|
||||
):
|
||||
"""Regression: for a model that publishes both service_tier and above_threshold rate
|
||||
variants, a priority request over the threshold must bill cached tokens at
|
||||
cache_read_input_token_cost_above_200k_tokens_priority (and analogously for
|
||||
input/output above-threshold), not the standard above-threshold rate."""
|
||||
usage = Usage(
|
||||
prompt_tokens=250_000,
|
||||
completion_tokens=1_000,
|
||||
total_tokens=251_000,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=200_000, text_tokens=50_000),
|
||||
completion_tokens_details=CompletionTokensDetailsWrapper(text_tokens=1_000),
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = generic_cost_per_token(
|
||||
model="gemini-3-pro-preview",
|
||||
usage=usage,
|
||||
custom_llm_provider="gemini",
|
||||
service_tier="priority",
|
||||
)
|
||||
|
||||
# gemini-3-pro-preview priority + above_200k rates from the pricing JSON:
|
||||
# input 7.2e-6, output 3.24e-5, cache_read 7.2e-7
|
||||
expected_prompt = 50_000 * 7.2e-6 + 200_000 * 7.2e-7
|
||||
expected_completion = 1_000 * 3.24e-5
|
||||
assert prompt_cost == pytest.approx(expected_prompt, rel=1e-9)
|
||||
assert completion_cost == pytest.approx(expected_completion, rel=1e-9)
|
||||
|
||||
|
||||
def test_service_tier_suffixes_constant_in_sync_with_enum():
|
||||
from litellm.litellm_core_utils.llm_cost_calc.utils import _SERVICE_TIER_SUFFIXES
|
||||
from litellm.types.utils import ServiceTier
|
||||
|
|
@ -3614,28 +3421,6 @@ def test_gemini_38_flash_matches_37_flash_promotional_pricing(prefix, _local_mod
|
|||
assert new_model[field] == old_model[field], field
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("model", "provider", "image_token_rate"),
|
||||
[
|
||||
("gpt-realtime-2.1", "openai", 5e-06),
|
||||
("gpt-realtime-2.1-mini", "openai", 8e-07),
|
||||
("azure/gpt-realtime-2.1", "azure", 5e-06),
|
||||
("azure/gpt-realtime-2.1-mini", "azure", 8e-07),
|
||||
],
|
||||
)
|
||||
def test_realtime_image_tokens_priced_per_token(model, provider, image_token_rate, _local_model_cost_map):
|
||||
"""Realtime image input is billed per 1M image tokens, not per image."""
|
||||
usage = Usage(
|
||||
prompt_tokens=1_100,
|
||||
completion_tokens=0,
|
||||
total_tokens=1_100,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=100, image_tokens=1_000),
|
||||
)
|
||||
prompt_cost, _ = generic_cost_per_token(model=model, usage=usage, custom_llm_provider=provider)
|
||||
text_rate = litellm.model_cost[model]["input_cost_per_token"]
|
||||
assert prompt_cost == pytest.approx(100 * text_rate + 1_000 * image_token_rate)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("response_quality", "requested_quality", "expected_cost"),
|
||||
[
|
||||
|
|
@ -3830,28 +3615,6 @@ def test_cached_audio_tokens_fall_back_to_cache_read_input_token_cost() -> None:
|
|||
assert prompt_cost == pytest.approx(expected)
|
||||
|
||||
|
||||
def test_cache_read_breakdown_splits_cached_audio_at_the_audio_cache_rate(_local_model_cost_map: None) -> None:
|
||||
usage = Usage(
|
||||
prompt_tokens=4863,
|
||||
completion_tokens=1087,
|
||||
total_tokens=5950,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=1693,
|
||||
audio_tokens=3170,
|
||||
cached_tokens=2816,
|
||||
cached_tokens_details={"text_tokens": 896, "audio_tokens": 1920},
|
||||
),
|
||||
)
|
||||
|
||||
breakdown = get_token_type_cost_breakdown(model="gpt-realtime-2.1-mini", custom_llm_provider="openai", usage=usage)
|
||||
prompt_cost, _ = generic_cost_per_token(model="gpt-realtime-2.1-mini", usage=usage, custom_llm_provider="openai")
|
||||
|
||||
assert breakdown.cache_read_cost == pytest.approx(896 * 6e-8 + 1920 * 3e-7)
|
||||
assert breakdown.rates is not None
|
||||
assert breakdown.rates.cache_read_input_audio_token_cost == pytest.approx(3e-7)
|
||||
assert prompt_cost == pytest.approx((1693 - 896) * 6e-7 + (3170 - 1920) * 1e-5 + breakdown.cache_read_cost)
|
||||
|
||||
|
||||
def test_generic_cost_per_token_bills_cache_creation_at_the_input_rate_without_a_write_price():
|
||||
"""Azure and OpenAI publish no cache-write price and bill cache writes as ordinary input.
|
||||
A deployment priced with only input, output, and cache-read rates must bill the creation
|
||||
|
|
|
|||
|
|
@ -309,102 +309,6 @@ def test_get_cost_for_gemini_web_search(model):
|
|||
assert cost > 0.0
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,custom_llm_provider",
|
||||
[
|
||||
("vertex_ai/gemini-2.5-flash", "vertex_ai"),
|
||||
("gemini-2.5-flash", "vertex_ai"),
|
||||
],
|
||||
)
|
||||
def test_get_cost_for_vertex_ai_gemini_web_search(model, custom_llm_provider):
|
||||
"""
|
||||
Test that Vertex AI Gemini web search costs are tracked when passing
|
||||
a ModelResponse with usage.prompt_tokens_details.web_search_requests.
|
||||
|
||||
This tests the fix for: https://github.com/BerriAI/litellm/issues/XXXXX
|
||||
|
||||
The issue: When a ModelResponse is passed, the detection logic only checks
|
||||
for url_citation annotations, not usage.prompt_tokens_details.web_search_requests.
|
||||
This causes Vertex AI grounding costs to not be tracked.
|
||||
"""
|
||||
from litellm.types.utils import Choices, Message, PromptTokensDetailsWrapper, Usage
|
||||
|
||||
# Create a realistic ModelResponse like what Vertex AI returns
|
||||
response = ModelResponse(
|
||||
id="test-id",
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason="stop",
|
||||
index=0,
|
||||
message=Message(
|
||||
content="Test response with grounding", role="assistant"
|
||||
),
|
||||
)
|
||||
],
|
||||
created=1234567890,
|
||||
model=model,
|
||||
object="chat.completion",
|
||||
system_fingerprint=None,
|
||||
)
|
||||
|
||||
# Add usage with web_search_requests (how Vertex AI indicates grounding was used)
|
||||
usage = Usage(
|
||||
prompt_tokens=11,
|
||||
completion_tokens=100,
|
||||
total_tokens=111,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=11, web_search_requests=1 # This should trigger grounding cost
|
||||
),
|
||||
)
|
||||
response.usage = usage
|
||||
|
||||
# Calculate cost - should include grounding cost
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
model=model,
|
||||
usage=usage,
|
||||
response_object=response, # Pass the ModelResponse
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
standard_built_in_tools_params=None,
|
||||
)
|
||||
|
||||
# Vertex AI charges $0.035 per grounded request
|
||||
assert cost == 0.035, f"Expected $0.035 grounding cost, got ${cost}"
|
||||
|
||||
|
||||
def test_azure_assistant_features_integrated_cost_tracking(monkeypatch):
|
||||
"""
|
||||
Test integrated cost tracking for Azure assistant features.
|
||||
"""
|
||||
# Force use of local model cost map for CI/CD consistency
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
litellm.model_cost = litellm.get_model_cost_map(url="")
|
||||
|
||||
model = "azure/gpt-4o"
|
||||
|
||||
# Test with multiple Azure assistant features
|
||||
standard_built_in_tools_params = StandardBuiltInToolsParams(
|
||||
vector_store_usage={"storage_gb": 1.0, "days": 10},
|
||||
computer_use_usage={"input_tokens": 1000, "output_tokens": 500},
|
||||
code_interpreter_sessions=2,
|
||||
)
|
||||
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
model=model,
|
||||
response_object=None,
|
||||
usage=None,
|
||||
custom_llm_provider="azure",
|
||||
standard_built_in_tools_params=standard_built_in_tools_params,
|
||||
)
|
||||
|
||||
# Should calculate costs for:
|
||||
# - Vector store: 1.0 * 10 * 0.1 = $1.00
|
||||
# - Computer use: (1000/1000 * 3.0) + (500/1000 * 12.0) = $9.00
|
||||
# - Code interpreter: 2 * 0.03 = $0.06
|
||||
# Total: $10.06
|
||||
expected_cost = 1.0 + 9.0 + 0.06
|
||||
assert abs(cost - expected_cost) < 0.01, f"Expected ~{expected_cost}, got {cost}"
|
||||
|
||||
|
||||
def test_completion_cost_includes_web_search_without_standard_built_in_tools_params():
|
||||
"""
|
||||
Test that completion_cost includes web search cost even when
|
||||
|
|
@ -510,68 +414,6 @@ def test_gemini_3x_web_search_billed_per_query(model, local_model_cost_map):
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,custom_llm_provider",
|
||||
[
|
||||
("gemini/gemini-2.5-flash", "gemini"),
|
||||
("vertex_ai/gemini-2.5-flash", "vertex_ai"),
|
||||
],
|
||||
)
|
||||
def test_gemini_2x_maps_grounding_billed_at_maps_rate(model, custom_llm_provider, local_model_cost_map):
|
||||
"""
|
||||
Grounding with Google Maps is its own SKU: a Maps-only grounded prompt on Gemini 2.x bills the
|
||||
$0.025 Maps per-prompt fee, not the $0.035 Google Search fee it was previously conflated with,
|
||||
and not $0 as on Vertex AI where webSearchQueries is never populated for Maps.
|
||||
Regression for https://github.com/BerriAI/litellm/issues/35906
|
||||
"""
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
||||
model_info = litellm.get_model_info(model)
|
||||
expected_cost = model_info["google_maps_grounding_cost_per_query"]
|
||||
assert expected_cost == pytest.approx(0.025)
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=15,
|
||||
completion_tokens=100,
|
||||
total_tokens=115,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=15, google_maps_grounding_requests=1),
|
||||
)
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
model=model,
|
||||
usage=usage,
|
||||
response_object=None,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
standard_built_in_tools_params=None,
|
||||
)
|
||||
assert cost == pytest.approx(expected_cost)
|
||||
|
||||
|
||||
def test_gemini_3x_maps_grounding_billed_per_query(local_model_cost_map):
|
||||
"""Gemini 3.x bills Maps grounding per executed query: N queries cost N * $0.014."""
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
||||
model = "vertex_ai/gemini-3.5-flash"
|
||||
model_info = litellm.get_model_info(model)
|
||||
assert model_info["web_search_billing_unit"] == "per_query"
|
||||
expected_cost = model_info["google_maps_grounding_cost_per_query"] * 2
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=15,
|
||||
completion_tokens=100,
|
||||
total_tokens=115,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=15, google_maps_grounding_requests=2),
|
||||
)
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
model=model,
|
||||
usage=usage,
|
||||
response_object=None,
|
||||
custom_llm_provider="vertex_ai",
|
||||
standard_built_in_tools_params=None,
|
||||
)
|
||||
assert cost == pytest.approx(expected_cost)
|
||||
assert cost == pytest.approx(0.028)
|
||||
|
||||
|
||||
def test_gemini_combined_search_and_maps_costs_are_additive(local_model_cost_map):
|
||||
"""A prompt grounded with both Google Search and Google Maps pays both fees."""
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
|
@ -708,35 +550,6 @@ def _openai_responses_with_web_search_calls(model, num_calls):
|
|||
)
|
||||
|
||||
|
||||
def test_openai_responses_web_search_priced_per_call(local_model_cost_map):
|
||||
"""
|
||||
Regression for LIT-5013 bug 1: OpenAI reasoning models (gpt-5 family, o-series, deep-research)
|
||||
carry supports_web_search but had no search_context_cost_per_query, so get_cost_for_web_search_request
|
||||
(no openai branch) returned None and the default fallback billed web search as $0. gpt-5-nano now
|
||||
prices at $0.01 per call, and two web_search_call items in the Responses output must bill 2 x $0.01.
|
||||
"""
|
||||
from litellm.types.utils import Usage
|
||||
|
||||
model = "gpt-5-nano"
|
||||
per_call = litellm.get_model_info(model)["search_context_cost_per_query"][
|
||||
"search_context_size_medium"
|
||||
]
|
||||
assert per_call == 0.01
|
||||
|
||||
response = _openai_responses_with_web_search_calls(model, num_calls=2)
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
model=model,
|
||||
response_object=response,
|
||||
usage=Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15),
|
||||
custom_llm_provider="openai",
|
||||
standard_built_in_tools_params=None,
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(2 * per_call), (
|
||||
f"gpt-5-nano web search must bill 2 x ${per_call}, got ${cost}"
|
||||
)
|
||||
|
||||
|
||||
def test_openai_responses_web_search_multiplied_by_call_count(local_model_cost_map):
|
||||
"""
|
||||
Regression for LIT-5013 bug 2: web_search_call detection was binary, so a Responses output with
|
||||
|
|
@ -808,88 +621,6 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map):
|
|||
)
|
||||
|
||||
|
||||
def test_dated_search_preview_entries_carry_search_pricing(local_model_cost_map):
|
||||
"""
|
||||
Regression for the live QA finding: OpenAI resolves gpt-4o-search-preview requests to the
|
||||
dated id gpt-4o-search-preview-2025-03-11, whose cost map entry lacked
|
||||
search_context_cost_per_query, so the default chat path silently billed the $0.035 search
|
||||
fee as $0. Dated entries must price identically to their undated siblings.
|
||||
"""
|
||||
from litellm.types.utils import Usage
|
||||
|
||||
for dated, undated in (
|
||||
("gpt-4o-search-preview-2025-03-11", "gpt-4o-search-preview"),
|
||||
("gpt-4o-mini-search-preview-2025-03-11", "gpt-4o-mini-search-preview"),
|
||||
):
|
||||
assert (
|
||||
litellm.get_model_info(dated)["search_context_cost_per_query"]
|
||||
== litellm.get_model_info(undated)["search_context_cost_per_query"]
|
||||
)
|
||||
|
||||
response = ModelResponse(
|
||||
model="gpt-4o-search-preview-2025-03-11",
|
||||
choices=[
|
||||
{
|
||||
"index": 0,
|
||||
"finish_reason": "stop",
|
||||
"message": {
|
||||
"role": "assistant",
|
||||
"content": "headlines",
|
||||
"annotations": [
|
||||
{
|
||||
"type": "url_citation",
|
||||
"url_citation": {
|
||||
"url": "https://example.com",
|
||||
"title": "t",
|
||||
"start_index": 0,
|
||||
"end_index": 1,
|
||||
},
|
||||
}
|
||||
],
|
||||
},
|
||||
}
|
||||
],
|
||||
)
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
model="gpt-4o-search-preview-2025-03-11",
|
||||
response_object=response,
|
||||
usage=Usage(prompt_tokens=14, completion_tokens=825, total_tokens=839),
|
||||
custom_llm_provider="openai",
|
||||
standard_built_in_tools_params=None,
|
||||
)
|
||||
assert cost == pytest.approx(0.025), (
|
||||
f"dated search-preview id must bill the $0.025 search fee, got ${cost}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"web_search_options",
|
||||
[
|
||||
None,
|
||||
WebSearchOptions(search_context_size="low"),
|
||||
WebSearchOptions(search_context_size="medium"),
|
||||
WebSearchOptions(search_context_size="high"),
|
||||
],
|
||||
)
|
||||
def test_gpt_4o_mini_snapshot_bills_web_search_like_its_alias(
|
||||
web_search_options: WebSearchOptions | None, local_model_cost_map: None
|
||||
) -> None:
|
||||
alias_info = litellm.get_model_info("gpt-4o-mini")
|
||||
snapshot_info = litellm.get_model_info("gpt-4o-mini-2024-07-18")
|
||||
|
||||
assert not snapshot_info["supports_web_search"]
|
||||
assert not alias_info["supports_web_search"]
|
||||
|
||||
snapshot_cost = StandardBuiltInToolCostTracking.get_cost_for_web_search(
|
||||
web_search_options=web_search_options, model_info=snapshot_info
|
||||
)
|
||||
alias_cost = StandardBuiltInToolCostTracking.get_cost_for_web_search(
|
||||
web_search_options=web_search_options, model_info=alias_info
|
||||
)
|
||||
|
||||
assert snapshot_cost == alias_cost == 0.025
|
||||
|
||||
|
||||
# Note: File search integration test removed due to complex annotation detection logic
|
||||
# The unit tests in test_azure_assistant_cost_tracking.py provide comprehensive coverage
|
||||
|
||||
|
|
@ -999,81 +730,3 @@ def _web_search_cost(model: str, response: ResponsesAPIResponse, custom_llm_prov
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", _BEDROCK_MANTLE_WEB_SEARCH_MODELS)
|
||||
def test_bedrock_mantle_web_search_billed_per_query(local_model_cost_map, model):
|
||||
"""Two Bedrock-reported web searches bill 2 x $0.012 under the prefixed and the bare model id alike."""
|
||||
pricing = litellm.get_model_info(model)["search_context_cost_per_query"]
|
||||
assert pricing == {
|
||||
"search_context_size_low": _BEDROCK_MANTLE_WEB_SEARCH_RATE,
|
||||
"search_context_size_medium": _BEDROCK_MANTLE_WEB_SEARCH_RATE,
|
||||
"search_context_size_high": _BEDROCK_MANTLE_WEB_SEARCH_RATE,
|
||||
}
|
||||
|
||||
response = _responses_with_web_search(
|
||||
model,
|
||||
actions=[{"type": "search", "query": "litellm"}, {"type": "search", "query": "bedrock web search"}],
|
||||
tool_usage={"web_search": {"num_requests": 2}},
|
||||
)
|
||||
for cost_model in (model, model.split("/", 1)[1]):
|
||||
cost = _web_search_cost(cost_model, response, "bedrock_mantle")
|
||||
assert cost == pytest.approx(2 * _BEDROCK_MANTLE_WEB_SEARCH_RATE), (
|
||||
f"{cost_model} must bill 2 x ${_BEDROCK_MANTLE_WEB_SEARCH_RATE} for 2 web searches, got ${cost}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("num_requests", [1, 0])
|
||||
def test_web_search_call_count_prefers_provider_reported_num_requests(local_model_cost_map, num_requests):
|
||||
"""A search plus an open_page fetch bills tool_usage.web_search.num_requests, never the two items."""
|
||||
model = "bedrock_mantle/openai.gpt-5.6-sol"
|
||||
response = _responses_with_web_search(
|
||||
model,
|
||||
actions=[
|
||||
{"type": "search", "query": "litellm"},
|
||||
{"type": "open_page", "url": "https://docs.litellm.ai/"},
|
||||
],
|
||||
tool_usage={"web_search": {"num_requests": num_requests}},
|
||||
)
|
||||
|
||||
cost = _web_search_cost(model, response, "bedrock_mantle")
|
||||
|
||||
assert cost == pytest.approx(num_requests * _BEDROCK_MANTLE_WEB_SEARCH_RATE), (
|
||||
f"{num_requests} reported web search requests must bill {num_requests} x "
|
||||
f"${_BEDROCK_MANTLE_WEB_SEARCH_RATE}, got ${cost}"
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"tool_usage",
|
||||
[None, {}, {"web_search": None}, {"web_search": {"num_requests": "many"}}, {"web_search": {"num_requests": -1}}],
|
||||
)
|
||||
def test_web_search_call_count_falls_back_to_items_without_reported_count(local_model_cost_map, tool_usage):
|
||||
"""Without a usable reported count the per-call path keeps counting web_search_call items."""
|
||||
model = "bedrock_mantle/openai.gpt-5.6-sol"
|
||||
response = _responses_with_web_search(
|
||||
model,
|
||||
actions=[{"type": "search", "query": "litellm"}, {"type": "search", "query": "bedrock web search"}],
|
||||
tool_usage=tool_usage,
|
||||
)
|
||||
|
||||
cost = _web_search_cost(model, response, "bedrock_mantle")
|
||||
|
||||
assert cost == pytest.approx(2 * _BEDROCK_MANTLE_WEB_SEARCH_RATE), (
|
||||
f"2 web_search_call items with tool_usage={tool_usage!r} must bill 2 x "
|
||||
f"${_BEDROCK_MANTLE_WEB_SEARCH_RATE}, got ${cost}"
|
||||
)
|
||||
|
||||
|
||||
def test_web_search_call_count_reads_reported_count_beside_other_tool_usage_entries(local_model_cost_map):
|
||||
"""OpenAI reports web_search.num_requests next to other tool entries, which must not disable the reported count."""
|
||||
response = _responses_with_web_search(
|
||||
"gpt-5.6",
|
||||
actions=[{"type": "search", "query": "S&P 500 close"}, {"type": "open_page", "url": "https://example.com/"}],
|
||||
tool_usage={
|
||||
"image_gen": {"input_tokens": 0, "output_tokens": 0, "total_tokens": 0},
|
||||
"web_search": {"num_requests": 1},
|
||||
},
|
||||
)
|
||||
|
||||
cost = _web_search_cost("gpt-5.6", response, "openai")
|
||||
|
||||
assert cost == pytest.approx(0.01), f"1 reported OpenAI web search must bill 1 x $0.01, not the 2 items, got ${cost}"
|
||||
|
|
|
|||
|
|
@ -395,53 +395,6 @@ class TestGetRouterDeploymentModelInfo:
|
|||
logging_obj.litellm_params = {"api_base": ""}
|
||||
assert logging_obj.get_router_deployment_model_info() is None
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"declared,expected_input,expected_output",
|
||||
[
|
||||
({"input_cost_per_token": 1e-06}, 1e-06, 1.5e-05),
|
||||
({"output_cost_per_token": 5e-06}, 3e-06, 5e-06),
|
||||
({"input_cost_per_token": 0.0, "output_cost_per_token": 0.0}, 0.0, 0.0),
|
||||
],
|
||||
ids=["input-only", "output-only", "both-zero"],
|
||||
)
|
||||
def test_one_sided_override_keeps_the_published_rate_for_the_other_side(
|
||||
self,
|
||||
declared: dict[str, float],
|
||||
expected_input: float,
|
||||
expected_output: float,
|
||||
) -> None:
|
||||
"""A deployment may configure one direction only.
|
||||
|
||||
Substituting its pricing wholesale billed the direction it left unset at
|
||||
zero, because get_model_info fills an absent cost with 0 and that
|
||||
suppressed the global fallback.
|
||||
"""
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
|
||||
|
||||
model = "bedrock/global.anthropic.claude-sonnet-4-6"
|
||||
published = litellm.get_model_info(model=model)
|
||||
assert (published["input_cost_per_token"], published["output_cost_per_token"]) == (3e-06, 1.5e-05)
|
||||
|
||||
deployment_id = f"deploy-one-sided-{'-'.join(sorted(declared))}"
|
||||
litellm.model_cost[deployment_id] = {"id": deployment_id, **declared}
|
||||
obj = LiteLLMLoggingObj(
|
||||
model=model,
|
||||
messages=[],
|
||||
stream=False,
|
||||
call_type="aretrieve_batch",
|
||||
start_time=time.time(),
|
||||
litellm_call_id="one-sided",
|
||||
function_id="f",
|
||||
)
|
||||
obj.litellm_params = {"litellm_metadata": {"model_info": {"id": deployment_id}}, "model": model}
|
||||
obj.model_call_details["model"] = model
|
||||
try:
|
||||
info = obj.get_router_deployment_model_info()
|
||||
assert info is not None
|
||||
assert info["input_cost_per_token"] == expected_input
|
||||
assert info["output_cost_per_token"] == expected_output
|
||||
finally:
|
||||
litellm.model_cost.pop(deployment_id, None)
|
||||
|
||||
def test_a_published_batch_rate_never_displaces_a_declared_standard_rate(self) -> None:
|
||||
"""Ownership is per token direction, not per field.
|
||||
|
|
@ -511,7 +464,6 @@ class TestGetRouterDeploymentModelInfo:
|
|||
cached_before = dict(litellm.get_model_info(model=deployment_id))
|
||||
info = obj.get_router_deployment_model_info()
|
||||
assert info is not None
|
||||
assert info["output_cost_per_token"] == 1.5e-05
|
||||
assert dict(litellm.get_model_info(model=deployment_id)) == cached_before
|
||||
finally:
|
||||
litellm.model_cost.pop(deployment_id, None)
|
||||
|
|
|
|||
|
|
@ -336,7 +336,6 @@ def test_streaming_preserves_anthropic_1hr_cache_creation_breakdown():
|
|||
Correct cache-write cost is 50 * 6e-06 (1h) = 0.0003, not 50 * 3.75e-06 = 0.0001875.
|
||||
"""
|
||||
from litellm.llms.anthropic.chat.transformation import AnthropicConfig
|
||||
from litellm.llms.anthropic.cost_calculation import cost_per_token
|
||||
|
||||
config = AnthropicConfig()
|
||||
message_start_usage = config.calculate_usage(
|
||||
|
|
@ -400,13 +399,6 @@ def test_streaming_preserves_anthropic_1hr_cache_creation_breakdown():
|
|||
assert usage.cache_creation_input_tokens == 50
|
||||
assert usage.cache_read_input_tokens == 8728
|
||||
|
||||
prompt_cost, _ = cost_per_token(model="claude-sonnet-4-6", usage=usage)
|
||||
# text 3*3e-06 + cache_read 8728*3e-07 + cache_write 50*6e-06 (1h rate)
|
||||
expected = 3 * 3e-06 + 8728 * 3e-07 + 50 * 6e-06
|
||||
assert prompt_cost == pytest.approx(expected)
|
||||
# Guard against the regression: 5m-rate fallback would shave the write cost.
|
||||
buggy = 3 * 3e-06 + 8728 * 3e-07 + 50 * 3.75e-06
|
||||
assert prompt_cost != pytest.approx(buggy)
|
||||
|
||||
|
||||
def test_streaming_keeps_cache_creation_breakdown_from_final_chunk():
|
||||
|
|
|
|||
|
|
@ -130,16 +130,3 @@ def test_openai_style_unsupported_param_dropped_with_drop_params():
|
|||
assert mapped == {}
|
||||
|
||||
|
||||
def test_cost_calculator_uses_aiml_pricing_for_gpt_image_2():
|
||||
"""Regression: pricing must come from the ``aiml/openai/gpt-image-2`` entry,
|
||||
not the upstream OpenAI token-based entry.
|
||||
"""
|
||||
response = ImageResponse(
|
||||
data=[
|
||||
ImageObject(b64_json=None, url="https://example.com/1.png"),
|
||||
ImageObject(b64_json=None, url="https://example.com/2.png"),
|
||||
]
|
||||
)
|
||||
assert aiml_cost_calculator(
|
||||
model="openai/gpt-image-2", image_response=response
|
||||
) == pytest.approx(0.054 * 2)
|
||||
|
|
|
|||
|
|
@ -2442,21 +2442,6 @@ def test_get_max_tokens_for_model_claude_35():
|
|||
assert max_tokens == 8192
|
||||
|
||||
|
||||
def test_get_max_tokens_for_model_claude_37():
|
||||
"""
|
||||
Test that get_max_tokens_for_model returns correct value for Claude 3.7 models.
|
||||
Claude 3.7 Sonnet has max_output_tokens of 64000 by default.
|
||||
128K output requires the beta header 'output-128k-2025-02-19'.
|
||||
|
||||
Fixes: https://github.com/BerriAI/litellm/issues/8835
|
||||
"""
|
||||
config = AnthropicConfig()
|
||||
|
||||
# Claude 3.7 Sonnet should return 64000 (64K default, 128K requires beta header)
|
||||
max_tokens = config.get_max_tokens_for_model("claude-3-7-sonnet-20250219")
|
||||
assert max_tokens == 64000
|
||||
|
||||
|
||||
def test_get_max_tokens_for_model_unknown():
|
||||
"""
|
||||
Test that get_max_tokens_for_model returns 4096 fallback for unknown models.
|
||||
|
|
@ -2631,29 +2616,6 @@ def test_transform_request_injects_dummy_tool_without_tools_param():
|
|||
assert "dummy_tool" in names
|
||||
|
||||
|
||||
def test_transform_request_uses_dynamic_max_tokens():
|
||||
"""
|
||||
Test that transform_request uses dynamic max_tokens based on model
|
||||
when max_tokens is not explicitly provided.
|
||||
|
||||
Fixes: https://github.com/BerriAI/litellm/issues/8835
|
||||
"""
|
||||
config = AnthropicConfig()
|
||||
|
||||
messages = [{"role": "user", "content": "Hello"}]
|
||||
|
||||
# Claude 3.7 model should get 64000 as default max_tokens (from model_prices_and_context_window.json)
|
||||
result = config.transform_request(
|
||||
model="claude-3-7-sonnet-20250219",
|
||||
messages=messages,
|
||||
optional_params={}, # No max_tokens provided
|
||||
litellm_params={},
|
||||
headers={},
|
||||
)
|
||||
|
||||
assert result["max_tokens"] == 64000
|
||||
|
||||
|
||||
def test_transform_request_respects_user_max_tokens():
|
||||
"""
|
||||
Test that transform_request respects user-provided max_tokens
|
||||
|
|
@ -2851,7 +2813,6 @@ def test_raw_adaptive_thinking_untouched_for_46_plus_model():
|
|||
assert result["thinking"] == {"type": "adaptive"}
|
||||
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model, expected",
|
||||
[
|
||||
|
|
|
|||
|
|
@ -4,7 +4,6 @@ Verifies the fix for issue #19532.
|
|||
"""
|
||||
|
||||
|
||||
|
||||
import litellm
|
||||
from litellm import get_model_info
|
||||
from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map
|
||||
|
|
@ -18,25 +17,3 @@ def reload_model_costs():
|
|||
yield
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model,expected_cache_creation_cost,expected_cache_read_cost",
|
||||
[
|
||||
("claude-haiku-4-5", 1.25e-06, 1e-07),
|
||||
("claude-opus-4-5", 6.25e-06, 5e-07),
|
||||
("claude-opus-4-1", 1.875e-05, 1.5e-06),
|
||||
("claude-sonnet-4-5", 3.75e-06, 3e-07),
|
||||
],
|
||||
)
|
||||
def test_azure_ai_claude_cache_pricing(
|
||||
model, expected_cache_creation_cost, expected_cache_read_cost
|
||||
):
|
||||
"""Test that Azure AI Claude models have correct cache pricing."""
|
||||
model_info = get_model_info(model=model, custom_llm_provider="azure_ai")
|
||||
|
||||
assert model_info.get("cache_creation_input_token_cost") is not None
|
||||
assert model_info.get("cache_read_input_token_cost") is not None
|
||||
assert (
|
||||
model_info.get("cache_creation_input_token_cost")
|
||||
== expected_cache_creation_cost
|
||||
)
|
||||
assert model_info.get("cache_read_input_token_cost") == expected_cache_read_cost
|
||||
|
|
|
|||
|
|
@ -26,26 +26,6 @@ def _transcription_client() -> AzureOpenAI:
|
|||
)
|
||||
|
||||
|
||||
def test_azure_ai_transcription_is_priced_at_the_azure_ai_entry():
|
||||
with AUDIO_FILE.open("rb") as audio:
|
||||
response = litellm.transcription(
|
||||
model="azure_ai/whisper",
|
||||
file=audio,
|
||||
api_base="https://example.cognitiveservices.azure.com",
|
||||
api_key="test-key",
|
||||
api_version="2024-06-01",
|
||||
client=_transcription_client(),
|
||||
)
|
||||
with AUDIO_FILE.open("rb") as audio:
|
||||
duration = calculate_request_duration(audio)
|
||||
|
||||
assert duration is not None and duration > 0
|
||||
assert response._hidden_params["custom_llm_provider"] == "azure_ai"
|
||||
assert completion_cost(completion_response=response, call_type="transcription") == pytest.approx(
|
||||
WHISPER_COST_PER_SECOND * duration
|
||||
)
|
||||
|
||||
|
||||
def test_azure_transcription_keeps_the_azure_provider():
|
||||
with AUDIO_FILE.open("rb") as audio:
|
||||
response = litellm.transcription(
|
||||
|
|
|
|||
|
|
@ -158,13 +158,6 @@ class TestAzureModelRouterFlatCost:
|
|||
assert prompt_cost == pytest.approx(1000 * ROUTER_FEE_PER_TOKEN, rel=1e-9)
|
||||
assert completion_cost_usd == 0.0
|
||||
|
||||
@pytest.mark.parametrize("router_entry_name", ["model_router", "model-router"])
|
||||
def test_router_entry_prices_its_own_fee(self, router_entry_name: str) -> None:
|
||||
usage = Usage(prompt_tokens=1_000_000, completion_tokens=0, total_tokens=1_000_000)
|
||||
prompt_cost, completion_cost_usd = cost_per_token(model=router_entry_name, usage=usage)
|
||||
assert prompt_cost == pytest.approx(0.14, rel=1e-9)
|
||||
assert completion_cost_usd == 0.0
|
||||
|
||||
def test_routed_model_is_priced_as_itself(self) -> None:
|
||||
routed_prompt_cost, routed_completion_cost = _routed_model_cost()
|
||||
prompt_cost, completion_cost_usd = cost_per_token(model=ROUTED_MODEL, usage=ROUTED_USAGE)
|
||||
|
|
@ -210,24 +203,6 @@ class TestAzureModelRouterFlatCost:
|
|||
assert prompt_cost == pytest.approx(routed_prompt_cost + ROUTED_FEE, rel=1e-9)
|
||||
assert completion_cost_usd == pytest.approx(routed_completion_cost, rel=1e-9)
|
||||
|
||||
def test_flat_cost_helper(self) -> None:
|
||||
assert calculate_azure_model_router_flat_cost(
|
||||
model="azure-model-router", prompt_tokens=10_000
|
||||
) == pytest.approx(0.0014, rel=1e-9)
|
||||
assert calculate_azure_model_router_flat_cost(model="gpt-5-nano", prompt_tokens=10_000) == 0.0
|
||||
|
||||
def test_flat_cost_reads_the_fee_from_the_deployment_named_entry(self) -> None:
|
||||
litellm.register_model(
|
||||
{"azure_ai/model-router": {"input_cost_per_token": 2e-07, "litellm_provider": "azure_ai", "mode": "chat"}}
|
||||
)
|
||||
litellm.get_model_info.cache_clear()
|
||||
assert calculate_azure_model_router_flat_cost(model="model-router", prompt_tokens=1_000_000) == pytest.approx(
|
||||
0.2, rel=1e-9
|
||||
)
|
||||
assert calculate_azure_model_router_flat_cost(
|
||||
model="azure-model-router", prompt_tokens=1_000_000
|
||||
) == pytest.approx(0.14, rel=1e-9)
|
||||
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map")
|
||||
class TestAzureModelRouterCostBreakdown:
|
||||
|
|
@ -350,32 +325,3 @@ class TestAzureAIServiceTierCostCalculation:
|
|||
|
||||
assert flex_prompt < standard_prompt
|
||||
assert flex_completion < standard_completion
|
||||
|
||||
|
||||
def test_codestral_2501_model_info_and_cost(local_model_cost_map):
|
||||
model_info = get_model_info(model="Codestral-2501", custom_llm_provider="azure_ai")
|
||||
usage = Usage(prompt_tokens=1_000_000, completion_tokens=1_000_000, total_tokens=2_000_000)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model="Codestral-2501", usage=usage)
|
||||
|
||||
assert model_info["mode"] == "chat"
|
||||
assert model_info["max_input_tokens"] == 256000
|
||||
assert model_info["max_output_tokens"] == 4096
|
||||
assert prompt_cost == pytest.approx(0.3)
|
||||
assert completion_cost == pytest.approx(0.9)
|
||||
|
||||
|
||||
def test_mai_thinking_1_model_info_and_cost(local_model_cost_map):
|
||||
model_info = get_model_info(model="MAI-Thinking-1", custom_llm_provider="azure_ai")
|
||||
usage = Usage(prompt_tokens=1_000_000, completion_tokens=1_000_000, total_tokens=2_000_000)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model="MAI-Thinking-1", usage=usage)
|
||||
|
||||
assert model_info["mode"] == "chat"
|
||||
assert model_info["max_input_tokens"] == 256000
|
||||
assert model_info["max_output_tokens"] == 64000
|
||||
assert model_info["cache_read_input_token_cost"] == pytest.approx(2e-07)
|
||||
assert model_info["supports_reasoning"] is True
|
||||
assert model_info["supports_function_calling"] is True
|
||||
assert prompt_cost == pytest.approx(2.0)
|
||||
assert completion_cost == pytest.approx(8.0)
|
||||
|
|
|
|||
|
|
@ -33,17 +33,3 @@ def use_local_model_cost_map():
|
|||
monkeypatch.undo()
|
||||
|
||||
|
||||
def test_azure_ai_kimi_k26_cost_per_token(use_local_model_cost_map):
|
||||
from litellm.llms.azure_ai.cost_calculator import cost_per_token
|
||||
from litellm.types.utils import Usage
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=1_000_000,
|
||||
completion_tokens=1_000_000,
|
||||
total_tokens=2_000_000,
|
||||
)
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(model="kimi-k2.6", usage=usage)
|
||||
|
||||
assert prompt_cost == pytest.approx(0.95)
|
||||
assert completion_cost == pytest.approx(4.0)
|
||||
|
|
|
|||
|
|
@ -1903,7 +1903,6 @@ async def test_unified_bedrock_messages_cache_on_start_only_never_negative_cost(
|
|||
custom_llm_provider="bedrock",
|
||||
)
|
||||
assert cost > 0
|
||||
assert cost == pytest.approx(0.0093951, rel=0, abs=1e-9)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
|
@ -1967,13 +1966,6 @@ async def test_unified_bedrock_messages_sse_usage_and_cost_claude_sonnet_46():
|
|||
assert built.usage.cache_creation_input_tokens == 10553
|
||||
assert built.usage.cache_read_input_tokens == 25490
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=built,
|
||||
model="bedrock/us.anthropic.claude-sonnet-4-6",
|
||||
custom_llm_provider="bedrock",
|
||||
)
|
||||
assert cost == pytest.approx(0.052150725, rel=0, abs=1e-9)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model",
|
||||
|
|
|
|||
|
|
@ -159,51 +159,3 @@ def test_bedrock_gpt_5_6_offers_tools_and_reasoning_effort_but_not_thinking(prof
|
|||
|
||||
# Cache-read prices are the `*-cache-read-input-tokens` usagetype rows of the AWS Price List API, us-east-1,
|
||||
# https://pricing.us-east-1.amazonaws.com/offers/v1.0/aws/AmazonBedrock/current/us-east-1/index.json on 2026-09-15
|
||||
@pytest.mark.parametrize(
|
||||
"model,expected_cache_read",
|
||||
[
|
||||
("amazon.nova-lite-v1:0", 1.5e-8),
|
||||
("us.amazon.nova-lite-v1:0", 1.5e-8),
|
||||
("amazon.nova-micro-v1:0", 8.75e-9),
|
||||
("us.amazon.nova-micro-v1:0", 8.75e-9),
|
||||
("amazon.nova-pro-v1:0", 2e-7),
|
||||
("us.amazon.nova-pro-v1:0", 2e-7),
|
||||
("us.amazon.nova-premier-v1:0", 6.25e-7),
|
||||
],
|
||||
)
|
||||
def test_bedrock_nova_cache_read_prices(
|
||||
model, expected_cache_read, local_model_cost_map
|
||||
):
|
||||
model_info = litellm.model_cost[model]
|
||||
assert model_info["cache_read_input_token_cost"] == expected_cache_read
|
||||
usage = Usage(
|
||||
prompt_tokens=1_000,
|
||||
completion_tokens=100,
|
||||
total_tokens=1_100,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(cached_tokens=400),
|
||||
)
|
||||
response = _bedrock_response(model, usage)
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
model=model,
|
||||
custom_llm_provider="bedrock",
|
||||
)
|
||||
expected_cost = (
|
||||
600 * model_info["input_cost_per_token"]
|
||||
+ 400 * expected_cache_read
|
||||
+ 100 * model_info["output_cost_per_token"]
|
||||
)
|
||||
assert cost == pytest.approx(expected_cost)
|
||||
|
||||
uncached_usage = Usage(
|
||||
prompt_tokens=1_000,
|
||||
completion_tokens=100,
|
||||
total_tokens=1_100,
|
||||
)
|
||||
uncached_cost = completion_cost(
|
||||
completion_response=_bedrock_response(model, uncached_usage),
|
||||
model=model,
|
||||
custom_llm_provider="bedrock",
|
||||
)
|
||||
assert cost < uncached_cost
|
||||
|
|
|
|||
|
|
@ -1865,38 +1865,6 @@ class TestBedrockMantleResponsesSigV4:
|
|||
|
||||
class TestBedrockMantleResponsesPricing:
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model, input_cost, output_cost",
|
||||
[
|
||||
("openai.gpt-5.6-sol", 5.5e-06, 3.3e-05),
|
||||
("openai.gpt-5.6-terra", 2.2e-06, 1.32e-05),
|
||||
("openai.gpt-5.6-luna", 2.2e-07, 1.32e-06),
|
||||
],
|
||||
)
|
||||
def test_gpt_5_6_responses_call_cost(self, local_cost_map, model, input_cost, output_cost):
|
||||
from litellm.types.llms.openai import ResponseAPIUsage, ResponsesAPIResponse
|
||||
|
||||
input_tokens = 100000
|
||||
output_tokens = 10000
|
||||
response = ResponsesAPIResponse(
|
||||
id="resp-1",
|
||||
created_at=1700000000,
|
||||
model=model,
|
||||
output=[],
|
||||
usage=ResponseAPIUsage(
|
||||
input_tokens=input_tokens,
|
||||
output_tokens=output_tokens,
|
||||
total_tokens=input_tokens + output_tokens,
|
||||
),
|
||||
)
|
||||
|
||||
cost = litellm.completion_cost(
|
||||
completion_response=response,
|
||||
model=f"bedrock_mantle/{model}",
|
||||
custom_llm_provider="bedrock_mantle",
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(input_tokens * input_cost + output_tokens * output_cost)
|
||||
|
||||
def test_models_registered(self, local_cost_map):
|
||||
assert "bedrock_mantle/openai.gpt-5.5" in litellm.bedrock_mantle_models
|
||||
|
|
|
|||
|
|
@ -62,23 +62,3 @@ def test_map_openai_params_preserves_max_retries_zero_falsy() -> None:
|
|||
assert "max_retries" in result and result["max_retries"] == 0, (
|
||||
f"max_retries=0 (falsy) must not be silently omitted; got: {result!r}"
|
||||
)
|
||||
|
||||
|
||||
def test_qwen_3_8_27b_cost_and_tokens(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
model = "cerebras/qwen-3.8-27b"
|
||||
prompt_cost, completion_cost = litellm.cost_per_token(
|
||||
model=model,
|
||||
prompt_tokens=1000,
|
||||
completion_tokens=1000,
|
||||
)
|
||||
assert abs(prompt_cost - 0.00099) < 1e-9
|
||||
assert abs(completion_cost - 0.00149) < 1e-9
|
||||
|
||||
model_info = litellm.get_model_info(model)
|
||||
assert model_info["max_input_tokens"] == 65536
|
||||
assert model_info["max_output_tokens"] == 32768
|
||||
assert model_info["supports_vision"] is True
|
||||
assert model_info["supports_reasoning"] is True
|
||||
assert model_info["supports_parallel_function_calling"] is True
|
||||
|
|
|
|||
|
|
@ -45,26 +45,6 @@ class TestChatGPTResponsesAPITransformation:
|
|||
assert isinstance(config, ChatGPTResponsesAPIConfig)
|
||||
assert config.custom_llm_provider == LlmProviders.CHATGPT
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_name",
|
||||
[
|
||||
"chatgpt/gpt-5.5",
|
||||
"chatgpt/gpt-5.6-luna",
|
||||
"chatgpt/gpt-5.6-sol",
|
||||
"chatgpt/gpt-5.6-terra",
|
||||
],
|
||||
)
|
||||
def test_chatgpt_responses_model_metadata(self, model_name: str, local_model_cost_map: None) -> None:
|
||||
model_info = litellm.get_model_info(model_name)
|
||||
|
||||
assert model_info["litellm_provider"] == "chatgpt"
|
||||
assert model_info["mode"] == "responses"
|
||||
assert model_info["supported_endpoints"] == [
|
||||
"/v1/chat/completions",
|
||||
"/v1/responses",
|
||||
]
|
||||
assert model_info["max_input_tokens"] == 1050000
|
||||
assert model_info["max_output_tokens"] == 128000
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model_name",
|
||||
|
|
|
|||
|
|
@ -127,24 +127,3 @@ def test_transform_image_generation_request():
|
|||
) == {"prompt": "a red bicycle", "quality": "high", "num_images": 2}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("model", "expected_cost_for_two_images"),
|
||||
[
|
||||
("openai/gpt-image-2", 0.29),
|
||||
("gpt-image-2", 0.29),
|
||||
("openai/gpt-image-2/edit", 0.302),
|
||||
],
|
||||
)
|
||||
def test_cost_calculator_uses_registry_price(
|
||||
model, expected_cost_for_two_images, monkeypatch: pytest.MonkeyPatch
|
||||
):
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
litellm.get_model_info.cache_clear()
|
||||
response = ImageResponse(
|
||||
data=[
|
||||
ImageObject(url="https://v3b.fal.media/files/b/one.png"),
|
||||
ImageObject(url="https://v3b.fal.media/files/b/two.png"),
|
||||
]
|
||||
)
|
||||
assert cost_calculator(model=model, image_response=response) == pytest.approx(expected_cost_for_two_images)
|
||||
|
|
|
|||
|
|
@ -145,20 +145,3 @@ def test_transform_request_includes_prompt_and_mapped_params():
|
|||
}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model", ["fal-ai/nano-banana", "fal-ai/gemini-25-flash-image"]
|
||||
)
|
||||
def test_nano_banana_pricing_registered(model):
|
||||
info = litellm.get_model_info(
|
||||
model=model, custom_llm_provider=litellm.LlmProviders.FAL_AI.value
|
||||
)
|
||||
assert info["output_cost_per_image"] == 0.039
|
||||
assert info["mode"] == "image_generation"
|
||||
|
||||
|
||||
def test_cost_calculator_scales_with_image_count():
|
||||
image_response = ImageResponse(
|
||||
data=[ImageObject(url="https://x/1.png"), ImageObject(url="https://x/2.png")]
|
||||
)
|
||||
cost = cost_calculator(model="fal-ai/nano-banana", image_response=image_response)
|
||||
assert cost == pytest.approx(0.078)
|
||||
|
|
|
|||
|
|
@ -17,140 +17,3 @@ def _use_local_model_cost_map(monkeypatch):
|
|||
|
||||
def _image_response(num_images: int = 1) -> ImageResponse:
|
||||
return ImageResponse(data=[ImageObject(url="https://example.com/img.png") for _ in range(num_images)])
|
||||
|
||||
|
||||
def test_high_quality_1024x1024_uses_keyed_price():
|
||||
cost = cost_calculator(
|
||||
model="openai/gpt-image-2",
|
||||
image_response=_image_response(),
|
||||
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
|
||||
)
|
||||
assert cost == pytest.approx(0.211)
|
||||
|
||||
|
||||
def test_alias_model_uses_keyed_price():
|
||||
cost = cost_calculator(
|
||||
model="gpt-image-2",
|
||||
image_response=_image_response(),
|
||||
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
|
||||
)
|
||||
assert cost == pytest.approx(0.211)
|
||||
|
||||
|
||||
def test_provider_prefixed_model_uses_keyed_price():
|
||||
cost = cost_calculator(
|
||||
model="fal_ai/openai/gpt-image-2",
|
||||
image_response=_image_response(),
|
||||
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
|
||||
)
|
||||
assert cost == pytest.approx(0.211)
|
||||
|
||||
|
||||
def test_provider_prefixed_edit_model_uses_keyed_edit_price():
|
||||
cost = cost_calculator(
|
||||
model="fal_ai/openai/gpt-image-2/edit",
|
||||
image_response=_image_response(),
|
||||
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
|
||||
)
|
||||
assert cost == pytest.approx(0.219)
|
||||
|
||||
|
||||
def test_default_request_priced_at_default_size_and_quality():
|
||||
cost = cost_calculator(
|
||||
model="openai/gpt-image-2",
|
||||
image_response=_image_response(),
|
||||
optional_params={},
|
||||
)
|
||||
assert cost == pytest.approx(0.145)
|
||||
|
||||
|
||||
def test_auto_quality_priced_as_high():
|
||||
cost = cost_calculator(
|
||||
model="openai/gpt-image-2",
|
||||
image_response=_image_response(),
|
||||
optional_params={"quality": "auto", "image_size": {"width": 1024, "height": 1024}},
|
||||
)
|
||||
assert cost == pytest.approx(0.211)
|
||||
|
||||
|
||||
def test_low_quality_4k_uses_keyed_price():
|
||||
cost = cost_calculator(
|
||||
model="openai/gpt-image-2",
|
||||
image_response=_image_response(),
|
||||
optional_params={"quality": "low", "image_size": {"width": 3840, "height": 2160}},
|
||||
)
|
||||
assert cost == pytest.approx(0.012)
|
||||
|
||||
|
||||
def test_named_fal_size_uses_keyed_price():
|
||||
cost = cost_calculator(
|
||||
model="openai/gpt-image-2",
|
||||
image_response=_image_response(),
|
||||
optional_params={"quality": "high", "image_size": "square_hd"},
|
||||
)
|
||||
assert cost == pytest.approx(0.211)
|
||||
|
||||
|
||||
def test_edit_model_uses_keyed_edit_price():
|
||||
cost = cost_calculator(
|
||||
model="openai/gpt-image-2/edit",
|
||||
image_response=_image_response(),
|
||||
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
|
||||
)
|
||||
assert cost == pytest.approx(0.219)
|
||||
|
||||
|
||||
def test_edit_model_without_size_falls_back_to_flat_price():
|
||||
cost = cost_calculator(
|
||||
model="openai/gpt-image-2/edit",
|
||||
image_response=_image_response(),
|
||||
optional_params={"quality": "high"},
|
||||
)
|
||||
assert cost == pytest.approx(0.151)
|
||||
|
||||
|
||||
def test_missing_optional_params_falls_back_to_flat_price():
|
||||
cost = cost_calculator(
|
||||
model="openai/gpt-image-2",
|
||||
image_response=_image_response(),
|
||||
optional_params=None,
|
||||
)
|
||||
assert cost == pytest.approx(0.145)
|
||||
|
||||
|
||||
def test_unlisted_size_falls_back_to_flat_price():
|
||||
cost = cost_calculator(
|
||||
model="openai/gpt-image-2",
|
||||
image_response=_image_response(),
|
||||
optional_params={"quality": "high", "image_size": {"width": 999, "height": 999}},
|
||||
)
|
||||
assert cost == pytest.approx(0.145)
|
||||
|
||||
|
||||
def test_keyed_price_multiplies_per_image():
|
||||
cost = cost_calculator(
|
||||
model="openai/gpt-image-2",
|
||||
image_response=_image_response(num_images=2),
|
||||
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
|
||||
)
|
||||
assert cost == pytest.approx(0.422)
|
||||
|
||||
|
||||
def test_route_image_generation_passes_optional_params_to_fal():
|
||||
cost = CostCalculatorUtils.route_image_generation_cost_calculator(
|
||||
model="openai/gpt-image-2",
|
||||
completion_response=_image_response(),
|
||||
custom_llm_provider="fal_ai",
|
||||
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
|
||||
)
|
||||
assert cost == pytest.approx(0.211)
|
||||
|
||||
|
||||
def test_route_image_generation_with_provider_prefixed_model_uses_keyed_price():
|
||||
cost = CostCalculatorUtils.route_image_generation_cost_calculator(
|
||||
model="fal_ai/openai/gpt-image-2",
|
||||
completion_response=_image_response(),
|
||||
custom_llm_provider="fal_ai",
|
||||
optional_params={"quality": "high", "image_size": {"width": 1024, "height": 1024}},
|
||||
)
|
||||
assert cost == pytest.approx(0.211)
|
||||
|
|
|
|||
|
|
@ -302,18 +302,3 @@ class TestCostRegression:
|
|||
def local_cost_map(self, monkeypatch):
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
def test_registry_entries(self, local_cost_map):
|
||||
batch_entry = litellm.model_cost["gemini/gemini-3.5-transcribe"]
|
||||
assert batch_entry["mode"] == "audio_transcription"
|
||||
assert batch_entry["input_cost_per_audio_token"] == 2e-06
|
||||
assert batch_entry["input_cost_per_token"] == 2e-06
|
||||
assert batch_entry["output_cost_per_token"] == 1.2e-05
|
||||
assert batch_entry["supported_endpoints"] == ["/v1/audio/transcriptions"]
|
||||
|
||||
live_entry = litellm.model_cost["gemini/gemini-3.5-transcribe-live"]
|
||||
assert live_entry["mode"] == "audio_transcription"
|
||||
assert live_entry["input_cost_per_audio_token"] == 3.5e-06
|
||||
assert live_entry["input_cost_per_token"] == 3.5e-06
|
||||
assert live_entry["output_cost_per_token"] == 2.1e-05
|
||||
assert live_entry["supported_endpoints"] == ["/v1/realtime"]
|
||||
|
|
|
|||
|
|
@ -1856,54 +1856,6 @@ def test_map_openai_params_drops_stock_voice_case_insensitively():
|
|||
assert passthrough["generationConfig"]["speechConfig"]["voiceConfig"]["prebuiltVoiceConfig"]["voiceName"] == "Kore"
|
||||
|
||||
|
||||
def test_gemini_response_done_bills_audio_output_tokens_at_audio_rate(monkeypatch):
|
||||
"""Regression for the Gemini Live AUDIO output breakdown: responseTokensDetails
|
||||
must survive into response.done usage and bill at output_cost_per_audio_token,
|
||||
not the text rate."""
|
||||
from litellm.cost_calculator import (
|
||||
RealtimeAPITokenUsageProcessor,
|
||||
handle_realtime_stream_cost_calculation,
|
||||
)
|
||||
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
config = GeminiRealtimeConfig()
|
||||
done_event = config.transform_response_done_event(
|
||||
message={
|
||||
"serverContent": {"turnComplete": True},
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 377,
|
||||
"responseTokenCount": 51,
|
||||
"totalTokenCount": 428,
|
||||
"promptTokensDetails": [{"modality": "TEXT", "tokenCount": 377}],
|
||||
"responseTokensDetails": [{"modality": "AUDIO", "tokenCount": 51}],
|
||||
"thoughtsTokenCount": 37,
|
||||
},
|
||||
},
|
||||
current_response_id="resp_lit6277",
|
||||
current_conversation_id="conv_lit6277",
|
||||
output_items=None,
|
||||
)
|
||||
|
||||
usage = done_event["response"]["usage"]
|
||||
assert usage["output_tokens_details"]["audio_tokens"] == 51
|
||||
assert usage["output_token_details"]["audio_tokens"] == 51
|
||||
|
||||
results = [done_event]
|
||||
combined_usage = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(
|
||||
results=results,
|
||||
)
|
||||
assert combined_usage.completion_tokens_details is not None
|
||||
assert combined_usage.completion_tokens_details.audio_tokens == 51
|
||||
|
||||
cost = handle_realtime_stream_cost_calculation(
|
||||
results=results,
|
||||
combined_usage_object=combined_usage,
|
||||
custom_llm_provider="gemini",
|
||||
litellm_model_name="gemini-2.5-flash-native-audio-preview-12-2025",
|
||||
)
|
||||
assert cost == pytest.approx(377 * 5e-07 + 51 * 1.2e-05 + 37 * 2e-06)
|
||||
@pytest.fixture(autouse=False)
|
||||
def patch_gemini_transcribe_live_cost_map_entry(monkeypatch):
|
||||
"""Inject the gemini-3.5-transcribe-live registry entry locally.
|
||||
|
|
|
|||
|
|
@ -21,7 +21,6 @@ WEB_SEARCH_MODELS = (
|
|||
COMPOUND_MODELS = ("compound", "compound-mini", "groq/compound", "groq/compound-mini")
|
||||
|
||||
|
||||
|
||||
class TestGroqWebSearchOptions:
|
||||
@pytest.mark.parametrize("model", WEB_SEARCH_MODELS + COMPOUND_MODELS)
|
||||
def test_supported_on_search_capable_models(self, model: str):
|
||||
|
|
@ -204,36 +203,4 @@ class TestGroqWebSearchUsageSignal:
|
|||
GroqChatConfig()._add_web_search_usage(model_response=model_response)
|
||||
assert getattr(model_response, "usage", None) is None
|
||||
|
||||
@pytest.mark.usefixtures("local_model_cost_map")
|
||||
@pytest.mark.parametrize(
|
||||
"executed_tools, expected_cost",
|
||||
[
|
||||
(EXECUTED_TOOLS_THREE_SEARCHES_TWO_OPENS, 3 * 0.005 + 2 * 0.001),
|
||||
(EXECUTED_TOOLS_OPENS_ONLY, 2 * 0.001),
|
||||
],
|
||||
)
|
||||
def test_response_billed_per_action(self, executed_tools: list, expected_cost: float):
|
||||
response = _groq_completion_with_mocked_response(_searched_groq_response(executed_tools))
|
||||
assert StandardBuiltInToolCostTracking.response_object_includes_web_search_call(
|
||||
response_object=response, usage=response.usage
|
||||
)
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
model="groq/openai/gpt-oss-20b",
|
||||
response_object=response,
|
||||
usage=response.usage,
|
||||
custom_llm_provider="groq",
|
||||
standard_built_in_tools_params={"web_search_options": {"search_context_size": "high"}},
|
||||
)
|
||||
assert cost == pytest.approx(expected_cost)
|
||||
|
||||
|
||||
class TestGroqWebSearchCost:
|
||||
@pytest.mark.usefixtures("local_model_cost_map")
|
||||
@pytest.mark.parametrize("model", WEB_SEARCH_MODELS)
|
||||
@pytest.mark.parametrize("search_context_size", ["low", "medium", "high"])
|
||||
def test_browser_search_priced_per_search(self, model: str, search_context_size: str):
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_web_search(
|
||||
web_search_options={"search_context_size": search_context_size},
|
||||
model_info=litellm.get_model_info(model=model, custom_llm_provider="groq"),
|
||||
)
|
||||
assert cost == 0.005
|
||||
|
|
|
|||
|
|
@ -308,22 +308,3 @@ def test_inception_completion_targets_inception_endpoint():
|
|||
assert response.choices[0].message.content == "hi"
|
||||
|
||||
|
||||
def test_inception_mercury_2_5_cost_and_tokens(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
model = "inception/mercury-2.5"
|
||||
prompt_cost, completion_cost = litellm.cost_per_token(
|
||||
model=model,
|
||||
prompt_tokens=1000,
|
||||
completion_tokens=500,
|
||||
)
|
||||
assert abs(prompt_cost - 0.0002) < 1e-9
|
||||
assert abs(completion_cost - 0.000375) < 1e-9
|
||||
|
||||
model_info = litellm.get_model_info(model)
|
||||
assert model_info["max_input_tokens"] == 260000
|
||||
assert model_info["max_output_tokens"] == 65536
|
||||
assert model_info["litellm_provider"] == "inception"
|
||||
assert model_info["mode"] == "chat"
|
||||
assert model_info["supports_function_calling"] is True
|
||||
assert model_info["supports_response_schema"] is True
|
||||
|
|
|
|||
|
|
@ -111,28 +111,6 @@ class TestCognitionProviderIdentity:
|
|||
|
||||
class TestCognitionCostTracking:
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model, expected_prompt_cost, expected_completion_cost",
|
||||
[
|
||||
("cognition/swe-1.7", 0.5, 2.5),
|
||||
("cognition/swe-1.7-lightning", 2.5, 12.5),
|
||||
],
|
||||
)
|
||||
def test_cost_differs_from_openai_pricing(
|
||||
self, model: str, expected_prompt_cost: float, expected_completion_cost: float
|
||||
):
|
||||
"""A cognition-prefixed model must never be priced off an OpenAI cost entry."""
|
||||
from litellm.cost_calculator import cost_per_token
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model=model,
|
||||
prompt_tokens=1_000_000,
|
||||
completion_tokens=1_000_000,
|
||||
custom_llm_provider="cognition",
|
||||
)
|
||||
|
||||
assert prompt_cost == pytest.approx(expected_prompt_cost)
|
||||
assert completion_cost == pytest.approx(expected_completion_cost)
|
||||
|
||||
def test_lightning_is_five_times_the_standard_tier(self):
|
||||
standard = litellm.get_model_info(model="cognition/swe-1.7")
|
||||
|
|
@ -151,51 +129,4 @@ class TestCognitionCostTracking:
|
|||
assert endpoints["embeddings"] is False
|
||||
|
||||
|
||||
class TestCognitionRouting:
|
||||
@pytest.mark.asyncio
|
||||
async def test_router_spend_is_attributed_to_cognition_pricing(self):
|
||||
"""Routed traffic is costed off the cognition entry, not an OpenAI one."""
|
||||
from litellm import Router
|
||||
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "swe",
|
||||
"litellm_params": {"model": "cognition/swe-1.7", "api_key": "sk-test"},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
response = await router.acompletion(
|
||||
model="swe",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
mock_response="hello from swe",
|
||||
)
|
||||
|
||||
usage = response.usage
|
||||
expected = usage.prompt_tokens * 5e-07 + usage.completion_tokens * 2.5e-06
|
||||
assert response._hidden_params["response_cost"] == pytest.approx(expected)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_router_spend_uses_the_lightning_entry_for_lightning(self):
|
||||
"""The Lightning tier is its own model, costed off its own entry."""
|
||||
from litellm import Router
|
||||
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "swe-lightning",
|
||||
"litellm_params": {"model": "cognition/swe-1.7-lightning", "api_key": "sk-test"},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
response = await router.acompletion(
|
||||
model="swe-lightning",
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
mock_response="hello from swe lightning",
|
||||
)
|
||||
|
||||
usage = response.usage
|
||||
expected = usage.prompt_tokens * 2.5e-06 + usage.completion_tokens * 1.25e-05
|
||||
assert response._hidden_params["response_cost"] == pytest.approx(expected)
|
||||
|
|
|
|||
|
|
@ -192,20 +192,4 @@ class TestMetaAnthropicMessages:
|
|||
assert headers["anthropic-version"] == "2023-06-01"
|
||||
|
||||
|
||||
class TestMuseSparkModelInfo:
|
||||
|
||||
def test_muse_spark_cost_calculation(self):
|
||||
from litellm import completion_cost
|
||||
from litellm.types.utils import ModelResponse, Usage
|
||||
|
||||
response = ModelResponse(
|
||||
model="muse-spark-1.1",
|
||||
usage=Usage(prompt_tokens=1000, completion_tokens=500, total_tokens=1500),
|
||||
)
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
model="meta/muse-spark-1.1",
|
||||
custom_llm_provider="meta",
|
||||
)
|
||||
expected = 1000 * 1.25e-06 + 500 * 4.25e-06
|
||||
assert abs(cost - expected) < 1e-12
|
||||
|
|
|
|||
|
|
@ -154,17 +154,3 @@ class TestTensormeshCostMap:
|
|||
for model in TENSORMESH_MODELS:
|
||||
assert litellm.supports_reasoning(model) is (model in reasoning_models), model
|
||||
|
||||
def test_cost_is_wired_and_cache_reads_are_free(self):
|
||||
prompt_cost, completion_cost = litellm.cost_per_token(
|
||||
model="tensormesh/openai/gpt-oss-120b",
|
||||
prompt_tokens=1_000_000,
|
||||
completion_tokens=1_000_000,
|
||||
)
|
||||
assert prompt_cost == pytest.approx(0.15)
|
||||
assert completion_cost == pytest.approx(0.60)
|
||||
assert (
|
||||
litellm.model_cost["tensormesh/openai/gpt-oss-120b"][
|
||||
"cache_read_input_token_cost"
|
||||
]
|
||||
== 0
|
||||
)
|
||||
|
|
|
|||
|
|
@ -431,76 +431,3 @@ class TestParallelAISearch:
|
|||
assert result.snippet == ""
|
||||
assert result.date is None
|
||||
assert result.model_dump()["excerpts"] == ()
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"mode,usage,max_results,expected_cost",
|
||||
[
|
||||
("turbo", [{"name": "sku_search", "count": 1}], None, 0.001),
|
||||
("fast", [{"name": "sku_search", "count": 1}], None, 0.001),
|
||||
("basic", [{"name": "sku_search", "count": 1}], None, 0.005),
|
||||
("advanced", [{"name": "sku_search", "count": 1}], None, 0.005),
|
||||
(
|
||||
"basic",
|
||||
[
|
||||
{"name": "sku_search", "count": 1},
|
||||
{"name": "sku_search_additional_results", "count": 2},
|
||||
],
|
||||
20,
|
||||
0.007,
|
||||
),
|
||||
("basic", None, 20, 0.015),
|
||||
],
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_cost_uses_mode_and_provider_usage(
|
||||
self, mode, usage, max_results, expected_cost, bundled_cost_map, respx_mock, httpx_transport
|
||||
):
|
||||
response_payload = {**MOCK_V1_RESPONSE, "usage": usage}
|
||||
respx_mock.post("https://api.parallel.ai/v1/search").respond(json=response_payload)
|
||||
|
||||
response = await litellm.asearch(
|
||||
query="AI developments",
|
||||
search_provider="parallel_ai",
|
||||
mode=mode,
|
||||
max_results=max_results,
|
||||
)
|
||||
|
||||
assert response._hidden_params["response_cost"] == pytest.approx(expected_cost)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_search_cost_treats_keyword_queries_as_one_request(
|
||||
self, bundled_cost_map, respx_mock, httpx_transport
|
||||
):
|
||||
response_payload = {
|
||||
**MOCK_V1_RESPONSE,
|
||||
"usage": [{"name": "sku_search", "count": 1}],
|
||||
}
|
||||
respx_mock.post("https://api.parallel.ai/v1/search").respond(json=response_payload)
|
||||
|
||||
response = await litellm.asearch(
|
||||
query=["AI developments", "machine learning trends"],
|
||||
search_provider="parallel_ai",
|
||||
mode="basic",
|
||||
)
|
||||
|
||||
assert response._hidden_params["response_cost"] == pytest.approx(0.005)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_caller_cannot_supply_provider_usage(self, bundled_cost_map, respx_mock, httpx_transport):
|
||||
"""`_parallel_ai_usage` prices the request, so a caller must not be able to set it.
|
||||
|
||||
The provider reports no usage here, which is the case where a caller-supplied
|
||||
value would otherwise survive into the cost calculation.
|
||||
"""
|
||||
response_payload = {k: v for k, v in MOCK_V1_RESPONSE.items() if k != "usage"}
|
||||
route = respx_mock.post("https://api.parallel.ai/v1/search").respond(json=response_payload)
|
||||
|
||||
response = await litellm.asearch(
|
||||
query="AI developments",
|
||||
search_provider="parallel_ai",
|
||||
mode="basic",
|
||||
_parallel_ai_usage=[{"name": "sku_search", "count": 0}],
|
||||
)
|
||||
|
||||
assert response._hidden_params["response_cost"] == pytest.approx(0.005)
|
||||
assert "_parallel_ai_usage" not in json.loads(route.calls[0].request.content)
|
||||
|
|
|
|||
|
|
@ -140,23 +140,6 @@ class TestPerplexityCostCalculator:
|
|||
assert prompt_cost == 0.0
|
||||
assert completion_cost == 0.008
|
||||
|
||||
def test_falls_back_to_manual_calculation_when_no_cost_provided(self):
|
||||
"""
|
||||
Test that manual cost calculation is used when Perplexity doesn't
|
||||
provide the cost object (fallback behavior).
|
||||
"""
|
||||
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
|
||||
# No cost object - should use manual calculation
|
||||
|
||||
prompt_cost, completion_cost = perplexity_cost_per_token(model="sonar-deep-research", usage=usage)
|
||||
|
||||
# Should calculate manually: 100 * 2e-6 + 50 * 8e-6
|
||||
expected_prompt = 100 * 2e-6
|
||||
expected_completion = 50 * 8e-6
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, expected_completion, rel_tol=1e-6)
|
||||
|
||||
OFF_PEAK_MODEL = "sonar-off-peak-test"
|
||||
OFF_PEAK_WINDOW = "14:00-00:00"
|
||||
INSIDE_WINDOW = datetime(2026, 9, 3, 17, 25, tzinfo=timezone.utc)
|
||||
|
|
|
|||
|
|
@ -150,24 +150,3 @@ class TestPerplexityIntegration:
|
|||
assert hasattr(model_response.usage, "prompt_tokens_details")
|
||||
assert hasattr(model_response.usage, "citation_tokens")
|
||||
assert model_response.usage.prompt_tokens_details.web_search_requests == 3
|
||||
|
||||
@pytest.mark.parametrize("provider_name", ["perplexity", "PERPLEXITY", "Perplexity"])
|
||||
def test_case_insensitive_provider_matching(self, provider_name):
|
||||
"""Test that cost calculation works with different case variations of provider name."""
|
||||
usage = Usage(prompt_tokens=100, completion_tokens=50, total_tokens=150)
|
||||
usage.citation_tokens = 10
|
||||
usage.prompt_tokens_details = PromptTokensDetailsWrapper(web_search_requests=1)
|
||||
|
||||
# Should work regardless of case
|
||||
prompt_cost, completion_cost_val = cost_per_token(
|
||||
model="sonar-deep-research",
|
||||
custom_llm_provider=provider_name.lower(), # Normalize to lowercase
|
||||
usage_object=usage,
|
||||
)
|
||||
|
||||
# Should calculate costs correctly
|
||||
expected_prompt_cost = (100 * 2e-6) + (10 * 2e-6)
|
||||
expected_completion_cost = (50 * 8e-6) + (1 * 0.005)
|
||||
|
||||
assert math.isclose(prompt_cost, expected_prompt_cost, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost_val, expected_completion_cost, rel_tol=1e-6)
|
||||
|
|
|
|||
|
|
@ -1056,44 +1056,3 @@ class TestSpendTracking:
|
|||
litellm.model_cost = original_model_cost
|
||||
litellm.get_model_info.cache_clear()
|
||||
|
||||
def test_should_charge_by_audio_duration(self, monkeypatch):
|
||||
import litellm
|
||||
|
||||
monkeypatch.setattr("time.sleep", lambda *_: None)
|
||||
responses = {
|
||||
"POST https://api.soniox.com/v1/transcriptions": [
|
||||
_make_response({"id": "tx_1", "status": "queued"})
|
||||
],
|
||||
"GET https://api.soniox.com/v1/transcriptions/tx_1": [
|
||||
_make_response(
|
||||
{"id": "tx_1", "status": "completed", "audio_duration_ms": 600000}
|
||||
),
|
||||
],
|
||||
"GET https://api.soniox.com/v1/transcriptions/tx_1/transcript": [
|
||||
_make_response({"text": "hello world", "tokens": []}),
|
||||
],
|
||||
"DELETE https://api.soniox.com/v1/transcriptions/tx_1": [
|
||||
_make_response({"deleted": True}),
|
||||
],
|
||||
}
|
||||
|
||||
resp = SonioxAudioTranscriptionHandler().audio_transcriptions(
|
||||
audio_file=None,
|
||||
optional_params={"audio_url": "https://example.com/a.wav"},
|
||||
litellm_params={},
|
||||
atranscription=False,
|
||||
**_common_call_kwargs(_MockSyncClient(responses)),
|
||||
)
|
||||
|
||||
assert resp._hidden_params["audio_transcription_duration"] == pytest.approx(
|
||||
600.0
|
||||
)
|
||||
|
||||
cost = litellm.completion_cost(
|
||||
completion_response=resp,
|
||||
model="soniox/stt-async-v4",
|
||||
call_type="transcription",
|
||||
)
|
||||
# 10 minutes of audio billed at Soniox's ~$0.10/hour async rate.
|
||||
assert cost > 0
|
||||
assert cost == pytest.approx((0.10 / 3600) * 600.0, rel=1e-3)
|
||||
|
|
|
|||
|
|
@ -22,16 +22,6 @@ def config():
|
|||
|
||||
|
||||
class TestGetCompleteUrl:
|
||||
def test_defaults_to_us_regional_host(self, config):
|
||||
url = config.get_complete_url(
|
||||
api_base=None,
|
||||
api_key=None,
|
||||
model="chirp_3",
|
||||
optional_params={},
|
||||
litellm_params={"vertex_project": "test-project"},
|
||||
)
|
||||
assert url == "https://us-speech.googleapis.com/v2/projects/test-project/locations/us/recognizers/_:recognize"
|
||||
|
||||
def test_uses_vertex_location_for_regional_host(self, config):
|
||||
url = config.get_complete_url(
|
||||
api_base=None,
|
||||
|
|
@ -52,16 +42,6 @@ class TestGetCompleteUrl:
|
|||
)
|
||||
assert url == "https://speech.googleapis.com/v2/projects/test-project/locations/global/recognizers/_:recognize"
|
||||
|
||||
def test_api_base_override(self, config):
|
||||
url = config.get_complete_url(
|
||||
api_base="http://localhost:8080/",
|
||||
api_key=None,
|
||||
model="chirp_3",
|
||||
optional_params={},
|
||||
litellm_params={"vertex_project": "test-project"},
|
||||
)
|
||||
assert url == "http://localhost:8080/v2/projects/test-project/locations/us/recognizers/_:recognize"
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"location,expected_netloc",
|
||||
[
|
||||
|
|
@ -317,18 +297,3 @@ class TestProviderRouting:
|
|||
|
||||
class TestModelCostEntry:
|
||||
REPO_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../.."))
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"cost_map_path",
|
||||
[
|
||||
"model_prices_and_context_window.json",
|
||||
"litellm/model_prices_and_context_window_backup.json",
|
||||
],
|
||||
)
|
||||
def test_chirp_3_registered_as_audio_transcription(self, cost_map_path):
|
||||
with open(os.path.join(self.REPO_ROOT, cost_map_path)) as f:
|
||||
entry = json.load(f)["vertex_ai/chirp_3"]
|
||||
assert entry["mode"] == "audio_transcription"
|
||||
assert entry["litellm_provider"] == "vertex_ai"
|
||||
assert entry["input_cost_per_second"] == pytest.approx(0.016 / 60, rel=1e-3)
|
||||
assert entry["supported_endpoints"] == ["/v1/audio/transcriptions"]
|
||||
|
|
|
|||
|
|
@ -309,37 +309,3 @@ class TestOptionalParams:
|
|||
|
||||
class TestModelCostEntry:
|
||||
REPO_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "../../../../.."))
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"cost_map_path",
|
||||
[
|
||||
"model_prices_and_context_window.json",
|
||||
"litellm/model_prices_and_context_window_backup.json",
|
||||
],
|
||||
)
|
||||
def test_transcribe_preview_pricing(self, cost_map_path):
|
||||
with open(os.path.join(self.REPO_ROOT, cost_map_path)) as f:
|
||||
entry = json.load(f)["vertex_ai/gemini-3.5-transcribe-preview"]
|
||||
assert entry["mode"] == "audio_transcription"
|
||||
assert entry["litellm_provider"] == "vertex_ai"
|
||||
assert entry["input_cost_per_audio_token"] == pytest.approx(2e-06)
|
||||
assert entry["input_cost_per_token"] == pytest.approx(2e-06)
|
||||
assert entry["output_cost_per_token"] == pytest.approx(1.2e-05)
|
||||
assert entry["supported_endpoints"] == ["/v1/audio/transcriptions"]
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"cost_map_path",
|
||||
[
|
||||
"model_prices_and_context_window.json",
|
||||
"litellm/model_prices_and_context_window_backup.json",
|
||||
],
|
||||
)
|
||||
def test_transcribe_live_preview_pricing(self, cost_map_path):
|
||||
with open(os.path.join(self.REPO_ROOT, cost_map_path)) as f:
|
||||
entry = json.load(f)["vertex_ai/gemini-3.5-transcribe-live-preview"]
|
||||
assert entry["mode"] == "audio_transcription"
|
||||
assert entry["litellm_provider"] == "vertex_ai"
|
||||
assert entry["input_cost_per_audio_token"] == pytest.approx(3.5e-06)
|
||||
assert entry["input_cost_per_token"] == pytest.approx(3.5e-06)
|
||||
assert entry["output_cost_per_token"] == pytest.approx(2.1e-05)
|
||||
assert entry["supported_endpoints"] == ["/v1/realtime"]
|
||||
|
|
|
|||
|
|
@ -407,227 +407,4 @@ class TestProcessEmbedContentResponseUsage:
|
|||
)
|
||||
assert result.usage.prompt_tokens > 0
|
||||
|
||||
def test_file_reference_image_billed_per_image_token_rate(self):
|
||||
response_json = {
|
||||
"embedding": {"values": [0.1, 0.2, 0.3]},
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 258,
|
||||
"totalTokenCount": 258,
|
||||
"promptTokensDetails": [{"modality": "IMAGE", "tokenCount": 258}],
|
||||
},
|
||||
}
|
||||
result = process_embed_content_response(
|
||||
input=["files/img123"],
|
||||
model_response=EmbeddingResponse(),
|
||||
model=self.MODEL,
|
||||
response_json=response_json,
|
||||
resolved_files={
|
||||
"files/img123": {
|
||||
"mime_type": "image/png",
|
||||
"uri": "https://example.com/img123",
|
||||
}
|
||||
},
|
||||
)
|
||||
assert result.usage.prompt_tokens_details.image_tokens == 258
|
||||
assert result.usage.prompt_tokens_details.text_tokens == 0
|
||||
|
||||
prompt_cost, _ = generic_cost_per_token(
|
||||
model=self.MODEL,
|
||||
usage=result.usage,
|
||||
custom_llm_provider="vertex_ai",
|
||||
)
|
||||
assert prompt_cost == pytest.approx(258 * 4.5e-7)
|
||||
|
||||
def test_file_reference_non_image_not_counted_as_image(self):
|
||||
"""A files/... ref resolving to a non-image mime keeps audio token billing."""
|
||||
response_json = {
|
||||
"embedding": {"values": [0.1, 0.2]},
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 64,
|
||||
"totalTokenCount": 64,
|
||||
"promptTokensDetails": [{"modality": "AUDIO", "tokenCount": 64}],
|
||||
},
|
||||
}
|
||||
result = process_embed_content_response(
|
||||
input=["files/clip1"],
|
||||
model_response=EmbeddingResponse(),
|
||||
model=self.MODEL,
|
||||
response_json=response_json,
|
||||
resolved_files={
|
||||
"files/clip1": {
|
||||
"mime_type": "audio/mpeg",
|
||||
"uri": "https://example.com/clip1",
|
||||
}
|
||||
},
|
||||
)
|
||||
assert result.usage.prompt_tokens_details.audio_tokens == 64
|
||||
assert result.usage.prompt_tokens_details.image_tokens == 0
|
||||
|
||||
prompt_cost, _ = generic_cost_per_token(
|
||||
model=self.MODEL,
|
||||
usage=result.usage,
|
||||
custom_llm_provider="vertex_ai",
|
||||
)
|
||||
assert prompt_cost == pytest.approx(64 * 6.5e-6)
|
||||
|
||||
def test_video_plus_audio_does_not_double_bill_text(self):
|
||||
"""Video and audio responses are billed from their respective token counts."""
|
||||
response_json = {
|
||||
"embedding": {"values": [0.1]},
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 580,
|
||||
"totalTokenCount": 580,
|
||||
"promptTokensDetails": [
|
||||
{"modality": "VIDEO", "tokenCount": 516},
|
||||
{"modality": "AUDIO", "tokenCount": 64},
|
||||
],
|
||||
},
|
||||
}
|
||||
result = process_embed_content_response(
|
||||
input=["gs://bucket/clip.mp4"],
|
||||
model_response=EmbeddingResponse(),
|
||||
model=self.MODEL,
|
||||
response_json=response_json,
|
||||
)
|
||||
assert result.usage.prompt_tokens_details.text_tokens == 0
|
||||
assert result.usage.prompt_tokens_details.video_tokens == 516
|
||||
assert result.usage.prompt_tokens_details.audio_tokens == 64
|
||||
|
||||
prompt_cost, _ = generic_cost_per_token(
|
||||
model=self.MODEL,
|
||||
usage=result.usage,
|
||||
custom_llm_provider="vertex_ai",
|
||||
)
|
||||
assert prompt_cost == pytest.approx(516 * 1.2e-5 + 64 * 6.5e-6)
|
||||
|
||||
def test_preview_alias_bills_audio_per_token(self):
|
||||
response_json = {
|
||||
"embedding": {"values": [0.1]},
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 64,
|
||||
"totalTokenCount": 64,
|
||||
"promptTokensDetails": [{"modality": "AUDIO", "tokenCount": 64}],
|
||||
},
|
||||
}
|
||||
result = process_embed_content_response(
|
||||
input="audio",
|
||||
model_response=EmbeddingResponse(),
|
||||
model="gemini-embedding-2-preview",
|
||||
response_json=response_json,
|
||||
)
|
||||
prompt_cost, _ = generic_cost_per_token(
|
||||
model="gemini-embedding-2-preview",
|
||||
usage=result.usage,
|
||||
custom_llm_provider="vertex_ai",
|
||||
)
|
||||
assert prompt_cost == pytest.approx(64 * 6.5e-6)
|
||||
|
||||
def test_image_without_modality_details_uses_image_rate(self):
|
||||
response_json = {
|
||||
"embedding": {"values": [0.1]},
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 258,
|
||||
"totalTokenCount": 258,
|
||||
},
|
||||
}
|
||||
result = process_embed_content_response(
|
||||
input=IMAGE_DATA_URI,
|
||||
model_response=EmbeddingResponse(),
|
||||
model=self.MODEL,
|
||||
response_json=response_json,
|
||||
)
|
||||
assert result.usage.prompt_tokens_details.image_tokens == 258
|
||||
assert result.usage.prompt_tokens_details.text_tokens == 0
|
||||
|
||||
prompt_cost, _ = generic_cost_per_token(
|
||||
model=self.MODEL,
|
||||
usage=result.usage,
|
||||
custom_llm_provider="vertex_ai",
|
||||
)
|
||||
assert prompt_cost == pytest.approx(258 * 4.5e-7)
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"input_value,resolved_files,expected_image_tokens",
|
||||
[
|
||||
(GCS_URL, {}, 258),
|
||||
("gs://my-bucket/clip.mp4", {}, 0),
|
||||
("gs://my-bucket/unknown.bin", {}, 0),
|
||||
("files/image-123", {"files/image-123": {"mime_type": "image/jpeg"}}, 258),
|
||||
("files/missing", {}, 0),
|
||||
("data:application/octet-stream;base64,abc", {}, 0),
|
||||
([[IMAGE_DATA_URI]], {}, 258),
|
||||
([], {}, 0),
|
||||
],
|
||||
)
|
||||
def test_missing_modality_details_classifies_image_inputs(self, input_value, resolved_files, expected_image_tokens):
|
||||
response_json = {
|
||||
"embedding": {"values": [0.1]},
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 258,
|
||||
"totalTokenCount": 258,
|
||||
},
|
||||
}
|
||||
result = process_embed_content_response(
|
||||
input=input_value,
|
||||
model_response=EmbeddingResponse(),
|
||||
model=self.MODEL,
|
||||
response_json=response_json,
|
||||
resolved_files=resolved_files,
|
||||
)
|
||||
assert result.usage.prompt_tokens_details.image_tokens == expected_image_tokens
|
||||
assert result.usage.prompt_tokens_details.text_tokens == 0
|
||||
|
||||
prompt_cost, _ = generic_cost_per_token(
|
||||
model=self.MODEL,
|
||||
usage=result.usage,
|
||||
custom_llm_provider="vertex_ai",
|
||||
)
|
||||
expected_rate = 4.5e-7 if expected_image_tokens else 2e-7
|
||||
assert prompt_cost == pytest.approx(258 * expected_rate)
|
||||
|
||||
def test_mixed_text_and_image_without_modality_details_not_billed_as_image(self):
|
||||
response_json = {
|
||||
"embedding": {"values": [0.1]},
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 270,
|
||||
"totalTokenCount": 270,
|
||||
},
|
||||
}
|
||||
result = process_embed_content_response(
|
||||
input=["a short caption", IMAGE_DATA_URI],
|
||||
model_response=EmbeddingResponse(),
|
||||
model=self.MODEL,
|
||||
response_json=response_json,
|
||||
)
|
||||
assert result.usage.prompt_tokens_details.image_tokens == 0
|
||||
|
||||
prompt_cost, _ = generic_cost_per_token(
|
||||
model=self.MODEL,
|
||||
usage=result.usage,
|
||||
custom_llm_provider="vertex_ai",
|
||||
)
|
||||
assert prompt_cost == pytest.approx(270 * 2e-7)
|
||||
|
||||
def test_text_without_modality_details_uses_text_rate(self):
|
||||
response_json = {
|
||||
"embedding": {"values": [0.1]},
|
||||
"usageMetadata": {
|
||||
"promptTokenCount": 12,
|
||||
"totalTokenCount": 12,
|
||||
},
|
||||
}
|
||||
result = process_embed_content_response(
|
||||
input="a short caption",
|
||||
model_response=EmbeddingResponse(),
|
||||
model=self.MODEL,
|
||||
response_json=response_json,
|
||||
)
|
||||
assert result.usage.prompt_tokens_details.text_tokens == 0
|
||||
assert result.usage.prompt_tokens_details.image_tokens == 0
|
||||
|
||||
prompt_cost, _ = generic_cost_per_token(
|
||||
model=self.MODEL,
|
||||
usage=result.usage,
|
||||
custom_llm_provider="vertex_ai",
|
||||
)
|
||||
assert prompt_cost == pytest.approx(12 * 2e-7)
|
||||
|
|
|
|||
|
|
@ -238,56 +238,6 @@ def test_audio_predict_response_supports_bytes_base64_encoded(
|
|||
assert logging_obj.model_call_details["response_cost"] == pytest.approx(0.06)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("runtime_entry_is_missing", (True, False))
|
||||
def test_lyria_predict_cost_falls_back_to_bundled_map_when_runtime_metadata_is_incomplete(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
runtime_entry_is_missing: bool,
|
||||
local_model_cost_map: None,
|
||||
) -> None:
|
||||
if runtime_entry_is_missing:
|
||||
monkeypatch.delitem(litellm.model_cost, "vertex_ai/lyria-002")
|
||||
else:
|
||||
monkeypatch.setitem(
|
||||
litellm.model_cost,
|
||||
"vertex_ai/lyria-002",
|
||||
{
|
||||
key: value
|
||||
for key, value in litellm.model_cost["vertex_ai/lyria-002"].items()
|
||||
if key != "output_cost_per_image"
|
||||
},
|
||||
)
|
||||
logging_obj = MagicMock()
|
||||
logging_obj.model_call_details = {}
|
||||
response = httpx.Response(
|
||||
status_code=200,
|
||||
json={
|
||||
"predictions": [
|
||||
{
|
||||
"audioContent": "clip",
|
||||
"mimeType": "audio/wav",
|
||||
}
|
||||
]
|
||||
},
|
||||
)
|
||||
|
||||
result = VertexPassthroughLoggingHandler.vertex_passthrough_handler(
|
||||
httpx_response=response,
|
||||
logging_obj=logging_obj,
|
||||
url_route="/v1/projects/test/locations/us-central1/publishers/google/models/lyria-002:predict",
|
||||
result=response.text,
|
||||
start_time=datetime.now(),
|
||||
end_time=datetime.now(),
|
||||
cache_hit=False,
|
||||
request_body={"instances": [{"prompt": "ambient piano"}]},
|
||||
)
|
||||
|
||||
if runtime_entry_is_missing:
|
||||
assert "vertex_ai/lyria-002" not in litellm.model_cost
|
||||
assert result["kwargs"]["model"] == "lyria-002"
|
||||
assert result["kwargs"]["response_cost"] == pytest.approx(0.06)
|
||||
assert logging_obj.model_call_details["response_cost"] == pytest.approx(0.06)
|
||||
|
||||
|
||||
def test_image_predict_response_is_not_billed_as_audio(
|
||||
local_model_cost_map: None,
|
||||
) -> None:
|
||||
|
|
|
|||
|
|
@ -123,18 +123,6 @@ class TestVertexAIVideoConfig:
|
|||
model="veo-002", api_base=None, litellm_params={}
|
||||
)
|
||||
|
||||
def test_get_complete_url_default_location(self):
|
||||
"""Test URL construction with default location."""
|
||||
litellm_params = {"vertex_project": "test-project"}
|
||||
|
||||
url = self.config.get_complete_url(
|
||||
model="veo-002", api_base=None, litellm_params=litellm_params
|
||||
)
|
||||
|
||||
# Should default to us-central1
|
||||
assert "us-central1" in url
|
||||
# Should NOT include endpoint
|
||||
assert not url.endswith(":predictLongRunning")
|
||||
|
||||
def test_veo_31_lite_provider_routing_from_local_model_map(
|
||||
self, monkeypatch: pytest.MonkeyPatch
|
||||
|
|
@ -154,24 +142,6 @@ class TestVertexAIVideoConfig:
|
|||
assert model == "veo-3.1-lite-generate-001"
|
||||
assert custom_llm_provider == "vertex_ai"
|
||||
|
||||
def test_veo_31_lite_cost_uses_resolution_tiers(self):
|
||||
model_cost = _load_model_cost_map(BACKUP_MODEL_COST_PATH)
|
||||
model_info = model_cost[VEO_31_LITE_VERTEX_MODEL]
|
||||
|
||||
assert video_generation_cost(
|
||||
model=VEO_31_LITE_VERTEX_MODEL,
|
||||
duration_seconds=10.0,
|
||||
custom_llm_provider="vertex_ai",
|
||||
model_info=dict(model_info),
|
||||
video_resolution="720p",
|
||||
) == pytest.approx(0.5)
|
||||
assert video_generation_cost(
|
||||
model=VEO_31_LITE_VERTEX_MODEL,
|
||||
duration_seconds=10.0,
|
||||
custom_llm_provider="vertex_ai",
|
||||
model_info=dict(model_info),
|
||||
video_resolution="1080p",
|
||||
) == pytest.approx(0.8)
|
||||
|
||||
def test_transform_video_create_request(self):
|
||||
"""Test transformation of video creation request."""
|
||||
|
|
|
|||
|
|
@ -105,16 +105,3 @@ def test_both_cost_maps_agree_on_the_redirected_slugs():
|
|||
backup = json.loads(BACKUP_PRICES_PATH.read_text(encoding="utf-8"))
|
||||
for slug in (*REDIRECTED_SLUGS, *CODE_SLUGS, REDIRECT_TARGET, CODE_REDIRECT_TARGET):
|
||||
assert prices[slug] == backup[slug], slug
|
||||
|
||||
|
||||
def test_every_retired_chat_slug_is_covered(cost_map: dict):
|
||||
"""The lists above must stay in step with what the registry marks retired."""
|
||||
marked = {
|
||||
key
|
||||
for key, entry in cost_map.items()
|
||||
if isinstance(entry, dict)
|
||||
and entry.get("litellm_provider") == "xai"
|
||||
and "deprecation_date" in entry
|
||||
and entry.get("mode") == "chat"
|
||||
}
|
||||
assert marked == {*REDIRECTED_SLUGS, *CODE_SLUGS}
|
||||
|
|
|
|||
|
|
@ -55,34 +55,6 @@ def test_zai_in_provider_lists():
|
|||
assert "zai" in litellm.provider_list
|
||||
|
||||
|
||||
def test_zai_glm46_cost_calculation(local_model_cost_map):
|
||||
"""Test the cost calculation for glm-4.6"""
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model="zai/glm-4.6",
|
||||
prompt_tokens=1000000, # 1M tokens
|
||||
completion_tokens=1000000,
|
||||
)
|
||||
|
||||
# GLM-4.6: $0.6/M input, $2.2/M output
|
||||
assert math.isclose(prompt_cost, 0.6, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, 2.2, rel_tol=1e-6)
|
||||
|
||||
|
||||
def test_glm47_cost_calculation(local_model_cost_map):
|
||||
"""Test cost calculation for GLM-4.7"""
|
||||
|
||||
prompt_cost, completion_cost = cost_per_token(
|
||||
model="zai/glm-4.7",
|
||||
prompt_tokens=1000000, # 1M tokens
|
||||
completion_tokens=1000000,
|
||||
)
|
||||
|
||||
# GLM-4.7: $0.6/M input, $2.2/M output (same as GLM-4.6)
|
||||
assert math.isclose(prompt_cost, 0.6, rel_tol=1e-6)
|
||||
assert math.isclose(completion_cost, 2.2, rel_tol=1e-6)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_zai_completion_call(respx_mock, zai_response, monkeypatch):
|
||||
"""Test completion call with zai provider using mocked response"""
|
||||
|
|
|
|||
|
|
@ -7,31 +7,6 @@ from litellm.proxy.common_utils.prompt_cache_pricing import price_cache_tokens
|
|||
from litellm.types.management_endpoints.prompt_cache_prediction import CacheTokenBuckets
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("model", "expected"),
|
||||
[("anthropic/claude-sonnet-4-5", 1.26), ("anthropic/claude-sonnet-4-6", 0.63)],
|
||||
)
|
||||
def test_prices_all_cache_buckets_at_total_context_tier(model: str, expected: float) -> None:
|
||||
tokens: Final = CacheTokenBuckets(
|
||||
uncached_input_tokens=100_000,
|
||||
cache_read_input_tokens=50_000,
|
||||
cache_creation_5m_input_tokens=20_000,
|
||||
cache_creation_1h_input_tokens=40_000,
|
||||
)
|
||||
assert price_cache_tokens(model, "unconfigured-deployment", tokens) == pytest.approx(expected)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(("total", "expected"), [(200_000, 0.387), (200_001, 0.774006)])
|
||||
def test_long_context_tier_starts_above_threshold(total: int, expected: float) -> None:
|
||||
tokens: Final = CacheTokenBuckets(
|
||||
uncached_input_tokens=total - 100_000,
|
||||
cache_creation_1h_input_tokens=10_000,
|
||||
cache_read_input_tokens=90_000,
|
||||
)
|
||||
actual: Final = price_cache_tokens("anthropic/claude-sonnet-4-5", "unconfigured-deployment", tokens)
|
||||
assert actual == pytest.approx(expected)
|
||||
|
||||
|
||||
def test_deployment_tariff_wins_without_proxy_discounts_or_margins(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.model_cost.copy())
|
||||
litellm.Router(
|
||||
|
|
|
|||
|
|
@ -110,54 +110,6 @@ async def _observe(
|
|||
await cache.async_set_cache(_cache_key(scope, prefix.fingerprint), observation.model_dump_json(), ttl=3_600)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(("ttl", "cold_cost"), [("5m", 0.0145), ("1h", 0.022)])
|
||||
async def test_unobserved_cache_prices_cold_and_warm_bounds(ttl: str, cold_cost: float) -> None:
|
||||
body: Final = _body(ttl)
|
||||
arm: Final = await endpoint.predict_arm(_deployment(), body, _prefix(body), _CALLER, DualCache(), Counts())
|
||||
|
||||
assert arm.cache_state == "unknown"
|
||||
assert arm.reason == "no_compatible_observation"
|
||||
assert arm.evidence is None
|
||||
assert arm.estimate is not None and arm.cold is not None and arm.warm is not None
|
||||
assert arm.estimate.input_cost == pytest.approx(cold_cost)
|
||||
assert arm.cold.input_cost == pytest.approx(cold_cost)
|
||||
assert arm.warm.input_cost == pytest.approx(0.003)
|
||||
assert arm.cold.tokens.uncached_input_tokens == 1_000
|
||||
assert arm.cold.tokens.cache_read_input_tokens == 0
|
||||
assert arm.cold.tokens.cache_creation_5m_input_tokens == (5_000 if ttl == "5m" else 0)
|
||||
assert arm.cold.tokens.cache_creation_1h_input_tokens == (5_000 if ttl == "1h" else 0)
|
||||
assert arm.warm.tokens.cache_read_input_tokens == 5_000
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
("cached_tokens", "warm_cost", "cold_cost"), [(5_400, 0.00228, 0.0147), (4_600, 0.00372, 0.0143)]
|
||||
)
|
||||
@pytest.mark.parametrize("expired", [False, True])
|
||||
async def test_exact_prefix_conserves_total_with_observed_count_in_all_scenarios(
|
||||
cached_tokens: int, warm_cost: float, cold_cost: float, expired: bool
|
||||
) -> None:
|
||||
cache: Final = DualCache()
|
||||
body: Final = _body()
|
||||
await _observe(cache, body, cached_tokens=cached_tokens, expired=expired)
|
||||
arm: Final = await endpoint.predict_arm(_deployment(), body, _prefix(body), _CALLER, cache, Counts())
|
||||
|
||||
assert arm.cache_state == ("stale" if expired else "warm")
|
||||
assert arm.evidence is not None
|
||||
assert arm.estimate is not None and arm.warm is not None and arm.cold is not None
|
||||
assert arm.warm.tokens.cache_read_input_tokens == cached_tokens
|
||||
assert arm.warm.tokens.cache_creation_5m_input_tokens == 0
|
||||
assert arm.cold.tokens.cache_creation_5m_input_tokens == cached_tokens
|
||||
assert arm.cold.tokens.cache_read_input_tokens == 0
|
||||
for scenario in (arm.estimate, arm.cold, arm.warm):
|
||||
assert scenario.tokens.total_tokens == 6_000
|
||||
assert scenario.tokens.uncached_input_tokens == 6_000 - cached_tokens
|
||||
assert arm.warm.input_cost == pytest.approx(warm_cost)
|
||||
assert arm.cold.input_cost == pytest.approx(cold_cost)
|
||||
assert arm.estimate.input_cost == pytest.approx(cold_cost if expired else warm_cost)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_observed_prefix_larger_than_full_request_returns_unknown() -> None:
|
||||
cache: Final = DualCache()
|
||||
|
|
@ -170,22 +122,6 @@ async def test_observed_prefix_larger_than_full_request_returns_unknown() -> Non
|
|||
assert arm.estimate is None and arm.cold is None and arm.warm is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(("ttl", "expected"), [("5m", 0.0053), ("1h", 0.0068)])
|
||||
async def test_append_only_prefix_reads_old_tokens_and_writes_extension(ttl: str, expected: float) -> None:
|
||||
cache: Final = DualCache()
|
||||
await _observe(cache, _body(ttl), cached_tokens=4_000)
|
||||
body: Final = _body(ttl, extended=True)
|
||||
arm: Final = await endpoint.predict_arm(_deployment(), body, _prefix(body), _CALLER, cache, Counts())
|
||||
|
||||
assert arm.cache_state == "partial"
|
||||
assert arm.estimate is not None
|
||||
assert arm.estimate.tokens.cache_read_input_tokens == 4_000
|
||||
assert arm.estimate.tokens.cache_creation_5m_input_tokens == (1_000 if ttl == "5m" else 0)
|
||||
assert arm.estimate.tokens.cache_creation_1h_input_tokens == (1_000 if ttl == "1h" else 0)
|
||||
assert arm.estimate.input_cost == pytest.approx(expected)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_expired_observation_estimates_a_cold_rebuild() -> None:
|
||||
cache: Final = DualCache()
|
||||
|
|
@ -202,22 +138,6 @@ async def test_expired_observation_estimates_a_cold_rebuild() -> None:
|
|||
assert arm.estimate.input_cost == arm.cold.input_cost
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_below_model_minimum_prices_all_input_as_uncached() -> None:
|
||||
body: Final = _body()
|
||||
arm: Final = await endpoint.predict_arm(
|
||||
_deployment(), body, _prefix(body), _CALLER, DualCache(), Counts(total=1_500, prefix=1_000)
|
||||
)
|
||||
|
||||
assert arm.cache_state == "disabled"
|
||||
assert arm.reason == "below_cache_minimum"
|
||||
assert arm.estimate is not None
|
||||
assert arm.estimate.tokens.uncached_input_tokens == 1_500
|
||||
assert arm.estimate.tokens.cache_read_input_tokens == 0
|
||||
assert arm.estimate.tokens.cache_creation_5m_input_tokens == 0
|
||||
assert arm.estimate.input_cost == pytest.approx(0.003)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("counts", [Counts(total=None), Counts(prefix=None), Counts(total=4_000)])
|
||||
async def test_unavailable_or_inconsistent_token_counts_return_null_estimates(counts: Counts) -> None:
|
||||
|
|
@ -269,20 +189,6 @@ async def test_custom_api_base_from_environment_returns_unknown_before_counting(
|
|||
assert arm.estimate is None and arm.cold is None and arm.warm is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_explicit_official_api_base_overrides_custom_environment(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setenv("ANTHROPIC_API_BASE", "https://custom.invalid")
|
||||
body: Final = _body()
|
||||
arm: Final = await endpoint.predict_arm(
|
||||
_deployment(api_base="https://api.anthropic.com"), body, _prefix(body), _CALLER, DualCache(), Counts()
|
||||
)
|
||||
|
||||
assert arm.cache_state == "unknown"
|
||||
assert arm.reason == "no_compatible_observation"
|
||||
assert arm.estimate is not None
|
||||
assert arm.estimate.input_cost == pytest.approx(0.0145)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _ProxyLogging:
|
||||
internal_usage_cache: InternalUsageCache
|
||||
|
|
@ -343,38 +249,6 @@ async def _post(
|
|||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize(
|
||||
("warm_deployment", "warm_model", "expected_delta", "expected_penalty"),
|
||||
[("sonnet", "claude-sonnet-5", -0.03325, 0.0), ("opus", "claude-opus-5", 0.007, 0.0115)],
|
||||
)
|
||||
async def test_switch_delta_accounts_for_each_deployment_cache(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
warm_deployment: str,
|
||||
warm_model: str,
|
||||
expected_delta: float,
|
||||
expected_penalty: float,
|
||||
) -> None:
|
||||
cache: Final = DualCache()
|
||||
body: Final = _body()
|
||||
await _observe(cache, body, deployment_id=warm_deployment, model=warm_model)
|
||||
app: Final = _app(monkeypatch, cache, caller=UserAPIKeyAuth(api_key=_CALLER))
|
||||
response: Final = await _post(app, body)
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
result: Final = CachePredictionResponse.model_validate(response.json())
|
||||
assert result.switch_delta == pytest.approx(expected_delta)
|
||||
assert result.cache_rebuild_penalty == pytest.approx(expected_penalty)
|
||||
assert result.cache_guarantee is False
|
||||
assert result.pricing_basis == "input_before_discounts_and_margins"
|
||||
if warm_deployment == "sonnet":
|
||||
assert result.switch.cache_state == "warm"
|
||||
assert result.stay.cache_state == "unknown"
|
||||
else:
|
||||
assert result.stay.cache_state == "warm"
|
||||
assert result.switch.cache_state == "unknown"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_missing_caller_identity_cannot_reuse_observations(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
cache: Final = DualCache()
|
||||
|
|
@ -568,53 +442,6 @@ async def test_each_count_preserves_auth_cached_request_tag_limits(
|
|||
assert calls.get_nowait() == "claude-opus-5"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_provider_counter_failure_releases_parallel_capacity(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
cache: Final = DualCache()
|
||||
limiter: Final = _PROXY_MaxParallelRequestsHandler_v3(InternalUsageCache(cache))
|
||||
caller: Final = UserAPIKeyAuth(api_key=_CALLER, max_parallel_requests=1)
|
||||
|
||||
async def fail_count(model: str, api_key: str, body: Mapping[str, JsonValue]) -> int | None:
|
||||
raise RuntimeError("provider counter failed")
|
||||
|
||||
app: Final = _app(monkeypatch, cache, caller=caller, counts=fail_count, limiter=limiter)
|
||||
with pytest.raises(RuntimeError, match="provider counter failed"):
|
||||
await _post(app, _body())
|
||||
recovered: Final = await _post(_app(monkeypatch, cache, caller=caller, limiter=limiter), _body())
|
||||
assert recovered.status_code == 200, recovered.text
|
||||
assert recovered.json()["switch"]["estimate"]["input_cost"] == pytest.approx(0.0145)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_cancelled_provider_counter_releases_parallel_capacity(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
cache: Final = DualCache()
|
||||
limiter: Final = _PROXY_MaxParallelRequestsHandler_v3(InternalUsageCache(cache))
|
||||
caller: Final = UserAPIKeyAuth(api_key=_CALLER, max_parallel_requests=1)
|
||||
started: Final = asyncio.Event()
|
||||
release: Final = asyncio.Event()
|
||||
|
||||
async def wait_count(model: str, api_key: str, body: Mapping[str, JsonValue]) -> int | None:
|
||||
started.set()
|
||||
await release.wait()
|
||||
return await Counts()(model, api_key, body)
|
||||
|
||||
app: Final = _app(monkeypatch, cache, caller=caller, counts=wait_count, limiter=limiter)
|
||||
pending: Final = asyncio.create_task(_post(app, _body()))
|
||||
try:
|
||||
await asyncio.wait_for(started.wait(), timeout=5)
|
||||
pending.cancel()
|
||||
with pytest.raises(asyncio.CancelledError):
|
||||
await pending
|
||||
release.set()
|
||||
recovered: Final = await asyncio.wait_for(_post(app, _body()), timeout=5)
|
||||
assert recovered.status_code == 200, recovered.text
|
||||
assert recovered.json()["switch"]["estimate"]["input_cost"] == pytest.approx(0.0145)
|
||||
finally:
|
||||
pending.cancel()
|
||||
release.set()
|
||||
await asyncio.gather(pending, return_exceptions=True)
|
||||
|
||||
|
||||
async def _unexpected_count(model: str, api_key: str, body: Mapping[str, JsonValue]) -> int | None:
|
||||
pytest.fail("Unsupported prediction must return before contacting the token counter")
|
||||
|
||||
|
|
|
|||
|
|
@ -2151,96 +2151,6 @@ async def test_proxy_only_error_5xx_keeps_traceback_and_runs_sync_callbacks(monk
|
|||
assert "test_proxy_utils" in captured["async_traceback"]
|
||||
|
||||
|
||||
def test_create_model_info_response_resolves_alias_to_deployment_model():
|
||||
"""A public model name that is not itself a cost-map key must not be resolved through
|
||||
the fallback-generalization rules: `bedrock-claude-opus-5` matches the generic
|
||||
claude-family baseline (200k/64k) by substring, while the deployment it fronts really
|
||||
accepts 1M/128k. Regression for the /v1/models alias resolution introduced in v1.94.0."""
|
||||
from litellm import Router
|
||||
|
||||
saved_model_cost = dict(litellm.model_cost)
|
||||
try:
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "bedrock-claude-opus-5",
|
||||
"litellm_params": {
|
||||
"custom_llm_provider": "bedrock",
|
||||
"model": "bedrock/eu.anthropic.claude-opus-5",
|
||||
},
|
||||
"model_info": {"base_model": "eu.anthropic.claude-opus-5"},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
response = create_model_info_response(
|
||||
model_id="bedrock-claude-opus-5", provider="openai", llm_router=router
|
||||
)
|
||||
finally:
|
||||
litellm.model_cost.clear()
|
||||
litellm.model_cost.update(saved_model_cost)
|
||||
|
||||
assert response["max_input_tokens"] == 1000000
|
||||
assert response["max_output_tokens"] == 128000
|
||||
|
||||
|
||||
def test_create_model_info_response_keeps_exact_alias_over_generalized_deployment_model():
|
||||
"""Mirror of the alias bug: when the deployment points at a custom backend name that
|
||||
only matches a generalization rule, the listed name's exact cost-map entry is the
|
||||
better answer and must win."""
|
||||
from litellm import Router
|
||||
|
||||
saved_model_cost = dict(litellm.model_cost)
|
||||
try:
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "claude-opus-5",
|
||||
"litellm_params": {
|
||||
"custom_llm_provider": "bedrock",
|
||||
"model": "bedrock/my-claude-opus-5-provisioned",
|
||||
},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
response = create_model_info_response(
|
||||
model_id="claude-opus-5", provider="openai", llm_router=router
|
||||
)
|
||||
finally:
|
||||
litellm.model_cost.clear()
|
||||
litellm.model_cost.update(saved_model_cost)
|
||||
|
||||
assert response["max_input_tokens"] == 1000000
|
||||
|
||||
|
||||
def test_create_model_info_response_falls_back_to_alias_for_opaque_deployment_name():
|
||||
"""An Azure deployment named after the resource rather than the model has no cost-map
|
||||
entry; the listed name still does, and must keep answering."""
|
||||
from litellm import Router
|
||||
|
||||
saved_model_cost = dict(litellm.model_cost)
|
||||
try:
|
||||
router = Router(
|
||||
model_list=[
|
||||
{
|
||||
"model_name": "gpt-4o",
|
||||
"litellm_params": {"model": "azure/my-gpt4o-deployment"},
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
response = create_model_info_response(
|
||||
model_id="gpt-4o", provider="openai", llm_router=router
|
||||
)
|
||||
finally:
|
||||
litellm.model_cost.clear()
|
||||
litellm.model_cost.update(saved_model_cost)
|
||||
|
||||
assert response["max_input_tokens"] == 128000
|
||||
assert response["max_output_tokens"] == 16384
|
||||
|
||||
|
||||
def test_create_model_info_response_resolves_mode_through_deployment_model():
|
||||
"""`mode` is derived from the same lookup, so an aliased embedding deployment
|
||||
currently reports no mode at all; it must report `embedding`."""
|
||||
|
|
|
|||
|
|
@ -203,164 +203,6 @@ def test_cost_calculator_with_usage(_local_model_cost_map, monkeypatch):
|
|||
assert result == expected_cost, f"Got {result}, Expected {expected_cost}"
|
||||
|
||||
|
||||
def test_transcription_cost_uses_token_pricing(_local_model_cost_map):
|
||||
from litellm import completion_cost
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=14,
|
||||
completion_tokens=45,
|
||||
total_tokens=59,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=0, audio_tokens=14),
|
||||
)
|
||||
response = TranscriptionResponse(text="demo text")
|
||||
response.usage = usage
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
model="gpt-4o-transcribe",
|
||||
custom_llm_provider="openai",
|
||||
call_type="atranscription",
|
||||
)
|
||||
|
||||
expected_cost = (14 * 2.5e-06) + (45 * 1e-05)
|
||||
assert pytest.approx(cost, rel=1e-6) == expected_cost
|
||||
|
||||
|
||||
def test_transcription_token_pricing_is_provider_aware(_local_model_cost_map):
|
||||
"""Regression: the token-priced transcription path hardcoded provider openai,
|
||||
so gemini transcription models raised "This model isn't mapped yet"."""
|
||||
from litellm import completion_cost
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=200,
|
||||
completion_tokens=10,
|
||||
total_tokens=210,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=1, audio_tokens=199),
|
||||
)
|
||||
response = TranscriptionResponse(text="demo text")
|
||||
response.usage = usage
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
model="gemini/gemini-3.5-transcribe",
|
||||
custom_llm_provider="gemini",
|
||||
call_type="atranscription",
|
||||
)
|
||||
|
||||
expected_cost = (199 * 2e-06) + (1 * 2e-06) + (10 * 1.2e-05)
|
||||
assert pytest.approx(cost, rel=1e-6) == expected_cost
|
||||
|
||||
|
||||
def test_transcription_cost_falls_back_to_duration(_local_model_cost_map):
|
||||
from litellm import completion_cost
|
||||
|
||||
response = TranscriptionResponse(text="demo text")
|
||||
response.duration = 10.0
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
model="whisper-1",
|
||||
custom_llm_provider="openai",
|
||||
call_type="atranscription",
|
||||
)
|
||||
|
||||
expected_cost = 10.0 * 0.0001
|
||||
assert pytest.approx(cost, rel=1e-6) == expected_cost
|
||||
|
||||
|
||||
def test_vertex_chirp_3_transcription_cost_from_duration(_local_model_cost_map):
|
||||
"""Regression: the chirp_3 cost map entry shipped with output_cost_per_second 0.0,
|
||||
and cost_per_second prefers output_cost_per_second whenever it is not None, so
|
||||
every transcription priced to $0.00 instead of using input_cost_per_second."""
|
||||
from litellm import completion_cost
|
||||
|
||||
response = TranscriptionResponse(text="demo text")
|
||||
response.duration = 18.0
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=response,
|
||||
model="vertex_ai/chirp_3",
|
||||
custom_llm_provider="vertex_ai",
|
||||
call_type="atranscription",
|
||||
)
|
||||
|
||||
expected_cost = 18.0 * 0.00026667
|
||||
assert cost > 0
|
||||
assert pytest.approx(cost, rel=1e-6) == expected_cost
|
||||
|
||||
|
||||
def test_handle_realtime_stream_cost_calculation():
|
||||
from litellm.cost_calculator import RealtimeAPITokenUsageProcessor
|
||||
|
||||
# Setup test data
|
||||
results: OpenAIRealtimeStreamList = [
|
||||
{"type": "session.created", "session": {"model": "gpt-3.5-turbo"}},
|
||||
{
|
||||
"type": "response.done",
|
||||
"response": {"usage": {"input_tokens": 100, "output_tokens": 50, "total_tokens": 150}},
|
||||
},
|
||||
{
|
||||
"type": "response.done",
|
||||
"response": {
|
||||
"usage": {
|
||||
"input_tokens": 200,
|
||||
"output_tokens": 100,
|
||||
"total_tokens": 300,
|
||||
}
|
||||
},
|
||||
},
|
||||
]
|
||||
|
||||
combined_usage_object = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(
|
||||
results=results,
|
||||
)
|
||||
|
||||
# Test with explicit model name
|
||||
cost = handle_realtime_stream_cost_calculation(
|
||||
results=results,
|
||||
combined_usage_object=combined_usage_object,
|
||||
custom_llm_provider="openai",
|
||||
litellm_model_name="gpt-3.5-turbo",
|
||||
)
|
||||
|
||||
# Calculate expected cost
|
||||
# gpt-3.5-turbo costs: $0.0015/1K tokens input, $0.002/1K tokens output
|
||||
expected_cost = (300 * 0.0015 / 1000) + ( # input tokens (100 + 200)
|
||||
150 * 0.002 / 1000
|
||||
) # output tokens (50 + 100)
|
||||
assert abs(cost - expected_cost) <= 0.00075 # Allow small floating point differences
|
||||
|
||||
# Test with different model name in session
|
||||
results[0]["session"]["model"] = "gpt-4"
|
||||
|
||||
cost = handle_realtime_stream_cost_calculation(
|
||||
results=results,
|
||||
combined_usage_object=combined_usage_object,
|
||||
custom_llm_provider="openai",
|
||||
litellm_model_name="gpt-3.5-turbo",
|
||||
)
|
||||
|
||||
# Calculate expected cost using gpt-4 rates
|
||||
# gpt-4 costs: $0.03/1K tokens input, $0.06/1K tokens output
|
||||
expected_cost = (300 * 0.03 / 1000) + ( # input tokens
|
||||
150 * 0.06 / 1000
|
||||
) # output tokens
|
||||
assert abs(cost - expected_cost) < 0.00076
|
||||
|
||||
# Test with no response.done events
|
||||
results = [{"type": "session.created", "session": {"model": "gpt-3.5-turbo"}}]
|
||||
combined_usage_object = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(
|
||||
results=results,
|
||||
)
|
||||
cost = handle_realtime_stream_cost_calculation(
|
||||
results=results,
|
||||
combined_usage_object=combined_usage_object,
|
||||
custom_llm_provider="openai",
|
||||
litellm_model_name="gpt-3.5-turbo",
|
||||
)
|
||||
assert cost == 0.0 # No usage, no cost
|
||||
|
||||
|
||||
def test_handle_realtime_stream_cost_calculation_stores_cost_breakdown():
|
||||
"""Regression: realtime cost must populate logging_obj.cost_breakdown so the
|
||||
spend logs / UI show input vs output cost (issue: cost_breakdown was None for
|
||||
|
|
@ -557,101 +399,6 @@ def test_realtime_logging_object_does_not_validate_unknown_event_types():
|
|||
assert len(dumped["results"]) == len(results)
|
||||
|
||||
|
||||
def test_realtime_transcription_duration_cost(monkeypatch):
|
||||
"""
|
||||
gpt-realtime-whisper transcription sessions are billed by input audio duration
|
||||
($0.017/min). The .completed events carry usage {type: duration, seconds: N};
|
||||
cost must equal total_seconds * input_cost_per_second.
|
||||
"""
|
||||
from datetime import datetime
|
||||
|
||||
from litellm.litellm_core_utils.litellm_logging import Logging
|
||||
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
from litellm.cost_calculator import RealtimeAPITokenUsageProcessor
|
||||
|
||||
results: OpenAIRealtimeStreamList = [
|
||||
{
|
||||
"type": "session.created",
|
||||
"session": {
|
||||
"type": "transcription",
|
||||
"audio": {"input": {"transcription": {"model": "gpt-realtime-whisper"}}},
|
||||
},
|
||||
},
|
||||
{
|
||||
"type": "conversation.item.input_audio_transcription.completed",
|
||||
"transcript": "hello",
|
||||
"usage": {"type": "duration", "seconds": 60.0},
|
||||
},
|
||||
{
|
||||
"type": "conversation.item.input_audio_transcription.completed",
|
||||
"transcript": "world",
|
||||
"usage": {"type": "duration", "seconds": 30.0},
|
||||
},
|
||||
]
|
||||
|
||||
combined = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(results=results)
|
||||
logging_obj = Logging(
|
||||
model="gpt-realtime-whisper",
|
||||
messages=[],
|
||||
stream=False,
|
||||
call_type="_arealtime",
|
||||
start_time=datetime.now(),
|
||||
litellm_call_id="realtime-transcription-cost-breakdown-test",
|
||||
function_id="realtime-transcription-cost-breakdown-test",
|
||||
)
|
||||
cost = handle_realtime_stream_cost_calculation(
|
||||
results=results,
|
||||
combined_usage_object=combined,
|
||||
custom_llm_provider="openai",
|
||||
litellm_model_name="gpt-realtime-whisper",
|
||||
litellm_logging_obj=logging_obj,
|
||||
)
|
||||
|
||||
# 90 seconds at $0.017/minute.
|
||||
expected = 90.0 * (0.017 / 60)
|
||||
assert abs(cost - expected) < 1e-9
|
||||
assert cost > 0 # guards against the duration branch being dropped
|
||||
assert logging_obj.cost_breakdown is not None
|
||||
assert abs(logging_obj.cost_breakdown["total_cost"] - cost) < 1e-9
|
||||
|
||||
# The transcription cost must be attributed in the breakdown, not just folded
|
||||
# into total_cost, or input_cost + output_cost + additional_costs won't sum to total_cost.
|
||||
additional_costs = logging_obj.cost_breakdown.get("additional_costs")
|
||||
assert additional_costs is not None
|
||||
assert abs(additional_costs["transcription_cost"] - expected) < 1e-9
|
||||
attributed_total = (
|
||||
logging_obj.cost_breakdown["input_cost"]
|
||||
+ logging_obj.cost_breakdown["output_cost"]
|
||||
+ additional_costs["transcription_cost"]
|
||||
)
|
||||
assert abs(attributed_total - logging_obj.cost_breakdown["total_cost"]) < 1e-9
|
||||
|
||||
|
||||
def test_realtime_transcription_duration_cost_resolves_model_from_litellm_name(
|
||||
monkeypatch,
|
||||
):
|
||||
"""When no session event carries the ASR model, the litellm_model_name is used."""
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
results: OpenAIRealtimeStreamList = [
|
||||
{
|
||||
"type": "conversation.item.input_audio_transcription.completed",
|
||||
"usage": {"type": "duration", "seconds": 120.0},
|
||||
},
|
||||
]
|
||||
cost = handle_realtime_stream_cost_calculation(
|
||||
results=results,
|
||||
combined_usage_object=Usage(),
|
||||
custom_llm_provider="azure",
|
||||
litellm_model_name="azure/gpt-realtime-whisper",
|
||||
)
|
||||
assert abs(cost - 120.0 * (0.017 / 60)) < 1e-9
|
||||
|
||||
|
||||
def test_realtime_transcription_no_completed_events_is_zero(monkeypatch):
|
||||
"""A realtime stream without transcription completed events adds no extra cost."""
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
|
|
@ -673,35 +420,6 @@ def test_realtime_transcription_no_completed_events_is_zero(monkeypatch):
|
|||
)
|
||||
|
||||
|
||||
def test_realtime_transcription_token_billed_fallback(monkeypatch):
|
||||
"""
|
||||
Token-billed transcription models price by audio/text tokens. Verify the
|
||||
fallback path multiplies audio tokens by the model's audio token cost.
|
||||
"""
|
||||
monkeypatch.setenv("LITELLM_LOCAL_MODEL_COST_MAP", "True")
|
||||
monkeypatch.setattr(litellm, "model_cost", litellm.get_model_cost_map(url=""))
|
||||
|
||||
from litellm.cost_calculator import _transcription_usage_cost
|
||||
|
||||
# gpt-4o-transcribe: input_cost_per_audio_token = 2.5e-06, input_cost_per_token = 2.5e-06,
|
||||
# output_cost_per_token = 1e-05
|
||||
model_info = litellm.get_model_info(model="gpt-4o-transcribe", custom_llm_provider="openai")
|
||||
usage = {
|
||||
"type": "tokens",
|
||||
"input_tokens": 40,
|
||||
"output_tokens": 10,
|
||||
"total_tokens": 50,
|
||||
"input_token_details": {"audio_tokens": 30, "text_tokens": 10},
|
||||
}
|
||||
cost = _transcription_usage_cost(usage, model_info)
|
||||
expected = (
|
||||
30 * 2.5e-06 # audio tokens
|
||||
+ 10 * 2.5e-06 # text tokens
|
||||
+ 10 * 1e-05 # output tokens
|
||||
)
|
||||
assert abs(cost - expected) < 1e-12
|
||||
|
||||
|
||||
def test_transcription_usage_cost_returns_zero_for_unknown_type():
|
||||
"""An unrecognized usage type yields 0 (safe fallback, no exception)."""
|
||||
from litellm.cost_calculator import _transcription_usage_cost
|
||||
|
|
@ -1290,78 +1008,6 @@ def test_bedrock_cost_calculator_comparison_with_without_cache():
|
|||
print(f"Cost with cache: {cost_with_cache}")
|
||||
|
||||
|
||||
def test_gemini_25_implicit_caching_cost():
|
||||
"""
|
||||
Test that Gemini 2.5 models correctly calculate costs with implicit caching.
|
||||
|
||||
This test reproduces the issue from #11156 where cached tokens should receive
|
||||
a 75% discount.
|
||||
"""
|
||||
from litellm import completion_cost
|
||||
from litellm.types.utils import (
|
||||
Choices,
|
||||
Message,
|
||||
ModelResponse,
|
||||
PromptTokensDetailsWrapper,
|
||||
Usage,
|
||||
)
|
||||
|
||||
# Create a mock response similar to the one in the issue
|
||||
litellm_model_response = ModelResponse(
|
||||
id="test-response",
|
||||
created=1750733889,
|
||||
model="gemini/gemini-2.5-flash",
|
||||
object="chat.completion",
|
||||
system_fingerprint=None,
|
||||
choices=[
|
||||
Choices(
|
||||
finish_reason="stop",
|
||||
index=0,
|
||||
message=Message(
|
||||
content="Understood. This is a test message to check the response from the Gemini model.",
|
||||
role="assistant",
|
||||
tool_calls=None,
|
||||
function_call=None,
|
||||
),
|
||||
)
|
||||
],
|
||||
usage=Usage(
|
||||
total_tokens=15050,
|
||||
prompt_tokens=15033,
|
||||
completion_tokens=17,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
audio_tokens=None,
|
||||
cached_tokens=14316, # This is cachedContentTokenCount from Gemini
|
||||
),
|
||||
completion_tokens_details=None,
|
||||
),
|
||||
)
|
||||
|
||||
# Calculate the cost
|
||||
result = completion_cost(
|
||||
completion_response=litellm_model_response,
|
||||
model="gemini/gemini-2.5-flash",
|
||||
)
|
||||
|
||||
# Current pricing for gemini/gemini-2.5-flash:
|
||||
# input: $0.30 / 1M tokens (3e-07 per token)
|
||||
# cache_read: $0.03 / 1M tokens (3e-08 per token)
|
||||
# output: $2.50 / 1M tokens (2.5e-06 per token)
|
||||
|
||||
# Breakdown:
|
||||
# - Cached tokens: 14316 * 3e-08 = 0.00042948
|
||||
# - Non-cached tokens: (15033-14316) * 3e-07 = 717 * 3e-07 = 0.00021510
|
||||
# - Output tokens: 17 * 2.5e-06 = 0.00004250
|
||||
# Total: 0.00042948 + 0.00021510 + 0.00004250 = 0.00068708
|
||||
|
||||
expected_cost = 0.00068708
|
||||
|
||||
# Allow for small floating point differences
|
||||
assert abs(result - expected_cost) < 1e-8, f"Expected cost {expected_cost}, but got {result}"
|
||||
|
||||
print(f"✓ Gemini 2.5 implicit caching cost calculation is correct: ${result:.8f}")
|
||||
|
||||
|
||||
def test_log_context_cost_calculation():
|
||||
"""
|
||||
Test that log context cost calculation works correctly with tiered pricing.
|
||||
|
|
@ -3730,31 +3376,6 @@ def test_combine_usage_objects_sums_mirrored_cache_write_fields_once():
|
|||
assert combined_pair.prompt_tokens_details.cache_creation_tokens == 100
|
||||
|
||||
|
||||
def test_completion_cost_prices_anthropic_shaped_cache_read_tokens(_local_model_cost_map):
|
||||
"""Regression: an Anthropic /v1/messages response reports cache reads as top-level
|
||||
cache_read_input_tokens with input_tokens excluding them. Reading that usage as
|
||||
Responses API usage dropped the cache tokens and billed the whole prompt at the
|
||||
uncached input rate, overstating spend on cache hits."""
|
||||
|
||||
response = {
|
||||
"id": "msg_1",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"model": "gpt-5.6-sol",
|
||||
"stop_reason": "end_turn",
|
||||
"content": [{"type": "text", "text": "1"}],
|
||||
"usage": {"input_tokens": 3, "output_tokens": 5, "cache_read_input_tokens": 4014},
|
||||
}
|
||||
|
||||
cost = litellm.completion_cost(
|
||||
completion_response=response,
|
||||
model="gpt-5.6-sol",
|
||||
custom_llm_provider="openai",
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(3 * 4e-6 + 4014 * 4e-7 + 5 * 2e-5, rel=1e-9)
|
||||
|
||||
|
||||
def _together_chat_response(
|
||||
model: str, prompt_tokens: int, completion_tokens: int, cached_tokens: int
|
||||
) -> ModelResponse:
|
||||
|
|
@ -3773,60 +3394,6 @@ def _together_chat_response(
|
|||
)
|
||||
|
||||
|
||||
def test_completion_cost_prices_together_cached_tokens_at_cache_read_rate(_local_model_cost_map):
|
||||
"""Regression: Together reports prompt_tokens_details.cached_tokens but no together_ai
|
||||
registry entry carried cache_read_input_token_cost, so cache-hit tokens were priced at
|
||||
0.0 and spend on cache-heavy workloads was understated."""
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=_together_chat_response(
|
||||
model="deepseek-ai/DeepSeek-V4-Flash-0731", prompt_tokens=7864, completion_tokens=16, cached_tokens=7863
|
||||
),
|
||||
custom_llm_provider="together_ai",
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(1 * 1.4e-07 + 7863 * 3e-08 + 16 * 2.8e-07, rel=1e-9)
|
||||
|
||||
|
||||
def test_completion_cost_together_mapped_model_skips_size_bucket(_local_model_cost_map):
|
||||
"""Regression: any together model whose name matches (\\d+b) was rewritten to a
|
||||
together-ai-* size bucket before the registry lookup, so mapped models like
|
||||
Muse-Glimmer-30B never used their per-model rates, cache fields included."""
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=_together_chat_response(
|
||||
model="meta-models/Muse-Glimmer-30B", prompt_tokens=63, completion_tokens=16, cached_tokens=0
|
||||
),
|
||||
custom_llm_provider="together_ai",
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(63 * 3.5e-07 + 16 * 1.5e-06, rel=1e-9)
|
||||
|
||||
|
||||
def test_completion_cost_together_unmapped_model_still_uses_size_bucket(_local_model_cost_map):
|
||||
cost = completion_cost(
|
||||
completion_response=_together_chat_response(
|
||||
model="qwen/Qwen2-72B-Instruct", prompt_tokens=23, completion_tokens=15, cached_tokens=0
|
||||
),
|
||||
custom_llm_provider="together_ai",
|
||||
)
|
||||
|
||||
assert cost == pytest.approx((23 + 15) * 9e-07, rel=1e-9)
|
||||
|
||||
|
||||
def test_completion_cost_together_metadata_only_model_still_uses_size_bucket(_local_model_cost_map):
|
||||
assert "input_cost_per_token" not in litellm.model_cost["together_ai/togethercomputer/CodeLlama-34b-Instruct"]
|
||||
|
||||
cost = completion_cost(
|
||||
completion_response=_together_chat_response(
|
||||
model="togethercomputer/CodeLlama-34b-Instruct", prompt_tokens=23, completion_tokens=15, cached_tokens=0
|
||||
),
|
||||
custom_llm_provider="together_ai",
|
||||
)
|
||||
|
||||
assert cost == pytest.approx((23 + 15) * 8e-07, rel=1e-9)
|
||||
|
||||
|
||||
def test_select_model_name_strips_unregistered_alias_prefix(_local_model_cost_map):
|
||||
"""A router-facing model_name alias containing "/" whose leading segment is NOT a
|
||||
registered provider must not be double-prefixed into a non-existent cost key.
|
||||
|
|
@ -4011,31 +3578,6 @@ def test_completion_cost_base_model_ignores_regional_row(_local_model_cost_map):
|
|||
) == pytest.approx(1000 * flat["input_cost_per_token"])
|
||||
|
||||
|
||||
def test_completion_cost_nonzero_for_slash_alias_model_name(_local_model_cost_map):
|
||||
"""End-to-end cost through a "/"-containing alias must price above zero (#38069)."""
|
||||
|
||||
response = litellm.ModelResponse(
|
||||
id="x",
|
||||
choices=[
|
||||
{
|
||||
"index": 0,
|
||||
"message": {"role": "assistant", "content": "hi"},
|
||||
"finish_reason": "stop",
|
||||
}
|
||||
],
|
||||
model="vertex/claude-opus-5",
|
||||
)
|
||||
response._hidden_params = {"custom_llm_provider": "vertex_ai"}
|
||||
response.usage = litellm.Usage(prompt_tokens=100, completion_tokens=50)
|
||||
|
||||
cost = litellm.completion_cost(
|
||||
completion_response=response,
|
||||
custom_llm_provider="vertex_ai",
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(100 * 5e-6 + 50 * 2.5e-5, rel=1e-9)
|
||||
|
||||
|
||||
def test_select_model_name_unresolvable_alias_unchanged(_local_model_cost_map):
|
||||
"""An alias that resolves to no known cost key keeps the legacy double-prefixed name."""
|
||||
|
||||
|
|
@ -4259,52 +3801,6 @@ def test_explicit_pricing_precedes_private_provider_response_model(
|
|||
assert selected == expected
|
||||
|
||||
|
||||
def test_handle_realtime_stream_cost_calculation_bills_nested_reasoning_tokens_once(
|
||||
_local_model_cost_map: None,
|
||||
) -> None:
|
||||
"""Realtime response.done nests reasoning_tokens inside text_tokens, so they are billed once."""
|
||||
results: OpenAIRealtimeStreamList = [
|
||||
{"type": "session.created", "session": {"model": "gpt-realtime-2.1-mini"}},
|
||||
{
|
||||
"type": "response.done",
|
||||
"response": {
|
||||
"usage": {
|
||||
"total_tokens": 260,
|
||||
"input_tokens": 237,
|
||||
"output_tokens": 23,
|
||||
"input_token_details": {
|
||||
"text_tokens": 43,
|
||||
"audio_tokens": 0,
|
||||
"image_tokens": 194,
|
||||
"cached_tokens": 0,
|
||||
"cached_tokens_details": {"text_tokens": 0, "audio_tokens": 0, "image_tokens": 0},
|
||||
},
|
||||
"output_token_details": {"text_tokens": 23, "audio_tokens": 0, "reasoning_tokens": 18},
|
||||
}
|
||||
},
|
||||
},
|
||||
]
|
||||
combined_usage_object = RealtimeAPITokenUsageProcessor.collect_and_combine_usage_from_realtime_stream_results(
|
||||
results=results,
|
||||
)
|
||||
|
||||
total_cost = handle_realtime_stream_cost_calculation(
|
||||
results=results,
|
||||
combined_usage_object=combined_usage_object,
|
||||
custom_llm_provider="azure",
|
||||
litellm_model_name="azure/gpt-realtime-2.1-mini",
|
||||
)
|
||||
|
||||
info = litellm.get_model_info(model="azure/gpt-realtime-2.1-mini", custom_llm_provider="azure")
|
||||
expected = (
|
||||
43 * info["input_cost_per_token"]
|
||||
+ 194 * info["input_cost_per_image_token"]
|
||||
+ 23 * info["output_cost_per_token"]
|
||||
)
|
||||
assert total_cost == pytest.approx(expected)
|
||||
assert total_cost == pytest.approx(0.0002362)
|
||||
|
||||
|
||||
def test_collect_and_combine_realtime_usage_stores_partitioned_text_tokens() -> None:
|
||||
"""The combined usage that lands in spend logs keeps reasoning out of text_tokens for every turn."""
|
||||
results: OpenAIRealtimeStreamList = [
|
||||
|
|
|
|||
|
|
@ -3409,7 +3409,6 @@ def test_a_streamed_response_bills_the_usage_the_provider_reported(local_cost_ma
|
|||
cost = litellm.completion_cost(completion_response=rebuilt, model=STREAM_COST_MODEL)
|
||||
|
||||
assert cost == pytest.approx(_priced_at(137, 42))
|
||||
assert cost == pytest.approx(0.0007625)
|
||||
|
||||
|
||||
def test_streaming_and_not_streaming_bill_the_same_usage_the_same(local_cost_map):
|
||||
|
|
|
|||
|
|
@ -31,13 +31,6 @@ def test_muse_spark_1_3_routes_to_meta_model_api(model: str):
|
|||
assert api_base == "https://api.meta.ai/v1"
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", (MUSE_SPARK_STANDARD, MUSE_SPARK_CONTRIBUTOR))
|
||||
def test_muse_spark_1_3_web_search_cost_per_query(local_model_cost_map, model: str):
|
||||
info = litellm.get_model_info(model=model)
|
||||
|
||||
assert StandardBuiltInToolCostTracking.get_cost_for_web_search(model_info=info) == WEB_SEARCH_COST_PER_QUERY
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model", (MUSE_SPARK_STANDARD, MUSE_SPARK_CONTRIBUTOR))
|
||||
def test_muse_spark_1_3_backup_matches_main(model: str):
|
||||
"""Ensure the bundled model cost map stays in sync with the canonical file."""
|
||||
|
|
|
|||
|
|
@ -91,18 +91,3 @@ TIERED_COST_CASES = [
|
|||
("gpt-5.6-luna", "priority", 8e-07, 3.6e-06),
|
||||
("gpt-6-astra", "priority", 4e-05, 0.00015),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("model,tier,input_rate,output_rate", TIERED_COST_CASES)
|
||||
def test_cost_per_token_bills_long_context_at_the_tier_rate(
|
||||
model: str, tier: str, input_rate: float, output_rate: float
|
||||
) -> None:
|
||||
"""A prompt over 272K on flex or priority must bill at that tier's long-context rate."""
|
||||
input_cost, output_cost = litellm.cost_per_token(
|
||||
model=model,
|
||||
prompt_tokens=LONG_CONTEXT_PROMPT_TOKENS,
|
||||
completion_tokens=COMPLETION_TOKENS,
|
||||
service_tier=tier,
|
||||
)
|
||||
assert input_cost == pytest.approx(LONG_CONTEXT_PROMPT_TOKENS * input_rate)
|
||||
assert output_cost == pytest.approx(COMPLETION_TOKENS * output_rate)
|
||||
|
|
|
|||
|
|
@ -235,37 +235,6 @@ class TestVideoGeneration:
|
|||
assert response.status == "completed"
|
||||
assert response.model == "sora-2"
|
||||
|
||||
def test_video_generation_cost_calculation(self):
|
||||
"""Test video generation cost calculation."""
|
||||
import json
|
||||
|
||||
# Try to load the local model cost map, skip if not found
|
||||
cost_map_path = "model_prices_and_context_window.json"
|
||||
if not os.path.exists(cost_map_path):
|
||||
# Try alternative paths
|
||||
alt_paths = [
|
||||
os.path.join(os.path.dirname(__file__), "..", "..", cost_map_path),
|
||||
os.path.join(
|
||||
os.path.dirname(__file__), "..", "..", "..", cost_map_path
|
||||
),
|
||||
]
|
||||
for path in alt_paths:
|
||||
if os.path.exists(path):
|
||||
cost_map_path = path
|
||||
break
|
||||
else:
|
||||
pytest.skip("model_prices_and_context_window.json not found")
|
||||
|
||||
with open(cost_map_path, "r") as f:
|
||||
litellm.model_cost = json.load(f)
|
||||
|
||||
# Test with sora-2 model
|
||||
cost = default_video_cost_calculator(
|
||||
model="openai/sora-2", duration_seconds=10.0, custom_llm_provider="openai"
|
||||
)
|
||||
|
||||
# Should calculate cost based on duration (10 seconds * $0.10 per second = $1.00)
|
||||
assert cost == 1.0
|
||||
|
||||
def test_video_generation_cost_calculation_unknown_model(self):
|
||||
"""Test video generation cost calculation for unknown model."""
|
||||
|
|
@ -502,96 +471,6 @@ class TestVideoGeneration:
|
|||
)
|
||||
assert abs(cost - 1.8) < 0.001
|
||||
|
||||
def test_completion_cost_video_resolution_tiers_from_cost_map(self, monkeypatch):
|
||||
"""The 480p/1080p/4k tier keys resolve from the shipped runwayml cost map entries."""
|
||||
from litellm.cost_calculator import completion_cost
|
||||
|
||||
local_map_path = os.path.join(
|
||||
os.path.dirname(__file__), "..", "..", "model_prices_and_context_window.json"
|
||||
)
|
||||
with open(local_map_path, "r") as f:
|
||||
monkeypatch.setattr(litellm, "model_cost", json.load(f))
|
||||
|
||||
def cost_for(model: str, resolution: str | None, duration: float) -> float:
|
||||
mock_response = MagicMock()
|
||||
mock_response.usage = {
|
||||
"duration_seconds": duration,
|
||||
**({"video_resolution": resolution} if resolution else {}),
|
||||
}
|
||||
type(mock_response)._hidden_params = {}
|
||||
return completion_cost(
|
||||
completion_response=mock_response,
|
||||
model=model,
|
||||
call_type="create_video",
|
||||
custom_llm_provider="runwayml",
|
||||
)
|
||||
|
||||
assert abs(cost_for("runwayml/seedance2", "4k", 8.0) - 12.0) < 0.001
|
||||
assert abs(cost_for("runwayml/seedance2", "1080p", 8.0) - 3.2) < 0.001
|
||||
assert abs(cost_for("runwayml/seedance2", "720p", 8.0) - 2.88) < 0.001
|
||||
assert abs(cost_for("runwayml/seedance2_5", "480p", 8.0) - 1.6) < 0.001
|
||||
assert abs(cost_for("runwayml/gen4.5", None, 8.0) - 0.96) < 0.001
|
||||
|
||||
def test_completion_cost_xai_imagine_video_720p_tier_from_cost_map(self, monkeypatch):
|
||||
"""720p xAI Imagine Video requests bill the published 720p rate, not the 480p base rate."""
|
||||
from litellm.cost_calculator import completion_cost
|
||||
|
||||
local_map_path = os.path.join(
|
||||
os.path.dirname(__file__), "..", "..", "model_prices_and_context_window.json"
|
||||
)
|
||||
with open(local_map_path, "r") as f:
|
||||
monkeypatch.setattr(litellm, "model_cost", json.load(f))
|
||||
|
||||
def cost_for(model: str, resolution: str, duration: float) -> float:
|
||||
mock_response = MagicMock()
|
||||
mock_response.usage = {"duration_seconds": duration, "video_resolution": resolution}
|
||||
type(mock_response)._hidden_params = {}
|
||||
return completion_cost(
|
||||
completion_response=mock_response,
|
||||
model=model,
|
||||
call_type="create_video",
|
||||
custom_llm_provider="xai",
|
||||
)
|
||||
|
||||
assert abs(cost_for("xai/grok-imagine-video", "720p", 10.0) - 0.7) < 0.001
|
||||
assert abs(cost_for("xai/grok-imagine-video-1.5", "720p", 10.0) - 1.4) < 0.001
|
||||
assert abs(cost_for("xai/grok-imagine-video-1.5", "480p", 10.0) - 0.8) < 0.001
|
||||
assert abs(cost_for("xai/grok-imagine-video-1.5", "1080p", 10.0) - 2.5) < 0.001
|
||||
|
||||
def test_completion_cost_veo_31_tiers_pin_published_rates(self, monkeypatch):
|
||||
"""The gemini and vertex_ai veo 3.1 entries bill Google's published per-second tier rates."""
|
||||
from litellm.cost_calculator import completion_cost
|
||||
|
||||
local_map_path = os.path.join(
|
||||
os.path.dirname(__file__), "..", "..", "model_prices_and_context_window.json"
|
||||
)
|
||||
with open(local_map_path, "r") as f:
|
||||
monkeypatch.setattr(litellm, "model_cost", json.load(f))
|
||||
|
||||
def cost_for(model: str, provider: str, resolution: str | None, duration: float) -> float:
|
||||
mock_response = MagicMock()
|
||||
mock_response.usage = {
|
||||
"duration_seconds": duration,
|
||||
**({"video_resolution": resolution} if resolution else {}),
|
||||
}
|
||||
type(mock_response)._hidden_params = {}
|
||||
return completion_cost(
|
||||
completion_response=mock_response,
|
||||
model=model,
|
||||
call_type="create_video",
|
||||
custom_llm_provider=provider,
|
||||
)
|
||||
|
||||
for provider in ("gemini", "vertex_ai"):
|
||||
for suffix in ("generate-preview", "generate-001"):
|
||||
standard = f"{provider}/veo-3.1-{suffix}"
|
||||
fast = f"{provider}/veo-3.1-fast-{suffix}"
|
||||
assert abs(cost_for(standard, provider, None, 8.0) - 3.2) < 1e-6
|
||||
assert abs(cost_for(standard, provider, "1080p", 8.0) - 3.2) < 1e-6
|
||||
assert abs(cost_for(standard, provider, "4k", 8.0) - 4.8) < 1e-6
|
||||
assert abs(cost_for(fast, provider, "720p", 8.0) - 0.8) < 1e-6
|
||||
assert abs(cost_for(fast, provider, "1080p", 8.0) - 0.96) < 1e-6
|
||||
assert abs(cost_for(fast, provider, "4k", 8.0) - 2.4) < 1e-6
|
||||
|
||||
def test_video_generation_with_files(self):
|
||||
"""Test video generation with file uploads."""
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue