test(cost): scope deepseek cache tests to the local cost map fixture and cover openrouter

Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
mateo 2026-09-07 19:36:26 +00:00
parent 106eb9e8d9
commit 84a30113f2
3 changed files with 113 additions and 54 deletions

View file

@ -1,5 +1,4 @@
import json
import os
import pytest
from fastapi.testclient import TestClient
@ -3984,10 +3983,11 @@ def test_token_type_cost_breakdown_applies_regional_uplift(_local_model_cost_map
("deepseek/deepseek-r1", "deepseek", 1.4e-07),
("deepseek/deepseek-v3.2", "deepseek", 2.8e-08),
("deepseek/deepseek-coder", "deepseek", 1.4e-08),
("openrouter/deepseek/deepseek-r1", "openrouter", 1.4e-07),
],
)
def test_deepseek_cache_read_cost_in_breakdown(
model, custom_llm_provider, expected_cache_read_rate
model, custom_llm_provider, expected_cache_read_rate, _local_model_cost_map
):
"""
DeepSeek models report cached tokens via prompt_cache_hit_tokens. The
@ -3996,9 +3996,6 @@ def test_deepseek_cache_read_cost_in_breakdown(
Regression for https://github.com/BerriAI/litellm/issues/31594
"""
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
cache_hit_tokens = 64
usage = Usage(
prompt_tokens=100,

View file

@ -13,6 +13,8 @@ from litellm.types.llms.openai import FileSearchTool, ResponsesAPIResponse, WebS
from litellm.types.utils import ModelResponse, StandardBuiltInToolsParams
def test_web_search_cost_low():
web_search_options = WebSearchOptions(search_context_size="low")
model_info = litellm.get_model_info("gpt-4o-search-preview")
@ -21,7 +23,9 @@ def test_web_search_cost_low():
web_search_options=web_search_options, model_info=model_info
)
assert cost == model_info["search_context_cost_per_query"]["search_context_size_low"]
assert (
cost == model_info["search_context_cost_per_query"]["search_context_size_low"]
)
def test_web_search_cost_medium():
@ -32,7 +36,10 @@ def test_web_search_cost_medium():
web_search_options=web_search_options, model_info=model_info
)
assert cost == model_info["search_context_cost_per_query"]["search_context_size_medium"]
assert (
cost
== model_info["search_context_cost_per_query"]["search_context_size_medium"]
)
def test_web_search_cost_high():
@ -43,21 +50,33 @@ def test_web_search_cost_high():
web_search_options=web_search_options, model_info=model_info
)
assert cost == model_info["search_context_cost_per_query"]["search_context_size_high"]
assert (
cost == model_info["search_context_cost_per_query"]["search_context_size_high"]
)
# Test file search cost calculation
def test_file_search_cost():
file_search = FileSearchTool(type="file_search")
cost = StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=file_search)
cost = StandardBuiltInToolCostTracking.get_cost_for_file_search(
file_search=file_search
)
assert cost == 0.0025 # $2.50/1000 calls = 0.0025 per call
# Test edge cases
def test_none_inputs():
# Test with None inputs
assert StandardBuiltInToolCostTracking.get_cost_for_web_search(web_search_options=None, model_info=None) == 0.0
assert StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=None) == 0.0
assert (
StandardBuiltInToolCostTracking.get_cost_for_web_search(
web_search_options=None, model_info=None
)
== 0.0
)
assert (
StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=None)
== 0.0
)
# Test the main get_cost_for_built_in_tools method
@ -82,7 +101,9 @@ def test_get_cost_for_built_in_tools_file_search():
Test that the cost for a file search is 0.00 when no response object is provided
"""
model = "gpt-4"
standard_built_in_tools_params = StandardBuiltInToolsParams(file_search=FileSearchTool(type="file_search"))
standard_built_in_tools_params = StandardBuiltInToolsParams(
file_search=FileSearchTool(type="file_search")
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model=model,
@ -125,7 +146,9 @@ def test_get_cost_for_anthropic_web_search_with_server_tool_use_dict():
usage = Usage(server_tool_use={"web_search_requests": 1})
assert isinstance(usage.server_tool_use, ServerToolUse)
assert StandardBuiltInToolCostTracking.response_object_includes_web_search_call(response_object=None, usage=usage)
assert StandardBuiltInToolCostTracking.response_object_includes_web_search_call(
response_object=None, usage=usage
)
def test_anthropic_web_search_cost_from_raw_response_dict_when_usage_drops_server_tool_use():
@ -164,7 +187,9 @@ def test_anthropic_web_search_cost_from_raw_response_dict_when_usage_drops_serve
standard_built_in_tools_params=None,
)
per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"]
per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"][
"search_context_size_medium"
]
assert cost == per_query_cost * web_search_requests
assert cost > 0.0
assert getattr(usage, "server_tool_use", None) is None
@ -202,7 +227,9 @@ def test_anthropic_web_search_cost_from_raw_response_dict_when_usage_is_none():
standard_built_in_tools_params=None,
)
per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"]
per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"][
"search_context_size_medium"
]
assert cost == per_query_cost * web_search_requests
@ -266,14 +293,18 @@ def test_anthropic_response_usage_block_preserves_server_tool_use():
assert dumped_usage["server_tool_use"] == {"web_search_requests": 2}
@pytest.mark.parametrize("model", ["gemini/gemini-2.0-flash-001", "gemini-2.0-flash-001"])
@pytest.mark.parametrize(
"model", ["gemini/gemini-2.0-flash-001", "gemini-2.0-flash-001"]
)
def test_get_cost_for_gemini_web_search(model):
"""
Test that the cost for a web search is 0.00 when no response object is provided
"""
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
usage = Usage(prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=1))
usage = Usage(
prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=1)
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model=model,
usage=usage,
@ -310,7 +341,9 @@ def test_get_cost_for_vertex_ai_gemini_web_search(model, custom_llm_provider):
Choices(
finish_reason="stop",
index=0,
message=Message(content="Test response with grounding", role="assistant"),
message=Message(
content="Test response with grounding", role="assistant"
),
)
],
created=1234567890,
@ -325,8 +358,7 @@ def test_get_cost_for_vertex_ai_gemini_web_search(model, custom_llm_provider):
completion_tokens=100,
total_tokens=111,
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=11,
web_search_requests=1, # This should trigger grounding cost
text_tokens=11, web_search_requests=1 # This should trigger grounding cost
),
)
response.usage = usage
@ -426,7 +458,9 @@ def test_completion_cost_includes_web_search_without_standard_built_in_tools_par
)
assert web_search_cost > 0, "Web search cost should be non-zero"
assert cost >= web_search_cost, f"completion_cost ({cost}) should include web search cost ({web_search_cost})"
assert (
cost >= web_search_cost
), f"completion_cost ({cost}) should include web search cost ({web_search_cost})"
@pytest.mark.parametrize(
@ -453,14 +487,18 @@ def test_gemini_3x_web_search_billed_per_query(model, local_model_cost_map):
web_search_requests = 2
model_info = litellm.get_model_info(model)
assert model_info["web_search_billing_unit"] == "per_query"
per_query_cost = model_info["search_context_cost_per_query"]["search_context_size_medium"]
per_query_cost = model_info["search_context_cost_per_query"][
"search_context_size_medium"
]
expected_cost = per_query_cost * web_search_requests
usage = Usage(
prompt_tokens=11,
completion_tokens=100,
total_tokens=111,
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=11, web_search_requests=web_search_requests),
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=11, web_search_requests=web_search_requests
),
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
@ -472,7 +510,8 @@ def test_gemini_3x_web_search_billed_per_query(model, local_model_cost_map):
)
assert cost == pytest.approx(expected_cost), (
f"Expected {web_search_requests} x ${per_query_cost} = ${expected_cost} per_query search fee, got ${cost}"
f"Expected {web_search_requests} x ${per_query_cost} = ${expected_cost} "
f"per_query search fee, got ${cost}"
)
@ -577,13 +616,17 @@ def test_gemini_2x_web_search_still_billed_per_prompt(local_model_cost_map):
model = "vertex_ai/gemini-2.5-flash"
model_info = litellm.get_model_info(model)
assert not model_info.get("web_search_billing_unit")
expected_cost = model_info["search_context_cost_per_query"]["search_context_size_medium"]
expected_cost = model_info["search_context_cost_per_query"][
"search_context_size_medium"
]
usage = Usage(
prompt_tokens=11,
completion_tokens=100,
total_tokens=111,
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=11, web_search_requests=2),
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=11, web_search_requests=2
),
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
@ -595,7 +638,8 @@ def test_gemini_2x_web_search_still_billed_per_prompt(local_model_cost_map):
)
assert cost == pytest.approx(expected_cost), (
f"Expected flat ${expected_cost} per_prompt search fee (2 queries clamped to 1), got ${cost}"
f"Expected flat ${expected_cost} per_prompt search fee (2 queries clamped to 1), "
f"got ${cost}"
)
@ -621,7 +665,9 @@ def test_web_search_provider_prefix_fallback_does_not_misprice_non_gemini_model(
prompt_tokens=11,
completion_tokens=100,
total_tokens=111,
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=11, web_search_requests=2),
prompt_tokens_details=PromptTokensDetailsWrapper(
text_tokens=11, web_search_requests=2
),
)
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
@ -639,13 +685,12 @@ def test_web_search_provider_prefix_fallback_does_not_misprice_non_gemini_model(
def _openai_responses_with_web_search_calls(model, num_calls):
from litellm.types.llms.openai import ResponsesAPIResponse
from openai.types.responses.response_function_web_search import (
ActionSearch,
ResponseFunctionWebSearch,
)
from litellm.types.llms.openai import ResponsesAPIResponse
output = [
ResponseFunctionWebSearch(
id=f"ws_{i}",
@ -677,7 +722,9 @@ def test_openai_responses_web_search_priced_per_call(local_model_cost_map):
from litellm.types.utils import Usage
model = "gpt-5-nano"
per_call = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"]
per_call = litellm.get_model_info(model)["search_context_cost_per_query"][
"search_context_size_medium"
]
assert per_call == 0.01
response = _openai_responses_with_web_search_calls(model, num_calls=2)
@ -689,7 +736,9 @@ def test_openai_responses_web_search_priced_per_call(local_model_cost_map):
standard_built_in_tools_params=None,
)
assert cost == pytest.approx(2 * per_call), f"gpt-5-nano web search must bill 2 x ${per_call}, got ${cost}"
assert cost == pytest.approx(2 * per_call), (
f"gpt-5-nano web search must bill 2 x ${per_call}, got ${cost}"
)
def test_openai_responses_web_search_multiplied_by_call_count(local_model_cost_map):
@ -701,7 +750,9 @@ def test_openai_responses_web_search_multiplied_by_call_count(local_model_cost_m
from litellm.types.utils import Usage
model = "gpt-4o-search-preview"
per_call = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"]
per_call = litellm.get_model_info(model)["search_context_cost_per_query"][
"search_context_size_medium"
]
usage = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15)
for num_calls in (1, 3):
@ -729,7 +780,9 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map):
from litellm.types.utils import Usage
model = "gpt-4o-search-preview"
per_call = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"]
per_call = litellm.get_model_info(model)["search_context_cost_per_query"][
"search_context_size_medium"
]
response = ResponsesAPIResponse.model_validate(
{
@ -738,7 +791,10 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map):
"model": model,
"object": "response",
"status": "completed",
"output": [{"type": "web_search_call", "id": f"ws_{i}", "status": "completed"} for i in range(3)],
"output": [
{"type": "web_search_call", "id": f"ws_{i}", "status": "completed"}
for i in range(3)
],
}
)
assert all(isinstance(item, dict) for item in response.output)
@ -751,7 +807,9 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map):
standard_built_in_tools_params=None,
)
assert cost == pytest.approx(3 * per_call), f"3 dict-shaped web searches must bill 3 x ${per_call}, got ${cost}"
assert cost == pytest.approx(3 * per_call), (
f"3 dict-shaped web searches must bill 3 x ${per_call}, got ${cost}"
)
def test_dated_search_preview_entries_carry_search_pricing(local_model_cost_map):
@ -803,7 +861,9 @@ def test_dated_search_preview_entries_carry_search_pricing(local_model_cost_map)
custom_llm_provider="openai",
standard_built_in_tools_params=None,
)
assert cost == pytest.approx(0.035), f"dated search-preview id must bill the $0.035 search fee, got ${cost}"
assert cost == pytest.approx(0.035), (
f"dated search-preview id must bill the $0.035 search fee, got ${cost}"
)
@pytest.mark.parametrize(
@ -815,7 +875,9 @@ def test_dated_search_preview_entries_carry_search_pricing(local_model_cost_map)
WebSearchOptions(search_context_size="high"),
],
)
def test_gpt_4o_mini_snapshot_bills_web_search_like_its_alias(web_search_options, local_model_cost_map):
def test_gpt_4o_mini_snapshot_bills_web_search_like_its_alias(
web_search_options, local_model_cost_map
):
"""
gpt-4o-mini-2024-07-18 is the dated snapshot of gpt-4o-mini and must bill web search
identically. It kept a search_context_cost_per_query from the March 2025 launch tiers that the
@ -851,18 +913,21 @@ def test_gpt_4o_mini_snapshot_web_search_price_absent_from_both_cost_maps():
differ in a handful of unrelated entries, which is how one name kept a price its alias had lost.
"""
repo_root = Path(__file__).parents[4]
map_paths = (
"model_prices_and_context_window.json",
"litellm/model_prices_and_context_window_backup.json",
)
assert all(os.path.isfile(repo_root / path) for path in map_paths)
entries = tuple(
json.loads((repo_root / path).read_text(encoding="utf-8"))["gpt-4o-mini-2024-07-18"] for path in map_paths
json.loads((repo_root / path).read_text(encoding="utf-8"))[
"gpt-4o-mini-2024-07-18"
]
for path in (
"model_prices_and_context_window.json",
"litellm/model_prices_and_context_window_backup.json",
)
)
canonical, backup = entries
assert "search_context_cost_per_query" not in canonical
assert canonical == backup, "gpt-4o-mini-2024-07-18 differs between the two cost maps"
assert (
canonical == backup
), "gpt-4o-mini-2024-07-18 differs between the two cost maps"
# Note: File search integration test removed due to complex annotation detection logic
@ -957,7 +1022,9 @@ def _responses_with_web_search(
for i, action in enumerate(actions)
],
}
return ResponsesAPIResponse.model_validate(payload if tool_usage is None else {**payload, "tool_usage": tool_usage})
return ResponsesAPIResponse.model_validate(
payload if tool_usage is None else {**payload, "tool_usage": tool_usage}
)
def _web_search_cost(model: str, response: ResponsesAPIResponse, custom_llm_provider: str) -> float:
@ -1049,6 +1116,4 @@ def test_web_search_call_count_reads_reported_count_beside_other_tool_usage_entr
cost = _web_search_cost("gpt-5.6", response, "openai")
assert cost == pytest.approx(0.01), (
f"1 reported OpenAI web search must bill 1 x $0.01, not the 2 items, got ${cost}"
)
assert cost == pytest.approx(0.01), f"1 reported OpenAI web search must bill 1 x $0.01, not the 2 items, got ${cost}"

View file

@ -1,6 +1,5 @@
import json
import os
from pathlib import Path
from typing import Final
@ -3744,10 +3743,11 @@ def test_completion_cost_logs_reasoning_and_cache_breakdown(_local_model_cost_ma
("deepseek/deepseek-r1", "deepseek", 1.4e-07),
("deepseek/deepseek-v3.2", "deepseek", 2.8e-08),
("deepseek/deepseek-coder", "deepseek", 1.4e-08),
("openrouter/deepseek/deepseek-r1", "openrouter", 1.4e-07),
],
)
def test_deepseek_cost_breakdown_includes_cache_read_cost(
model, custom_llm_provider, cache_read_rate
model, custom_llm_provider, cache_read_rate, _local_model_cost_map
):
"""
DeepSeek reports cached tokens via prompt_cache_hit_tokens. The cost
@ -3761,9 +3761,6 @@ def test_deepseek_cost_breakdown_includes_cache_read_cost(
from litellm.litellm_core_utils.litellm_logging import Logging
from litellm.types.utils import Choices, Message
os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True"
litellm.model_cost = litellm.get_model_cost_map(url="")
cache_hit_tokens = 64
logging_obj = Logging(