diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py index b90c6a90452..c0ab24b88d4 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_llm_cost_calc_utils.py @@ -1,5 +1,4 @@ import json -import os import pytest from fastapi.testclient import TestClient @@ -3984,10 +3983,11 @@ def test_token_type_cost_breakdown_applies_regional_uplift(_local_model_cost_map ("deepseek/deepseek-r1", "deepseek", 1.4e-07), ("deepseek/deepseek-v3.2", "deepseek", 2.8e-08), ("deepseek/deepseek-coder", "deepseek", 1.4e-08), + ("openrouter/deepseek/deepseek-r1", "openrouter", 1.4e-07), ], ) def test_deepseek_cache_read_cost_in_breakdown( - model, custom_llm_provider, expected_cache_read_rate + model, custom_llm_provider, expected_cache_read_rate, _local_model_cost_map ): """ DeepSeek models report cached tokens via prompt_cache_hit_tokens. The @@ -3996,9 +3996,6 @@ def test_deepseek_cache_read_cost_in_breakdown( Regression for https://github.com/BerriAI/litellm/issues/31594 """ - os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" - litellm.model_cost = litellm.get_model_cost_map(url="") - cache_hit_tokens = 64 usage = Usage( prompt_tokens=100, diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py index 78ac2b4f3ef..681708e6619 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py @@ -13,6 +13,8 @@ from litellm.types.llms.openai import FileSearchTool, ResponsesAPIResponse, WebS from litellm.types.utils import ModelResponse, StandardBuiltInToolsParams + + def test_web_search_cost_low(): web_search_options = WebSearchOptions(search_context_size="low") model_info = litellm.get_model_info("gpt-4o-search-preview") @@ -21,7 +23,9 @@ def test_web_search_cost_low(): web_search_options=web_search_options, model_info=model_info ) - assert cost == model_info["search_context_cost_per_query"]["search_context_size_low"] + assert ( + cost == model_info["search_context_cost_per_query"]["search_context_size_low"] + ) def test_web_search_cost_medium(): @@ -32,7 +36,10 @@ def test_web_search_cost_medium(): web_search_options=web_search_options, model_info=model_info ) - assert cost == model_info["search_context_cost_per_query"]["search_context_size_medium"] + assert ( + cost + == model_info["search_context_cost_per_query"]["search_context_size_medium"] + ) def test_web_search_cost_high(): @@ -43,21 +50,33 @@ def test_web_search_cost_high(): web_search_options=web_search_options, model_info=model_info ) - assert cost == model_info["search_context_cost_per_query"]["search_context_size_high"] + assert ( + cost == model_info["search_context_cost_per_query"]["search_context_size_high"] + ) # Test file search cost calculation def test_file_search_cost(): file_search = FileSearchTool(type="file_search") - cost = StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=file_search) + cost = StandardBuiltInToolCostTracking.get_cost_for_file_search( + file_search=file_search + ) assert cost == 0.0025 # $2.50/1000 calls = 0.0025 per call # Test edge cases def test_none_inputs(): # Test with None inputs - assert StandardBuiltInToolCostTracking.get_cost_for_web_search(web_search_options=None, model_info=None) == 0.0 - assert StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=None) == 0.0 + assert ( + StandardBuiltInToolCostTracking.get_cost_for_web_search( + web_search_options=None, model_info=None + ) + == 0.0 + ) + assert ( + StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=None) + == 0.0 + ) # Test the main get_cost_for_built_in_tools method @@ -82,7 +101,9 @@ def test_get_cost_for_built_in_tools_file_search(): Test that the cost for a file search is 0.00 when no response object is provided """ model = "gpt-4" - standard_built_in_tools_params = StandardBuiltInToolsParams(file_search=FileSearchTool(type="file_search")) + standard_built_in_tools_params = StandardBuiltInToolsParams( + file_search=FileSearchTool(type="file_search") + ) cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( model=model, @@ -125,7 +146,9 @@ def test_get_cost_for_anthropic_web_search_with_server_tool_use_dict(): usage = Usage(server_tool_use={"web_search_requests": 1}) assert isinstance(usage.server_tool_use, ServerToolUse) - assert StandardBuiltInToolCostTracking.response_object_includes_web_search_call(response_object=None, usage=usage) + assert StandardBuiltInToolCostTracking.response_object_includes_web_search_call( + response_object=None, usage=usage + ) def test_anthropic_web_search_cost_from_raw_response_dict_when_usage_drops_server_tool_use(): @@ -164,7 +187,9 @@ def test_anthropic_web_search_cost_from_raw_response_dict_when_usage_drops_serve standard_built_in_tools_params=None, ) - per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"] + per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"][ + "search_context_size_medium" + ] assert cost == per_query_cost * web_search_requests assert cost > 0.0 assert getattr(usage, "server_tool_use", None) is None @@ -202,7 +227,9 @@ def test_anthropic_web_search_cost_from_raw_response_dict_when_usage_is_none(): standard_built_in_tools_params=None, ) - per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"] + per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"][ + "search_context_size_medium" + ] assert cost == per_query_cost * web_search_requests @@ -266,14 +293,18 @@ def test_anthropic_response_usage_block_preserves_server_tool_use(): assert dumped_usage["server_tool_use"] == {"web_search_requests": 2} -@pytest.mark.parametrize("model", ["gemini/gemini-2.0-flash-001", "gemini-2.0-flash-001"]) +@pytest.mark.parametrize( + "model", ["gemini/gemini-2.0-flash-001", "gemini-2.0-flash-001"] +) def test_get_cost_for_gemini_web_search(model): """ Test that the cost for a web search is 0.00 when no response object is provided """ from litellm.types.utils import PromptTokensDetailsWrapper, Usage - usage = Usage(prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=1)) + usage = Usage( + prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=1) + ) cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( model=model, usage=usage, @@ -310,7 +341,9 @@ def test_get_cost_for_vertex_ai_gemini_web_search(model, custom_llm_provider): Choices( finish_reason="stop", index=0, - message=Message(content="Test response with grounding", role="assistant"), + message=Message( + content="Test response with grounding", role="assistant" + ), ) ], created=1234567890, @@ -325,8 +358,7 @@ def test_get_cost_for_vertex_ai_gemini_web_search(model, custom_llm_provider): completion_tokens=100, total_tokens=111, prompt_tokens_details=PromptTokensDetailsWrapper( - text_tokens=11, - web_search_requests=1, # This should trigger grounding cost + text_tokens=11, web_search_requests=1 # This should trigger grounding cost ), ) response.usage = usage @@ -426,7 +458,9 @@ def test_completion_cost_includes_web_search_without_standard_built_in_tools_par ) assert web_search_cost > 0, "Web search cost should be non-zero" - assert cost >= web_search_cost, f"completion_cost ({cost}) should include web search cost ({web_search_cost})" + assert ( + cost >= web_search_cost + ), f"completion_cost ({cost}) should include web search cost ({web_search_cost})" @pytest.mark.parametrize( @@ -453,14 +487,18 @@ def test_gemini_3x_web_search_billed_per_query(model, local_model_cost_map): web_search_requests = 2 model_info = litellm.get_model_info(model) assert model_info["web_search_billing_unit"] == "per_query" - per_query_cost = model_info["search_context_cost_per_query"]["search_context_size_medium"] + per_query_cost = model_info["search_context_cost_per_query"][ + "search_context_size_medium" + ] expected_cost = per_query_cost * web_search_requests usage = Usage( prompt_tokens=11, completion_tokens=100, total_tokens=111, - prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=11, web_search_requests=web_search_requests), + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=11, web_search_requests=web_search_requests + ), ) cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( @@ -472,7 +510,8 @@ def test_gemini_3x_web_search_billed_per_query(model, local_model_cost_map): ) assert cost == pytest.approx(expected_cost), ( - f"Expected {web_search_requests} x ${per_query_cost} = ${expected_cost} per_query search fee, got ${cost}" + f"Expected {web_search_requests} x ${per_query_cost} = ${expected_cost} " + f"per_query search fee, got ${cost}" ) @@ -577,13 +616,17 @@ def test_gemini_2x_web_search_still_billed_per_prompt(local_model_cost_map): model = "vertex_ai/gemini-2.5-flash" model_info = litellm.get_model_info(model) assert not model_info.get("web_search_billing_unit") - expected_cost = model_info["search_context_cost_per_query"]["search_context_size_medium"] + expected_cost = model_info["search_context_cost_per_query"][ + "search_context_size_medium" + ] usage = Usage( prompt_tokens=11, completion_tokens=100, total_tokens=111, - prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=11, web_search_requests=2), + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=11, web_search_requests=2 + ), ) cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( @@ -595,7 +638,8 @@ def test_gemini_2x_web_search_still_billed_per_prompt(local_model_cost_map): ) assert cost == pytest.approx(expected_cost), ( - f"Expected flat ${expected_cost} per_prompt search fee (2 queries clamped to 1), got ${cost}" + f"Expected flat ${expected_cost} per_prompt search fee (2 queries clamped to 1), " + f"got ${cost}" ) @@ -621,7 +665,9 @@ def test_web_search_provider_prefix_fallback_does_not_misprice_non_gemini_model( prompt_tokens=11, completion_tokens=100, total_tokens=111, - prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=11, web_search_requests=2), + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=11, web_search_requests=2 + ), ) cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( @@ -639,13 +685,12 @@ def test_web_search_provider_prefix_fallback_does_not_misprice_non_gemini_model( def _openai_responses_with_web_search_calls(model, num_calls): + from litellm.types.llms.openai import ResponsesAPIResponse from openai.types.responses.response_function_web_search import ( ActionSearch, ResponseFunctionWebSearch, ) - from litellm.types.llms.openai import ResponsesAPIResponse - output = [ ResponseFunctionWebSearch( id=f"ws_{i}", @@ -677,7 +722,9 @@ def test_openai_responses_web_search_priced_per_call(local_model_cost_map): from litellm.types.utils import Usage model = "gpt-5-nano" - per_call = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"] + per_call = litellm.get_model_info(model)["search_context_cost_per_query"][ + "search_context_size_medium" + ] assert per_call == 0.01 response = _openai_responses_with_web_search_calls(model, num_calls=2) @@ -689,7 +736,9 @@ def test_openai_responses_web_search_priced_per_call(local_model_cost_map): standard_built_in_tools_params=None, ) - assert cost == pytest.approx(2 * per_call), f"gpt-5-nano web search must bill 2 x ${per_call}, got ${cost}" + assert cost == pytest.approx(2 * per_call), ( + f"gpt-5-nano web search must bill 2 x ${per_call}, got ${cost}" + ) def test_openai_responses_web_search_multiplied_by_call_count(local_model_cost_map): @@ -701,7 +750,9 @@ def test_openai_responses_web_search_multiplied_by_call_count(local_model_cost_m from litellm.types.utils import Usage model = "gpt-4o-search-preview" - per_call = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"] + per_call = litellm.get_model_info(model)["search_context_cost_per_query"][ + "search_context_size_medium" + ] usage = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15) for num_calls in (1, 3): @@ -729,7 +780,9 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map): from litellm.types.utils import Usage model = "gpt-4o-search-preview" - per_call = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"] + per_call = litellm.get_model_info(model)["search_context_cost_per_query"][ + "search_context_size_medium" + ] response = ResponsesAPIResponse.model_validate( { @@ -738,7 +791,10 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map): "model": model, "object": "response", "status": "completed", - "output": [{"type": "web_search_call", "id": f"ws_{i}", "status": "completed"} for i in range(3)], + "output": [ + {"type": "web_search_call", "id": f"ws_{i}", "status": "completed"} + for i in range(3) + ], } ) assert all(isinstance(item, dict) for item in response.output) @@ -751,7 +807,9 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map): standard_built_in_tools_params=None, ) - assert cost == pytest.approx(3 * per_call), f"3 dict-shaped web searches must bill 3 x ${per_call}, got ${cost}" + assert cost == pytest.approx(3 * per_call), ( + f"3 dict-shaped web searches must bill 3 x ${per_call}, got ${cost}" + ) def test_dated_search_preview_entries_carry_search_pricing(local_model_cost_map): @@ -803,7 +861,9 @@ def test_dated_search_preview_entries_carry_search_pricing(local_model_cost_map) custom_llm_provider="openai", standard_built_in_tools_params=None, ) - assert cost == pytest.approx(0.035), f"dated search-preview id must bill the $0.035 search fee, got ${cost}" + assert cost == pytest.approx(0.035), ( + f"dated search-preview id must bill the $0.035 search fee, got ${cost}" + ) @pytest.mark.parametrize( @@ -815,7 +875,9 @@ def test_dated_search_preview_entries_carry_search_pricing(local_model_cost_map) WebSearchOptions(search_context_size="high"), ], ) -def test_gpt_4o_mini_snapshot_bills_web_search_like_its_alias(web_search_options, local_model_cost_map): +def test_gpt_4o_mini_snapshot_bills_web_search_like_its_alias( + web_search_options, local_model_cost_map +): """ gpt-4o-mini-2024-07-18 is the dated snapshot of gpt-4o-mini and must bill web search identically. It kept a search_context_cost_per_query from the March 2025 launch tiers that the @@ -851,18 +913,21 @@ def test_gpt_4o_mini_snapshot_web_search_price_absent_from_both_cost_maps(): differ in a handful of unrelated entries, which is how one name kept a price its alias had lost. """ repo_root = Path(__file__).parents[4] - map_paths = ( - "model_prices_and_context_window.json", - "litellm/model_prices_and_context_window_backup.json", - ) - assert all(os.path.isfile(repo_root / path) for path in map_paths) entries = tuple( - json.loads((repo_root / path).read_text(encoding="utf-8"))["gpt-4o-mini-2024-07-18"] for path in map_paths + json.loads((repo_root / path).read_text(encoding="utf-8"))[ + "gpt-4o-mini-2024-07-18" + ] + for path in ( + "model_prices_and_context_window.json", + "litellm/model_prices_and_context_window_backup.json", + ) ) canonical, backup = entries assert "search_context_cost_per_query" not in canonical - assert canonical == backup, "gpt-4o-mini-2024-07-18 differs between the two cost maps" + assert ( + canonical == backup + ), "gpt-4o-mini-2024-07-18 differs between the two cost maps" # Note: File search integration test removed due to complex annotation detection logic @@ -957,7 +1022,9 @@ def _responses_with_web_search( for i, action in enumerate(actions) ], } - return ResponsesAPIResponse.model_validate(payload if tool_usage is None else {**payload, "tool_usage": tool_usage}) + return ResponsesAPIResponse.model_validate( + payload if tool_usage is None else {**payload, "tool_usage": tool_usage} + ) def _web_search_cost(model: str, response: ResponsesAPIResponse, custom_llm_provider: str) -> float: @@ -1049,6 +1116,4 @@ def test_web_search_call_count_reads_reported_count_beside_other_tool_usage_entr cost = _web_search_cost("gpt-5.6", response, "openai") - assert cost == pytest.approx(0.01), ( - f"1 reported OpenAI web search must bill 1 x $0.01, not the 2 items, got ${cost}" - ) + assert cost == pytest.approx(0.01), f"1 reported OpenAI web search must bill 1 x $0.01, not the 2 items, got ${cost}" diff --git a/tests/test_litellm/test_cost_calculator.py b/tests/test_litellm/test_cost_calculator.py index 71d5753f007..10203891e24 100644 --- a/tests/test_litellm/test_cost_calculator.py +++ b/tests/test_litellm/test_cost_calculator.py @@ -1,6 +1,5 @@ import json -import os from pathlib import Path from typing import Final @@ -3744,10 +3743,11 @@ def test_completion_cost_logs_reasoning_and_cache_breakdown(_local_model_cost_ma ("deepseek/deepseek-r1", "deepseek", 1.4e-07), ("deepseek/deepseek-v3.2", "deepseek", 2.8e-08), ("deepseek/deepseek-coder", "deepseek", 1.4e-08), + ("openrouter/deepseek/deepseek-r1", "openrouter", 1.4e-07), ], ) def test_deepseek_cost_breakdown_includes_cache_read_cost( - model, custom_llm_provider, cache_read_rate + model, custom_llm_provider, cache_read_rate, _local_model_cost_map ): """ DeepSeek reports cached tokens via prompt_cache_hit_tokens. The cost @@ -3761,9 +3761,6 @@ def test_deepseek_cost_breakdown_includes_cache_read_cost( from litellm.litellm_core_utils.litellm_logging import Logging from litellm.types.utils import Choices, Message - os.environ["LITELLM_LOCAL_MODEL_COST_MAP"] = "True" - litellm.model_cost = litellm.get_model_cost_map(url="") - cache_hit_tokens = 64 logging_obj = Logging(