From dc4a6118ee06352f0897fe0a773b003ad1c70f66 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Fri, 18 Sep 2026 20:32:52 +0000 Subject: [PATCH] test(cost): scope image generation tool cost tests to the new cases Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../test_tool_call_cost_tracking.py | 123 ++++++++++++------ 1 file changed, 83 insertions(+), 40 deletions(-) diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py index c3a5698bf69..1979229f0fc 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py @@ -1,3 +1,4 @@ + import pytest import litellm @@ -16,7 +17,9 @@ def test_web_search_cost_low(): web_search_options=web_search_options, model_info=model_info ) - assert cost == model_info["search_context_cost_per_query"]["search_context_size_low"] + assert ( + cost == model_info["search_context_cost_per_query"]["search_context_size_low"] + ) def test_web_search_cost_medium(): @@ -27,7 +30,10 @@ def test_web_search_cost_medium(): web_search_options=web_search_options, model_info=model_info ) - assert cost == model_info["search_context_cost_per_query"]["search_context_size_medium"] + assert ( + cost + == model_info["search_context_cost_per_query"]["search_context_size_medium"] + ) def test_web_search_cost_high(): @@ -38,21 +44,33 @@ def test_web_search_cost_high(): web_search_options=web_search_options, model_info=model_info ) - assert cost == model_info["search_context_cost_per_query"]["search_context_size_high"] + assert ( + cost == model_info["search_context_cost_per_query"]["search_context_size_high"] + ) # Test file search cost calculation def test_file_search_cost(): file_search = FileSearchTool(type="file_search") - cost = StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=file_search) + cost = StandardBuiltInToolCostTracking.get_cost_for_file_search( + file_search=file_search + ) assert cost == 0.0025 # $2.50/1000 calls = 0.0025 per call # Test edge cases def test_none_inputs(): # Test with None inputs - assert StandardBuiltInToolCostTracking.get_cost_for_web_search(web_search_options=None, model_info=None) == 0.0 - assert StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=None) == 0.0 + assert ( + StandardBuiltInToolCostTracking.get_cost_for_web_search( + web_search_options=None, model_info=None + ) + == 0.0 + ) + assert ( + StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=None) + == 0.0 + ) # Test the main get_cost_for_built_in_tools method @@ -77,7 +95,9 @@ def test_get_cost_for_built_in_tools_file_search(): Test that the cost for a file search is 0.00 when no response object is provided """ model = "gpt-4" - standard_built_in_tools_params = StandardBuiltInToolsParams(file_search=FileSearchTool(type="file_search")) + standard_built_in_tools_params = StandardBuiltInToolsParams( + file_search=FileSearchTool(type="file_search") + ) cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( model=model, @@ -120,7 +140,9 @@ def test_get_cost_for_anthropic_web_search_with_server_tool_use_dict(): usage = Usage(server_tool_use={"web_search_requests": 1}) assert isinstance(usage.server_tool_use, ServerToolUse) - assert StandardBuiltInToolCostTracking.response_object_includes_web_search_call(response_object=None, usage=usage) + assert StandardBuiltInToolCostTracking.response_object_includes_web_search_call( + response_object=None, usage=usage + ) def test_anthropic_web_search_cost_from_raw_response_dict_when_usage_drops_server_tool_use(): @@ -159,7 +181,9 @@ def test_anthropic_web_search_cost_from_raw_response_dict_when_usage_drops_serve standard_built_in_tools_params=None, ) - per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"] + per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"][ + "search_context_size_medium" + ] assert cost == per_query_cost * web_search_requests assert cost > 0.0 assert getattr(usage, "server_tool_use", None) is None @@ -197,7 +221,9 @@ def test_anthropic_web_search_cost_from_raw_response_dict_when_usage_is_none(): standard_built_in_tools_params=None, ) - per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"] + per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"][ + "search_context_size_medium" + ] assert cost == per_query_cost * web_search_requests @@ -261,14 +287,18 @@ def test_anthropic_response_usage_block_preserves_server_tool_use(): assert dumped_usage["server_tool_use"] == {"web_search_requests": 2} -@pytest.mark.parametrize("model", ["gemini/gemini-2.0-flash-001", "gemini-2.0-flash-001"]) +@pytest.mark.parametrize( + "model", ["gemini/gemini-2.0-flash-001", "gemini-2.0-flash-001"] +) def test_get_cost_for_gemini_web_search(model): """ Test that the cost for a web search is 0.00 when no response object is provided """ from litellm.types.utils import PromptTokensDetailsWrapper, Usage - usage = Usage(prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=1)) + usage = Usage( + prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=1) + ) cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( model=model, usage=usage, @@ -326,7 +356,9 @@ def test_completion_cost_includes_web_search_without_standard_built_in_tools_par ) assert web_search_cost > 0, "Web search cost should be non-zero" - assert cost >= web_search_cost, f"completion_cost ({cost}) should include web search cost ({web_search_cost})" + assert ( + cost >= web_search_cost + ), f"completion_cost ({cost}) should include web search cost ({web_search_cost})" @pytest.mark.parametrize( @@ -353,14 +385,18 @@ def test_gemini_3x_web_search_billed_per_query(model, local_model_cost_map): web_search_requests = 2 model_info = litellm.get_model_info(model) assert model_info["web_search_billing_unit"] == "per_query" - per_query_cost = model_info["search_context_cost_per_query"]["search_context_size_medium"] + per_query_cost = model_info["search_context_cost_per_query"][ + "search_context_size_medium" + ] expected_cost = per_query_cost * web_search_requests usage = Usage( prompt_tokens=11, completion_tokens=100, total_tokens=111, - prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=11, web_search_requests=web_search_requests), + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=11, web_search_requests=web_search_requests + ), ) cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( @@ -372,7 +408,8 @@ def test_gemini_3x_web_search_billed_per_query(model, local_model_cost_map): ) assert cost == pytest.approx(expected_cost), ( - f"Expected {web_search_requests} x ${per_query_cost} = ${expected_cost} per_query search fee, got ${cost}" + f"Expected {web_search_requests} x ${per_query_cost} = ${expected_cost} " + f"per_query search fee, got ${cost}" ) @@ -415,13 +452,17 @@ def test_gemini_2x_web_search_still_billed_per_prompt(local_model_cost_map): model = "vertex_ai/gemini-2.5-flash" model_info = litellm.get_model_info(model) assert not model_info.get("web_search_billing_unit") - expected_cost = model_info["search_context_cost_per_query"]["search_context_size_medium"] + expected_cost = model_info["search_context_cost_per_query"][ + "search_context_size_medium" + ] usage = Usage( prompt_tokens=11, completion_tokens=100, total_tokens=111, - prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=11, web_search_requests=2), + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=11, web_search_requests=2 + ), ) cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( @@ -433,7 +474,8 @@ def test_gemini_2x_web_search_still_billed_per_prompt(local_model_cost_map): ) assert cost == pytest.approx(expected_cost), ( - f"Expected flat ${expected_cost} per_prompt search fee (2 queries clamped to 1), got ${cost}" + f"Expected flat ${expected_cost} per_prompt search fee (2 queries clamped to 1), " + f"got ${cost}" ) @@ -459,7 +501,9 @@ def test_web_search_provider_prefix_fallback_does_not_misprice_non_gemini_model( prompt_tokens=11, completion_tokens=100, total_tokens=111, - prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=11, web_search_requests=2), + prompt_tokens_details=PromptTokensDetailsWrapper( + text_tokens=11, web_search_requests=2 + ), ) cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( @@ -482,6 +526,7 @@ def _openai_responses_with_web_search_calls(model, num_calls): ResponseFunctionWebSearch, ) + output = [ ResponseFunctionWebSearch( id=f"ws_{i}", @@ -512,7 +557,9 @@ def test_openai_responses_web_search_multiplied_by_call_count(local_model_cost_m from litellm.types.utils import Usage model = "gpt-4o-search-preview" - per_call = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"] + per_call = litellm.get_model_info(model)["search_context_cost_per_query"][ + "search_context_size_medium" + ] usage = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15) for num_calls in (1, 3): @@ -539,7 +586,9 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map): from litellm.types.utils import Usage model = "gpt-4o-search-preview" - per_call = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"] + per_call = litellm.get_model_info(model)["search_context_cost_per_query"][ + "search_context_size_medium" + ] response = ResponsesAPIResponse.model_validate( { @@ -548,7 +597,10 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map): "model": model, "object": "response", "status": "completed", - "output": [{"type": "web_search_call", "id": f"ws_{i}", "status": "completed"} for i in range(3)], + "output": [ + {"type": "web_search_call", "id": f"ws_{i}", "status": "completed"} + for i in range(3) + ], } ) assert all(isinstance(item, dict) for item in response.output) @@ -561,7 +613,9 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map): standard_built_in_tools_params=None, ) - assert cost == pytest.approx(3 * per_call), f"3 dict-shaped web searches must bill 3 x ${per_call}, got ${cost}" + assert cost == pytest.approx(3 * per_call), ( + f"3 dict-shaped web searches must bill 3 x ${per_call}, got ${cost}" + ) # Note: File search integration test removed due to complex annotation detection logic @@ -641,6 +695,8 @@ _BEDROCK_MANTLE_WEB_SEARCH_MODELS = ( _BEDROCK_MANTLE_WEB_SEARCH_RATE = 0.012 + + def _openai_responses_response(model, output): return ResponsesAPIResponse.model_validate( { @@ -667,12 +723,7 @@ _GPT_IMAGE_1_HIGH_1024_COST_KEY = "high/1024-x-1024/gpt-image-1" def test_responses_image_generation_call_billed_as_tool_usage_cost(local_model_cost_map): - """ - Regression: a Responses API output carrying an image_generation_call item was - charged $0 of tool usage because get_cost_for_built_in_tools only looked for - web/file search calls. A completed image_generation_call must be billed at the - gpt-image-1 rate for its reported quality and size. - """ + """A completed image_generation_call in the Responses output bills at the gpt-image-1 rate for its quality/size.""" expected_image_cost = litellm.model_cost[_GPT_IMAGE_1_HIGH_1024_COST_KEY]["input_cost_per_image"] response = _openai_responses_response( "gpt-5", @@ -702,11 +753,7 @@ def test_responses_image_generation_call_billed_as_tool_usage_cost(local_model_c def test_responses_web_search_and_image_generation_costs_are_additive(local_model_cost_map): - """ - Regression: get_cost_for_built_in_tools returned early after the web search - branch, so a response billed for web search never reached image generation - pricing. A response with both must bill both. - """ + """A response billed for web search must still also bill its image_generation_call items.""" model = "gpt-4o-search-preview" image_cost = litellm.model_cost[_GPT_IMAGE_1_HIGH_1024_COST_KEY]["input_cost_per_image"] image_item = { @@ -775,11 +822,7 @@ def test_responses_incomplete_image_generation_call_not_billed(local_model_cost_ def test_completion_cost_includes_responses_image_generation_tool_cost(local_model_cost_map): - """ - An image_generation_call in the Responses output must flow through - completion_cost: the billed total for the same response without the image - item is the token-only baseline the image item must exceed by its tool cost. - """ + """The image tool fee must flow through completion_cost on top of the token-only baseline.""" image_item = { "type": "image_generation_call", "id": "ig_1",