mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-10 03:28:53 +00:00
style(cost): tidy image generation tool cost helper
Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
parent
24026ae0fe
commit
771355a707
2 changed files with 30 additions and 92 deletions
|
|
@ -2,7 +2,7 @@
|
|||
Helper utilities for tracking the cost of built-in tools.
|
||||
"""
|
||||
|
||||
from collections.abc import Callable, Mapping
|
||||
from collections.abc import Mapping
|
||||
from typing import Final, Literal, cast
|
||||
|
||||
from pydantic import ValidationError
|
||||
|
|
@ -133,8 +133,8 @@ class StandardBuiltInToolCostTracking:
|
|||
google_maps_grounding_cost
|
||||
+ image_generation_cost
|
||||
+ StandardBuiltInToolCostTracking._handle_azure_assistant_costs(
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
model=model,
|
||||
custom_llm_provider=custom_llm_provider,
|
||||
standard_built_in_tools_params=standard_built_in_tools_params,
|
||||
)
|
||||
)
|
||||
|
|
@ -256,9 +256,7 @@ class StandardBuiltInToolCostTracking:
|
|||
quality: Final = _output_item_field(output_item, "quality")
|
||||
size: Final = _output_item_field(output_item, "size")
|
||||
try:
|
||||
# the Responses image_generation tool item does not report the model, so price with
|
||||
# gpt-image-1, OpenAI's default model for that tool
|
||||
return cast(Callable[..., float], default_image_cost_calculator)(
|
||||
return default_image_cost_calculator(
|
||||
model="gpt-image-1",
|
||||
custom_llm_provider=custom_llm_provider or "openai",
|
||||
quality=quality if isinstance(quality, str) and quality != "auto" else None,
|
||||
|
|
|
|||
|
|
@ -1,4 +1,3 @@
|
|||
|
||||
import pytest
|
||||
|
||||
import litellm
|
||||
|
|
@ -17,9 +16,7 @@ def test_web_search_cost_low():
|
|||
web_search_options=web_search_options, model_info=model_info
|
||||
)
|
||||
|
||||
assert (
|
||||
cost == model_info["search_context_cost_per_query"]["search_context_size_low"]
|
||||
)
|
||||
assert cost == model_info["search_context_cost_per_query"]["search_context_size_low"]
|
||||
|
||||
|
||||
def test_web_search_cost_medium():
|
||||
|
|
@ -30,10 +27,7 @@ def test_web_search_cost_medium():
|
|||
web_search_options=web_search_options, model_info=model_info
|
||||
)
|
||||
|
||||
assert (
|
||||
cost
|
||||
== model_info["search_context_cost_per_query"]["search_context_size_medium"]
|
||||
)
|
||||
assert cost == model_info["search_context_cost_per_query"]["search_context_size_medium"]
|
||||
|
||||
|
||||
def test_web_search_cost_high():
|
||||
|
|
@ -44,33 +38,21 @@ def test_web_search_cost_high():
|
|||
web_search_options=web_search_options, model_info=model_info
|
||||
)
|
||||
|
||||
assert (
|
||||
cost == model_info["search_context_cost_per_query"]["search_context_size_high"]
|
||||
)
|
||||
assert cost == model_info["search_context_cost_per_query"]["search_context_size_high"]
|
||||
|
||||
|
||||
# Test file search cost calculation
|
||||
def test_file_search_cost():
|
||||
file_search = FileSearchTool(type="file_search")
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_file_search(
|
||||
file_search=file_search
|
||||
)
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=file_search)
|
||||
assert cost == 0.0025 # $2.50/1000 calls = 0.0025 per call
|
||||
|
||||
|
||||
# Test edge cases
|
||||
def test_none_inputs():
|
||||
# Test with None inputs
|
||||
assert (
|
||||
StandardBuiltInToolCostTracking.get_cost_for_web_search(
|
||||
web_search_options=None, model_info=None
|
||||
)
|
||||
== 0.0
|
||||
)
|
||||
assert (
|
||||
StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=None)
|
||||
== 0.0
|
||||
)
|
||||
assert StandardBuiltInToolCostTracking.get_cost_for_web_search(web_search_options=None, model_info=None) == 0.0
|
||||
assert StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=None) == 0.0
|
||||
|
||||
|
||||
# Test the main get_cost_for_built_in_tools method
|
||||
|
|
@ -95,9 +77,7 @@ def test_get_cost_for_built_in_tools_file_search():
|
|||
Test that the cost for a file search is 0.00 when no response object is provided
|
||||
"""
|
||||
model = "gpt-4"
|
||||
standard_built_in_tools_params = StandardBuiltInToolsParams(
|
||||
file_search=FileSearchTool(type="file_search")
|
||||
)
|
||||
standard_built_in_tools_params = StandardBuiltInToolsParams(file_search=FileSearchTool(type="file_search"))
|
||||
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
model=model,
|
||||
|
|
@ -140,9 +120,7 @@ def test_get_cost_for_anthropic_web_search_with_server_tool_use_dict():
|
|||
usage = Usage(server_tool_use={"web_search_requests": 1})
|
||||
|
||||
assert isinstance(usage.server_tool_use, ServerToolUse)
|
||||
assert StandardBuiltInToolCostTracking.response_object_includes_web_search_call(
|
||||
response_object=None, usage=usage
|
||||
)
|
||||
assert StandardBuiltInToolCostTracking.response_object_includes_web_search_call(response_object=None, usage=usage)
|
||||
|
||||
|
||||
def test_anthropic_web_search_cost_from_raw_response_dict_when_usage_drops_server_tool_use():
|
||||
|
|
@ -181,9 +159,7 @@ def test_anthropic_web_search_cost_from_raw_response_dict_when_usage_drops_serve
|
|||
standard_built_in_tools_params=None,
|
||||
)
|
||||
|
||||
per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"][
|
||||
"search_context_size_medium"
|
||||
]
|
||||
per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"]
|
||||
assert cost == per_query_cost * web_search_requests
|
||||
assert cost > 0.0
|
||||
assert getattr(usage, "server_tool_use", None) is None
|
||||
|
|
@ -221,9 +197,7 @@ def test_anthropic_web_search_cost_from_raw_response_dict_when_usage_is_none():
|
|||
standard_built_in_tools_params=None,
|
||||
)
|
||||
|
||||
per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"][
|
||||
"search_context_size_medium"
|
||||
]
|
||||
per_query_cost = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"]
|
||||
assert cost == per_query_cost * web_search_requests
|
||||
|
||||
|
||||
|
|
@ -287,18 +261,14 @@ def test_anthropic_response_usage_block_preserves_server_tool_use():
|
|||
assert dumped_usage["server_tool_use"] == {"web_search_requests": 2}
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"model", ["gemini/gemini-2.0-flash-001", "gemini-2.0-flash-001"]
|
||||
)
|
||||
@pytest.mark.parametrize("model", ["gemini/gemini-2.0-flash-001", "gemini-2.0-flash-001"])
|
||||
def test_get_cost_for_gemini_web_search(model):
|
||||
"""
|
||||
Test that the cost for a web search is 0.00 when no response object is provided
|
||||
"""
|
||||
from litellm.types.utils import PromptTokensDetailsWrapper, Usage
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=1)
|
||||
)
|
||||
usage = Usage(prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=1))
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
model=model,
|
||||
usage=usage,
|
||||
|
|
@ -356,9 +326,7 @@ def test_completion_cost_includes_web_search_without_standard_built_in_tools_par
|
|||
)
|
||||
|
||||
assert web_search_cost > 0, "Web search cost should be non-zero"
|
||||
assert (
|
||||
cost >= web_search_cost
|
||||
), f"completion_cost ({cost}) should include web search cost ({web_search_cost})"
|
||||
assert cost >= web_search_cost, f"completion_cost ({cost}) should include web search cost ({web_search_cost})"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
|
|
@ -385,18 +353,14 @@ def test_gemini_3x_web_search_billed_per_query(model, local_model_cost_map):
|
|||
web_search_requests = 2
|
||||
model_info = litellm.get_model_info(model)
|
||||
assert model_info["web_search_billing_unit"] == "per_query"
|
||||
per_query_cost = model_info["search_context_cost_per_query"][
|
||||
"search_context_size_medium"
|
||||
]
|
||||
per_query_cost = model_info["search_context_cost_per_query"]["search_context_size_medium"]
|
||||
expected_cost = per_query_cost * web_search_requests
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=11,
|
||||
completion_tokens=100,
|
||||
total_tokens=111,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=11, web_search_requests=web_search_requests
|
||||
),
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=11, web_search_requests=web_search_requests),
|
||||
)
|
||||
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
|
|
@ -408,8 +372,7 @@ def test_gemini_3x_web_search_billed_per_query(model, local_model_cost_map):
|
|||
)
|
||||
|
||||
assert cost == pytest.approx(expected_cost), (
|
||||
f"Expected {web_search_requests} x ${per_query_cost} = ${expected_cost} "
|
||||
f"per_query search fee, got ${cost}"
|
||||
f"Expected {web_search_requests} x ${per_query_cost} = ${expected_cost} per_query search fee, got ${cost}"
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -452,17 +415,13 @@ def test_gemini_2x_web_search_still_billed_per_prompt(local_model_cost_map):
|
|||
model = "vertex_ai/gemini-2.5-flash"
|
||||
model_info = litellm.get_model_info(model)
|
||||
assert not model_info.get("web_search_billing_unit")
|
||||
expected_cost = model_info["search_context_cost_per_query"][
|
||||
"search_context_size_medium"
|
||||
]
|
||||
expected_cost = model_info["search_context_cost_per_query"]["search_context_size_medium"]
|
||||
|
||||
usage = Usage(
|
||||
prompt_tokens=11,
|
||||
completion_tokens=100,
|
||||
total_tokens=111,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=11, web_search_requests=2
|
||||
),
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=11, web_search_requests=2),
|
||||
)
|
||||
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
|
|
@ -474,8 +433,7 @@ def test_gemini_2x_web_search_still_billed_per_prompt(local_model_cost_map):
|
|||
)
|
||||
|
||||
assert cost == pytest.approx(expected_cost), (
|
||||
f"Expected flat ${expected_cost} per_prompt search fee (2 queries clamped to 1), "
|
||||
f"got ${cost}"
|
||||
f"Expected flat ${expected_cost} per_prompt search fee (2 queries clamped to 1), got ${cost}"
|
||||
)
|
||||
|
||||
|
||||
|
|
@ -501,9 +459,7 @@ def test_web_search_provider_prefix_fallback_does_not_misprice_non_gemini_model(
|
|||
prompt_tokens=11,
|
||||
completion_tokens=100,
|
||||
total_tokens=111,
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(
|
||||
text_tokens=11, web_search_requests=2
|
||||
),
|
||||
prompt_tokens_details=PromptTokensDetailsWrapper(text_tokens=11, web_search_requests=2),
|
||||
)
|
||||
|
||||
cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
|
||||
|
|
@ -526,7 +482,6 @@ def _openai_responses_with_web_search_calls(model, num_calls):
|
|||
ResponseFunctionWebSearch,
|
||||
)
|
||||
|
||||
|
||||
output = [
|
||||
ResponseFunctionWebSearch(
|
||||
id=f"ws_{i}",
|
||||
|
|
@ -557,9 +512,7 @@ def test_openai_responses_web_search_multiplied_by_call_count(local_model_cost_m
|
|||
from litellm.types.utils import Usage
|
||||
|
||||
model = "gpt-4o-search-preview"
|
||||
per_call = litellm.get_model_info(model)["search_context_cost_per_query"][
|
||||
"search_context_size_medium"
|
||||
]
|
||||
per_call = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"]
|
||||
usage = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15)
|
||||
|
||||
for num_calls in (1, 3):
|
||||
|
|
@ -586,9 +539,7 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map):
|
|||
from litellm.types.utils import Usage
|
||||
|
||||
model = "gpt-4o-search-preview"
|
||||
per_call = litellm.get_model_info(model)["search_context_cost_per_query"][
|
||||
"search_context_size_medium"
|
||||
]
|
||||
per_call = litellm.get_model_info(model)["search_context_cost_per_query"]["search_context_size_medium"]
|
||||
|
||||
response = ResponsesAPIResponse.model_validate(
|
||||
{
|
||||
|
|
@ -597,10 +548,7 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map):
|
|||
"model": model,
|
||||
"object": "response",
|
||||
"status": "completed",
|
||||
"output": [
|
||||
{"type": "web_search_call", "id": f"ws_{i}", "status": "completed"}
|
||||
for i in range(3)
|
||||
],
|
||||
"output": [{"type": "web_search_call", "id": f"ws_{i}", "status": "completed"} for i in range(3)],
|
||||
}
|
||||
)
|
||||
assert all(isinstance(item, dict) for item in response.output)
|
||||
|
|
@ -613,9 +561,7 @@ def test_web_search_call_count_reads_dict_output_items(local_model_cost_map):
|
|||
standard_built_in_tools_params=None,
|
||||
)
|
||||
|
||||
assert cost == pytest.approx(3 * per_call), (
|
||||
f"3 dict-shaped web searches must bill 3 x ${per_call}, got ${cost}"
|
||||
)
|
||||
assert cost == pytest.approx(3 * per_call), f"3 dict-shaped web searches must bill 3 x ${per_call}, got ${cost}"
|
||||
|
||||
|
||||
# Note: File search integration test removed due to complex annotation detection logic
|
||||
|
|
@ -695,8 +641,6 @@ _BEDROCK_MANTLE_WEB_SEARCH_MODELS = (
|
|||
_BEDROCK_MANTLE_WEB_SEARCH_RATE = 0.012
|
||||
|
||||
|
||||
|
||||
|
||||
def _openai_responses_response(model, output):
|
||||
return ResponsesAPIResponse.model_validate(
|
||||
{
|
||||
|
|
@ -844,12 +788,8 @@ def test_completion_cost_includes_responses_image_generation_tool_cost(local_mod
|
|||
"size": "1024x1024",
|
||||
"result": "AAAA",
|
||||
}
|
||||
response_with_image = _openai_responses_response(
|
||||
"gpt-5", [image_item, dict(_ASSISTANT_MESSAGE_OUTPUT_ITEM)]
|
||||
)
|
||||
response_without_image = _openai_responses_response(
|
||||
"gpt-5", [dict(_ASSISTANT_MESSAGE_OUTPUT_ITEM)]
|
||||
)
|
||||
response_with_image = _openai_responses_response("gpt-5", [image_item, dict(_ASSISTANT_MESSAGE_OUTPUT_ITEM)])
|
||||
response_without_image = _openai_responses_response("gpt-5", [dict(_ASSISTANT_MESSAGE_OUTPUT_ITEM)])
|
||||
|
||||
cost_with_image = litellm.completion_cost(
|
||||
completion_response=response_with_image,
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue