diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py index bf99035a6b1..c241577c538 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -2,12 +2,13 @@ Helper utilities for tracking the cost of built-in tools. """ -from collections.abc import Mapping -from typing import Final, Literal +from collections.abc import Callable, Mapping +from typing import Final, Literal, cast from pydantic import ValidationError import litellm +from litellm._logging import verbose_logger from litellm.constants import OPENAI_FILE_SEARCH_COST_PER_1K_CALLS from litellm.litellm_core_utils.llm_cost_calc.utils import ( get_web_search_requests_from_usage, @@ -30,8 +31,15 @@ from litellm.types.utils import ( ) +def _output_item_field(output_item: object, field: str) -> object: + if isinstance(output_item, dict): + fields: Final[Mapping[str, object]] = cast(Mapping[str, object], output_item) + return fields.get(field) + return getattr(output_item, field, None) + + def _output_item_type(output_item: object) -> str | None: - item_type: Final = output_item.get("type") if isinstance(output_item, dict) else getattr(output_item, "type", None) + item_type: Final = _output_item_field(output_item, "type") return item_type if isinstance(item_type, str) else None @@ -87,31 +95,48 @@ class StandardBuiltInToolCostTracking: usage=usage, ) + image_generation_cost: Final = StandardBuiltInToolCostTracking._handle_image_generation_cost( + response_object=response_object, + custom_llm_provider=custom_llm_provider, + ) + # Handle web search if StandardBuiltInToolCostTracking.response_object_includes_web_search_call( response_object=response_object, usage=usage ): - return google_maps_grounding_cost + StandardBuiltInToolCostTracking._handle_web_search_cost( - model=model, - custom_llm_provider=custom_llm_provider, - usage=usage, - standard_built_in_tools_params=standard_built_in_tools_params, - response_object=response_object, + return ( + google_maps_grounding_cost + + image_generation_cost + + StandardBuiltInToolCostTracking._handle_web_search_cost( + model=model, + custom_llm_provider=custom_llm_provider, + usage=usage, + standard_built_in_tools_params=standard_built_in_tools_params, + response_object=response_object, + ) ) # Handle file search if StandardBuiltInToolCostTracking.response_object_includes_file_search_call(response_object=response_object): - return google_maps_grounding_cost + StandardBuiltInToolCostTracking._handle_file_search_cost( - model=model, - custom_llm_provider=custom_llm_provider, - standard_built_in_tools_params=standard_built_in_tools_params, + return ( + google_maps_grounding_cost + + image_generation_cost + + StandardBuiltInToolCostTracking._handle_file_search_cost( + model=model, + custom_llm_provider=custom_llm_provider, + standard_built_in_tools_params=standard_built_in_tools_params, + ) ) # Handle Azure assistant features - return google_maps_grounding_cost + StandardBuiltInToolCostTracking._handle_azure_assistant_costs( + return ( + google_maps_grounding_cost + + image_generation_cost + + StandardBuiltInToolCostTracking._handle_azure_assistant_costs( model=model, custom_llm_provider=custom_llm_provider, - standard_built_in_tools_params=standard_built_in_tools_params, + standard_built_in_tools_params=standard_built_in_tools_params, + ) ) @staticmethod @@ -210,6 +235,51 @@ class StandardBuiltInToolCostTracking: ) return max(count, 1) + @staticmethod + def response_object_includes_image_generation_call(response_object: object) -> bool: + """Check if the response object includes an image generation call (Responses API).""" + if not isinstance(response_object, ResponsesAPIResponse): + return False + return StandardBuiltInToolCostTracking.response_includes_output_type( + response_object=response_object, output_type="image_generation_call" + ) + + @staticmethod + def _image_generation_call_cost(output_item: object, custom_llm_provider: str | None) -> float: + from litellm.cost_calculator import ( + default_image_cost_calculator, # pyright: ignore[reportUnknownVariableType] # optional_params param is untyped + ) + + status: Final = _output_item_field(output_item, "status") + if status != "completed": + return 0.0 + quality: Final = _output_item_field(output_item, "quality") + size: Final = _output_item_field(output_item, "size") + try: + # the Responses image_generation tool item does not report the model, so price with + # gpt-image-1, OpenAI's default model for that tool + return cast(Callable[..., float], default_image_cost_calculator)( + model="gpt-image-1", + custom_llm_provider=custom_llm_provider or "openai", + quality=quality if isinstance(quality, str) and quality != "auto" else None, + n=1, + size=size if isinstance(size, str) else None, + ) + except Exception as e: + verbose_logger.debug("Could not price Responses API image_generation_call item: %s", e) + return 0.0 + + @staticmethod + def _handle_image_generation_cost(response_object: object, custom_llm_provider: str | None) -> float: + if not isinstance(response_object, ResponsesAPIResponse): + return 0.0 + output: Final[list[object]] = cast(list[object], response_object.output) + return sum( + StandardBuiltInToolCostTracking._image_generation_call_cost(output_item, custom_llm_provider) + for output_item in output + if _output_item_type(output_item) == "image_generation_call" + ) + @staticmethod def _handle_file_search_cost( model: str, @@ -505,7 +575,7 @@ class StandardBuiltInToolCostTracking: @staticmethod def response_includes_output_type( response_object: ResponsesAPIResponse, - output_type: Literal["web_search_call", "file_search_call"], + output_type: Literal["web_search_call", "file_search_call", "image_generation_call"], ) -> bool: """ Check if the ResponsesAPIResponse includes one of the specified output types. diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py index 7bae2eaa338..3ca7f153033 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py @@ -695,3 +695,176 @@ _BEDROCK_MANTLE_WEB_SEARCH_MODELS = ( _BEDROCK_MANTLE_WEB_SEARCH_RATE = 0.012 + + +def _openai_responses_response(model, output): + return ResponsesAPIResponse.model_validate( + { + "id": "resp_1", + "created_at": 1754900000, + "model": model, + "object": "response", + "status": "completed", + "output": output, + "usage": {"input_tokens": 10, "output_tokens": 5, "total_tokens": 15}, + } + ) + + +_ASSISTANT_MESSAGE_OUTPUT_ITEM = { + "type": "message", + "id": "msg_1", + "role": "assistant", + "status": "completed", + "content": [{"type": "output_text", "text": "done", "annotations": []}], +} + +_GPT_IMAGE_1_HIGH_1024_COST_KEY = "high/1024-x-1024/gpt-image-1" + + +def test_responses_image_generation_call_billed_as_tool_usage_cost(local_model_cost_map): + """ + Regression: a Responses API output carrying an image_generation_call item was + charged $0 of tool usage because get_cost_for_built_in_tools only looked for + web/file search calls. A completed image_generation_call must be billed at the + gpt-image-1 rate for its reported quality and size. + """ + expected_image_cost = litellm.model_cost[_GPT_IMAGE_1_HIGH_1024_COST_KEY]["input_cost_per_image"] + response = _openai_responses_response( + "gpt-5", + [ + { + "type": "image_generation_call", + "id": "ig_1", + "status": "completed", + "quality": "high", + "size": "1024x1024", + "result": "AAAA", + }, + dict(_ASSISTANT_MESSAGE_OUTPUT_ITEM), + ], + ) + + cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( + model="gpt-5", + response_object=response, + usage=None, + custom_llm_provider="openai", + standard_built_in_tools_params=None, + ) + + assert cost > 0 + assert cost == pytest.approx(expected_image_cost) + + +def test_responses_web_search_and_image_generation_costs_are_additive(local_model_cost_map): + """ + Regression: get_cost_for_built_in_tools returned early after the web search + branch, so a response billed for web search never reached image generation + pricing. A response with both must bill both. + """ + model = "gpt-4o-search-preview" + image_cost = litellm.model_cost[_GPT_IMAGE_1_HIGH_1024_COST_KEY]["input_cost_per_image"] + image_item = { + "type": "image_generation_call", + "id": "ig_1", + "status": "completed", + "quality": "high", + "size": "1024x1024", + "result": "AAAA", + } + web_search_item = {"type": "web_search_call", "id": "ws_1", "status": "completed"} + + combined = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( + model=model, + response_object=_openai_responses_response(model, [web_search_item, image_item]), + usage=None, + custom_llm_provider="openai", + standard_built_in_tools_params=None, + ) + web_search_only = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( + model=model, + response_object=_openai_responses_response(model, [web_search_item]), + usage=None, + custom_llm_provider="openai", + standard_built_in_tools_params=None, + ) + image_only = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( + model=model, + response_object=_openai_responses_response(model, [image_item]), + usage=None, + custom_llm_provider="openai", + standard_built_in_tools_params=None, + ) + + assert image_only == pytest.approx(image_cost) + assert web_search_only > 0 + assert combined == pytest.approx(web_search_only + image_only) + + +def test_responses_incomplete_image_generation_call_not_billed(local_model_cost_map): + """A failed image_generation_call produced no billable image, so it must cost $0.""" + response = _openai_responses_response( + "gpt-5", + [ + { + "type": "image_generation_call", + "id": "ig_1", + "status": "failed", + "quality": "high", + "size": "1024x1024", + "result": None, + }, + dict(_ASSISTANT_MESSAGE_OUTPUT_ITEM), + ], + ) + + cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( + model="gpt-5", + response_object=response, + usage=None, + custom_llm_provider="openai", + standard_built_in_tools_params=None, + ) + + assert cost == 0.0 + + +def test_completion_cost_includes_responses_image_generation_tool_cost(local_model_cost_map): + """ + An image_generation_call in the Responses output must flow through + completion_cost: the billed total for the same response without the image + item is the token-only baseline the image item must exceed by its tool cost. + """ + image_item = { + "type": "image_generation_call", + "id": "ig_1", + "status": "completed", + "quality": "high", + "size": "1024x1024", + "result": "AAAA", + } + response_with_image = _openai_responses_response( + "gpt-5", [image_item, dict(_ASSISTANT_MESSAGE_OUTPUT_ITEM)] + ) + response_without_image = _openai_responses_response( + "gpt-5", [dict(_ASSISTANT_MESSAGE_OUTPUT_ITEM)] + ) + + cost_with_image = litellm.completion_cost( + completion_response=response_with_image, + model="gpt-5", + custom_llm_provider="openai", + call_type="aresponses", + ) + cost_without_image = litellm.completion_cost( + completion_response=response_without_image, + model="gpt-5", + custom_llm_provider="openai", + call_type="aresponses", + ) + + assert cost_with_image > cost_without_image + assert cost_with_image - cost_without_image == pytest.approx( + litellm.model_cost[_GPT_IMAGE_1_HIGH_1024_COST_KEY]["input_cost_per_image"] + )