From 1d7accce9ed4c4ad26c7c128410dcf84c1032ed9 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 13:49:35 -0700 Subject: [PATCH 01/33] test_supports_web_search --- litellm/__init__.py | 1 + litellm/utils.py | 2 +- tests/litellm_utils_tests/test_utils.py | 19 +++++++++++++++++++ 3 files changed, 21 insertions(+), 1 deletion(-) diff --git a/litellm/__init__.py b/litellm/__init__.py index 25da6504405..4f0b0a16be7 100644 --- a/litellm/__init__.py +++ b/litellm/__init__.py @@ -756,6 +756,7 @@ from .utils import ( create_pretrained_tokenizer, create_tokenizer, supports_function_calling, + supports_web_search, supports_response_schema, supports_parallel_function_calling, supports_vision, diff --git a/litellm/utils.py b/litellm/utils.py index 03e69acf4e3..052857cd94d 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -1975,7 +1975,7 @@ def supports_system_messages(model: str, custom_llm_provider: Optional[str]) -> ) -def supports_web_search(model: str, custom_llm_provider: Optional[str]) -> bool: +def supports_web_search(model: str, custom_llm_provider: Optional[str] = None) -> bool: """ Check if the given model supports web search and return a boolean value. diff --git a/tests/litellm_utils_tests/test_utils.py b/tests/litellm_utils_tests/test_utils.py index 42df7c495c2..fea225e4a3b 100644 --- a/tests/litellm_utils_tests/test_utils.py +++ b/tests/litellm_utils_tests/test_utils.py @@ -477,6 +477,25 @@ def test_supports_function_calling(model, expected_bool): pytest.fail(f"Error occurred: {e}") +@pytest.mark.parametrize( + "model, expected_bool", + [ + ("gpt-4o-mini-search-preview", True), + ("openai/gpt-4o-mini-search-preview", True), + ("gpt-4o-search-preview", True), + ("openai/gpt-4o-search-preview", True), + ("groq/deepseek-r1-distill-llama-70b", False), + ("groq/llama-3.3-70b-versatile", False), + ("codestral/codestral-latest", False), + ], +) +def test_supports_web_search(model, expected_bool): + try: + assert litellm.supports_web_search(model=model) == expected_bool + except Exception as e: + pytest.fail(f"Error occurred: {e}") + + def test_get_max_token_unit_test(): """ More complete testing in `test_completion_cost.py` From 7dd37a5b181e1632da239c7bed4b4826545a302a Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 14:02:51 -0700 Subject: [PATCH 02/33] fix supports_web_search --- litellm/router.py | 5 +++++ litellm/types/router.py | 1 + litellm/types/utils.py | 1 + litellm/utils.py | 2 ++ 4 files changed, 9 insertions(+) diff --git a/litellm/router.py b/litellm/router.py index a395c851dd8..49ce815a246 100644 --- a/litellm/router.py +++ b/litellm/router.py @@ -4928,6 +4928,11 @@ class Router: and model_info["supports_function_calling"] is True # type: ignore ): model_group_info.supports_function_calling = True + if ( + model_info.get("supports_web_search", None) is not None + and model_info["supports_web_search"] is True # type: ignore + ): + model_group_info.supports_web_search = True if ( model_info.get("supported_openai_params", None) is not None and model_info["supported_openai_params"] is not None diff --git a/litellm/types/router.py b/litellm/types/router.py index e34366aa229..dcd547def29 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -559,6 +559,7 @@ class ModelGroupInfo(BaseModel): rpm: Optional[int] = None supports_parallel_function_calling: bool = Field(default=False) supports_vision: bool = Field(default=False) + supports_web_search: bool = Field(default=False) supports_function_calling: bool = Field(default=False) supported_openai_params: Optional[List[str]] = Field(default=[]) configurable_clientside_auth_params: CONFIGURABLE_CLIENTSIDE_AUTH_PARAMS = None diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 8821d2c80bd..2af0c791d88 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -97,6 +97,7 @@ class ProviderSpecificModelInfo(TypedDict, total=False): supports_pdf_input: Optional[bool] supports_native_streaming: Optional[bool] supports_parallel_function_calling: Optional[bool] + supports_web_search: Optional[bool] class ModelInfoBase(ProviderSpecificModelInfo, total=False): diff --git a/litellm/utils.py b/litellm/utils.py index 052857cd94d..317599f864b 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -4544,6 +4544,7 @@ def _get_model_info_helper( # noqa: PLR0915 supports_native_streaming=_model_info.get( "supports_native_streaming", None ), + supports_web_search=_model_info.get("supports_web_search", False), tpm=_model_info.get("tpm", None), rpm=_model_info.get("rpm", None), ) @@ -4612,6 +4613,7 @@ def get_model_info(model: str, custom_llm_provider: Optional[str] = None) -> Mod supports_audio_input: Optional[bool] supports_audio_output: Optional[bool] supports_pdf_input: Optional[bool] + supports_web_search: Optional[bool] Raises: Exception: If the model is not mapped yet. From a4f91c518a1e044c67084069b2a63ace90383176 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 14:06:14 -0700 Subject: [PATCH 03/33] fix supports web search --- .../model_prices_and_context_window_backup.json | 16 ++++++++-------- model_prices_and_context_window.json | 16 ++++++++-------- 2 files changed, 16 insertions(+), 16 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index c07607d2ba3..6131afdd51c 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -80,13 +80,7 @@ "supports_vision": true, "supports_prompt_caching": true, "supports_system_messages": true, - "supports_tool_choice": true, - "supports_web_search": true, - "search_context_cost_per_query": { - "search_context_size_low": 0.030, - "search_context_size_medium": 0.035, - "search_context_size_high": 0.050 - } + "supports_tool_choice": true }, "gpt-4o-search-preview": { "max_tokens": 16384, @@ -105,7 +99,13 @@ "supports_vision": true, "supports_prompt_caching": true, "supports_system_messages": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.030, + "search_context_size_medium": 0.035, + "search_context_size_high": 0.050 + } }, "gpt-4.5-preview": { "max_tokens": 16384, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index c07607d2ba3..6131afdd51c 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -80,13 +80,7 @@ "supports_vision": true, "supports_prompt_caching": true, "supports_system_messages": true, - "supports_tool_choice": true, - "supports_web_search": true, - "search_context_cost_per_query": { - "search_context_size_low": 0.030, - "search_context_size_medium": 0.035, - "search_context_size_high": 0.050 - } + "supports_tool_choice": true }, "gpt-4o-search-preview": { "max_tokens": 16384, @@ -105,7 +99,13 @@ "supports_vision": true, "supports_prompt_caching": true, "supports_system_messages": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.030, + "search_context_size_medium": 0.035, + "search_context_size_high": 0.050 + } }, "gpt-4.5-preview": { "max_tokens": 16384, From bddbeff7173863c0fb6ccb812c0b33f88eab1978 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 14:29:02 -0700 Subject: [PATCH 04/33] test_openai_web_search_logging_cost_tracking --- .../test_token_counting.py | 34 +++++++++++++++++++ 1 file changed, 34 insertions(+) diff --git a/tests/logging_callback_tests/test_token_counting.py b/tests/logging_callback_tests/test_token_counting.py index 341ef2a5451..d4da465041a 100644 --- a/tests/logging_callback_tests/test_token_counting.py +++ b/tests/logging_callback_tests/test_token_counting.py @@ -244,3 +244,37 @@ async def test_stream_token_counting_anthropic_with_include_usage(): ) <= 10 ) + + +class TestCustomLogger(CustomLogger): + def __init__(self): + self.standard_logging_payload: Optional[StandardLoggingPayload] = None + + async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): + print("kwargs: ", kwargs) + self.standard_logging_payload = kwargs.get("standard_logging_object", None) + + +@pytest.mark.asyncio +async def test_openai_web_search_logging_cost_tracking(): + """Makes a simple web search request and validates the response contains web search annotations and all expected fields are present""" + litellm._turn_on_debug() + test_custom_logger = TestCustomLogger() + litellm.callbacks = [test_custom_logger] + response = await litellm.acompletion( + model="openai/gpt-4o-search-preview", + messages=[ + { + "role": "user", + "content": "What was a positive news story from today?", + } + ], + ) + print("litellm response: ", response.model_dump_json(indent=4)) + + await asyncio.sleep(1) + + print( + "logged standard logging payload: ", + json.dumps(test_custom_logger.standard_logging_payload, indent=4), + ) From cf22d31b2b01f220756d909e38dc6556c8496c2b Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 14:52:58 -0700 Subject: [PATCH 05/33] search_context_cost_per_query --- litellm/types/utils.py | 9 +++++ litellm/utils.py | 3 ++ .../test_token_counting.py | 33 +++++++++++++------ 3 files changed, 35 insertions(+), 10 deletions(-) diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 2af0c791d88..a7894e424c8 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -100,6 +100,12 @@ class ProviderSpecificModelInfo(TypedDict, total=False): supports_web_search: Optional[bool] +class SearchContextCostPerQuery(TypedDict, total=False): + search_context_size_low: float + search_context_size_medium: float + search_context_size_high: float + + class ModelInfoBase(ProviderSpecificModelInfo, total=False): key: Required[str] # the key in litellm.model_cost which is returned @@ -136,6 +142,9 @@ class ModelInfoBase(ProviderSpecificModelInfo, total=False): output_cost_per_video_per_second: Optional[float] # only for vertex ai models output_cost_per_audio_per_second: Optional[float] # only for vertex ai models output_cost_per_second: Optional[float] # for OpenAI Speech models + search_context_cost_per_query: Optional[ + SearchContextCostPerQuery + ] # Cost for using web search tool litellm_provider: Required[str] mode: Required[ diff --git a/litellm/utils.py b/litellm/utils.py index 317599f864b..dc97c4d898f 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -4545,6 +4545,9 @@ def _get_model_info_helper( # noqa: PLR0915 "supports_native_streaming", None ), supports_web_search=_model_info.get("supports_web_search", False), + search_context_cost_per_query=_model_info.get( + "search_context_cost_per_query", None + ), tpm=_model_info.get("tpm", None), rpm=_model_info.get("rpm", None), ) diff --git a/tests/logging_callback_tests/test_token_counting.py b/tests/logging_callback_tests/test_token_counting.py index d4da465041a..6821c7d345c 100644 --- a/tests/logging_callback_tests/test_token_counting.py +++ b/tests/logging_callback_tests/test_token_counting.py @@ -21,16 +21,18 @@ sys.path.insert( import litellm import asyncio from typing import Optional -from litellm.types.utils import StandardLoggingPayload, Usage +from litellm.types.utils import StandardLoggingPayload, Usage, ModelInfoBase from litellm.integrations.custom_logger import CustomLogger class TestCustomLogger(CustomLogger): def __init__(self): self.recorded_usage: Optional[Usage] = None + self.standard_logging_payload: Optional[StandardLoggingPayload] = None async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): standard_logging_payload = kwargs.get("standard_logging_object") + self.standard_logging_payload = standard_logging_payload print( "standard_logging_payload", json.dumps(standard_logging_payload, indent=4, default=str), @@ -246,15 +248,6 @@ async def test_stream_token_counting_anthropic_with_include_usage(): ) -class TestCustomLogger(CustomLogger): - def __init__(self): - self.standard_logging_payload: Optional[StandardLoggingPayload] = None - - async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): - print("kwargs: ", kwargs) - self.standard_logging_payload = kwargs.get("standard_logging_object", None) - - @pytest.mark.asyncio async def test_openai_web_search_logging_cost_tracking(): """Makes a simple web search request and validates the response contains web search annotations and all expected fields are present""" @@ -278,3 +271,23 @@ async def test_openai_web_search_logging_cost_tracking(): "logged standard logging payload: ", json.dumps(test_custom_logger.standard_logging_payload, indent=4), ) + standard_logging_payload = test_custom_logger.standard_logging_payload + response_cost = standard_logging_payload.get("response_cost") + assert response_cost is not None + + # Assert the cost = Token Usage + Web Search Cost + model_map_information = standard_logging_payload["model_map_information"] + model_map_value: ModelInfoBase = model_map_information["model_map_value"] + total_token_cost = ( + standard_logging_payload["prompt_tokens"] + * model_map_value["input_cost_per_token"] + ) + ( + standard_logging_payload["completion_tokens"] + * model_map_value["output_cost_per_token"] + ) + print("total token cost:", total_token_cost) + assert ( + response_cost + == total_token_cost + + model_map_value["search_context_cost_per_query"]["search_context_size_low"] + ) From e19b82f202f56dd13ad5e98592653b953b7f4429 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 15:37:34 -0700 Subject: [PATCH 06/33] add WebSearchOptions as supported chat completion param --- litellm/types/llms/openai.py | 39 ++++++++++++++++++++++++++++++++++++ 1 file changed, 39 insertions(+) diff --git a/litellm/types/llms/openai.py b/litellm/types/llms/openai.py index e58f5732271..d80ba350d16 100644 --- a/litellm/types/llms/openai.py +++ b/litellm/types/llms/openai.py @@ -382,6 +382,45 @@ class ChatCompletionThinkingBlock(TypedDict, total=False): cache_control: Optional[Union[dict, ChatCompletionCachedContent]] +class WebSearchOptionsUserLocationApproximate(TypedDict, total=False): + city: str + """Free text input for the city of the user, e.g. `San Francisco`.""" + + country: str + """ + The two-letter [ISO country code](https://en.wikipedia.org/wiki/ISO_3166-1) of + the user, e.g. `US`. + """ + + region: str + """Free text input for the region of the user, e.g. `California`.""" + + timezone: str + """ + The [IANA timezone](https://timeapi.io/documentation/iana-timezones) of the + user, e.g. `America/Los_Angeles`. + """ + + +class WebSearchOptionsUserLocation(TypedDict, total=False): + approximate: Required[WebSearchOptionsUserLocationApproximate] + """Approximate location parameters for the search.""" + + type: Required[Literal["approximate"]] + """The type of location approximation. Always `approximate`.""" + + +class WebSearchOptions(TypedDict, total=False): + search_context_size: Literal["low", "medium", "high"] + """ + High level guidance for the amount of context window space to use for the + search. One of `low`, `medium`, or `high`. `medium` is the default. + """ + + user_location: Optional[WebSearchOptionsUserLocation] + """Approximate location parameters for the search.""" + + class ChatCompletionAnnotationURLCitation(TypedDict, total=False): end_index: int """The index of the last character of the URL citation in the message.""" From 1910ed6027971709c24073f2ad5244c243626d6d Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 15:39:04 -0700 Subject: [PATCH 07/33] WebSearchOptions --- litellm/types/utils.py | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/litellm/types/utils.py b/litellm/types/utils.py index a7894e424c8..6b027ab8b77 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -33,6 +33,7 @@ from .llms.openai import ( ChatCompletionToolCallChunk, ChatCompletionUsageBlock, OpenAIChatCompletionChunk, + WebSearchOptions, ) from .rerank import RerankResponse @@ -1622,6 +1623,18 @@ class StandardLoggingUserAPIKeyMetadata(TypedDict): user_api_key_end_user_id: Optional[str] +class StandardBuiltInToolsParams(TypedDict, total=False): + """ + Standard built-in OpenAItools parameters + + This is used to calculate the cost of built-in tools, insert any standard built-in tools parameters here + + OpenAI charges users based on the `web_search_options` parameter + """ + + web_search_options: Optional[WebSearchOptions] + + class StandardLoggingPromptManagementMetadata(TypedDict): prompt_id: str prompt_variables: Optional[dict] From ded9a2e4817d8b0ae7ab5b9458b7e11d42e07695 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 16:01:12 -0700 Subject: [PATCH 08/33] add cost tracking for StandardBuiltInToolCostTracking --- litellm/cost_calculator.py | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index e17a94c87ec..55736772af1 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -9,6 +9,9 @@ from pydantic import BaseModel import litellm import litellm._logging from litellm import verbose_logger +from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import ( + StandardBuiltInToolCostTracking, +) from litellm.litellm_core_utils.llm_cost_calc.utils import _generic_cost_per_character from litellm.llms.anthropic.cost_calculation import ( cost_per_token as anthropic_cost_per_token, @@ -57,6 +60,7 @@ from litellm.types.utils import ( LlmProvidersSet, ModelInfo, PassthroughCallTypes, + StandardBuiltInToolsParams, Usage, ) from litellm.utils import ( @@ -524,6 +528,7 @@ def completion_cost( # noqa: PLR0915 optional_params: Optional[dict] = None, custom_pricing: Optional[bool] = None, base_model: Optional[str] = None, + standard_built_in_tools_params: Optional[StandardBuiltInToolsParams] = None, ) -> float: """ Calculate the cost of a given completion call fot GPT-3.5-turbo, llama2, any litellm supported llm. @@ -802,6 +807,12 @@ def completion_cost( # noqa: PLR0915 rerank_billed_units=rerank_billed_units, ) _final_cost = prompt_tokens_cost_usd_dollar + completion_tokens_cost_usd_dollar + _final_cost += StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( + model=model, + response_object=completion_response, + standard_built_in_tools_params=standard_built_in_tools_params, + custom_llm_provider=custom_llm_provider, + ) return _final_cost except Exception as e: @@ -861,6 +872,7 @@ def response_cost_calculator( base_model: Optional[str] = None, custom_pricing: Optional[bool] = None, prompt: str = "", + standard_built_in_tools_params: Optional[StandardBuiltInToolsParams] = None, ) -> float: """ Returns @@ -890,6 +902,7 @@ def response_cost_calculator( custom_pricing=custom_pricing, base_model=base_model, prompt=prompt, + standard_built_in_tools_params=standard_built_in_tools_params, ) return response_cost except Exception as e: From 10da28722512fcaaafed46367611b346d056769b Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 16:03:53 -0700 Subject: [PATCH 09/33] initialize_standard_built_in_tools_params --- litellm/litellm_core_utils/litellm_logging.py | 25 ++++++++++++++++++- 1 file changed, 24 insertions(+), 1 deletion(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 3e694220a54..f437755771c 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -35,6 +35,9 @@ from litellm.integrations.custom_logger import CustomLogger from litellm.integrations.mlflow import MlflowLogger from litellm.integrations.pagerduty.pagerduty import PagerDutyAlerting from litellm.litellm_core_utils.get_litellm_params import get_litellm_params +from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import ( + StandardBuiltInToolCostTracking, +) from litellm.litellm_core_utils.model_param_helper import ModelParamHelper from litellm.litellm_core_utils.redact_messages import ( redact_message_input_output_from_custom_logger, @@ -48,6 +51,7 @@ from litellm.types.llms.openai import ( HttpxBinaryResponseContent, ResponseCompletedEvent, ResponsesAPIResponse, + WebSearchOptions, ) from litellm.types.rerank import RerankResponse from litellm.types.router import SPECIAL_MODEL_INFO_PARAMS @@ -60,6 +64,7 @@ from litellm.types.utils import ( ModelResponse, ModelResponseStream, RawRequestTypedDict, + StandardBuiltInToolsParams, StandardCallbackDynamicParams, StandardLoggingAdditionalHeaders, StandardLoggingHiddenParams, @@ -264,7 +269,9 @@ class Logging(LiteLLMLoggingBaseClass): self.standard_callback_dynamic_params: StandardCallbackDynamicParams = ( self.initialize_standard_callback_dynamic_params(kwargs) ) - + self.standard_built_in_tools_params: StandardBuiltInToolsParams = ( + self.initialize_standard_built_in_tools_params(kwargs) + ) ## TIME TO FIRST TOKEN LOGGING ## self.completion_start_time: Optional[datetime.datetime] = None self._llm_caching_handler: Optional[LLMCachingHandler] = None @@ -369,6 +376,20 @@ class Logging(LiteLLMLoggingBaseClass): """ return _initialize_standard_callback_dynamic_params(kwargs) + def initialize_standard_built_in_tools_params( + self, kwargs: Optional[Dict] = None + ) -> StandardBuiltInToolsParams: + """ + Initialize the standard built-in tools params from the kwargs + + checks if web_search_options in kwargs or tools and sets the corresponding attribute in StandardBuiltInToolsParams + """ + return StandardBuiltInToolsParams( + web_search_options=StandardBuiltInToolCostTracking._get_web_search_options( + kwargs or {} + ) + ) + def update_environment_variables( self, litellm_params: Dict, @@ -903,6 +924,7 @@ class Logging(LiteLLMLoggingBaseClass): "optional_params": self.optional_params, "custom_pricing": custom_pricing, "prompt": prompt, + "standard_built_in_tools_params": self.standard_built_in_tools_params, } except Exception as e: # error creating kwargs for cost calculation debug_info = StandardLoggingModelCostFailureDebugInformation( @@ -1102,6 +1124,7 @@ class Logging(LiteLLMLoggingBaseClass): prompt="", completion=getattr(result, "content", ""), total_time=float_diff, + standard_built_in_tools_params=self.standard_built_in_tools_params, ) return start_time, end_time, result From bfe3132bb64f7393fdf93dd2928fe8ff160fc627 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 16:06:47 -0700 Subject: [PATCH 10/33] StandardBuiltInToolCostTracking --- .../llm_cost_calc/tool_call_cost_tracking.py | 102 ++++++++++++++++++ 1 file changed, 102 insertions(+) create mode 100644 litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py new file mode 100644 index 00000000000..81bf4fd5e71 --- /dev/null +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -0,0 +1,102 @@ +""" +Helper utilities for tracking the cost of built-in tools. +""" + +from typing import Any, Dict, Optional + +import litellm +from litellm.types.llms.openai import WebSearchOptions +from litellm.types.utils import ( + ModelInfo, + ModelResponse, + SearchContextCostPerQuery, + StandardBuiltInToolsParams, +) + + +class StandardBuiltInToolCostTracking: + """ + Helper class for tracking the cost of built-in tools + + Example: Web Search + """ + + @staticmethod + def get_cost_for_built_in_tools( + model: str, + response_object: Any, + custom_llm_provider: Optional[str] = None, + standard_built_in_tools_params: Optional[StandardBuiltInToolsParams] = None, + ) -> float: + """ + Get the cost of using built-in tools. + + Supported tools: + - Web Search + + """ + model_info = litellm.get_model_info( + model=model, custom_llm_provider=custom_llm_provider + ) + if isinstance(response_object, ModelResponse): + if ( + standard_built_in_tools_params is not None + and standard_built_in_tools_params.get("web_search_options", None) + is not None + ): + return StandardBuiltInToolCostTracking.get_cost_for_web_search( + web_search_options=standard_built_in_tools_params.get( + "web_search_options", None + ), + model_info=model_info, + ) + # elif isinstance(response_object, ResponsesAPIResponse): + # if response_object.web_search_options is not None: + # return 0.0 + return 0.0 + + @staticmethod + def get_cost_for_web_search( + web_search_options: Optional[WebSearchOptions] = None, + model_info: Optional[ModelInfo] = None, + ) -> float: + """ + If request includes `web_search_options`, calculate the cost of the web search. + """ + if web_search_options is None: + return 0.0 + if model_info is None: + return 0.0 + + search_context_pricing: SearchContextCostPerQuery = ( + model_info.get("search_context_cost_per_query", {}) or {} + ) + if web_search_options.get("search_context_size", None) == "low": + return search_context_pricing.get("search_context_size_low", 0.0) + elif web_search_options.get("search_context_size", None) == "medium": + return search_context_pricing.get("search_context_size_medium", 0.0) + elif web_search_options.get("search_context_size", None) == "high": + return search_context_pricing.get("search_context_size_high", 0.0) + else: + return 0.0 + + @staticmethod + def response_includes_annotations(response_object: ModelResponse) -> bool: + for _choice in response_object.choices: + message = getattr(_choice, "message", None) + if message is not None and hasattr(message, "annotations"): + return True + return False + + @staticmethod + def _get_web_search_options(kwargs: Dict) -> Optional[WebSearchOptions]: + if "web_search_options" in kwargs: + return WebSearchOptions(**kwargs.get("web_search_options", {})) + if "tools" in kwargs: + tools = kwargs.get("tools", []) + # Look for web search tool in the tools array + for tool in tools: + if isinstance(tool, dict): + if "search_context_size" in tool: + return WebSearchOptions(**tool) + return None From 475dfaa15626b480dde86cfacd9e94cc4f327499 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 16:56:32 -0700 Subject: [PATCH 11/33] fixes for web search cost tracking --- litellm/litellm_core_utils/litellm_logging.py | 7 ++ .../llm_cost_calc/tool_call_cost_tracking.py | 21 ++++- litellm/types/utils.py | 1 + .../test_token_counting.py | 77 ++++++++++++++----- 4 files changed, 83 insertions(+), 23 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index f437755771c..8ce6b156686 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -1089,6 +1089,7 @@ class Logging(LiteLLMLoggingBaseClass): end_time=end_time, logging_obj=self, status="success", + standard_built_in_tools_params=self.standard_built_in_tools_params, ) ) elif isinstance(result, dict): # pass-through endpoints @@ -1101,6 +1102,7 @@ class Logging(LiteLLMLoggingBaseClass): end_time=end_time, logging_obj=self, status="success", + standard_built_in_tools_params=self.standard_built_in_tools_params, ) ) elif standard_logging_object is not None: @@ -1178,6 +1180,7 @@ class Logging(LiteLLMLoggingBaseClass): end_time=end_time, logging_obj=self, status="success", + standard_built_in_tools_params=self.standard_built_in_tools_params, ) ) callbacks = self.get_combined_callback_list( @@ -1718,6 +1721,7 @@ class Logging(LiteLLMLoggingBaseClass): end_time=end_time, logging_obj=self, status="success", + standard_built_in_tools_params=self.standard_built_in_tools_params, ) ) callbacks = self.get_combined_callback_list( @@ -1934,6 +1938,7 @@ class Logging(LiteLLMLoggingBaseClass): status="failure", error_str=str(exception), original_exception=exception, + standard_built_in_tools_params=self.standard_built_in_tools_params, ) ) return start_time, end_time @@ -3390,6 +3395,7 @@ def get_standard_logging_object_payload( status: StandardLoggingPayloadStatus, error_str: Optional[str] = None, original_exception: Optional[Exception] = None, + standard_built_in_tools_params: Optional[StandardBuiltInToolsParams] = None, ) -> Optional[StandardLoggingPayload]: try: kwargs = kwargs or {} @@ -3565,6 +3571,7 @@ def get_standard_logging_object_payload( guardrail_information=metadata.get( "standard_logging_guardrail_information", None ), + standard_built_in_tools_params=standard_built_in_tools_params, ) emit_standard_logging_payload(payload) diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py index 81bf4fd5e71..1717565831d 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -50,9 +50,12 @@ class StandardBuiltInToolCostTracking: ), model_info=model_info, ) - # elif isinstance(response_object, ResponsesAPIResponse): - # if response_object.web_search_options is not None: - # return 0.0 + elif StandardBuiltInToolCostTracking.response_includes_annotations( + response_object + ): + return StandardBuiltInToolCostTracking.get_default_cost_for_web_search( + model_info + ) return 0.0 @staticmethod @@ -80,6 +83,18 @@ class StandardBuiltInToolCostTracking: else: return 0.0 + @staticmethod + def get_default_cost_for_web_search(model_info: ModelInfo) -> float: + """ + If no web search options are provided, use the `search_context_size_medium` pricing. + + https://platform.openai.com/docs/pricing#web-search + """ + search_context_pricing: SearchContextCostPerQuery = ( + model_info.get("search_context_cost_per_query", {}) or {} + ) or {} + return search_context_pricing.get("search_context_size_medium", 0.0) + @staticmethod def response_includes_annotations(response_object: ModelResponse) -> bool: for _choice in response_object.choices: diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 6b027ab8b77..1c1ee24bfbf 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -1752,6 +1752,7 @@ class StandardLoggingPayload(TypedDict): model_parameters: dict hidden_params: StandardLoggingHiddenParams guardrail_information: Optional[StandardLoggingGuardrailInformation] + standard_built_in_tools_params: Optional[StandardBuiltInToolsParams] from typing import AsyncIterator, Iterator diff --git a/tests/logging_callback_tests/test_token_counting.py b/tests/logging_callback_tests/test_token_counting.py index 6821c7d345c..05e21efe0d9 100644 --- a/tests/logging_callback_tests/test_token_counting.py +++ b/tests/logging_callback_tests/test_token_counting.py @@ -248,34 +248,23 @@ async def test_stream_token_counting_anthropic_with_include_usage(): ) -@pytest.mark.asyncio -async def test_openai_web_search_logging_cost_tracking(): - """Makes a simple web search request and validates the response contains web search annotations and all expected fields are present""" +async def _setup_web_search_test(): + """Helper function to setup common test requirements""" litellm._turn_on_debug() test_custom_logger = TestCustomLogger() litellm.callbacks = [test_custom_logger] - response = await litellm.acompletion( - model="openai/gpt-4o-search-preview", - messages=[ - { - "role": "user", - "content": "What was a positive news story from today?", - } - ], - ) - print("litellm response: ", response.model_dump_json(indent=4)) + return test_custom_logger + +async def _verify_web_search_cost(test_custom_logger, expected_context_size): + """Helper function to verify web search costs""" await asyncio.sleep(1) - print( - "logged standard logging payload: ", - json.dumps(test_custom_logger.standard_logging_payload, indent=4), - ) standard_logging_payload = test_custom_logger.standard_logging_payload response_cost = standard_logging_payload.get("response_cost") assert response_cost is not None - # Assert the cost = Token Usage + Web Search Cost + # Calculate token cost model_map_information = standard_logging_payload["model_map_information"] model_map_value: ModelInfoBase = model_map_information["model_map_value"] total_token_cost = ( @@ -285,9 +274,57 @@ async def test_openai_web_search_logging_cost_tracking(): standard_logging_payload["completion_tokens"] * model_map_value["output_cost_per_token"] ) - print("total token cost:", total_token_cost) + + # Verify total cost assert ( response_cost == total_token_cost - + model_map_value["search_context_cost_per_query"]["search_context_size_low"] + + model_map_value["search_context_cost_per_query"][expected_context_size] ) + + +@pytest.mark.asyncio +async def test_openai_web_search_logging_cost_tracking_no_explicit_search_context_size(): + """Cost is tracked as `search_context_size_medium` when no `search_context_size` is passed in""" + test_custom_logger = await _setup_web_search_test() + + response = await litellm.acompletion( + model="openai/gpt-4o-search-preview", + messages=[ + {"role": "user", "content": "What was a positive news story from today?"} + ], + ) + + await _verify_web_search_cost(test_custom_logger, "search_context_size_medium") + + +@pytest.mark.asyncio +async def test_openai_web_search_logging_cost_tracking_explicit_search_context_size(): + """search_context_size=low passed in, so cost tracked as `search_context_size_low`""" + test_custom_logger = await _setup_web_search_test() + + response = await litellm.acompletion( + model="openai/gpt-4o-search-preview", + messages=[ + {"role": "user", "content": "What was a positive news story from today?"} + ], + web_search_options={"search_context_size": "low"}, + ) + + await _verify_web_search_cost(test_custom_logger, "search_context_size_low") + + +@pytest.mark.asyncio +async def test_openai_web_search_with_tool_call_logging_cost_tracking(): + """search_context_size=high passed in tool call, so cost tracked as `search_context_size_high`""" + test_custom_logger = await _setup_web_search_test() + + response = await litellm.aresponses( + model="openai/gpt-4o", + input=[ + {"role": "user", "content": "What was a positive news story from today?"} + ], + tools=[{"type": "web_search_preview", "search_context_size": "high"}], + ) + + await _verify_web_search_cost(test_custom_logger, "search_context_size_high") From b9ee316e23e1211fd3c6378dd56ea4d48d7da249 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 16:59:04 -0700 Subject: [PATCH 12/33] fix model cost map --- ...odel_prices_and_context_window_backup.json | 60 +++++++++++-------- model_prices_and_context_window.json | 60 +++++++++++-------- 2 files changed, 72 insertions(+), 48 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 330fe890e4a..61f2703f55a 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -80,32 +80,38 @@ "supports_vision": true, "supports_prompt_caching": true, "supports_system_messages": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.030, + "search_context_size_medium": 0.035, + "search_context_size_high": 0.050 + } }, "gpt-4o-search-preview-2025-03-11": { "max_tokens": 16384, - "max_input_tokens": 128000, - "max_output_tokens": 16384, - "input_cost_per_token": 0.0000025, - "output_cost_per_token": 0.000010, - "input_cost_per_token_batches": 0.00000125, - "output_cost_per_token_batches": 0.00000500, - "cache_read_input_token_cost": 0.00000125, - "litellm_provider": "openai", - "mode": "chat", - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_web_search": true, - "search_context_cost_per_query": { - "search_context_size_low": 0.030, - "search_context_size_medium": 0.035, - "search_context_size_high": 0.050 - } + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.0000025, + "output_cost_per_token": 0.000010, + "input_cost_per_token_batches": 0.00000125, + "output_cost_per_token_batches": 0.00000500, + "cache_read_input_token_cost": 0.00000125, + "litellm_provider": "openai", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.030, + "search_context_size_medium": 0.035, + "search_context_size_high": 0.050 + } }, "gpt-4o-search-preview": { "max_tokens": 16384, @@ -255,7 +261,13 @@ "supports_vision": true, "supports_prompt_caching": true, "supports_system_messages": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.025, + "search_context_size_medium": 0.0275, + "search_context_size_high": 0.030 + } }, "gpt-4o-mini-search-preview-2025-03-11":{ "max_tokens": 16384, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 330fe890e4a..61f2703f55a 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -80,32 +80,38 @@ "supports_vision": true, "supports_prompt_caching": true, "supports_system_messages": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.030, + "search_context_size_medium": 0.035, + "search_context_size_high": 0.050 + } }, "gpt-4o-search-preview-2025-03-11": { "max_tokens": 16384, - "max_input_tokens": 128000, - "max_output_tokens": 16384, - "input_cost_per_token": 0.0000025, - "output_cost_per_token": 0.000010, - "input_cost_per_token_batches": 0.00000125, - "output_cost_per_token_batches": 0.00000500, - "cache_read_input_token_cost": 0.00000125, - "litellm_provider": "openai", - "mode": "chat", - "supports_function_calling": true, - "supports_parallel_function_calling": true, - "supports_response_schema": true, - "supports_vision": true, - "supports_prompt_caching": true, - "supports_system_messages": true, - "supports_tool_choice": true, - "supports_web_search": true, - "search_context_cost_per_query": { - "search_context_size_low": 0.030, - "search_context_size_medium": 0.035, - "search_context_size_high": 0.050 - } + "max_input_tokens": 128000, + "max_output_tokens": 16384, + "input_cost_per_token": 0.0000025, + "output_cost_per_token": 0.000010, + "input_cost_per_token_batches": 0.00000125, + "output_cost_per_token_batches": 0.00000500, + "cache_read_input_token_cost": 0.00000125, + "litellm_provider": "openai", + "mode": "chat", + "supports_function_calling": true, + "supports_parallel_function_calling": true, + "supports_response_schema": true, + "supports_vision": true, + "supports_prompt_caching": true, + "supports_system_messages": true, + "supports_tool_choice": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.030, + "search_context_size_medium": 0.035, + "search_context_size_high": 0.050 + } }, "gpt-4o-search-preview": { "max_tokens": 16384, @@ -255,7 +261,13 @@ "supports_vision": true, "supports_prompt_caching": true, "supports_system_messages": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.025, + "search_context_size_medium": 0.0275, + "search_context_size_high": 0.030 + } }, "gpt-4o-mini-search-preview-2025-03-11":{ "max_tokens": 16384, From bdefc9fe0630c69a1c328a5c9c257c4d16168333 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 17:02:28 -0700 Subject: [PATCH 13/33] add web search to gpt-4o-2024-08-06 --- litellm/model_prices_and_context_window_backup.json | 8 +++++++- model_prices_and_context_window.json | 8 +++++++- 2 files changed, 14 insertions(+), 2 deletions(-) diff --git a/litellm/model_prices_and_context_window_backup.json b/litellm/model_prices_and_context_window_backup.json index 61f2703f55a..1d4353e3ed8 100644 --- a/litellm/model_prices_and_context_window_backup.json +++ b/litellm/model_prices_and_context_window_backup.json @@ -548,7 +548,13 @@ "supports_vision": true, "supports_prompt_caching": true, "supports_system_messages": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.030, + "search_context_size_medium": 0.035, + "search_context_size_high": 0.050 + } }, "gpt-4o-2024-11-20": { "max_tokens": 16384, diff --git a/model_prices_and_context_window.json b/model_prices_and_context_window.json index 61f2703f55a..1d4353e3ed8 100644 --- a/model_prices_and_context_window.json +++ b/model_prices_and_context_window.json @@ -548,7 +548,13 @@ "supports_vision": true, "supports_prompt_caching": true, "supports_system_messages": true, - "supports_tool_choice": true + "supports_tool_choice": true, + "supports_web_search": true, + "search_context_cost_per_query": { + "search_context_size_low": 0.030, + "search_context_size_medium": 0.035, + "search_context_size_high": 0.050 + } }, "gpt-4o-2024-11-20": { "max_tokens": 16384, From beb68037d0c6c229f971284c9367a6ed0e69d9d9 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 17:02:43 -0700 Subject: [PATCH 14/33] move web search cost tracking --- .../test_built_in_tools_cost_tracking.py | 128 ++++++++++++++++++ .../test_token_counting.py | 82 ----------- 2 files changed, 128 insertions(+), 82 deletions(-) create mode 100644 tests/logging_callback_tests/test_built_in_tools_cost_tracking.py diff --git a/tests/logging_callback_tests/test_built_in_tools_cost_tracking.py b/tests/logging_callback_tests/test_built_in_tools_cost_tracking.py new file mode 100644 index 00000000000..726d270d55c --- /dev/null +++ b/tests/logging_callback_tests/test_built_in_tools_cost_tracking.py @@ -0,0 +1,128 @@ +import os +import sys +import traceback +import uuid +import pytest +from dotenv import load_dotenv +from fastapi import Request +from fastapi.routing import APIRoute + +load_dotenv() +import io +import os +import time +import json + +# this file is to test litellm/proxy + +sys.path.insert( + 0, os.path.abspath("../..") +) # Adds the parent directory to the system path +import litellm +import asyncio +from typing import Optional +from litellm.types.utils import StandardLoggingPayload, Usage, ModelInfoBase +from litellm.integrations.custom_logger import CustomLogger + + +class TestCustomLogger(CustomLogger): + def __init__(self): + self.recorded_usage: Optional[Usage] = None + self.standard_logging_payload: Optional[StandardLoggingPayload] = None + + async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): + standard_logging_payload = kwargs.get("standard_logging_object") + self.standard_logging_payload = standard_logging_payload + print( + "standard_logging_payload", + json.dumps(standard_logging_payload, indent=4, default=str), + ) + + self.recorded_usage = Usage( + prompt_tokens=standard_logging_payload.get("prompt_tokens"), + completion_tokens=standard_logging_payload.get("completion_tokens"), + total_tokens=standard_logging_payload.get("total_tokens"), + ) + pass + + +async def _setup_web_search_test(): + """Helper function to setup common test requirements""" + litellm._turn_on_debug() + test_custom_logger = TestCustomLogger() + litellm.callbacks = [test_custom_logger] + return test_custom_logger + + +async def _verify_web_search_cost(test_custom_logger, expected_context_size): + """Helper function to verify web search costs""" + await asyncio.sleep(1) + + standard_logging_payload = test_custom_logger.standard_logging_payload + response_cost = standard_logging_payload.get("response_cost") + assert response_cost is not None + + # Calculate token cost + model_map_information = standard_logging_payload["model_map_information"] + model_map_value: ModelInfoBase = model_map_information["model_map_value"] + total_token_cost = ( + standard_logging_payload["prompt_tokens"] + * model_map_value["input_cost_per_token"] + ) + ( + standard_logging_payload["completion_tokens"] + * model_map_value["output_cost_per_token"] + ) + + # Verify total cost + assert ( + response_cost + == total_token_cost + + model_map_value["search_context_cost_per_query"][expected_context_size] + ) + + +@pytest.mark.asyncio +async def test_openai_web_search_logging_cost_tracking_no_explicit_search_context_size(): + """Cost is tracked as `search_context_size_medium` when no `search_context_size` is passed in""" + test_custom_logger = await _setup_web_search_test() + + response = await litellm.acompletion( + model="openai/gpt-4o-search-preview", + messages=[ + {"role": "user", "content": "What was a positive news story from today?"} + ], + ) + + await _verify_web_search_cost(test_custom_logger, "search_context_size_medium") + + +@pytest.mark.asyncio +async def test_openai_web_search_logging_cost_tracking_explicit_search_context_size(): + """search_context_size=low passed in, so cost tracked as `search_context_size_low`""" + test_custom_logger = await _setup_web_search_test() + + response = await litellm.acompletion( + model="openai/gpt-4o-search-preview", + messages=[ + {"role": "user", "content": "What was a positive news story from today?"} + ], + web_search_options={"search_context_size": "low"}, + ) + + await _verify_web_search_cost(test_custom_logger, "search_context_size_low") + + +@pytest.mark.asyncio +async def test_openai_web_search_with_tool_call_logging_cost_tracking(): + """search_context_size=high passed in tool call, so cost tracked as `search_context_size_high`""" + test_custom_logger = await _setup_web_search_test() + + response = await litellm.aresponses( + model="openai/gpt-4o", + input=[ + {"role": "user", "content": "What was a positive news story from today?"} + ], + tools=[{"type": "web_search_preview", "search_context_size": "high"}], + ) + + await _verify_web_search_cost(test_custom_logger, "search_context_size_high") diff --git a/tests/logging_callback_tests/test_token_counting.py b/tests/logging_callback_tests/test_token_counting.py index 05e21efe0d9..ed778fa86b2 100644 --- a/tests/logging_callback_tests/test_token_counting.py +++ b/tests/logging_callback_tests/test_token_counting.py @@ -246,85 +246,3 @@ async def test_stream_token_counting_anthropic_with_include_usage(): ) <= 10 ) - - -async def _setup_web_search_test(): - """Helper function to setup common test requirements""" - litellm._turn_on_debug() - test_custom_logger = TestCustomLogger() - litellm.callbacks = [test_custom_logger] - return test_custom_logger - - -async def _verify_web_search_cost(test_custom_logger, expected_context_size): - """Helper function to verify web search costs""" - await asyncio.sleep(1) - - standard_logging_payload = test_custom_logger.standard_logging_payload - response_cost = standard_logging_payload.get("response_cost") - assert response_cost is not None - - # Calculate token cost - model_map_information = standard_logging_payload["model_map_information"] - model_map_value: ModelInfoBase = model_map_information["model_map_value"] - total_token_cost = ( - standard_logging_payload["prompt_tokens"] - * model_map_value["input_cost_per_token"] - ) + ( - standard_logging_payload["completion_tokens"] - * model_map_value["output_cost_per_token"] - ) - - # Verify total cost - assert ( - response_cost - == total_token_cost - + model_map_value["search_context_cost_per_query"][expected_context_size] - ) - - -@pytest.mark.asyncio -async def test_openai_web_search_logging_cost_tracking_no_explicit_search_context_size(): - """Cost is tracked as `search_context_size_medium` when no `search_context_size` is passed in""" - test_custom_logger = await _setup_web_search_test() - - response = await litellm.acompletion( - model="openai/gpt-4o-search-preview", - messages=[ - {"role": "user", "content": "What was a positive news story from today?"} - ], - ) - - await _verify_web_search_cost(test_custom_logger, "search_context_size_medium") - - -@pytest.mark.asyncio -async def test_openai_web_search_logging_cost_tracking_explicit_search_context_size(): - """search_context_size=low passed in, so cost tracked as `search_context_size_low`""" - test_custom_logger = await _setup_web_search_test() - - response = await litellm.acompletion( - model="openai/gpt-4o-search-preview", - messages=[ - {"role": "user", "content": "What was a positive news story from today?"} - ], - web_search_options={"search_context_size": "low"}, - ) - - await _verify_web_search_cost(test_custom_logger, "search_context_size_low") - - -@pytest.mark.asyncio -async def test_openai_web_search_with_tool_call_logging_cost_tracking(): - """search_context_size=high passed in tool call, so cost tracked as `search_context_size_high`""" - test_custom_logger = await _setup_web_search_test() - - response = await litellm.aresponses( - model="openai/gpt-4o", - input=[ - {"role": "user", "content": "What was a positive news story from today?"} - ], - tools=[{"type": "web_search_preview", "search_context_size": "high"}], - ) - - await _verify_web_search_cost(test_custom_logger, "search_context_size_high") From 904795443d97aa0d1fe656f65fbdcc1e3b8e0079 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 17:06:34 -0700 Subject: [PATCH 15/33] get_cost_for_built_in_tools --- .../llm_cost_calc/tool_call_cost_tracking.py | 27 ++++++++++--------- 1 file changed, 15 insertions(+), 12 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py index 1717565831d..b1cdc119d63 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -38,24 +38,27 @@ class StandardBuiltInToolCostTracking: model_info = litellm.get_model_info( model=model, custom_llm_provider=custom_llm_provider ) + + if ( + standard_built_in_tools_params is not None + and standard_built_in_tools_params.get("web_search_options", None) + is not None + ): + return StandardBuiltInToolCostTracking.get_cost_for_web_search( + web_search_options=standard_built_in_tools_params.get( + "web_search_options", None + ), + model_info=model_info, + ) + if isinstance(response_object, ModelResponse): - if ( - standard_built_in_tools_params is not None - and standard_built_in_tools_params.get("web_search_options", None) - is not None - ): - return StandardBuiltInToolCostTracking.get_cost_for_web_search( - web_search_options=standard_built_in_tools_params.get( - "web_search_options", None - ), - model_info=model_info, - ) - elif StandardBuiltInToolCostTracking.response_includes_annotations( + if StandardBuiltInToolCostTracking.response_includes_annotations( response_object ): return StandardBuiltInToolCostTracking.get_default_cost_for_web_search( model_info ) + return 0.0 @staticmethod From 81033dc8c9465f4c10209ee7a2c456d9df079615 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 17:21:41 -0700 Subject: [PATCH 16/33] test_openai_responses_api_web_search_cost_tracking --- .../llm_cost_calc/tool_call_cost_tracking.py | 23 +++++++++++++++---- .../test_built_in_tools_cost_tracking.py | 20 ++++++++++++---- 2 files changed, 34 insertions(+), 9 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py index b1cdc119d63..629d6c9e311 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -5,7 +5,11 @@ Helper utilities for tracking the cost of built-in tools. from typing import Any, Dict, Optional import litellm -from litellm.types.llms.openai import WebSearchOptions +from litellm.types.llms.openai import ( + ResponseOutputItem, + ResponsesAPIResponse, + WebSearchOptions, +) from litellm.types.utils import ( ModelInfo, ModelResponse, @@ -52,13 +56,12 @@ class StandardBuiltInToolCostTracking: ) if isinstance(response_object, ModelResponse): - if StandardBuiltInToolCostTracking.response_includes_annotations( + if StandardBuiltInToolCostTracking.chat_completion_response_includes_annotations( response_object ): return StandardBuiltInToolCostTracking.get_default_cost_for_web_search( model_info ) - return 0.0 @staticmethod @@ -99,7 +102,9 @@ class StandardBuiltInToolCostTracking: return search_context_pricing.get("search_context_size_medium", 0.0) @staticmethod - def response_includes_annotations(response_object: ModelResponse) -> bool: + def chat_completion_response_includes_annotations( + response_object: ModelResponse, + ) -> bool: for _choice in response_object.choices: message = getattr(_choice, "message", None) if message is not None and hasattr(message, "annotations"): @@ -115,6 +120,14 @@ class StandardBuiltInToolCostTracking: # Look for web search tool in the tools array for tool in tools: if isinstance(tool, dict): - if "search_context_size" in tool: + if StandardBuiltInToolCostTracking._web_search_tool_call(tool): return WebSearchOptions(**tool) return None + + @staticmethod + def _web_search_tool_call(tool: Dict) -> bool: + if tool.get("type", None) == "web_search_preview": + return True + if "search_context_size" in tool: + return True + return False diff --git a/tests/logging_callback_tests/test_built_in_tools_cost_tracking.py b/tests/logging_callback_tests/test_built_in_tools_cost_tracking.py index 726d270d55c..c269cd8cb22 100644 --- a/tests/logging_callback_tests/test_built_in_tools_cost_tracking.py +++ b/tests/logging_callback_tests/test_built_in_tools_cost_tracking.py @@ -113,8 +113,20 @@ async def test_openai_web_search_logging_cost_tracking_explicit_search_context_s @pytest.mark.asyncio -async def test_openai_web_search_with_tool_call_logging_cost_tracking(): - """search_context_size=high passed in tool call, so cost tracked as `search_context_size_high`""" +@pytest.mark.parametrize( + "tools_config,expected_context_size", + [ + ( + [{"type": "web_search_preview", "search_context_size": "high"}], + "search_context_size_high", + ), + ([{"type": "web_search_preview"}], "search_context_size_medium"), + ], +) +async def test_openai_responses_api_web_search_cost_tracking( + tools_config, expected_context_size +): + """Test web search cost tracking with different search context sizes""" test_custom_logger = await _setup_web_search_test() response = await litellm.aresponses( @@ -122,7 +134,7 @@ async def test_openai_web_search_with_tool_call_logging_cost_tracking(): input=[ {"role": "user", "content": "What was a positive news story from today?"} ], - tools=[{"type": "web_search_preview", "search_context_size": "high"}], + tools=tools_config, ) - await _verify_web_search_cost(test_custom_logger, "search_context_size_high") + await _verify_web_search_cost(test_custom_logger, expected_context_size) From f74640cb99cbd473def18f7bee7958418a46f3e4 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 17:24:29 -0700 Subject: [PATCH 17/33] test_openai_responses_api_web_search_cost_tracking --- .../llm_cost_calc/tool_call_cost_tracking.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py index 629d6c9e311..8f7d14b20fd 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -86,8 +86,9 @@ class StandardBuiltInToolCostTracking: return search_context_pricing.get("search_context_size_medium", 0.0) elif web_search_options.get("search_context_size", None) == "high": return search_context_pricing.get("search_context_size_high", 0.0) - else: - return 0.0 + return StandardBuiltInToolCostTracking.get_default_cost_for_web_search( + model_info + ) @staticmethod def get_default_cost_for_web_search(model_info: ModelInfo) -> float: From 43a6c4c6351e3c37f433bf8627751508a8876841 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 17:29:17 -0700 Subject: [PATCH 18/33] test_openai_responses_api_web_search_cost_tracking --- .../test_built_in_tools_cost_tracking.py | 63 +++++++++++-------- 1 file changed, 37 insertions(+), 26 deletions(-) diff --git a/tests/logging_callback_tests/test_built_in_tools_cost_tracking.py b/tests/logging_callback_tests/test_built_in_tools_cost_tracking.py index c269cd8cb22..d4013f2e8db 100644 --- a/tests/logging_callback_tests/test_built_in_tools_cost_tracking.py +++ b/tests/logging_callback_tests/test_built_in_tools_cost_tracking.py @@ -82,51 +82,56 @@ async def _verify_web_search_cost(test_custom_logger, expected_context_size): @pytest.mark.asyncio -async def test_openai_web_search_logging_cost_tracking_no_explicit_search_context_size(): - """Cost is tracked as `search_context_size_medium` when no `search_context_size` is passed in""" +@pytest.mark.parametrize( + "web_search_options,expected_context_size", + [ + (None, "search_context_size_medium"), + ({"search_context_size": "low"}, "search_context_size_low"), + ({"search_context_size": "high"}, "search_context_size_high"), + ], +) +async def test_openai_web_search_logging_cost_tracking( + web_search_options, expected_context_size +): + """Test web search cost tracking with different search context sizes""" test_custom_logger = await _setup_web_search_test() - response = await litellm.acompletion( - model="openai/gpt-4o-search-preview", - messages=[ + request_kwargs = { + "model": "openai/gpt-4o-search-preview", + "messages": [ {"role": "user", "content": "What was a positive news story from today?"} ], - ) + } + if web_search_options is not None: + request_kwargs["web_search_options"] = web_search_options - await _verify_web_search_cost(test_custom_logger, "search_context_size_medium") + response = await litellm.acompletion(**request_kwargs) - -@pytest.mark.asyncio -async def test_openai_web_search_logging_cost_tracking_explicit_search_context_size(): - """search_context_size=low passed in, so cost tracked as `search_context_size_low`""" - test_custom_logger = await _setup_web_search_test() - - response = await litellm.acompletion( - model="openai/gpt-4o-search-preview", - messages=[ - {"role": "user", "content": "What was a positive news story from today?"} - ], - web_search_options={"search_context_size": "low"}, - ) - - await _verify_web_search_cost(test_custom_logger, "search_context_size_low") + await _verify_web_search_cost(test_custom_logger, expected_context_size) @pytest.mark.asyncio @pytest.mark.parametrize( - "tools_config,expected_context_size", + "tools_config,expected_context_size,stream", [ ( [{"type": "web_search_preview", "search_context_size": "high"}], "search_context_size_high", + True, ), - ([{"type": "web_search_preview"}], "search_context_size_medium"), + ( + [{"type": "web_search_preview", "search_context_size": "high"}], + "search_context_size_high", + False, + ), + ([{"type": "web_search_preview"}], "search_context_size_medium", True), + ([{"type": "web_search_preview"}], "search_context_size_medium", False), ], ) async def test_openai_responses_api_web_search_cost_tracking( - tools_config, expected_context_size + tools_config, expected_context_size, stream ): - """Test web search cost tracking with different search context sizes""" + """Test web search cost tracking with different search context sizes and streaming options""" test_custom_logger = await _setup_web_search_test() response = await litellm.aresponses( @@ -135,6 +140,12 @@ async def test_openai_responses_api_web_search_cost_tracking( {"role": "user", "content": "What was a positive news story from today?"} ], tools=tools_config, + stream=stream, ) + if stream is True: + async for chunk in response: + print("chunk", chunk) + else: + print("response", response) await _verify_web_search_cost(test_custom_logger, expected_context_size) From 7f807cd6baf1991ce525dd1036b9255b6b260f39 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 17:31:47 -0700 Subject: [PATCH 19/33] fix code quality check --- litellm/litellm_core_utils/litellm_logging.py | 1 - .../llm_cost_calc/tool_call_cost_tracking.py | 6 +----- 2 files changed, 1 insertion(+), 6 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 8ce6b156686..88e8c9cfe71 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -51,7 +51,6 @@ from litellm.types.llms.openai import ( HttpxBinaryResponseContent, ResponseCompletedEvent, ResponsesAPIResponse, - WebSearchOptions, ) from litellm.types.rerank import RerankResponse from litellm.types.router import SPECIAL_MODEL_INFO_PARAMS diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py index 8f7d14b20fd..93e5be63110 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -5,11 +5,7 @@ Helper utilities for tracking the cost of built-in tools. from typing import Any, Dict, Optional import litellm -from litellm.types.llms.openai import ( - ResponseOutputItem, - ResponsesAPIResponse, - WebSearchOptions, -) +from litellm.types.llms.openai import WebSearchOptions from litellm.types.utils import ( ModelInfo, ModelResponse, From 0f36b8afc2ffbdc6854a83462480be586be6b3c8 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 17:36:22 -0700 Subject: [PATCH 20/33] chat_completion_response_includes_annotations --- .../llm_cost_calc/tool_call_cost_tracking.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py index 93e5be63110..594b59989ae 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -104,7 +104,12 @@ class StandardBuiltInToolCostTracking: ) -> bool: for _choice in response_object.choices: message = getattr(_choice, "message", None) - if message is not None and hasattr(message, "annotations"): + if ( + message is not None + and hasattr(message, "annotations") + and message.annotations is not None + and len(message.annotations) > 0 + ): return True return False From 8766b43a4bd824405b3e19bab5bb79f6a37a963f Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 17:53:06 -0700 Subject: [PATCH 21/33] fix StandardBuiltInToolsParams --- litellm/litellm_core_utils/litellm_logging.py | 5 +- .../llm_cost_calc/tool_call_cost_tracking.py | 63 ++++++++++++++----- litellm/types/llms/openai.py | 8 +++ litellm/types/utils.py | 2 + 4 files changed, 63 insertions(+), 15 deletions(-) diff --git a/litellm/litellm_core_utils/litellm_logging.py b/litellm/litellm_core_utils/litellm_logging.py index 88e8c9cfe71..67511968e21 100644 --- a/litellm/litellm_core_utils/litellm_logging.py +++ b/litellm/litellm_core_utils/litellm_logging.py @@ -386,7 +386,10 @@ class Logging(LiteLLMLoggingBaseClass): return StandardBuiltInToolsParams( web_search_options=StandardBuiltInToolCostTracking._get_web_search_options( kwargs or {} - ) + ), + file_search=StandardBuiltInToolCostTracking._get_file_search_tool_call( + kwargs or {} + ), ) def update_environment_variables( diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py index 594b59989ae..723910dbb0c 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -5,7 +5,7 @@ Helper utilities for tracking the cost of built-in tools. from typing import Any, Dict, Optional import litellm -from litellm.types.llms.openai import WebSearchOptions +from litellm.types.llms.openai import FileSearchTool, WebSearchOptions from litellm.types.utils import ( ModelInfo, ModelResponse, @@ -39,17 +39,22 @@ class StandardBuiltInToolCostTracking: model=model, custom_llm_provider=custom_llm_provider ) - if ( - standard_built_in_tools_params is not None - and standard_built_in_tools_params.get("web_search_options", None) - is not None - ): - return StandardBuiltInToolCostTracking.get_cost_for_web_search( - web_search_options=standard_built_in_tools_params.get( - "web_search_options", None - ), - model_info=model_info, - ) + if standard_built_in_tools_params is not None: + if ( + standard_built_in_tools_params.get("web_search_options", None) + is not None + ): + return StandardBuiltInToolCostTracking.get_cost_for_web_search( + web_search_options=standard_built_in_tools_params.get( + "web_search_options", None + ), + model_info=model_info, + ) + + if standard_built_in_tools_params.get("file_search", None) is not None: + return StandardBuiltInToolCostTracking.get_cost_for_file_search( + file_search=standard_built_in_tools_params.get("file_search", None), + ) if isinstance(response_object, ModelResponse): if StandardBuiltInToolCostTracking.chat_completion_response_includes_annotations( @@ -98,6 +103,19 @@ class StandardBuiltInToolCostTracking: ) or {} return search_context_pricing.get("search_context_size_medium", 0.0) + @staticmethod + def get_cost_for_file_search( + file_search: Optional[FileSearchTool] = None, + ) -> float: + """ " + Charged at $2.50/1k calls + + Doc: https://platform.openai.com/docs/pricing#built-in-tools + """ + if file_search is None: + return 0.0 + return 2.5 / 1000 + @staticmethod def chat_completion_response_includes_annotations( response_object: ModelResponse, @@ -122,14 +140,31 @@ class StandardBuiltInToolCostTracking: # Look for web search tool in the tools array for tool in tools: if isinstance(tool, dict): - if StandardBuiltInToolCostTracking._web_search_tool_call(tool): + if StandardBuiltInToolCostTracking._is_web_search_tool_call(tool): return WebSearchOptions(**tool) return None @staticmethod - def _web_search_tool_call(tool: Dict) -> bool: + def _get_file_search_tool_call(kwargs: Dict) -> Optional[FileSearchTool]: + if "tools" in kwargs: + tools = kwargs.get("tools", []) + # Look for web search tool in the tools array + for tool in tools: + if isinstance(tool, dict): + if StandardBuiltInToolCostTracking._is_file_search_tool_call(tool): + return FileSearchTool(**tool) + return None + + @staticmethod + def _is_web_search_tool_call(tool: Dict) -> bool: if tool.get("type", None) == "web_search_preview": return True if "search_context_size" in tool: return True return False + + @staticmethod + def _is_file_search_tool_call(tool: Dict) -> bool: + if tool.get("type", None) == "file_search": + return True + return False diff --git a/litellm/types/llms/openai.py b/litellm/types/llms/openai.py index d80ba350d16..19899648f58 100644 --- a/litellm/types/llms/openai.py +++ b/litellm/types/llms/openai.py @@ -421,6 +421,14 @@ class WebSearchOptions(TypedDict, total=False): """Approximate location parameters for the search.""" +class FileSearchTool(TypedDict, total=False): + type: Literal["file_search"] + """The type of tool being defined: `file_search`""" + + vector_store_ids: Optional[List[str]] + """The IDs of the vector stores to search.""" + + class ChatCompletionAnnotationURLCitation(TypedDict, total=False): end_index: int """The index of the last character of the URL citation in the message.""" diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 1c1ee24bfbf..cec59c0b3b9 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -13,6 +13,7 @@ from openai.types.completion_usage import ( CompletionUsage, PromptTokensDetails, ) +from openai.types.llms.openai import FileSearchTool from openai.types.moderation import ( Categories, CategoryAppliedInputTypes, @@ -1633,6 +1634,7 @@ class StandardBuiltInToolsParams(TypedDict, total=False): """ web_search_options: Optional[WebSearchOptions] + file_search: Optional[FileSearchTool] class StandardLoggingPromptManagementMetadata(TypedDict): From 4d4e0d8a6dc545c4905dee2c18aa9fcba4f3eb19 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 17:56:14 -0700 Subject: [PATCH 22/33] FileSearchTool --- litellm/types/utils.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/types/utils.py b/litellm/types/utils.py index cec59c0b3b9..83e8525fa9e 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -13,7 +13,6 @@ from openai.types.completion_usage import ( CompletionUsage, PromptTokensDetails, ) -from openai.types.llms.openai import FileSearchTool from openai.types.moderation import ( Categories, CategoryAppliedInputTypes, @@ -33,6 +32,7 @@ from .llms.openai import ( ChatCompletionThinkingBlock, ChatCompletionToolCallChunk, ChatCompletionUsageBlock, + FileSearchTool, OpenAIChatCompletionChunk, WebSearchOptions, ) From ded612563dd50145a605b22bf7a481c208d28fd5 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 18:02:22 -0700 Subject: [PATCH 23/33] _get_file_search_tool_call --- .../llm_cost_calc/tool_call_cost_tracking.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py index 723910dbb0c..ca20bce8bfa 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -137,6 +137,8 @@ class StandardBuiltInToolCostTracking: return WebSearchOptions(**kwargs.get("web_search_options", {})) if "tools" in kwargs: tools = kwargs.get("tools", []) + if tools is None: + return None # Look for web search tool in the tools array for tool in tools: if isinstance(tool, dict): @@ -148,6 +150,8 @@ class StandardBuiltInToolCostTracking: def _get_file_search_tool_call(kwargs: Dict) -> Optional[FileSearchTool]: if "tools" in kwargs: tools = kwargs.get("tools", []) + if tools is None: + return None # Look for web search tool in the tools array for tool in tools: if isinstance(tool, dict): From 1b9f1837cd1a703f083dab98fd451797bf8d8b43 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 18:04:33 -0700 Subject: [PATCH 24/33] _get_file_search_tool_call --- .../llm_cost_calc/tool_call_cost_tracking.py | 25 ++++++++++++------- 1 file changed, 16 insertions(+), 9 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py index ca20bce8bfa..1919f757233 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -2,7 +2,7 @@ Helper utilities for tracking the cost of built-in tools. """ -from typing import Any, Dict, Optional +from typing import Any, Dict, List, Optional import litellm from litellm.types.llms.openai import FileSearchTool, WebSearchOptions @@ -135,10 +135,11 @@ class StandardBuiltInToolCostTracking: def _get_web_search_options(kwargs: Dict) -> Optional[WebSearchOptions]: if "web_search_options" in kwargs: return WebSearchOptions(**kwargs.get("web_search_options", {})) - if "tools" in kwargs: - tools = kwargs.get("tools", []) - if tools is None: - return None + + tools = StandardBuiltInToolCostTracking._get_tools_from_kwargs( + kwargs, "web_search_preview" + ) + if tools: # Look for web search tool in the tools array for tool in tools: if isinstance(tool, dict): @@ -147,12 +148,18 @@ class StandardBuiltInToolCostTracking: return None @staticmethod - def _get_file_search_tool_call(kwargs: Dict) -> Optional[FileSearchTool]: + def _get_tools_from_kwargs(kwargs: Dict, tool_type: str) -> Optional[List[Dict]]: if "tools" in kwargs: tools = kwargs.get("tools", []) - if tools is None: - return None - # Look for web search tool in the tools array + return tools + return None + + @staticmethod + def _get_file_search_tool_call(kwargs: Dict) -> Optional[FileSearchTool]: + tools = StandardBuiltInToolCostTracking._get_tools_from_kwargs( + kwargs, "file_search" + ) + if tools: for tool in tools: if isinstance(tool, dict): if StandardBuiltInToolCostTracking._is_file_search_tool_call(tool): From cf03b2a797a606c64a2305cd8c7410d8a6253fa6 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 18:09:04 -0700 Subject: [PATCH 25/33] test_langfuse_logging_completion --- litellm/types/utils.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/litellm/types/utils.py b/litellm/types/utils.py index 83e8525fa9e..2cc06eecbf3 100644 --- a/litellm/types/utils.py +++ b/litellm/types/utils.py @@ -598,6 +598,11 @@ class Message(OpenAIObject): # OpenAI compatible APIs like mistral API will raise an error if audio is passed in del self.audio + if annotations is None: + # ensure default response matches OpenAI spec + # Some OpenAI compatible APIs raise an error if annotations are passed in + del self.annotations + if reasoning_content is None: # ensure default response matches OpenAI spec del self.reasoning_content From edfcff47b950f82704c464179ace88a063f1d458 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 18:24:47 -0700 Subject: [PATCH 26/33] fix bug when unable to look up model info --- .../llm_cost_calc/tool_call_cost_tracking.py | 28 +++++++++++++++---- 1 file changed, 23 insertions(+), 5 deletions(-) diff --git a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py index 1919f757233..74d15e9a015 100644 --- a/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py +++ b/litellm/litellm_core_utils/llm_cost_calc/tool_call_cost_tracking.py @@ -35,15 +35,15 @@ class StandardBuiltInToolCostTracking: - Web Search """ - model_info = litellm.get_model_info( - model=model, custom_llm_provider=custom_llm_provider - ) - if standard_built_in_tools_params is not None: if ( standard_built_in_tools_params.get("web_search_options", None) is not None ): + model_info = StandardBuiltInToolCostTracking._safe_get_model_info( + model=model, custom_llm_provider=custom_llm_provider + ) + return StandardBuiltInToolCostTracking.get_cost_for_web_search( web_search_options=standard_built_in_tools_params.get( "web_search_options", None @@ -60,11 +60,25 @@ class StandardBuiltInToolCostTracking: if StandardBuiltInToolCostTracking.chat_completion_response_includes_annotations( response_object ): + model_info = StandardBuiltInToolCostTracking._safe_get_model_info( + model=model, custom_llm_provider=custom_llm_provider + ) return StandardBuiltInToolCostTracking.get_default_cost_for_web_search( model_info ) return 0.0 + @staticmethod + def _safe_get_model_info( + model: str, custom_llm_provider: Optional[str] = None + ) -> Optional[ModelInfo]: + try: + return litellm.get_model_info( + model=model, custom_llm_provider=custom_llm_provider + ) + except Exception: + return None + @staticmethod def get_cost_for_web_search( web_search_options: Optional[WebSearchOptions] = None, @@ -92,12 +106,16 @@ class StandardBuiltInToolCostTracking: ) @staticmethod - def get_default_cost_for_web_search(model_info: ModelInfo) -> float: + def get_default_cost_for_web_search( + model_info: Optional[ModelInfo] = None, + ) -> float: """ If no web search options are provided, use the `search_context_size_medium` pricing. https://platform.openai.com/docs/pricing#web-search """ + if model_info is None: + return 0.0 search_context_pricing: SearchContextCostPerQuery = ( model_info.get("search_context_cost_per_query", {}) or {} ) or {} From a10b3164be69fdd8368399039dea89312e667ea9 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 19:02:39 -0700 Subject: [PATCH 27/33] docs web search --- docs/my-website/sidebars.js | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 96267d09dab..b0f6db7c444 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -244,6 +244,7 @@ const sidebars = { "completion/provider_specific_params", "guides/finetuned_models", "completion/audio", + "completion/web_search", "completion/document_understanding", "completion/vision", "completion/json_mode", From 4bfa6ba186dbe3a034b296f16586d2a4077f4875 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 19:07:17 -0700 Subject: [PATCH 28/33] docs web search --- docs/my-website/docs/completion/web_search.md | 128 ++++++++++++++++++ 1 file changed, 128 insertions(+) create mode 100644 docs/my-website/docs/completion/web_search.md diff --git a/docs/my-website/docs/completion/web_search.md b/docs/my-website/docs/completion/web_search.md new file mode 100644 index 00000000000..7d0ec101754 --- /dev/null +++ b/docs/my-website/docs/completion/web_search.md @@ -0,0 +1,128 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Using Web Search + +How to use web search capabilities with the `/chat/completions` endpoint + + + +Works for: +- OpenAI GPT-4 with web search + +## Quick Start + + + + +```python +from litellm import completion + +response = completion( + model="openai/gpt-4-turbo-preview", + messages=[ + { + "role": "user", + "content": "What was a positive news story from today?", + } + ], + tool_choice="auto", # Enable web search capability +) +``` + + + +1. Setup config.yaml + +```yaml +model_list: + - model_name: gpt-4-turbo + litellm_params: + model: openai/gpt-4-turbo-preview + api_key: os.environ/OPENAI_API_KEY +``` + +2. Start the proxy + +```bash +litellm --config /path/to/config.yaml +``` + +3. Test it! + +```bash +curl -X POST 'http://0.0.0.0:4000/chat/completions' \ +-H 'Content-Type: application/json' \ +-H 'Authorization: Bearer sk-1234' \ +-d '{ + "model": "gpt-4-turbo", + "messages": [ + { + "role": "user", + "content": "What was a positive news story from today?" + } + ], + "tool_choice": "auto" +}' +``` + + + +## Checking if a model supports web search + + + + +Use `litellm.supports_web_search(model="openai/gpt-4-turbo-preview")` -> returns `True` if model can perform web searches + +```python +assert litellm.supports_web_search(model="openai/gpt-4-turbo-preview") == True +``` + + + + +1. Define OpenAI models in config.yaml + +```yaml +model_list: + - model_name: gpt-4-turbo + litellm_params: + model: openai/gpt-4-turbo-preview + api_key: os.environ/OPENAI_API_KEY + model_info: + supports_web_search: True +``` + +2. Run proxy server + +```bash +litellm --config config.yaml +``` + +3. Call `/model_group/info` to check if a model supports web search + +```shell +curl -X 'GET' \ + 'http://localhost:4000/model_group/info' \ + -H 'accept: application/json' \ + -H 'x-api-key: sk-1234' +``` + +Expected Response + +```json +{ + "data": [ + { + "model_group": "gpt-4-turbo", + "providers": ["openai"], + "max_tokens": 128000, + "supports_web_search": true, # 👈 supports_web_search is true + } + ] +} +``` + + + From 5ae11007b684f688ce99f217b43b998066cc28d0 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 19:13:26 -0700 Subject: [PATCH 29/33] docs web search --- docs/my-website/docs/completion/web_search.md | 19 ++++++++++--------- 1 file changed, 10 insertions(+), 9 deletions(-) diff --git a/docs/my-website/docs/completion/web_search.md b/docs/my-website/docs/completion/web_search.md index 7d0ec101754..f8ec0098b6e 100644 --- a/docs/my-website/docs/completion/web_search.md +++ b/docs/my-website/docs/completion/web_search.md @@ -3,19 +3,20 @@ import TabItem from '@theme/TabItem'; # Using Web Search -How to use web search capabilities with the `/chat/completions` endpoint +Use web search with litellm - - -Works for: -- OpenAI GPT-4 with web search +| Feature | Details | +|---------|---------| +| Supported Endpoints | - `/chat/completions`
- `/responses` | +| Supported Providers | `openai` | +| LiteLLM Version | `v1.63.15-nightly` or higher | ## Quick Start -```python +```python showLineNumbers from litellm import completion response = completion( @@ -50,7 +51,7 @@ litellm --config /path/to/config.yaml 3. Test it! -```bash +```bash showLineNumbers curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ @@ -75,7 +76,7 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ Use `litellm.supports_web_search(model="openai/gpt-4-turbo-preview")` -> returns `True` if model can perform web searches -```python +```python showLineNumbers assert litellm.supports_web_search(model="openai/gpt-4-turbo-preview") == True ``` @@ -111,7 +112,7 @@ curl -X 'GET' \ Expected Response -```json +```json showLineNumbers { "data": [ { From e0133889e53d868f45bbe9f1b1cf7b800d1cc852 Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 19:15:52 -0700 Subject: [PATCH 30/33] docs web search --- docs/my-website/docs/completion/web_search.md | 21 +++++++++---------- 1 file changed, 10 insertions(+), 11 deletions(-) diff --git a/docs/my-website/docs/completion/web_search.md b/docs/my-website/docs/completion/web_search.md index f8ec0098b6e..803d080b108 100644 --- a/docs/my-website/docs/completion/web_search.md +++ b/docs/my-website/docs/completion/web_search.md @@ -9,6 +9,7 @@ Use web search with litellm |---------|---------| | Supported Endpoints | - `/chat/completions`
- `/responses` | | Supported Providers | `openai` | +| LiteLLM Cost Tracking | ✅ Supported | | LiteLLM Version | `v1.63.15-nightly` or higher | ## Quick Start @@ -20,14 +21,13 @@ Use web search with litellm from litellm import completion response = completion( - model="openai/gpt-4-turbo-preview", + model="openai/gpt-4o-search-preview", messages=[ { "role": "user", "content": "What was a positive news story from today?", } ], - tool_choice="auto", # Enable web search capability ) ``` @@ -37,9 +37,9 @@ response = completion( ```yaml model_list: - - model_name: gpt-4-turbo + - model_name: gpt-4o-search-preview litellm_params: - model: openai/gpt-4-turbo-preview + model: openai/gpt-4o-search-preview api_key: os.environ/OPENAI_API_KEY ``` @@ -56,14 +56,13 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -H 'Content-Type: application/json' \ -H 'Authorization: Bearer sk-1234' \ -d '{ - "model": "gpt-4-turbo", + "model": "gpt-4o-search-preview", "messages": [ { "role": "user", "content": "What was a positive news story from today?" } ], - "tool_choice": "auto" }' ``` @@ -74,10 +73,10 @@ curl -X POST 'http://0.0.0.0:4000/chat/completions' \ -Use `litellm.supports_web_search(model="openai/gpt-4-turbo-preview")` -> returns `True` if model can perform web searches +Use `litellm.supports_web_search(model="openai/gpt-4o-search-preview")` -> returns `True` if model can perform web searches ```python showLineNumbers -assert litellm.supports_web_search(model="openai/gpt-4-turbo-preview") == True +assert litellm.supports_web_search(model="openai/gpt-4o-search-preview") == True ``` @@ -87,9 +86,9 @@ assert litellm.supports_web_search(model="openai/gpt-4-turbo-preview") == True ```yaml model_list: - - model_name: gpt-4-turbo + - model_name: gpt-4o-search-preview litellm_params: - model: openai/gpt-4-turbo-preview + model: openai/gpt-4o-search-preview api_key: os.environ/OPENAI_API_KEY model_info: supports_web_search: True @@ -116,7 +115,7 @@ Expected Response { "data": [ { - "model_group": "gpt-4-turbo", + "model_group": "gpt-4o-search-preview", "providers": ["openai"], "max_tokens": 128000, "supports_web_search": true, # 👈 supports_web_search is true From 64ba5c8d8cfc1699d0128d9447ee10f3106a022b Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 19:24:31 -0700 Subject: [PATCH 31/33] docs web search --- docs/my-website/docs/completion/web_search.md | 72 ++++++++++++++++--- 1 file changed, 64 insertions(+), 8 deletions(-) diff --git a/docs/my-website/docs/completion/web_search.md b/docs/my-website/docs/completion/web_search.md index 803d080b108..e9aebc96ed7 100644 --- a/docs/my-website/docs/completion/web_search.md +++ b/docs/my-website/docs/completion/web_search.md @@ -51,19 +51,75 @@ litellm --config /path/to/config.yaml 3. Test it! -```bash showLineNumbers -curl -X POST 'http://0.0.0.0:4000/chat/completions' \ --H 'Content-Type: application/json' \ --H 'Authorization: Bearer sk-1234' \ --d '{ - "model": "gpt-4o-search-preview", - "messages": [ +```python showLineNumbers +from openai import OpenAI + +# Point to your proxy server +client = OpenAI( + api_key="sk-1234", + base_url="http://0.0.0.0:4000" +) + +response = client.chat.completions.create( + model="gpt-4o-search-preview", + messages=[ + { + "role": "user", + "content": "What was a positive news story from today?" + } + ] +) +``` + + + +## Search context size + + + + +```python showLineNumbers +from litellm import completion + +# Customize search context size +response = completion( + model="openai/gpt-4o-search-preview", + messages=[ + { + "role": "user", + "content": "What was a positive news story from today?", + } + ], + web_search_options={ + "search_context_size": "low" # Options: "low", "medium" (default), "high" + } +) +``` + + + +```python showLineNumbers +from openai import OpenAI + +# Point to your proxy server +client = OpenAI( + api_key="sk-1234", + base_url="http://0.0.0.0:4000" +) + +# Customize search context size +response = client.chat.completions.create( + model="gpt-4o-search-preview", + messages=[ { "role": "user", "content": "What was a positive news story from today?" } ], -}' + web_search_options={ + "search_context_size": "low" # Options: "low", "medium" (default), "high" + } +) ``` From b76e0dcc648ff38718e5ae0b8890035f05da64af Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 19:28:24 -0700 Subject: [PATCH 32/33] docs web search --- docs/my-website/docs/completion/web_search.md | 128 +++++++++++++++++- 1 file changed, 126 insertions(+), 2 deletions(-) diff --git a/docs/my-website/docs/completion/web_search.md b/docs/my-website/docs/completion/web_search.md index e9aebc96ed7..7a67dc265e4 100644 --- a/docs/my-website/docs/completion/web_search.md +++ b/docs/my-website/docs/completion/web_search.md @@ -12,7 +12,10 @@ Use web search with litellm | LiteLLM Cost Tracking | ✅ Supported | | LiteLLM Version | `v1.63.15-nightly` or higher | -## Quick Start + +## `/chat/completions` (litellm.completion) + +### Quick Start @@ -73,7 +76,7 @@ response = client.chat.completions.create( -## Search context size +### Search context size @@ -124,6 +127,127 @@ response = client.chat.completions.create( +## `/responses` (litellm.responses) + +### Quick Start + + + + +```python showLineNumbers +from litellm import responses + +response = responses( + model="openai/gpt-4o", + input=[ + { + "role": "user", + "content": "What was a positive news story from today?" + } + ], + tools=[{ + "type": "web_search_preview" # enables web search with default medium context size + }] +) +``` + + + +1. Setup config.yaml + +```yaml +model_list: + - model_name: gpt-4o + litellm_params: + model: openai/gpt-4o + api_key: os.environ/OPENAI_API_KEY +``` + +2. Start the proxy + +```bash +litellm --config /path/to/config.yaml +``` + +3. Test it! + +```python showLineNumbers +from openai import OpenAI + +# Point to your proxy server +client = OpenAI( + api_key="sk-1234", + base_url="http://0.0.0.0:4000" +) + +response = client.responses.create( + model="gpt-4o", + tools=[{ + "type": "web_search_preview" + }], + input="What was a positive news story from today?", +) + +print(response.output_text) +``` + + + +### Search context size + + + + +```python showLineNumbers +from litellm import responses + +# Customize search context size +response = responses( + model="openai/gpt-4o", + input=[ + { + "role": "user", + "content": "What was a positive news story from today?" + } + ], + tools=[{ + "type": "web_search_preview", + "search_context_size": "low" # Options: "low", "medium" (default), "high" + }] +) +``` + + + +```python showLineNumbers +from openai import OpenAI + +# Point to your proxy server +client = OpenAI( + api_key="sk-1234", + base_url="http://0.0.0.0:4000" +) + +# Customize search context size +response = client.responses.create( + model="gpt-4o", + tools=[{ + "type": "web_search_preview", + "search_context_size": "low" # Options: "low", "medium" (default), "high" + }], + input="What was a positive news story from today?", +) + +print(response.output_text) +``` + + + + + + + + ## Checking if a model supports web search From 6f7d618918ce44189352e0c849c300dbca51b99e Mon Sep 17 00:00:00 2001 From: Ishaan Jaff Date: Sat, 22 Mar 2025 19:47:13 -0700 Subject: [PATCH 33/33] test tool call cost tracking --- .../test_tool_call_cost_tracking.py | 113 ++++++++++++++++++ 1 file changed, 113 insertions(+) create mode 100644 tests/litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py diff --git a/tests/litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py b/tests/litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py new file mode 100644 index 00000000000..0d6bcb7cce0 --- /dev/null +++ b/tests/litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py @@ -0,0 +1,113 @@ +import json +import os +import sys + +import pytest +from fastapi.testclient import TestClient + +import litellm +from litellm.litellm_core_utils.llm_cost_calc.tool_call_cost_tracking import ( + StandardBuiltInToolCostTracking, +) +from litellm.types.llms.openai import FileSearchTool, WebSearchOptions +from litellm.types.utils import ModelInfo, ModelResponse, StandardBuiltInToolsParams + +sys.path.insert( + 0, os.path.abspath("../../..") +) # Adds the parent directory to the system path + + +# Test basic web search cost calculations +def test_web_search_cost_low(): + web_search_options = WebSearchOptions(search_context_size="low") + model_info = litellm.get_model_info("gpt-4o-search-preview") + + cost = StandardBuiltInToolCostTracking.get_cost_for_web_search( + web_search_options=web_search_options, model_info=model_info + ) + + assert ( + cost == model_info["search_context_cost_per_query"]["search_context_size_low"] + ) + + +def test_web_search_cost_medium(): + web_search_options = WebSearchOptions(search_context_size="medium") + model_info = litellm.get_model_info("gpt-4o-search-preview") + + cost = StandardBuiltInToolCostTracking.get_cost_for_web_search( + web_search_options=web_search_options, model_info=model_info + ) + + assert ( + cost + == model_info["search_context_cost_per_query"]["search_context_size_medium"] + ) + + +def test_web_search_cost_high(): + web_search_options = WebSearchOptions(search_context_size="high") + model_info = litellm.get_model_info("gpt-4o-search-preview") + + cost = StandardBuiltInToolCostTracking.get_cost_for_web_search( + web_search_options=web_search_options, model_info=model_info + ) + + assert ( + cost == model_info["search_context_cost_per_query"]["search_context_size_high"] + ) + + +# Test file search cost calculation +def test_file_search_cost(): + file_search = FileSearchTool(type="file_search") + cost = StandardBuiltInToolCostTracking.get_cost_for_file_search( + file_search=file_search + ) + assert cost == 0.0025 # $2.50/1000 calls = 0.0025 per call + + +# Test edge cases +def test_none_inputs(): + # Test with None inputs + assert ( + StandardBuiltInToolCostTracking.get_cost_for_web_search( + web_search_options=None, model_info=None + ) + == 0.0 + ) + assert ( + StandardBuiltInToolCostTracking.get_cost_for_file_search(file_search=None) + == 0.0 + ) + + +# Test the main get_cost_for_built_in_tools method +def test_get_cost_for_built_in_tools_web_search(): + model = "gpt-4" + standard_built_in_tools_params = StandardBuiltInToolsParams( + web_search_options=WebSearchOptions(search_context_size="medium") + ) + + cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( + model=model, + response_object=None, + standard_built_in_tools_params=standard_built_in_tools_params, + ) + + assert isinstance(cost, float) + + +def test_get_cost_for_built_in_tools_file_search(): + model = "gpt-4" + standard_built_in_tools_params = StandardBuiltInToolsParams( + file_search=FileSearchTool(type="file_search") + ) + + cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( + model=model, + response_object=None, + standard_built_in_tools_params=standard_built_in_tools_params, + ) + + assert cost == 0.0025