From 874bafbbb4db28b039d58e4a305e33e3abbf1a15 Mon Sep 17 00:00:00 2001 From: ryan-crabbe <128659760+ryan-crabbe@users.noreply.github.com> Date: Sat, 7 Feb 2026 10:10:00 -0800 Subject: [PATCH] perf: add early-exit guards in completion_cost for unused features (#20020) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * perf: add early-exit guards in completion_cost for unused features Skip function calls to get_cost_for_built_in_tools, _apply_cost_discount, _apply_cost_margin, and _store_cost_breakdown_in_logging_obj when their respective features are not configured. Reduces completion_cost() time by ~20% (4.39s → 3.53s over 6K requests) for the common case where built-in tools, discounts, margins, and logging object are not active. * fix: always call get_cost_for_built_in_tools regardless of standard_built_in_tools_params The function can detect web search usage from the usage object (e.g. server_tool_use.web_search_requests, prompt_tokens_details.web_search_requests) even when standard_built_in_tools_params is None, so guarding on it can under-count cost for providers like Vertex AI and Anthropic. Adds regression test for completion_cost with web search in usage but no standard_built_in_tools_params. --- litellm/cost_calculator.py | 64 +++++++++++-------- .../test_tool_call_cost_tracking.py | 53 +++++++++++++++ 2 files changed, 90 insertions(+), 27 deletions(-) diff --git a/litellm/cost_calculator.py b/litellm/cost_calculator.py index bef4d52ce49..824180ac36f 100644 --- a/litellm/cost_calculator.py +++ b/litellm/cost_calculator.py @@ -1411,37 +1411,47 @@ def completion_cost( # noqa: PLR0915 # Apply discount from module-level config if configured original_cost = _final_cost - _final_cost, discount_percent, discount_amount = _apply_cost_discount( - base_cost=_final_cost, - custom_llm_provider=custom_llm_provider, - ) + if litellm.cost_discount_config: + _final_cost, discount_percent, discount_amount = _apply_cost_discount( + base_cost=_final_cost, + custom_llm_provider=custom_llm_provider, + ) + else: + discount_percent = 0.0 + discount_amount = 0.0 # Apply margin from module-level config if configured - ( - _final_cost, - margin_percent, - margin_fixed_amount, - margin_total_amount, - ) = _apply_cost_margin( - base_cost=_final_cost, - custom_llm_provider=custom_llm_provider, - ) + if litellm.cost_margin_config: + ( + _final_cost, + margin_percent, + margin_fixed_amount, + margin_total_amount, + ) = _apply_cost_margin( + base_cost=_final_cost, + custom_llm_provider=custom_llm_provider, + ) + else: + margin_percent = 0.0 + margin_fixed_amount = 0.0 + margin_total_amount = 0.0 # Store cost breakdown in logging object if available - _store_cost_breakdown_in_logging_obj( - litellm_logging_obj=litellm_logging_obj, - prompt_tokens_cost_usd_dollar=prompt_tokens_cost_usd_dollar, - completion_tokens_cost_usd_dollar=completion_tokens_cost_usd_dollar, - cost_for_built_in_tools_cost_usd_dollar=cost_for_built_in_tools, - total_cost_usd_dollar=_final_cost, - additional_costs=additional_costs, - original_cost=original_cost, - discount_percent=discount_percent, - discount_amount=discount_amount, - margin_percent=margin_percent, - margin_fixed_amount=margin_fixed_amount, - margin_total_amount=margin_total_amount, - ) + if litellm_logging_obj is not None: + _store_cost_breakdown_in_logging_obj( + litellm_logging_obj=litellm_logging_obj, + prompt_tokens_cost_usd_dollar=prompt_tokens_cost_usd_dollar, + completion_tokens_cost_usd_dollar=completion_tokens_cost_usd_dollar, + cost_for_built_in_tools_cost_usd_dollar=cost_for_built_in_tools, + total_cost_usd_dollar=_final_cost, + original_cost=original_cost, + additional_costs=additional_costs, + discount_percent=discount_percent, + discount_amount=discount_amount, + margin_percent=margin_percent, + margin_fixed_amount=margin_fixed_amount, + margin_total_amount=margin_total_amount, + ) return _final_cost except Exception as e: diff --git a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py index 776c41697fe..9eb8ae542e0 100644 --- a/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py +++ b/tests/test_litellm/litellm_core_utils/llm_cost_calc/test_tool_call_cost_tracking.py @@ -255,5 +255,58 @@ def test_azure_assistant_features_integrated_cost_tracking(): assert abs(cost - expected_cost) < 0.01, f"Expected ~{expected_cost}, got {cost}" +def test_completion_cost_includes_web_search_without_standard_built_in_tools_params(): + """ + Test that completion_cost includes web search cost even when + standard_built_in_tools_params is None. + + Regression test: the early-exit guard `if standard_built_in_tools_params:` + in completion_cost was skipping get_cost_for_built_in_tools entirely, + causing under-counted costs for providers like Vertex AI Gemini that + report web search usage via usage.prompt_tokens_details.web_search_requests. + """ + from litellm.types.utils import Choices, Message, PromptTokensDetailsWrapper, Usage + + response = ModelResponse( + id="test-id", + choices=[ + Choices( + finish_reason="stop", + index=0, + message=Message(content="test", role="assistant"), + ) + ], + created=1234567890, + model="gemini-2.5-flash", + object="chat.completion", + ) + response.usage = Usage( + prompt_tokens=100, + completion_tokens=50, + total_tokens=150, + prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=1), + ) + + cost = litellm.completion_cost( + completion_response=response, + model="gemini-2.5-flash", + custom_llm_provider="vertex_ai", + standard_built_in_tools_params=None, + ) + + web_search_cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools( + model="gemini-2.5-flash", + usage=response.usage, + response_object=response, + standard_built_in_tools_params=None, + custom_llm_provider="vertex_ai", + ) + + assert web_search_cost > 0, "Web search cost should be non-zero" + assert cost >= web_search_cost, ( + f"completion_cost ({cost}) should include web search cost ({web_search_cost})" + ) + + # Note: File search integration test removed due to complex annotation detection logic # The unit tests in test_azure_assistant_cost_tracking.py provide comprehensive coverage