perf: add early-exit guards in completion_cost for unused features (#20020)

* perf: add early-exit guards in completion_cost for unused features

Skip function calls to get_cost_for_built_in_tools, _apply_cost_discount,
_apply_cost_margin, and _store_cost_breakdown_in_logging_obj when their
respective features are not configured. Reduces completion_cost() time
by ~20% (4.39s → 3.53s over 6K requests) for the common case where
built-in tools, discounts, margins, and logging object are not active.

* fix: always call get_cost_for_built_in_tools regardless of standard_built_in_tools_params

The function can detect web search usage from the usage object (e.g.
server_tool_use.web_search_requests, prompt_tokens_details.web_search_requests)
even when standard_built_in_tools_params is None, so guarding on it can
under-count cost for providers like Vertex AI and Anthropic.

Adds regression test for completion_cost with web search in usage but
no standard_built_in_tools_params.
This commit is contained in:
ryan-crabbe 2026-02-07 10:10:00 -08:00 • committed by GitHub
parent 1780b1716f
commit 874bafbbb4
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 90 additions and 27 deletions

View file

@ -1411,37 +1411,47 @@ def completion_cost( # noqa: PLR0915
# Apply discount from module-level config if configured
original_cost = _final_cost
_final_cost, discount_percent, discount_amount = _apply_cost_discount(
base_cost=_final_cost,
custom_llm_provider=custom_llm_provider,
)
if litellm.cost_discount_config:
_final_cost, discount_percent, discount_amount = _apply_cost_discount(
base_cost=_final_cost,
custom_llm_provider=custom_llm_provider,
)
else:
discount_percent = 0.0
discount_amount = 0.0
# Apply margin from module-level config if configured
(
_final_cost,
margin_percent,
margin_fixed_amount,
margin_total_amount,
) = _apply_cost_margin(
base_cost=_final_cost,
custom_llm_provider=custom_llm_provider,
)
if litellm.cost_margin_config:
(
_final_cost,
margin_percent,
margin_fixed_amount,
margin_total_amount,
) = _apply_cost_margin(
base_cost=_final_cost,
custom_llm_provider=custom_llm_provider,
)
else:
margin_percent = 0.0
margin_fixed_amount = 0.0
margin_total_amount = 0.0
# Store cost breakdown in logging object if available
_store_cost_breakdown_in_logging_obj(
litellm_logging_obj=litellm_logging_obj,
prompt_tokens_cost_usd_dollar=prompt_tokens_cost_usd_dollar,
completion_tokens_cost_usd_dollar=completion_tokens_cost_usd_dollar,
cost_for_built_in_tools_cost_usd_dollar=cost_for_built_in_tools,
total_cost_usd_dollar=_final_cost,
additional_costs=additional_costs,
original_cost=original_cost,
discount_percent=discount_percent,
discount_amount=discount_amount,
margin_percent=margin_percent,
margin_fixed_amount=margin_fixed_amount,
margin_total_amount=margin_total_amount,
)
if litellm_logging_obj is not None:
_store_cost_breakdown_in_logging_obj(
litellm_logging_obj=litellm_logging_obj,
prompt_tokens_cost_usd_dollar=prompt_tokens_cost_usd_dollar,
completion_tokens_cost_usd_dollar=completion_tokens_cost_usd_dollar,
cost_for_built_in_tools_cost_usd_dollar=cost_for_built_in_tools,
total_cost_usd_dollar=_final_cost,
original_cost=original_cost,
additional_costs=additional_costs,
discount_percent=discount_percent,
discount_amount=discount_amount,
margin_percent=margin_percent,
margin_fixed_amount=margin_fixed_amount,
margin_total_amount=margin_total_amount,
)
return _final_cost
except Exception as e:

View file

@ -255,5 +255,58 @@ def test_azure_assistant_features_integrated_cost_tracking():
assert abs(cost - expected_cost) < 0.01, f"Expected ~{expected_cost}, got {cost}"
def test_completion_cost_includes_web_search_without_standard_built_in_tools_params():
"""
Test that completion_cost includes web search cost even when
standard_built_in_tools_params is None.
Regression test: the early-exit guard `if standard_built_in_tools_params:`
in completion_cost was skipping get_cost_for_built_in_tools entirely,
causing under-counted costs for providers like Vertex AI Gemini that
report web search usage via usage.prompt_tokens_details.web_search_requests.
"""
from litellm.types.utils import Choices, Message, PromptTokensDetailsWrapper, Usage
response = ModelResponse(
id="test-id",
choices=[
Choices(
finish_reason="stop",
index=0,
message=Message(content="test", role="assistant"),
)
],
created=1234567890,
model="gemini-2.5-flash",
object="chat.completion",
)
response.usage = Usage(
prompt_tokens=100,
completion_tokens=50,
total_tokens=150,
prompt_tokens_details=PromptTokensDetailsWrapper(web_search_requests=1),
)
cost = litellm.completion_cost(
completion_response=response,
model="gemini-2.5-flash",
custom_llm_provider="vertex_ai",
standard_built_in_tools_params=None,
)
web_search_cost = StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model="gemini-2.5-flash",
usage=response.usage,
response_object=response,
standard_built_in_tools_params=None,
custom_llm_provider="vertex_ai",
)
assert web_search_cost > 0, "Web search cost should be non-zero"
assert cost >= web_search_cost, (
f"completion_cost ({cost}) should include web search cost ({web_search_cost})"
)
# Note: File search integration test removed due to complex annotation detection logic
# The unit tests in test_azure_assistant_cost_tracking.py provide comprehensive coverage