revert: remove xAI-specific web search gate from shared cost tracking

Gate web search like OpenAI (output/annotations/web_search_requests).
xAI uses server_side_tool_usage_details only for per-call cost math, with
web_search_requests mirrored in llms/xai for existing gate compatibility.
This commit is contained in:
Yang Yang 2026-06-20 10:57:36 -07:00
parent 8ad0a57387
commit 74100989a2
2 changed files with 0 additions and 50 deletions

View file

@ -311,22 +311,6 @@ class StandardBuiltInToolCostTracking:
return Usage(server_tool_use=server_tool_use)
return usage.model_copy(update={"server_tool_use": server_tool_use})
@staticmethod
def _usage_has_server_side_web_search_calls(usage: Usage | None) -> bool:
"""True when usage.server_side_tool_usage_details.web_search_calls > 0."""
if usage is None:
return False
details = getattr(usage, "server_side_tool_usage_details", None)
if details is None:
return False
try:
web_search_calls = (
details.get("web_search_calls") if isinstance(details, dict) else getattr(details, "web_search_calls", None)
)
return int(web_search_calls or 0) > 0
except (TypeError, ValueError):
return False
@staticmethod
def response_object_includes_web_search_call(response_object: Any, usage: Usage | None = None) -> bool:
"""
@ -344,8 +328,6 @@ class StandardBuiltInToolCostTracking:
if get_anthropic_web_search_requests_from_response(response_object) is not None:
return True
if StandardBuiltInToolCostTracking._usage_has_server_side_web_search_calls(usage):
return True
if isinstance(response_object, ModelResponse):
# chat completions only include url_citation annotations when a web search call is made

View file

@ -27,9 +27,6 @@ from litellm.llms.xai.cost_calculator import (
cost_per_token,
cost_per_web_search_request,
)
from litellm.types.llms.openai import ResponsesAPIResponse
class TestXAICostCalculator:
"""Test suite for XAI cost calculation functionality."""
@ -411,35 +408,6 @@ class TestXAICostCalculator:
response_object=object(), usage=usage
)
def test_gate_detects_server_side_tool_usage_details_without_web_search_output(
self,
):
usage = Usage(prompt_tokens=10, completion_tokens=5, total_tokens=15)
setattr(
usage,
"server_side_tool_usage_details",
{"web_search_calls": 1},
)
response = ResponsesAPIResponse.model_construct(
id="resp_test",
created_at=0,
output=[{"type": "message", "role": "assistant", "content": []}],
usage=None,
)
assert StandardBuiltInToolCostTracking.response_object_includes_web_search_call(
response_object=response, usage=usage
)
assert (
StandardBuiltInToolCostTracking.get_cost_for_built_in_tools(
model="grok-4.3",
response_object=response,
usage=usage,
standard_built_in_tools_params={},
custom_llm_provider="xai",
)
== 5.0 / 1000.0
)
def test_grok_4_20_beta_reasoning_cost_calculation(self):
"""Test cost calculation for grok-4.20-beta-0309-reasoning model."""
usage = Usage(prompt_tokens=100, completion_tokens=200, total_tokens=300)