From 9d660ab9aad3b2b114fe1265fef71f66ceb3d070 Mon Sep 17 00:00:00 2001 From: Vigilans Date: Sun, 17 May 2026 12:16:40 +0800 Subject: [PATCH 1/4] feat(proxy): add per-deployment force_websearch_model setting When a request contains only web_search tools (e.g. Claude Code's step-2 contextless search request), redirect it to the model group specified in the deployment's `force_websearch_model` litellm_param. --- litellm/proxy/common_request_processing.py | 18 ++++++++++++++++++ litellm/types/router.py | 3 +++ 2 files changed, 21 insertions(+) diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index ef1d64335b4..9d69f4c97ba 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -32,6 +32,7 @@ from litellm.constants import ( STREAM_SSE_DATA_PREFIX, ) from litellm.integrations.custom_guardrail import CustomGuardrail +from litellm.integrations.websearch_interception.tools import is_web_search_tool from litellm.litellm_core_utils.dd_tracing import NullTracer, tracer from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj from litellm.litellm_core_utils.llm_response_utils.get_headers import ( @@ -922,6 +923,23 @@ class ProxyBaseLLMRequestProcessing: ): self.data["model"] = user_api_key_dict.aliases[self.data["model"]] + ### WEB SEARCH REDIRECT (per-deployment force) ### + # if request only contains a web_search tool and has any deployment that has + # `force_websearch_model`, redirect to that model. + if ( + isinstance(self.data.get("model"), str) + and llm_router is not None + and self.data.get("tools") + and all( + isinstance(t, dict) and is_web_search_tool(t) + for t in self.data["tools"] + ) + ): + for dep in llm_router.get_model_list(model_name=self.data["model"]) or []: + if model := dep.get("litellm_params", {}).get("force_websearch_model"): + self.data["model"] = model + break + self.data["litellm_call_id"] = request.headers.get( "x-litellm-call-id", str(uuid.uuid4()) ) diff --git a/litellm/types/router.py b/litellm/types/router.py index 6601f552b52..bd6e96e567a 100644 --- a/litellm/types/router.py +++ b/litellm/types/router.py @@ -222,6 +222,7 @@ class GenericLiteLLMParams(CredentialLiteLLMParams, CustomPricingLiteLLMParams): use_chat_completions_api: Optional[bool] = None model_config = ConfigDict(extra="allow", arbitrary_types_allowed=True) merge_reasoning_content_in_choices: Optional[bool] = False + force_websearch_model: Optional[str] = None model_info: Optional[Dict] = None mock_response: Optional[Union[str, ModelResponse, Exception, Any]] = None @@ -351,6 +352,8 @@ class LiteLLMParamsTypedDict(TypedDict, total=False): drop_params: Optional[bool] ## RESPONSES API → CHAT COMPLETIONS BRIDGE ## use_chat_completions_api: Optional[bool] + ## WEB SEARCH REDIRECT ## + force_websearch_model: Optional[str] ## UNIFIED PROJECT/REGION ## region_name: Optional[str] ## VERTEX AI ## From 6544beae3408a1bd67229b4ba34bde369b1ca91b Mon Sep 17 00:00:00 2001 From: Vigilans Date: Sun, 17 May 2026 12:16:58 +0800 Subject: [PATCH 2/4] feat(proxy): add global websearch_fallback_model setting MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When a web-search-only request targets a model that does not support web search, redirect it to the model group specified in litellm_settings.websearch_fallback_model. Uses litellm.supports_web_search() for capability detection, which correctly strips provider prefixes (e.g. github_copilot/gpt-5.5 → gpt-5.5) when looking up the model registry. Also add github_copilot to _check_provider_match exemption list, aligning it with the existing github exemption. --- litellm/proxy/common_request_processing.py | 15 ++++++++++++--- litellm/utils.py | 2 +- 2 files changed, 13 insertions(+), 4 deletions(-) diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 9d69f4c97ba..0b985c6b785 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -923,9 +923,11 @@ class ProxyBaseLLMRequestProcessing: ): self.data["model"] = user_api_key_dict.aliases[self.data["model"]] - ### WEB SEARCH REDIRECT (per-deployment force) ### - # if request only contains a web_search tool and has any deployment that has - # `force_websearch_model`, redirect to that model. + ### WEB SEARCH REDIRECT ### + # if request only contains web_search tools, check for redirect: + # 1. per-deployment force: always redirect to `force_websearch_model` + # 2. global fallback: redirect to `websearch_fallback_model` only when + # the current model group does not support web search if ( isinstance(self.data.get("model"), str) and llm_router is not None @@ -939,6 +941,13 @@ class ProxyBaseLLMRequestProcessing: if model := dep.get("litellm_params", {}).get("force_websearch_model"): self.data["model"] = model break + if ( + (dep_model := dep.get("litellm_params", {}).get("model", "")) + and not litellm.supports_web_search(dep_model) + and (model := getattr(litellm, "websearch_fallback_model", None)) + ): + self.data["model"] = model + break self.data["litellm_call_id"] = request.headers.get( "x-litellm-call-id", str(uuid.uuid4()) diff --git a/litellm/utils.py b/litellm/utils.py index 2ba6ef9cae8..9121473918c 100644 --- a/litellm/utils.py +++ b/litellm/utils.py @@ -5579,7 +5579,7 @@ def _check_provider_match(model_info: dict, custom_llm_provider: Optional[str]) # as a last attempt if the model is not on Azure AI, Azure then fallback to OpenAI cost # tracking the cost is better than attributing 0 cost to it. return True - elif custom_llm_provider == "github": + elif custom_llm_provider in ("github", "github_copilot"): # Allow github/ aliases to reuse existing provider metadata. return True else: From b89425e948f85db04f67adc8066e11ca51e76965 Mon Sep 17 00:00:00 2001 From: Vigilans Date: Mon, 25 May 2026 01:14:09 +0800 Subject: [PATCH 3/4] test(proxy): add tests for web search model redirect --- .../proxy/test_websearch_redirect.py | 273 ++++++++++++++++++ 1 file changed, 273 insertions(+) create mode 100644 tests/test_litellm/proxy/test_websearch_redirect.py diff --git a/tests/test_litellm/proxy/test_websearch_redirect.py b/tests/test_litellm/proxy/test_websearch_redirect.py new file mode 100644 index 00000000000..150d14cbbce --- /dev/null +++ b/tests/test_litellm/proxy/test_websearch_redirect.py @@ -0,0 +1,273 @@ +""" +Tests for web search model redirect in common_processing_pre_call_logic. + +Covers per-deployment force_websearch_model and global websearch_fallback_model. +The redirect logic lives inline in common_processing_pre_call_logic, so tests +construct a ProxyBaseLLMRequestProcessing instance and call the method with +mocked dependencies, then verify self.data["model"] after the redirect block. +""" + +import asyncio +from unittest.mock import AsyncMock, MagicMock, patch + +import pytest + +import litellm +from litellm.proxy.common_request_processing import ProxyBaseLLMRequestProcessing + +WEB_SEARCH_TOOL = {"type": "web_search_20250305", "name": "web_search"} +REGULAR_TOOL = { + "type": "custom", + "name": "get_weather", + "description": "Get weather", + "input_schema": {"type": "object", "properties": {"location": {"type": "string"}}}, +} + + +def _make_mock_router(deployments): + """Build a mock router that returns the given deployments for get_model_list.""" + router = MagicMock() + router.get_model_list.return_value = deployments + router.get_model_group_info.return_value = None + return router + + +def _make_mock_request(): + request = MagicMock() + request.headers = {} + request.url = MagicMock() + request.url.path = "/v1/messages" + return request + + +def _make_user_api_key_dict(): + user_api_key_dict = MagicMock() + user_api_key_dict.aliases = {} + user_api_key_dict.models = [] + user_api_key_dict.api_key = "sk-test" + user_api_key_dict.user_id = "test" + user_api_key_dict.team_id = None + user_api_key_dict.metadata = {} + return user_api_key_dict + + +async def _run_pre_call_until_redirect(data, llm_router): + """Run common_processing_pre_call_logic far enough to trigger the redirect, + then let it fail on subsequent steps — we only care about data['model'].""" + proc = ProxyBaseLLMRequestProcessing(data=data) + + async def _passthrough_add_litellm_data(data, **kwargs): + return data + + with patch( + "litellm.proxy.common_request_processing.add_litellm_data_to_request", + side_effect=_passthrough_add_litellm_data, + ): + try: + await proc.common_processing_pre_call_logic( + request=_make_mock_request(), + user_api_key_dict=_make_user_api_key_dict(), + llm_router=llm_router, + proxy_config=MagicMock(), + general_settings={}, + proxy_logging_obj=MagicMock(), + route_type="anthropic_messages", + version=None, + ) + except Exception: + pass + return proc.data.get("model") + + +class TestForceWebsearchModel: + @pytest.mark.asyncio + async def test_pure_websearch_redirected(self): + router = _make_mock_router( + [ + { + "litellm_params": { + "model": "openai/gpt-5.4", + "force_websearch_model": "gpt-5.5", + } + } + ] + ) + data = { + "model": "my-model", + "messages": [{"role": "user", "content": "search"}], + "tools": [WEB_SEARCH_TOOL], + } + result = await _run_pre_call_until_redirect(data, router) + assert result == "gpt-5.5" + + @pytest.mark.asyncio + async def test_mixed_tools_not_redirected(self): + router = _make_mock_router( + [ + { + "litellm_params": { + "model": "openai/gpt-5.4", + "force_websearch_model": "gpt-5.5", + } + } + ] + ) + data = { + "model": "my-model", + "messages": [{"role": "user", "content": "search"}], + "tools": [WEB_SEARCH_TOOL, REGULAR_TOOL], + } + result = await _run_pre_call_until_redirect(data, router) + assert result == "my-model" + + @pytest.mark.asyncio + async def test_multiple_websearch_tools_redirected(self): + router = _make_mock_router( + [ + { + "litellm_params": { + "model": "openai/gpt-5.4", + "force_websearch_model": "gpt-5.5", + } + } + ] + ) + data = { + "model": "my-model", + "messages": [{"role": "user", "content": "search"}], + "tools": [ + WEB_SEARCH_TOOL, + {"name": "WebSearch", "description": "search"}, + ], + } + result = await _run_pre_call_until_redirect(data, router) + assert result == "gpt-5.5" + + @pytest.mark.asyncio + async def test_no_force_no_redirect(self): + router = _make_mock_router([{"litellm_params": {"model": "openai/gpt-5.5"}}]) + data = { + "model": "gpt-5.5", + "messages": [{"role": "user", "content": "search"}], + "tools": [WEB_SEARCH_TOOL], + } + with patch("litellm.supports_web_search", return_value=True): + result = await _run_pre_call_until_redirect(data, router) + assert result == "gpt-5.5" + + +class TestWebsearchFallbackModel: + @pytest.mark.asyncio + async def test_fallback_when_model_lacks_search(self): + router = _make_mock_router( + [{"litellm_params": {"model": "openai/my-local-llm"}}] + ) + data = { + "model": "my-local-llm", + "messages": [{"role": "user", "content": "search"}], + "tools": [WEB_SEARCH_TOOL], + } + with ( + patch("litellm.supports_web_search", return_value=False), + patch.object( + litellm, + "websearch_fallback_model", + "gpt-5.4-mini", + create=True, + ), + ): + result = await _run_pre_call_until_redirect(data, router) + assert result == "gpt-5.4-mini" + + @pytest.mark.asyncio + async def test_no_fallback_when_model_supports_search(self): + router = _make_mock_router([{"litellm_params": {"model": "openai/gpt-5.5"}}]) + data = { + "model": "gpt-5.5", + "messages": [{"role": "user", "content": "search"}], + "tools": [WEB_SEARCH_TOOL], + } + with patch("litellm.supports_web_search", return_value=True): + result = await _run_pre_call_until_redirect(data, router) + assert result == "gpt-5.5" + + @pytest.mark.asyncio + async def test_no_fallback_when_setting_not_configured(self): + router = _make_mock_router( + [{"litellm_params": {"model": "openai/my-local-llm"}}] + ) + data = { + "model": "my-local-llm", + "messages": [{"role": "user", "content": "search"}], + "tools": [WEB_SEARCH_TOOL], + } + with ( + patch("litellm.supports_web_search", return_value=False), + patch.object(litellm, "websearch_fallback_model", None, create=True), + ): + result = await _run_pre_call_until_redirect(data, router) + assert result == "my-local-llm" + + +class TestWebsearchForceOverFallbackPriority: + @pytest.mark.asyncio + async def test_force_takes_priority_over_fallback(self): + router = _make_mock_router( + [ + { + "litellm_params": { + "model": "openai/gpt-5.4", + "force_websearch_model": "gpt-5.5", + } + } + ] + ) + data = { + "model": "my-model", + "messages": [{"role": "user", "content": "search"}], + "tools": [WEB_SEARCH_TOOL], + } + with ( + patch("litellm.supports_web_search", return_value=False), + patch.object( + litellm, + "websearch_fallback_model", + "gpt-5.4-mini", + create=True, + ), + ): + result = await _run_pre_call_until_redirect(data, router) + assert result == "gpt-5.5" + + +class TestWebsearchRedirectGuardConditions: + @pytest.mark.asyncio + async def test_no_tools(self): + router = _make_mock_router([]) + data = { + "model": "my-model", + "messages": [{"role": "user", "content": "hi"}], + } + result = await _run_pre_call_until_redirect(data, router) + assert result == "my-model" + + @pytest.mark.asyncio + async def test_no_router(self): + data = { + "model": "my-model", + "messages": [{"role": "user", "content": "search"}], + "tools": [WEB_SEARCH_TOOL], + } + result = await _run_pre_call_until_redirect(data, None) + assert result == "my-model" + + @pytest.mark.asyncio + async def test_empty_tools(self): + router = _make_mock_router([]) + data = { + "model": "my-model", + "messages": [{"role": "user", "content": "hi"}], + "tools": [], + } + result = await _run_pre_call_until_redirect(data, router) + assert result == "my-model" From 0a06f137b83255d7d47ee613716f82bddd3ecae6 Mon Sep 17 00:00:00 2001 From: Vigilans Date: Mon, 25 May 2026 01:42:18 +0800 Subject: [PATCH 4/4] style: add noqa PLR0915 for common_processing_pre_call_logic --- litellm/proxy/common_request_processing.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/litellm/proxy/common_request_processing.py b/litellm/proxy/common_request_processing.py index 0b985c6b785..b923fdd2590 100644 --- a/litellm/proxy/common_request_processing.py +++ b/litellm/proxy/common_request_processing.py @@ -744,7 +744,7 @@ class ProxyBaseLLMRequestProcessing: return custom_headers - async def common_processing_pre_call_logic( + async def common_processing_pre_call_logic( # noqa: PLR0915 self, request: Request, general_settings: dict,