From 394b015f5e7138a96a8147bf8509c7c7f73a650a Mon Sep 17 00:00:00 2001 From: jesset Date: Sun, 3 May 2026 15:46:24 +0800 Subject: [PATCH] fix: pass api_base/api_key to agentic hook follow-up requests When using third-party Anthropic-compatible providers (e.g. Tencent Cloud) with custom api_base, agentic hooks (websearch_interception) and count_tokens handler sent follow-up requests to api.anthropic.com instead of the configured endpoint, causing 401 authentication errors and deployment cooldown loops. - llm_http_handler: _execute_anthropic_agentic_plan now reads api_key and api_base from agentic_loop_params and passes them to anthropic_messages.acreate() - token_counter: extract api_base from litellm_params and build the count_tokens endpoint URL instead of hardcoding api.anthropic.com - websearch_interception handler: resolve merge conflict and fix legacy _execute_agentic_loop path to also pass api_key/api_base Co-Authored-By: Claude Sonnet 4.6 --- .../integrations/websearch_interception/handler.py | 13 +++++++++++++ .../llms/anthropic/count_tokens/token_counter.py | 11 ++++++++++- .../experimental_pass_through/messages/handler.py | 10 +++++++++- litellm/llms/custom_httpx/llm_http_handler.py | 6 ++++++ 4 files changed, 38 insertions(+), 2 deletions(-) diff --git a/litellm/integrations/websearch_interception/handler.py b/litellm/integrations/websearch_interception/handler.py index 41618c72627..d45e46cfab3 100644 --- a/litellm/integrations/websearch_interception/handler.py +++ b/litellm/integrations/websearch_interception/handler.py @@ -755,10 +755,23 @@ class WebSearchInterceptionLogger(CustomLogger): if max_tokens is None: max_tokens = cast(int, kwargs.get("max_tokens", 1024)) + # Pass api_key/api_base from agentic_loop_params so follow-up requests + # route to the same backend instead of falling back to api.anthropic.com + _followup_api_key: Optional[str] = None + _followup_api_base: Optional[str] = None + if logging_obj is not None: + agentic_params = logging_obj.model_call_details.get( + "agentic_loop_params", {} + ) + _followup_api_key = agentic_params.get("api_key") + _followup_api_base = agentic_params.get("api_base") + return await anthropic_messages.acreate( max_tokens=max_tokens, messages=request_patch.messages, model=request_patch.model or model, + api_key=_followup_api_key, + api_base=_followup_api_base, **optional_params, **request_patch.kwargs, ) diff --git a/litellm/llms/anthropic/count_tokens/token_counter.py b/litellm/llms/anthropic/count_tokens/token_counter.py index 93989c58547..7bca2cb7160 100644 --- a/litellm/llms/anthropic/count_tokens/token_counter.py +++ b/litellm/llms/anthropic/count_tokens/token_counter.py @@ -54,8 +54,9 @@ class AnthropicTokenCounter(BaseTokenCounter): deployment = deployment or {} litellm_params = deployment.get("litellm_params", {}) - # Get Anthropic API key from deployment config or environment + # Get Anthropic API key and api_base from deployment config or environment api_key = litellm_params.get("api_key") + api_base = litellm_params.get("api_base") if not api_key: api_key = os.getenv("ANTHROPIC_API_KEY") @@ -63,11 +64,19 @@ class AnthropicTokenCounter(BaseTokenCounter): verbose_logger.warning("No Anthropic API key found for token counting") return None + # Build count_tokens endpoint from api_base if available + count_tokens_api_base: Optional[str] = None + if api_base: + # e.g. https://api.lkeap.cloud.tencent.com/plan/anthropic -> https://api.lkeap.cloud.tencent.com/plan/anthropic/v1/messages/count_tokens + base = api_base.rstrip("/") + count_tokens_api_base = f"{base}/v1/messages/count_tokens" + try: result = await anthropic_count_tokens_handler.handle_count_tokens_request( model=model_to_use, messages=messages, api_key=api_key, + api_base=count_tokens_api_base, tools=tools, system=system, ) diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py index 0c59e812e0b..00f6f7bb796 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/handler.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/handler.py @@ -368,10 +368,18 @@ def anthropic_messages_handler( # Store agentic loop params in logging object for agentic hooks # This provides original request context needed for follow-up calls if litellm_logging_obj is not None: - litellm_logging_obj.model_call_details["agentic_loop_params"] = { + agentic_loop_params: Dict[str, Any] = { "model": original_model, "custom_llm_provider": custom_llm_provider, } + # Preserve api_base and api_key so that agentic hooks (e.g. websearch + # interception follow-up requests) can reuse the same credentials + # instead of falling back to defaults (api.anthropic.com / ANTHROPIC_API_KEY). + if dynamic_api_key is not None: + agentic_loop_params["api_key"] = dynamic_api_key + if dynamic_api_base is not None: + agentic_loop_params["api_base"] = dynamic_api_base + litellm_logging_obj.model_call_details["agentic_loop_params"] = agentic_loop_params # Check if stream was converted for WebSearch interception # This is set in the async wrapper above when stream=True is converted to stream=False diff --git a/litellm/llms/custom_httpx/llm_http_handler.py b/litellm/llms/custom_httpx/llm_http_handler.py index 0c4816fcda2..2e6417974cb 100644 --- a/litellm/llms/custom_httpx/llm_http_handler.py +++ b/litellm/llms/custom_httpx/llm_http_handler.py @@ -4642,11 +4642,15 @@ class BaseLLMHTTPHandler: raise ValueError("Agentic loop plan missing patched messages") full_model_name = model + agentic_api_key: Optional[str] = None + agentic_api_base: Optional[str] = None if logging_obj is not None: agentic_params = logging_obj.model_call_details.get( "agentic_loop_params", {} ) full_model_name = cast(str, agentic_params.get("model", model)) + agentic_api_key = agentic_params.get("api_key") + agentic_api_base = agentic_params.get("api_base") optional_params = dict(anthropic_messages_optional_request_params) optional_params.update(patch.optional_params) @@ -4681,6 +4685,8 @@ class BaseLLMHTTPHandler: "messages": patch.messages, "model": patch.model or full_model_name, "stream": stream, + **({"api_key": agentic_api_key} if agentic_api_key else {}), + **({"api_base": agentic_api_base} if agentic_api_base else {}), **optional_params, **kwargs_for_followup, }