mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix: pass api_base/api_key to agentic hook follow-up requests
When using third-party Anthropic-compatible providers (e.g. Tencent Cloud) with custom api_base, agentic hooks (websearch_interception) and count_tokens handler sent follow-up requests to api.anthropic.com instead of the configured endpoint, causing 401 authentication errors and deployment cooldown loops. - llm_http_handler: _execute_anthropic_agentic_plan now reads api_key and api_base from agentic_loop_params and passes them to anthropic_messages.acreate() - token_counter: extract api_base from litellm_params and build the count_tokens endpoint URL instead of hardcoding api.anthropic.com - websearch_interception handler: resolve merge conflict and fix legacy _execute_agentic_loop path to also pass api_key/api_base Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
parent
c94a8d6514
commit
394b015f5e
4 changed files with 38 additions and 2 deletions
|
|
@ -755,10 +755,23 @@ class WebSearchInterceptionLogger(CustomLogger):
|
|||
if max_tokens is None:
|
||||
max_tokens = cast(int, kwargs.get("max_tokens", 1024))
|
||||
|
||||
# Pass api_key/api_base from agentic_loop_params so follow-up requests
|
||||
# route to the same backend instead of falling back to api.anthropic.com
|
||||
_followup_api_key: Optional[str] = None
|
||||
_followup_api_base: Optional[str] = None
|
||||
if logging_obj is not None:
|
||||
agentic_params = logging_obj.model_call_details.get(
|
||||
"agentic_loop_params", {}
|
||||
)
|
||||
_followup_api_key = agentic_params.get("api_key")
|
||||
_followup_api_base = agentic_params.get("api_base")
|
||||
|
||||
return await anthropic_messages.acreate(
|
||||
max_tokens=max_tokens,
|
||||
messages=request_patch.messages,
|
||||
model=request_patch.model or model,
|
||||
api_key=_followup_api_key,
|
||||
api_base=_followup_api_base,
|
||||
**optional_params,
|
||||
**request_patch.kwargs,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -54,8 +54,9 @@ class AnthropicTokenCounter(BaseTokenCounter):
|
|||
deployment = deployment or {}
|
||||
litellm_params = deployment.get("litellm_params", {})
|
||||
|
||||
# Get Anthropic API key from deployment config or environment
|
||||
# Get Anthropic API key and api_base from deployment config or environment
|
||||
api_key = litellm_params.get("api_key")
|
||||
api_base = litellm_params.get("api_base")
|
||||
if not api_key:
|
||||
api_key = os.getenv("ANTHROPIC_API_KEY")
|
||||
|
||||
|
|
@ -63,11 +64,19 @@ class AnthropicTokenCounter(BaseTokenCounter):
|
|||
verbose_logger.warning("No Anthropic API key found for token counting")
|
||||
return None
|
||||
|
||||
# Build count_tokens endpoint from api_base if available
|
||||
count_tokens_api_base: Optional[str] = None
|
||||
if api_base:
|
||||
# e.g. https://api.lkeap.cloud.tencent.com/plan/anthropic -> https://api.lkeap.cloud.tencent.com/plan/anthropic/v1/messages/count_tokens
|
||||
base = api_base.rstrip("/")
|
||||
count_tokens_api_base = f"{base}/v1/messages/count_tokens"
|
||||
|
||||
try:
|
||||
result = await anthropic_count_tokens_handler.handle_count_tokens_request(
|
||||
model=model_to_use,
|
||||
messages=messages,
|
||||
api_key=api_key,
|
||||
api_base=count_tokens_api_base,
|
||||
tools=tools,
|
||||
system=system,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -368,10 +368,18 @@ def anthropic_messages_handler(
|
|||
# Store agentic loop params in logging object for agentic hooks
|
||||
# This provides original request context needed for follow-up calls
|
||||
if litellm_logging_obj is not None:
|
||||
litellm_logging_obj.model_call_details["agentic_loop_params"] = {
|
||||
agentic_loop_params: Dict[str, Any] = {
|
||||
"model": original_model,
|
||||
"custom_llm_provider": custom_llm_provider,
|
||||
}
|
||||
# Preserve api_base and api_key so that agentic hooks (e.g. websearch
|
||||
# interception follow-up requests) can reuse the same credentials
|
||||
# instead of falling back to defaults (api.anthropic.com / ANTHROPIC_API_KEY).
|
||||
if dynamic_api_key is not None:
|
||||
agentic_loop_params["api_key"] = dynamic_api_key
|
||||
if dynamic_api_base is not None:
|
||||
agentic_loop_params["api_base"] = dynamic_api_base
|
||||
litellm_logging_obj.model_call_details["agentic_loop_params"] = agentic_loop_params
|
||||
|
||||
# Check if stream was converted for WebSearch interception
|
||||
# This is set in the async wrapper above when stream=True is converted to stream=False
|
||||
|
|
|
|||
|
|
@ -4642,11 +4642,15 @@ class BaseLLMHTTPHandler:
|
|||
raise ValueError("Agentic loop plan missing patched messages")
|
||||
|
||||
full_model_name = model
|
||||
agentic_api_key: Optional[str] = None
|
||||
agentic_api_base: Optional[str] = None
|
||||
if logging_obj is not None:
|
||||
agentic_params = logging_obj.model_call_details.get(
|
||||
"agentic_loop_params", {}
|
||||
)
|
||||
full_model_name = cast(str, agentic_params.get("model", model))
|
||||
agentic_api_key = agentic_params.get("api_key")
|
||||
agentic_api_base = agentic_params.get("api_base")
|
||||
|
||||
optional_params = dict(anthropic_messages_optional_request_params)
|
||||
optional_params.update(patch.optional_params)
|
||||
|
|
@ -4681,6 +4685,8 @@ class BaseLLMHTTPHandler:
|
|||
"messages": patch.messages,
|
||||
"model": patch.model or full_model_name,
|
||||
"stream": stream,
|
||||
**({"api_key": agentic_api_key} if agentic_api_key else {}),
|
||||
**({"api_base": agentic_api_base} if agentic_api_base else {}),
|
||||
**optional_params,
|
||||
**kwargs_for_followup,
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue