diff --git a/litellm/llms/anthropic/experimental_pass_through/messages/interceptors/advisor.py b/litellm/llms/anthropic/experimental_pass_through/messages/interceptors/advisor.py index adf96db61c2..6f732db1548 100644 --- a/litellm/llms/anthropic/experimental_pass_through/messages/interceptors/advisor.py +++ b/litellm/llms/anthropic/experimental_pass_through/messages/interceptors/advisor.py @@ -376,17 +376,11 @@ def _inject_max_uses_error( def _resolve_advisor_router(advisor_model: str) -> "Router | None": - """Return the proxy router when it can resolve ``advisor_model``. + """Return the proxy router when it serves ``advisor_model`` directly or via a wildcard. - The advisor sub-call must honor the proxy's ``model_list`` (and its - fallbacks / credentials) exactly like a direct call to that model group - would. Without this, provider resolution falls back to the bare model - name, which for a ``claude-*`` advisor model means the public Anthropic - API, bypassing the configured deployment entirely. - - Returns ``None`` for SDK callers (no proxy router) and for advisor models - the router doesn't know about, so those keep resolving through - ``litellm.anthropic_messages()`` provider inference. + Returns ``None`` for SDK callers (no proxy router) and for advisor models the router + doesn't know about, so those keep resolving through ``litellm.anthropic_messages()`` + provider inference. """ try: from litellm.proxy.proxy_server import llm_router @@ -394,11 +388,7 @@ def _resolve_advisor_router(advisor_model: str) -> "Router | None": return None if llm_router is None: return None - if llm_router.get_model_list(model_name=advisor_model): - return llm_router - if llm_router.model_group_alias and advisor_model in llm_router.model_group_alias: - return llm_router - if llm_router.pattern_router.route(advisor_model) is not None: + if llm_router.is_recognized_model(advisor_model) or llm_router.pattern_router.route(advisor_model): return llm_router return None diff --git a/tests/test_litellm/llms/anthropic/messages/test_advisor_orchestration.py b/tests/test_litellm/llms/anthropic/messages/test_advisor_orchestration.py index bcb1843f408..019eb4355c2 100644 --- a/tests/test_litellm/llms/anthropic/messages/test_advisor_orchestration.py +++ b/tests/test_litellm/llms/anthropic/messages/test_advisor_orchestration.py @@ -1050,7 +1050,9 @@ async def test_executor_failure_is_not_tagged(): # --------------------------------------------------------------------------- -def _router_with_advisor_deployment(recorder, advisor_model="claude-opus-4-8"): +def _router_with_advisor_deployment( + recorder, advisor_model="claude-opus-4-8", deployment_model=None, model_group_alias=None +): """Build a Router whose only deployment is the advisor model on Foundry. The recorder replaces ``litellm.anthropic_messages`` before construction @@ -1066,12 +1068,13 @@ def _router_with_advisor_deployment(recorder, advisor_model="claude-opus-4-8"): { "model_name": advisor_model, "litellm_params": { - "model": f"azure_ai/{advisor_model}", + "model": deployment_model or f"azure_ai/{advisor_model}", "api_base": "http://127.0.0.1:1/foundry", "api_key": "fake-foundry-key", }, } ], + model_group_alias=model_group_alias, num_retries=0, ) @@ -1125,6 +1128,66 @@ async def test_advisor_sub_call_routes_through_proxy_router(): assert "Final answer." in result["content"][0]["text"] +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("router_kwargs", "advisor_model"), + [ + pytest.param({"model_group_alias": {"advisor": "claude-opus-4-8"}}, "advisor", id="model_group_alias"), + pytest.param( + {"advisor_model": "azure_ai/*", "deployment_model": "azure_ai/*"}, + "azure_ai/claude-opus-4-8", + id="wildcard", + ), + ], +) +async def test_advisor_sub_call_routes_through_router_for_alias_and_wildcard(router_kwargs, advisor_model): + """Alias and wildcard advisor models resolve through the router like exact model_list matches.""" + import litellm.proxy.proxy_server as proxy_server + from litellm.llms.anthropic.experimental_pass_through.messages.interceptors.advisor import ( + AdvisorOrchestrationHandler, + ) + + router_calls = [] + + async def recorder(**kwargs): + router_calls.append(kwargs) + return _make_text_response("Use trial division.", model="claude-opus-4-8") + + router = _router_with_advisor_deployment(recorder, **router_kwargs) + + call_count = 0 + + async def mock_call(model, messages, tools, stream, max_tokens, **kwargs): + nonlocal call_count + call_count += 1 + if call_count == 1: + return _make_advisor_tool_use_response() + return _make_text_response("Final answer.") + + with ( + patch( + "litellm.llms.anthropic.experimental_pass_through.messages.interceptors.advisor._call_messages_handler", + side_effect=mock_call, + ), + patch.object(proxy_server, "llm_router", router), + ): + h = AdvisorOrchestrationHandler() + await h.handle( + model="executor-model", + messages=MESSAGES, + tools=[{**ADVISOR_TOOL, "model": advisor_model}], + stream=False, + max_tokens=512, + custom_llm_provider="azure_ai", + ) + + assert call_count == 2 + assert len(router_calls) == 1 + assert router_calls[0]["model"] == "azure_ai/claude-opus-4-8" + assert router_calls[0]["api_base"] == "http://127.0.0.1:1/foundry" + assert router_calls[0]["api_key"] == "fake-foundry-key" + + @pytest.mark.asyncio async def test_advisor_sub_call_bypasses_router_for_unconfigured_model(): """An advisor model the router doesn't know about keeps the SDK-level path."""