fix(usage_ai_chat): route through llm_router so proxy model aliases work

The Ask AI panel called litellm.acompletion directly, which bypasses the
proxy's model_list and dispatches straight to OpenAI using OPENAI_API_KEY
from the env. Any deployment whose credentials live in the proxy config
(Bedrock, Azure, Anthropic-only, etc.) hit AuthenticationError on every
request, surfaced to the user as the generic "An internal error occurred."

Now we look up llm_router from proxy_server first; only fall back to
litellm.acompletion when no router is initialized (unit tests).
This commit is contained in:
Claude 2026-05-06 14:11:13 +00:00
parent e9fb29061a
commit ff4d175dd3
No known key found for this signature in database
2 changed files with 89 additions and 2 deletions

View file

@ -426,6 +426,22 @@ def _sse(event: SSEEvent) -> str:
return f"data: {json.dumps(event)}\n\n"
async def _acompletion(**kwargs: Any) -> Any:
"""Run completion through the proxy router so model aliases and credentials
configured in the proxy's ``model_list`` are honored. Falls back to
``litellm.acompletion`` directly when no router is initialized — primarily
for unit tests that don't bootstrap a full proxy. Without this, the AI
chat endpoint dispatches straight to OpenAI using ``OPENAI_API_KEY`` from
the env, breaking every deployment whose credentials live in the proxy
config (Bedrock, Azure, Anthropic-only setups, etc.).
"""
from litellm.proxy.proxy_server import llm_router
if llm_router is not None:
return await llm_router.acompletion(**kwargs)
return await litellm.acompletion(**kwargs)
def _resolve_fetch_kwargs(
fn_name: str,
fn_args: Dict[str, str],
@ -524,7 +540,7 @@ async def _stream_final_response(
"""Stream the final LLM response after tool results are appended."""
yield _sse({"type": "status", "message": "Analyzing results..."})
response = await litellm.acompletion(
response = await _acompletion(
model=model,
messages=chat_messages,
stream=True,
@ -555,7 +571,7 @@ async def stream_usage_ai_chat(
try:
yield _sse({"type": "status", "message": "Thinking..."})
tools = get_tools_for_role(is_admin)
response = await litellm.acompletion(
response = await _acompletion(
model=resolved_model,
messages=chat_messages,
tools=tools,

View file

@ -466,3 +466,74 @@ class TestUsageAiChatServiceAccountGuard:
is_admin=False,
)
assert "Endpoint-level guard missing" in str(exc_info.value)
class TestUsageAiChatRouterDispatch:
"""
Regression: the AI chat endpoint must dispatch through the proxy's
llm_router when one is configured, so model aliases and credentials from
the proxy config (Bedrock, Azure, etc.) are honored. The previous code
called litellm.acompletion directly, which only worked when OPENAI_API_KEY
happened to be set on the proxy env.
"""
@pytest.mark.asyncio
async def test_uses_llm_router_when_configured(self):
from litellm.proxy import proxy_server
mock_router = MagicMock()
mock_router.acompletion = AsyncMock()
mock_no_tools_response = MagicMock()
mock_no_tools_response.choices = [MagicMock()]
mock_no_tools_response.choices[0].message.tool_calls = None
mock_no_tools_response.choices[0].message.content = "Hello from router."
mock_router.acompletion.return_value = mock_no_tools_response
original_router = getattr(proxy_server, "llm_router", None)
try:
proxy_server.llm_router = mock_router
events = []
async for event in stream_usage_ai_chat(
messages=[{"role": "user", "content": "hi"}],
model="my-proxy-alias",
user_id="user-123",
is_admin=True,
):
events.append(event)
finally:
proxy_server.llm_router = original_router
mock_router.acompletion.assert_awaited_once()
call_kwargs = mock_router.acompletion.await_args.kwargs
assert call_kwargs["model"] == "my-proxy-alias"
@pytest.mark.asyncio
async def test_falls_back_to_litellm_when_no_router(self):
from litellm.proxy import proxy_server
mock_no_tools_response = MagicMock()
mock_no_tools_response.choices = [MagicMock()]
mock_no_tools_response.choices[0].message.tool_calls = None
mock_no_tools_response.choices[0].message.content = "Hello."
original_router = getattr(proxy_server, "llm_router", None)
try:
proxy_server.llm_router = None
with patch(
"litellm.proxy.management_endpoints.usage_endpoints.ai_usage_chat.litellm"
) as mock_litellm:
mock_litellm.acompletion = AsyncMock(
return_value=mock_no_tools_response
)
events = []
async for event in stream_usage_ai_chat(
messages=[{"role": "user", "content": "hi"}],
model="gpt-4o-mini",
user_id="user-123",
is_admin=True,
):
events.append(event)
mock_litellm.acompletion.assert_awaited_once()
finally:
proxy_server.llm_router = original_router