mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-04 02:31:27 +00:00
fix(usage_ai_chat): route through llm_router so proxy model aliases work
The Ask AI panel called litellm.acompletion directly, which bypasses the proxy's model_list and dispatches straight to OpenAI using OPENAI_API_KEY from the env. Any deployment whose credentials live in the proxy config (Bedrock, Azure, Anthropic-only, etc.) hit AuthenticationError on every request, surfaced to the user as the generic "An internal error occurred." Now we look up llm_router from proxy_server first; only fall back to litellm.acompletion when no router is initialized (unit tests).
This commit is contained in:
parent
e9fb29061a
commit
ff4d175dd3
2 changed files with 89 additions and 2 deletions
|
|
@ -426,6 +426,22 @@ def _sse(event: SSEEvent) -> str:
|
|||
return f"data: {json.dumps(event)}\n\n"
|
||||
|
||||
|
||||
async def _acompletion(**kwargs: Any) -> Any:
|
||||
"""Run completion through the proxy router so model aliases and credentials
|
||||
configured in the proxy's ``model_list`` are honored. Falls back to
|
||||
``litellm.acompletion`` directly when no router is initialized — primarily
|
||||
for unit tests that don't bootstrap a full proxy. Without this, the AI
|
||||
chat endpoint dispatches straight to OpenAI using ``OPENAI_API_KEY`` from
|
||||
the env, breaking every deployment whose credentials live in the proxy
|
||||
config (Bedrock, Azure, Anthropic-only setups, etc.).
|
||||
"""
|
||||
from litellm.proxy.proxy_server import llm_router
|
||||
|
||||
if llm_router is not None:
|
||||
return await llm_router.acompletion(**kwargs)
|
||||
return await litellm.acompletion(**kwargs)
|
||||
|
||||
|
||||
def _resolve_fetch_kwargs(
|
||||
fn_name: str,
|
||||
fn_args: Dict[str, str],
|
||||
|
|
@ -524,7 +540,7 @@ async def _stream_final_response(
|
|||
"""Stream the final LLM response after tool results are appended."""
|
||||
yield _sse({"type": "status", "message": "Analyzing results..."})
|
||||
|
||||
response = await litellm.acompletion(
|
||||
response = await _acompletion(
|
||||
model=model,
|
||||
messages=chat_messages,
|
||||
stream=True,
|
||||
|
|
@ -555,7 +571,7 @@ async def stream_usage_ai_chat(
|
|||
try:
|
||||
yield _sse({"type": "status", "message": "Thinking..."})
|
||||
tools = get_tools_for_role(is_admin)
|
||||
response = await litellm.acompletion(
|
||||
response = await _acompletion(
|
||||
model=resolved_model,
|
||||
messages=chat_messages,
|
||||
tools=tools,
|
||||
|
|
|
|||
|
|
@ -466,3 +466,74 @@ class TestUsageAiChatServiceAccountGuard:
|
|||
is_admin=False,
|
||||
)
|
||||
assert "Endpoint-level guard missing" in str(exc_info.value)
|
||||
|
||||
|
||||
class TestUsageAiChatRouterDispatch:
|
||||
"""
|
||||
Regression: the AI chat endpoint must dispatch through the proxy's
|
||||
llm_router when one is configured, so model aliases and credentials from
|
||||
the proxy config (Bedrock, Azure, etc.) are honored. The previous code
|
||||
called litellm.acompletion directly, which only worked when OPENAI_API_KEY
|
||||
happened to be set on the proxy env.
|
||||
"""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_uses_llm_router_when_configured(self):
|
||||
from litellm.proxy import proxy_server
|
||||
|
||||
mock_router = MagicMock()
|
||||
mock_router.acompletion = AsyncMock()
|
||||
|
||||
mock_no_tools_response = MagicMock()
|
||||
mock_no_tools_response.choices = [MagicMock()]
|
||||
mock_no_tools_response.choices[0].message.tool_calls = None
|
||||
mock_no_tools_response.choices[0].message.content = "Hello from router."
|
||||
mock_router.acompletion.return_value = mock_no_tools_response
|
||||
|
||||
original_router = getattr(proxy_server, "llm_router", None)
|
||||
try:
|
||||
proxy_server.llm_router = mock_router
|
||||
events = []
|
||||
async for event in stream_usage_ai_chat(
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
model="my-proxy-alias",
|
||||
user_id="user-123",
|
||||
is_admin=True,
|
||||
):
|
||||
events.append(event)
|
||||
finally:
|
||||
proxy_server.llm_router = original_router
|
||||
|
||||
mock_router.acompletion.assert_awaited_once()
|
||||
call_kwargs = mock_router.acompletion.await_args.kwargs
|
||||
assert call_kwargs["model"] == "my-proxy-alias"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_falls_back_to_litellm_when_no_router(self):
|
||||
from litellm.proxy import proxy_server
|
||||
|
||||
mock_no_tools_response = MagicMock()
|
||||
mock_no_tools_response.choices = [MagicMock()]
|
||||
mock_no_tools_response.choices[0].message.tool_calls = None
|
||||
mock_no_tools_response.choices[0].message.content = "Hello."
|
||||
|
||||
original_router = getattr(proxy_server, "llm_router", None)
|
||||
try:
|
||||
proxy_server.llm_router = None
|
||||
with patch(
|
||||
"litellm.proxy.management_endpoints.usage_endpoints.ai_usage_chat.litellm"
|
||||
) as mock_litellm:
|
||||
mock_litellm.acompletion = AsyncMock(
|
||||
return_value=mock_no_tools_response
|
||||
)
|
||||
events = []
|
||||
async for event in stream_usage_ai_chat(
|
||||
messages=[{"role": "user", "content": "hi"}],
|
||||
model="gpt-4o-mini",
|
||||
user_id="user-123",
|
||||
is_admin=True,
|
||||
):
|
||||
events.append(event)
|
||||
mock_litellm.acompletion.assert_awaited_once()
|
||||
finally:
|
||||
proxy_server.llm_router = original_router
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue