From 5fe3ed8124fa55f73f4657829febb6a975ed3892 Mon Sep 17 00:00:00 2001 From: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Sat, 1 Aug 2026 17:23:17 +0000 Subject: [PATCH] fix(usage): route Usage dashboard Ask AI chat through the proxy Router The Ask AI model dropdown is populated from proxy model groups, but the endpoint called litellm.acompletion() directly, which only resolves provider-prefixed model strings; every dropdown selection failed with 'LLM Provider NOT provided' and surfaced as a generic internal error Co-Authored-By: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../usage_endpoints/ai_usage_chat.py | 69 +++++++++++++---- .../usage_endpoints/test_ai_usage_chat.py | 74 +++++++++++++++++++ 2 files changed, 129 insertions(+), 14 deletions(-) diff --git a/litellm/proxy/management_endpoints/usage_endpoints/ai_usage_chat.py b/litellm/proxy/management_endpoints/usage_endpoints/ai_usage_chat.py index efb714e4da0..457bf6a418a 100644 --- a/litellm/proxy/management_endpoints/usage_endpoints/ai_usage_chat.py +++ b/litellm/proxy/management_endpoints/usage_endpoints/ai_usage_chat.py @@ -5,16 +5,18 @@ usage/spend data by querying the aggregated daily activity endpoints. import json from datetime import date -from typing import Any, AsyncIterator, Callable, Dict, List, Literal, Optional, cast +from typing import Any, AsyncIterator, Callable, Dict, List, Literal, Mapping, Optional, Sequence, Union, cast, overload from typing_extensions import TypedDict import litellm from litellm._logging import verbose_proxy_logger from litellm.constants import DEFAULT_COMPETITOR_DISCOVERY_MODEL +from litellm.litellm_core_utils.streaming_handler import CustomStreamWrapper from litellm.types.proxy.management_endpoints.common_daily_activity import ( SpendAnalyticsPaginatedResponse, ) +from litellm.types.utils import ModelResponse # --------------------------------------------------------------------------- # Constants @@ -494,16 +496,60 @@ async def _process_tool_call( chat_messages.append({"role": "tool", "tool_call_id": tc.id, "content": tool_result}) +@overload +async def _acompletion( + model: str, + messages: Sequence[Mapping[str, Any]], + *, + stream: Literal[True], + tools: Optional[Sequence[Mapping[str, Any]]] = None, +) -> CustomStreamWrapper: ... + + +@overload +async def _acompletion( + model: str, + messages: Sequence[Mapping[str, Any]], + *, + stream: Literal[False] = False, + tools: Optional[Sequence[Mapping[str, Any]]] = None, +) -> ModelResponse: ... + + +async def _acompletion( + model: str, + messages: Sequence[Mapping[str, Any]], + *, + stream: bool = False, + tools: Optional[Sequence[Mapping[str, Any]]] = None, +) -> Union[ModelResponse, CustomStreamWrapper]: + """ + Route through the proxy Router when `model` is a configured model group. + + Bare `litellm.acompletion` only understands provider-prefixed model strings, so the + virtual model names the dashboard offers would fail with "LLM Provider NOT provided". + """ + from litellm.proxy.proxy_server import llm_router + + completion = ( + llm_router.acompletion + if llm_router is not None and llm_router.get_model_list(model_name=model) + else litellm.acompletion + ) + return await completion( + model=model, + messages=messages, + tools=tools, + stream=stream, + temperature=USAGE_AI_TEMPERATURE, + ) + + async def _stream_final_response(model: str, chat_messages: List[Dict[str, Any]]) -> AsyncIterator[str]: """Stream the final LLM response after tool results are appended.""" yield _sse({"type": "status", "message": "Analyzing results..."}) - response = await litellm.acompletion( - model=model, - messages=chat_messages, - stream=True, - temperature=USAGE_AI_TEMPERATURE, - ) + response = await _acompletion(model, chat_messages, stream=True) async for chunk in response: delta = chunk.choices[0].delta.content if delta: @@ -527,13 +573,8 @@ async def stream_usage_ai_chat( try: yield _sse({"type": "status", "message": "Thinking..."}) tools = get_tools_for_role(is_admin) - response = await litellm.acompletion( - model=resolved_model, - messages=chat_messages, - tools=tools, - temperature=USAGE_AI_TEMPERATURE, - ) - choice = response.choices[0] # type: ignore + response = await _acompletion(resolved_model, chat_messages, tools=tools) + choice = response.choices[0] if not choice.message.tool_calls: if choice.message.content: diff --git a/tests/test_litellm/proxy/management_endpoints/usage_endpoints/test_ai_usage_chat.py b/tests/test_litellm/proxy/management_endpoints/usage_endpoints/test_ai_usage_chat.py index e8a74e41dae..af2091dce32 100644 --- a/tests/test_litellm/proxy/management_endpoints/usage_endpoints/test_ai_usage_chat.py +++ b/tests/test_litellm/proxy/management_endpoints/usage_endpoints/test_ai_usage_chat.py @@ -466,3 +466,77 @@ class TestUsageAiChatServiceAccountGuard: is_admin=False, ) assert "Endpoint-level guard missing" in str(exc_info.value) + + +class TestUsageAiChatModelResolution: + """ + Regression: the dashboard's model dropdown is populated from proxy model + groups (virtual aliases), which bare `litellm.acompletion` cannot resolve. + Those must be dispatched through the Router instead. + """ + + @staticmethod + def _text_response(content: str): + response = MagicMock() + response.choices = [MagicMock()] + response.choices[0].message.tool_calls = None + response.choices[0].message.content = content + return response + + @pytest.mark.asyncio + async def test_configured_model_group_is_routed_through_router(self): + mock_router = MagicMock() + mock_router.get_model_list.return_value = [{"model_name": "my-usage-alias"}] + mock_router.acompletion = AsyncMock(return_value=self._text_response("Spend is $1.00")) + + with ( + patch("litellm.proxy.proxy_server.llm_router", mock_router), + patch( + "litellm.proxy.management_endpoints.usage_endpoints.ai_usage_chat.litellm" + ) as mock_litellm, + ): + mock_litellm.acompletion = AsyncMock() + + events = [ + json.loads(event.replace("data: ", "").strip()) + async for event in stream_usage_ai_chat( + messages=[{"role": "user", "content": "What is my spend?"}], + model="my-usage-alias", + user_id="user-123", + is_admin=True, + ) + ] + + assert [e for e in events if e["type"] == "error"] == [] + assert [e["content"] for e in events if e["type"] == "chunk"] == ["Spend is $1.00"] + mock_litellm.acompletion.assert_not_called() + mock_router.get_model_list.assert_called_once_with(model_name="my-usage-alias") + assert mock_router.acompletion.call_args.kwargs["model"] == "my-usage-alias" + + @pytest.mark.asyncio + async def test_unknown_model_falls_back_to_sdk(self): + mock_router = MagicMock() + mock_router.get_model_list.return_value = None + mock_router.acompletion = AsyncMock() + + with ( + patch("litellm.proxy.proxy_server.llm_router", mock_router), + patch( + "litellm.proxy.management_endpoints.usage_endpoints.ai_usage_chat.litellm" + ) as mock_litellm, + ): + mock_litellm.acompletion = AsyncMock(return_value=self._text_response("hi")) + + events = [ + json.loads(event.replace("data: ", "").strip()) + async for event in stream_usage_ai_chat( + messages=[{"role": "user", "content": "hi"}], + model="gpt-4o-mini", + user_id="user-123", + is_admin=True, + ) + ] + + assert [e for e in events if e["type"] == "error"] == [] + mock_router.acompletion.assert_not_called() + assert mock_litellm.acompletion.call_args.kwargs["model"] == "gpt-4o-mini"