From 0ae62c5d02f7d24ba779a7047d34ec16fd8c2b15 Mon Sep 17 00:00:00 2001 From: "devin-ai-integration[bot]" <158243242+devin-ai-integration[bot]@users.noreply.github.com> Date: Thu, 8 Oct 2026 19:53:27 -0700 Subject: [PATCH] feat(ui): add the evaluation mode to the Add Model form (#45481) Co-authored-by: kerry Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com> --- .../health_endpoints/_health_endpoints.py | 1 + .../providers/test_decisions_wire.py | 17 +++++ .../health_endpoints/test_health_endpoints.py | 65 ++++++++++++++++++- .../AddModelForm.integration.test.tsx | 12 ++++ .../components/add_model/add_model_modes.tsx | 1 + .../handle_add_model_submit.test.tsx | 14 ++++ ui/litellm-dashboard/src/lib/http/schema.d.ts | 2 +- 7 files changed, 109 insertions(+), 3 deletions(-) diff --git a/litellm/proxy/health_endpoints/_health_endpoints.py b/litellm/proxy/health_endpoints/_health_endpoints.py index 4180ca1a656..557ff2680b4 100644 --- a/litellm/proxy/health_endpoints/_health_endpoints.py +++ b/litellm/proxy/health_endpoints/_health_endpoints.py @@ -2078,6 +2078,7 @@ async def test_model_connection( "responses", "anthropic_messages", "ocr", + "evaluation", ] | None = fastapi.Body( None, diff --git a/tests/integration/providers/test_decisions_wire.py b/tests/integration/providers/test_decisions_wire.py index db6629a43d2..07b02480fc8 100644 --- a/tests/integration/providers/test_decisions_wire.py +++ b/tests/integration/providers/test_decisions_wire.py @@ -252,6 +252,23 @@ def test_each_provider_gets_its_own_path_key_and_body_and_is_billed_from_the_cos assert math.isclose(_number(row["spend"]), expected_spend, rel_tol=1e-9), row +def test_test_connection_evaluation_mode_uses_typesafe_decisions_path(gateway: Gateway) -> None: + provider: Final = _PROVIDERS[1] + with gateway.scenario() as scenario: + handle: Final = _register(scenario, _answer_body(provider)) + model: Final = _deployment(scenario, handle, provider) + response: Final = gateway.request( + "POST", + "/health/test_connection", + {"litellm_params": {"model": model}, "mode": "evaluation"}, + ) + + assert response.status_code == 200, response.text + assert response.json()["status"] == "success", response.text + (call,) = _upstream_calls(gateway, handle) + assert call["path"] == f"/{handle.scenario_id}{provider.path}" + + def test_repeated_identical_requests_each_reach_the_upstream_and_are_each_billed(gateway: Gateway) -> None: with gateway.scenario() as scenario: handle: Final = _register(scenario, _answer_body(_PERPLEXITY)) diff --git a/tests/unit/proxy/health_endpoints/test_health_endpoints.py b/tests/unit/proxy/health_endpoints/test_health_endpoints.py index 266fd06333c..5aa763fcc9c 100644 --- a/tests/unit/proxy/health_endpoints/test_health_endpoints.py +++ b/tests/unit/proxy/health_endpoints/test_health_endpoints.py @@ -2,7 +2,7 @@ import asyncio import copy import json import time -from collections.abc import Iterator, Mapping, Sequence +from collections.abc import Awaitable, Iterator, Mapping, Sequence from contextlib import contextmanager from datetime import datetime, timedelta from types import MappingProxyType, SimpleNamespace @@ -705,6 +705,10 @@ def _test_connection_probe( ) -> Iterator[AsyncMock]: from litellm.types.router import Deployment, LiteLLM_Params + async def run_health_check(awaitable: Awaitable[object], _timeout: float) -> dict[str, str]: + await awaitable + return {"status": "healthy"} + router: Final = MagicMock() router.get_deployment.side_effect = lambda model_id: ( Deployment( @@ -727,7 +731,7 @@ def _test_connection_probe( patch("litellm.proxy.health_endpoints._health_endpoints.litellm.ahealth_check", ahealth_check), patch( "litellm.proxy.health_endpoints._health_endpoints.run_with_timeout", - AsyncMock(return_value={"status": "healthy"}), + AsyncMock(side_effect=run_health_check), ), ): yield ahealth_check @@ -874,6 +878,28 @@ async def test_test_model_connection_request_mode_wins_over_resolved_mode(): assert ahealth_check.call_args.kwargs["mode"] == "chat" +@pytest.mark.asyncio +async def test_test_model_connection_evaluation_mode_uses_decisions_handler(): + deployment: Final = MappingProxyType( + { + "model_name": "typesafe/jev-latest", + "litellm_params": {"model": "typesafe/jev-latest", "api_key": "fake-typesafe-key"}, + "model_info": {"id": "typesafe-jev-id"}, + } + ) + with _test_connection_probe(deployment) as ahealth_check: + result: Final = await health_test_model_connection( + request=MagicMock(), + mode="evaluation", + litellm_params={"model": "typesafe/jev-latest"}, + model_info={"id": "typesafe-jev-id"}, + user_api_key_dict=UserAPIKeyAuth(user_id="test-user", token="test-token"), + ) + + assert result["status"] == "success" + assert ahealth_check.call_args.kwargs["mode"] == "evaluation" + + @pytest.mark.asyncio async def test_test_model_connection_uses_loaded_deployment_team_id(): """ @@ -4665,6 +4691,41 @@ def test_test_model_connection_accepts_image_edit_mode(monkeypatch): assert response.json()["status"] == "success" +def test_test_model_connection_accepts_evaluation_mode(monkeypatch): + monkeypatch.setattr(litellm, "disable_aiohttp_transport", True) + litellm.in_memory_llm_clients_cache.flush_cache() + + app = FastAPI() + app.include_router(_health_endpoints_module.router) + app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(user_role=LitellmUserRoles.PROXY_ADMIN) + client = TestClient(app) + + with ( + patch( # test-quality-ok: endpoint reads the proxy-global DB client and 500s when it is None; it has no injection seam + "litellm.proxy.proxy_server.prisma_client", MagicMock() + ), + respx.mock(assert_all_called=True) as respx_mock, + ): + upstream = respx_mock.post("https://api.typesafe.ai/v1/systemone").respond( + json={ + "model": "jev-latest", + "answers": {"reachable": {"type": "noul", "noul": 1.0}}, + "usage": {"input_tokens": 12, "output_tokens": 1}, + } + ) + response = client.post( + "/health/test_connection", + json={ + "mode": "evaluation", + "litellm_params": {"model": "typesafe/jev-latest", "api_key": "sk-test"}, + }, + ) + + assert response.status_code == 200, response.text + assert response.json()["status"] == "success" + assert upstream.called + + def _pointfive_admin() -> UserAPIKeyAuth: return UserAPIKeyAuth(token="admin-token", user_id="admin-user", user_role=LitellmUserRoles.PROXY_ADMIN) diff --git a/ui/litellm-dashboard/src/components/add_model/AddModelForm.integration.test.tsx b/ui/litellm-dashboard/src/components/add_model/AddModelForm.integration.test.tsx index 220d46647f6..5fb95331114 100644 --- a/ui/litellm-dashboard/src/components/add_model/AddModelForm.integration.test.tsx +++ b/ui/litellm-dashboard/src/components/add_model/AddModelForm.integration.test.tsx @@ -312,6 +312,18 @@ describe("AddModelForm", () => { expect(await screen.findByRole("button", { name: "Add Model" })).toBeInTheDocument(); }); + it("offers the Evaluation decisions mode", async () => { + const mockUseAuthorized = vi.mocked(await import("@/app/(dashboard)/hooks/useAuthorized")); + mockUseAuthorized.default.mockReturnValue(mockAuthorizedUser("proxy_admin", "user-1", true)); + + renderWithProviders(); + + await screen.findByText("Provider"); + await userEvent.click(screen.getByRole("combobox", { name: "Mode" })); + + expect(await screen.findByRole("option", { name: "Evaluation - /v1/decisions", exact: true })).toBeInTheDocument(); + }); + it("shows only the Close button in the connection test dialog footer", async () => { const mockUseAuthorized = vi.mocked(await import("@/app/(dashboard)/hooks/useAuthorized")); mockUseAuthorized.default.mockReturnValue(mockAuthorizedUser("proxy_admin", "user-1", true)); diff --git a/ui/litellm-dashboard/src/components/add_model/add_model_modes.tsx b/ui/litellm-dashboard/src/components/add_model/add_model_modes.tsx index 05da3f96109..289fd390c62 100644 --- a/ui/litellm-dashboard/src/components/add_model/add_model_modes.tsx +++ b/ui/litellm-dashboard/src/components/add_model/add_model_modes.tsx @@ -13,6 +13,7 @@ export const TEST_MODES = [ { value: "batch", label: "Batch - /batch" }, { value: "anthropic_messages", label: "Anthropic Messages - /v1/messages" }, { value: "ocr", label: "OCR - /ocr" }, + { value: "evaluation", label: "Evaluation - /v1/decisions" }, ]; // Define the available auto router routing strategies diff --git a/ui/litellm-dashboard/src/components/add_model/handle_add_model_submit.test.tsx b/ui/litellm-dashboard/src/components/add_model/handle_add_model_submit.test.tsx index 923cf2aa0e3..6190b6aef78 100644 --- a/ui/litellm-dashboard/src/components/add_model/handle_add_model_submit.test.tsx +++ b/ui/litellm-dashboard/src/components/add_model/handle_add_model_submit.test.tsx @@ -102,6 +102,20 @@ describe("prepareModelAddRequest", () => { expect(deployment.litellmParamsObj.timeout).toBe(5); }); + it("saves the selected mode under model_info", async () => { + const formValues = { + model_mappings: [{ public_name: "Jev", litellm_model: "typesafe/jev-latest" }], + mode: "evaluation", + }; + + const deployments = await prepareModelAddRequest({ ...formValues }, "token", null); + + expect(deployments).toHaveLength(1); + const [deployment] = deployments!; + expect(deployment.modelInfoObj.mode).toBe("evaluation"); + expect(deployment.litellmParamsObj).not.toHaveProperty("mode"); + }); + it.each([ ["OpenAI", "openai/*"], ["Azure_AI_Studio", "azure_ai/*"], diff --git a/ui/litellm-dashboard/src/lib/http/schema.d.ts b/ui/litellm-dashboard/src/lib/http/schema.d.ts index fdc63268fe5..fa3c22c0713 100644 --- a/ui/litellm-dashboard/src/lib/http/schema.d.ts +++ b/ui/litellm-dashboard/src/lib/http/schema.d.ts @@ -27717,7 +27717,7 @@ export interface components { * Mode * @description The mode to test the model with. If not provided, resolved the way /health does: the deployment's model_info.mode (only while the request tests the deployment's own model), then the mode the provider requires for that model, then the model cost map. */ - mode?: ("chat" | "completion" | "embedding" | "audio_speech" | "audio_transcription" | "image_generation" | "image_edit" | "video_generation" | "batch" | "rerank" | "realtime" | "responses" | "anthropic_messages" | "ocr") | null; + mode?: ("chat" | "completion" | "embedding" | "audio_speech" | "audio_transcription" | "image_generation" | "image_edit" | "video_generation" | "batch" | "rerank" | "realtime" | "responses" | "anthropic_messages" | "ocr" | "evaluation") | null; /** * Model Info * @description Model info for the health check