feat(ui): add the evaluation mode to the Add Model form (#45481)

Co-authored-by: kerry <kerry@berri.ai>
Co-authored-by: Devin AI <158243242+devin-ai-integration[bot]@users.noreply.github.com>
This commit is contained in:
devin-ai-integration[bot] 2026-10-08 19:53:27 -07:00 • committed by GitHub
parent ea43e27485
commit 0ae62c5d02
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
7 changed files with 109 additions and 3 deletions

View file

@ -2078,6 +2078,7 @@ async def test_model_connection(
"responses",
"anthropic_messages",
"ocr",
"evaluation",
]
| None = fastapi.Body(
None,

View file

@ -252,6 +252,23 @@ def test_each_provider_gets_its_own_path_key_and_body_and_is_billed_from_the_cos
assert math.isclose(_number(row["spend"]), expected_spend, rel_tol=1e-9), row
def test_test_connection_evaluation_mode_uses_typesafe_decisions_path(gateway: Gateway) -> None:
provider: Final = _PROVIDERS[1]
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _answer_body(provider))
model: Final = _deployment(scenario, handle, provider)
response: Final = gateway.request(
"POST",
"/health/test_connection",
{"litellm_params": {"model": model}, "mode": "evaluation"},
)
assert response.status_code == 200, response.text
assert response.json()["status"] == "success", response.text
(call,) = _upstream_calls(gateway, handle)
assert call["path"] == f"/{handle.scenario_id}{provider.path}"
def test_repeated_identical_requests_each_reach_the_upstream_and_are_each_billed(gateway: Gateway) -> None:
with gateway.scenario() as scenario:
handle: Final = _register(scenario, _answer_body(_PERPLEXITY))

View file

@ -2,7 +2,7 @@ import asyncio
import copy
import json
import time
from collections.abc import Iterator, Mapping, Sequence
from collections.abc import Awaitable, Iterator, Mapping, Sequence
from contextlib import contextmanager
from datetime import datetime, timedelta
from types import MappingProxyType, SimpleNamespace
@ -705,6 +705,10 @@ def _test_connection_probe(
) -> Iterator[AsyncMock]:
from litellm.types.router import Deployment, LiteLLM_Params
async def run_health_check(awaitable: Awaitable[object], _timeout: float) -> dict[str, str]:
await awaitable
return {"status": "healthy"}
router: Final = MagicMock()
router.get_deployment.side_effect = lambda model_id: (
Deployment(
@ -727,7 +731,7 @@ def _test_connection_probe(
patch("litellm.proxy.health_endpoints._health_endpoints.litellm.ahealth_check", ahealth_check),
patch(
"litellm.proxy.health_endpoints._health_endpoints.run_with_timeout",
AsyncMock(return_value={"status": "healthy"}),
AsyncMock(side_effect=run_health_check),
),
):
yield ahealth_check
@ -874,6 +878,28 @@ async def test_test_model_connection_request_mode_wins_over_resolved_mode():
assert ahealth_check.call_args.kwargs["mode"] == "chat"
@pytest.mark.asyncio
async def test_test_model_connection_evaluation_mode_uses_decisions_handler():
deployment: Final = MappingProxyType(
{
"model_name": "typesafe/jev-latest",
"litellm_params": {"model": "typesafe/jev-latest", "api_key": "fake-typesafe-key"},
"model_info": {"id": "typesafe-jev-id"},
}
)
with _test_connection_probe(deployment) as ahealth_check:
result: Final = await health_test_model_connection(
request=MagicMock(),
mode="evaluation",
litellm_params={"model": "typesafe/jev-latest"},
model_info={"id": "typesafe-jev-id"},
user_api_key_dict=UserAPIKeyAuth(user_id="test-user", token="test-token"),
)
assert result["status"] == "success"
assert ahealth_check.call_args.kwargs["mode"] == "evaluation"
@pytest.mark.asyncio
async def test_test_model_connection_uses_loaded_deployment_team_id():
"""
@ -4665,6 +4691,41 @@ def test_test_model_connection_accepts_image_edit_mode(monkeypatch):
assert response.json()["status"] == "success"
def test_test_model_connection_accepts_evaluation_mode(monkeypatch):
monkeypatch.setattr(litellm, "disable_aiohttp_transport", True)
litellm.in_memory_llm_clients_cache.flush_cache()
app = FastAPI()
app.include_router(_health_endpoints_module.router)
app.dependency_overrides[user_api_key_auth] = lambda: UserAPIKeyAuth(user_role=LitellmUserRoles.PROXY_ADMIN)
client = TestClient(app)
with (
patch( # test-quality-ok: endpoint reads the proxy-global DB client and 500s when it is None; it has no injection seam
"litellm.proxy.proxy_server.prisma_client", MagicMock()
),
respx.mock(assert_all_called=True) as respx_mock,
):
upstream = respx_mock.post("https://api.typesafe.ai/v1/systemone").respond(
json={
"model": "jev-latest",
"answers": {"reachable": {"type": "noul", "noul": 1.0}},
"usage": {"input_tokens": 12, "output_tokens": 1},
}
)
response = client.post(
"/health/test_connection",
json={
"mode": "evaluation",
"litellm_params": {"model": "typesafe/jev-latest", "api_key": "sk-test"},
},
)
assert response.status_code == 200, response.text
assert response.json()["status"] == "success"
assert upstream.called
def _pointfive_admin() -> UserAPIKeyAuth:
return UserAPIKeyAuth(token="admin-token", user_id="admin-user", user_role=LitellmUserRoles.PROXY_ADMIN)

View file

@ -312,6 +312,18 @@ describe("AddModelForm", () => {
expect(await screen.findByRole("button", { name: "Add Model" })).toBeInTheDocument();
});
it("offers the Evaluation decisions mode", async () => {
const mockUseAuthorized = vi.mocked(await import("@/app/(dashboard)/hooks/useAuthorized"));
mockUseAuthorized.default.mockReturnValue(mockAuthorizedUser("proxy_admin", "user-1", true));
renderWithProviders(<AddModelForm {...createTestProps()} />);
await screen.findByText("Provider");
await userEvent.click(screen.getByRole("combobox", { name: "Mode" }));
expect(await screen.findByRole("option", { name: "Evaluation - /v1/decisions", exact: true })).toBeInTheDocument();
});
it("shows only the Close button in the connection test dialog footer", async () => {
const mockUseAuthorized = vi.mocked(await import("@/app/(dashboard)/hooks/useAuthorized"));
mockUseAuthorized.default.mockReturnValue(mockAuthorizedUser("proxy_admin", "user-1", true));

View file

@ -13,6 +13,7 @@ export const TEST_MODES = [
{ value: "batch", label: "Batch - /batch" },
{ value: "anthropic_messages", label: "Anthropic Messages - /v1/messages" },
{ value: "ocr", label: "OCR - /ocr" },
{ value: "evaluation", label: "Evaluation - /v1/decisions" },
];
// Define the available auto router routing strategies

View file

@ -102,6 +102,20 @@ describe("prepareModelAddRequest", () => {
expect(deployment.litellmParamsObj.timeout).toBe(5);
});
it("saves the selected mode under model_info", async () => {
const formValues = {
model_mappings: [{ public_name: "Jev", litellm_model: "typesafe/jev-latest" }],
mode: "evaluation",
};
const deployments = await prepareModelAddRequest({ ...formValues }, "token", null);
expect(deployments).toHaveLength(1);
const [deployment] = deployments!;
expect(deployment.modelInfoObj.mode).toBe("evaluation");
expect(deployment.litellmParamsObj).not.toHaveProperty("mode");
});
it.each([
["OpenAI", "openai/*"],
["Azure_AI_Studio", "azure_ai/*"],

View file

@ -27717,7 +27717,7 @@ export interface components {
* Mode
* @description The mode to test the model with. If not provided, resolved the way /health does: the deployment's model_info.mode (only while the request tests the deployment's own model), then the mode the provider requires for that model, then the model cost map.
*/
mode?: ("chat" | "completion" | "embedding" | "audio_speech" | "audio_transcription" | "image_generation" | "image_edit" | "video_generation" | "batch" | "rerank" | "realtime" | "responses" | "anthropic_messages" | "ocr") | null;
mode?: ("chat" | "completion" | "embedding" | "audio_speech" | "audio_transcription" | "image_generation" | "image_edit" | "video_generation" | "batch" | "rerank" | "realtime" | "responses" | "anthropic_messages" | "ocr" | "evaluation") | null;
/**
* Model Info
* @description Model info for the health check