From a80f85692b8feec7f25c1459f0da0ab51632bd10 Mon Sep 17 00:00:00 2001 From: Abhimanyu Kapur <38531241+akapur99@users.noreply.github.com> Date: Sat, 11 Jul 2026 16:43:36 -0700 Subject: [PATCH] fix(ui): drop max_tokens from the auto-router connection probe max_tokens=1 makes reasoning models (o1/o3/...) return a 400 "max_tokens reached" because reasoning tokens count against the cap, so a reachable reasoning tier showed a false failure in Test Connection. Live-verified: o3 400s with the cap and succeeds without it. Extract the request shape into a pure buildModelGroupTestRequest and cover it with a test asserting the chat body carries no max_tokens (or max_completion_tokens), so this regression is caught in unit tests instead of only against a live reasoning model. --- .../src/components/networking.test.ts | 16 +++++++++++++ .../src/components/networking.tsx | 24 ++++++++++++++----- 2 files changed, 34 insertions(+), 6 deletions(-) diff --git a/ui/litellm-dashboard/src/components/networking.test.ts b/ui/litellm-dashboard/src/components/networking.test.ts index 43f85cc8674..b9d04a00f61 100644 --- a/ui/litellm-dashboard/src/components/networking.test.ts +++ b/ui/litellm-dashboard/src/components/networking.test.ts @@ -514,3 +514,19 @@ describe("sessionSpendLogsCall", () => { expect(parsed.searchParams.get("page_size")).toBe("100"); }); }); + +describe("buildModelGroupTestRequest", () => { + it("builds a chat completion request with NO max_tokens (reasoning models 400 on a tiny cap)", () => { + const { path, body } = Networking.buildModelGroupTestRequest("o3", "chat"); + expect(path).toBe("/v1/chat/completions"); + expect(body).toEqual({ model: "o3", messages: [{ role: "user", content: "test from litellm" }] }); + expect(body).not.toHaveProperty("max_tokens"); + expect(body).not.toHaveProperty("max_completion_tokens"); + }); + + it("builds an embeddings request for embedding mode", () => { + const { path, body } = Networking.buildModelGroupTestRequest("text-embedding-3-small", "embedding"); + expect(path).toBe("/v1/embeddings"); + expect(body).toEqual({ model: "text-embedding-3-small", input: "test from litellm" }); + }); +}); diff --git a/ui/litellm-dashboard/src/components/networking.tsx b/ui/litellm-dashboard/src/components/networking.tsx index ae263727569..10d7f12604d 100644 --- a/ui/litellm-dashboard/src/components/networking.tsx +++ b/ui/litellm-dashboard/src/components/networking.tsx @@ -2324,17 +2324,29 @@ export type ModelGroupConnectionResult = { status: "success" } | { status: "erro * resolves the group, credentials, and provider. Used by the auto-router Test * Connection to probe each tier's model group and the embedding model. */ +/** + * Build the minimal request that probes a model group by public name. No + * max_tokens: reasoning models (o1/o3/...) reject a tiny cap with "max_tokens + * reached" because reasoning tokens count against it, which would show a false + * failure for a reachable tier. + */ +export const buildModelGroupTestRequest = ( + modelGroup: string, + mode: "chat" | "embedding", +): { path: string; body: Record } => + mode === "embedding" + ? { path: "/v1/embeddings", body: { model: modelGroup, input: "test from litellm" } } + : { + path: "/v1/chat/completions", + body: { model: modelGroup, messages: [{ role: "user", content: "test from litellm" }] }, + }; + export const testModelGroupConnection = async ( accessToken: string, modelGroup: string, mode: "chat" | "embedding", ): Promise => { - const path = mode === "embedding" ? "/v1/embeddings" : "/v1/chat/completions"; - const body = - mode === "embedding" - ? { model: modelGroup, input: "test from litellm" } - : { model: modelGroup, messages: [{ role: "user", content: "test from litellm" }], max_tokens: 1 }; - + const { path, body } = buildModelGroupTestRequest(modelGroup, mode); try { await apiClient.post(path, { accessToken, body }); return { status: "success" };