fix(ui): drop max_tokens from the auto-router connection probe

max_tokens=1 makes reasoning models (o1/o3/...) return a 400 "max_tokens
reached" because reasoning tokens count against the cap, so a reachable
reasoning tier showed a false failure in Test Connection. Live-verified: o3
400s with the cap and succeeds without it.

Extract the request shape into a pure buildModelGroupTestRequest and cover it
with a test asserting the chat body carries no max_tokens (or
max_completion_tokens), so this regression is caught in unit tests instead of
only against a live reasoning model.
This commit is contained in:
Abhimanyu Kapur 2026-07-11 16:43:36 -07:00
parent ddc13b331a
commit a80f85692b
2 changed files with 34 additions and 6 deletions

View file

@ -514,3 +514,19 @@ describe("sessionSpendLogsCall", () => {
expect(parsed.searchParams.get("page_size")).toBe("100");
});
});
describe("buildModelGroupTestRequest", () => {
it("builds a chat completion request with NO max_tokens (reasoning models 400 on a tiny cap)", () => {
const { path, body } = Networking.buildModelGroupTestRequest("o3", "chat");
expect(path).toBe("/v1/chat/completions");
expect(body).toEqual({ model: "o3", messages: [{ role: "user", content: "test from litellm" }] });
expect(body).not.toHaveProperty("max_tokens");
expect(body).not.toHaveProperty("max_completion_tokens");
});
it("builds an embeddings request for embedding mode", () => {
const { path, body } = Networking.buildModelGroupTestRequest("text-embedding-3-small", "embedding");
expect(path).toBe("/v1/embeddings");
expect(body).toEqual({ model: "text-embedding-3-small", input: "test from litellm" });
});
});

View file

@ -2324,17 +2324,29 @@ export type ModelGroupConnectionResult = { status: "success" } | { status: "erro
* resolves the group, credentials, and provider. Used by the auto-router Test
* Connection to probe each tier's model group and the embedding model.
*/
/**
* Build the minimal request that probes a model group by public name. No
* max_tokens: reasoning models (o1/o3/...) reject a tiny cap with "max_tokens
* reached" because reasoning tokens count against it, which would show a false
* failure for a reachable tier.
*/
export const buildModelGroupTestRequest = (
modelGroup: string,
mode: "chat" | "embedding",
): { path: string; body: Record<string, unknown> } =>
mode === "embedding"
? { path: "/v1/embeddings", body: { model: modelGroup, input: "test from litellm" } }
: {
path: "/v1/chat/completions",
body: { model: modelGroup, messages: [{ role: "user", content: "test from litellm" }] },
};
export const testModelGroupConnection = async (
accessToken: string,
modelGroup: string,
mode: "chat" | "embedding",
): Promise<ModelGroupConnectionResult> => {
const path = mode === "embedding" ? "/v1/embeddings" : "/v1/chat/completions";
const body =
mode === "embedding"
? { model: modelGroup, input: "test from litellm" }
: { model: modelGroup, messages: [{ role: "user", content: "test from litellm" }], max_tokens: 1 };
const { path, body } = buildModelGroupTestRequest(modelGroup, mode);
try {
await apiClient.post(path, { accessToken, body });
return { status: "success" };