mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-08 03:08:45 +00:00
fix(ui): drop max_tokens from the auto-router connection probe
max_tokens=1 makes reasoning models (o1/o3/...) return a 400 "max_tokens reached" because reasoning tokens count against the cap, so a reachable reasoning tier showed a false failure in Test Connection. Live-verified: o3 400s with the cap and succeeds without it. Extract the request shape into a pure buildModelGroupTestRequest and cover it with a test asserting the chat body carries no max_tokens (or max_completion_tokens), so this regression is caught in unit tests instead of only against a live reasoning model.
This commit is contained in:
parent
ddc13b331a
commit
a80f85692b
2 changed files with 34 additions and 6 deletions
|
|
@ -514,3 +514,19 @@ describe("sessionSpendLogsCall", () => {
|
|||
expect(parsed.searchParams.get("page_size")).toBe("100");
|
||||
});
|
||||
});
|
||||
|
||||
describe("buildModelGroupTestRequest", () => {
|
||||
it("builds a chat completion request with NO max_tokens (reasoning models 400 on a tiny cap)", () => {
|
||||
const { path, body } = Networking.buildModelGroupTestRequest("o3", "chat");
|
||||
expect(path).toBe("/v1/chat/completions");
|
||||
expect(body).toEqual({ model: "o3", messages: [{ role: "user", content: "test from litellm" }] });
|
||||
expect(body).not.toHaveProperty("max_tokens");
|
||||
expect(body).not.toHaveProperty("max_completion_tokens");
|
||||
});
|
||||
|
||||
it("builds an embeddings request for embedding mode", () => {
|
||||
const { path, body } = Networking.buildModelGroupTestRequest("text-embedding-3-small", "embedding");
|
||||
expect(path).toBe("/v1/embeddings");
|
||||
expect(body).toEqual({ model: "text-embedding-3-small", input: "test from litellm" });
|
||||
});
|
||||
});
|
||||
|
|
|
|||
|
|
@ -2324,17 +2324,29 @@ export type ModelGroupConnectionResult = { status: "success" } | { status: "erro
|
|||
* resolves the group, credentials, and provider. Used by the auto-router Test
|
||||
* Connection to probe each tier's model group and the embedding model.
|
||||
*/
|
||||
/**
|
||||
* Build the minimal request that probes a model group by public name. No
|
||||
* max_tokens: reasoning models (o1/o3/...) reject a tiny cap with "max_tokens
|
||||
* reached" because reasoning tokens count against it, which would show a false
|
||||
* failure for a reachable tier.
|
||||
*/
|
||||
export const buildModelGroupTestRequest = (
|
||||
modelGroup: string,
|
||||
mode: "chat" | "embedding",
|
||||
): { path: string; body: Record<string, unknown> } =>
|
||||
mode === "embedding"
|
||||
? { path: "/v1/embeddings", body: { model: modelGroup, input: "test from litellm" } }
|
||||
: {
|
||||
path: "/v1/chat/completions",
|
||||
body: { model: modelGroup, messages: [{ role: "user", content: "test from litellm" }] },
|
||||
};
|
||||
|
||||
export const testModelGroupConnection = async (
|
||||
accessToken: string,
|
||||
modelGroup: string,
|
||||
mode: "chat" | "embedding",
|
||||
): Promise<ModelGroupConnectionResult> => {
|
||||
const path = mode === "embedding" ? "/v1/embeddings" : "/v1/chat/completions";
|
||||
const body =
|
||||
mode === "embedding"
|
||||
? { model: modelGroup, input: "test from litellm" }
|
||||
: { model: modelGroup, messages: [{ role: "user", content: "test from litellm" }], max_tokens: 1 };
|
||||
|
||||
const { path, body } = buildModelGroupTestRequest(modelGroup, mode);
|
||||
try {
|
||||
await apiClient.post(path, { accessToken, body });
|
||||
return { status: "success" };
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue