mirror of
https://github.com/BerriAI/litellm.git
synced 2026-10-09 03:18:44 +00:00
fix(ui): respect reported reasoning efforts
This commit is contained in:
parent
901e312b17
commit
d5481ca037
3 changed files with 57 additions and 23 deletions
|
|
@ -273,8 +273,8 @@ const AddAutoRouterTab: React.FC<AddAutoRouterTabProps> = ({
|
|||
[presets, availability],
|
||||
);
|
||||
const automaticRouterConfig = React.useMemo(
|
||||
() => buildAutomaticRouterConfig(modelInfo, deployments ?? [], preferredTierModels, availability),
|
||||
[modelInfo, deployments, preferredTierModels, availability],
|
||||
() => buildAutomaticRouterConfig(modelInfo, deployments ?? [], preferredTierModels),
|
||||
[modelInfo, deployments, preferredTierModels],
|
||||
);
|
||||
|
||||
// A preset's models can only be trusted against a successfully loaded list. Selection and the
|
||||
|
|
|
|||
|
|
@ -4,6 +4,12 @@ import { buildModelAvailability } from "@/lib/autorouter_presets";
|
|||
import { buildAutomaticRouterConfig, buildPreferredTierModels, type PreferredTierModels } from "./auto_setup";
|
||||
|
||||
const models = (...names: string[]) => names.map((model_group) => ({ model_group, mode: "chat" }));
|
||||
const reasoningModel = (model_group: string, supported_reasoning_efforts: string[]) => ({
|
||||
model_group,
|
||||
mode: "chat",
|
||||
supports_reasoning: true,
|
||||
supported_reasoning_efforts,
|
||||
});
|
||||
const deployment = (model_name: string, model = model_name): AutoRouterDeployment => ({
|
||||
model_name,
|
||||
litellm_params: { model },
|
||||
|
|
@ -66,42 +72,79 @@ describe("buildAutomaticRouterConfig", () => {
|
|||
provider: "OpenAI",
|
||||
available: ["gpt-5.6-luna", "gpt-5.6-terra", "gpt-6-astra"],
|
||||
expected: ["gpt-5.6-luna", "gpt-5.6-terra", "gpt-6-astra", "gpt-6-astra"],
|
||||
supportedEfforts: ["low", "medium", "high", "xhigh", "max"],
|
||||
effort: "max",
|
||||
},
|
||||
{
|
||||
provider: "Anthropic",
|
||||
available: ["claude-haiku-4-5", "claude-sonnet-5", "claude-opus-5"],
|
||||
expected: ["claude-haiku-4-5", "claude-sonnet-5", "claude-opus-5", "claude-opus-5"],
|
||||
supportedEfforts: ["low", "medium", "high", "max"],
|
||||
effort: "max",
|
||||
},
|
||||
{
|
||||
provider: "Google",
|
||||
available: ["gemini-3.5-flash-lite", "gemini-3.8-flash", "gemini-3.1-pro-preview"],
|
||||
expected: ["gemini-3.5-flash-lite", "gemini-3.8-flash", "gemini-3.1-pro-preview", "gemini-3.1-pro-preview"],
|
||||
supportedEfforts: ["low", "medium", "high"],
|
||||
effort: "high",
|
||||
},
|
||||
{
|
||||
provider: "DeepSeek",
|
||||
available: ["deepseek-v4-flash", "deepseek-v4-pro"],
|
||||
expected: ["deepseek-v4-flash", "deepseek-v4-flash", "deepseek-v4-pro", "deepseek-v4-pro"],
|
||||
supportedEfforts: ["none", "high"],
|
||||
effort: "high",
|
||||
},
|
||||
{
|
||||
provider: "xAI",
|
||||
available: ["grok-4.6"],
|
||||
expected: ["grok-4.6", "grok-4.6", "grok-4.6", "grok-4.6"],
|
||||
supportedEfforts: ["low", "medium", "high", "xhigh"],
|
||||
effort: "xhigh",
|
||||
},
|
||||
])("uses the current $provider ladder and maximum reasoning effort", ({ available, expected, effort }) => {
|
||||
])(
|
||||
"uses the current $provider ladder and strongest advertised reasoning effort",
|
||||
({ available, expected, supportedEfforts, effort }) => {
|
||||
const availability = buildModelAvailability(available, []);
|
||||
const preferred = buildPreferredTierModels([], availability);
|
||||
const modelInfo = models(...available).map((model) =>
|
||||
model.model_group === expected[3] ? reasoningModel(model.model_group, supportedEfforts) : model,
|
||||
);
|
||||
|
||||
const config = buildAutomaticRouterConfig(modelInfo, [], preferred);
|
||||
|
||||
expect(tierModels(config)).toEqual(expected);
|
||||
expect(config?.tier_model_params).toEqual({
|
||||
REASONING: { [expected[3]]: { reasoning_effort: effort } },
|
||||
});
|
||||
},
|
||||
);
|
||||
|
||||
it("never exceeds the selected model group's advertised reasoning efforts", () => {
|
||||
const available = ["gpt-5.6-luna", "gpt-5.6-terra", "gpt-5.6-sol"];
|
||||
const availability = buildModelAvailability(available, []);
|
||||
const preferred = buildPreferredTierModels([], availability);
|
||||
const modelInfo = [
|
||||
...models("gpt-5.6-luna", "gpt-5.6-terra"),
|
||||
reasoningModel("gpt-5.6-sol", ["none", "low", "medium", "high", "xhigh"]),
|
||||
];
|
||||
|
||||
const config = buildAutomaticRouterConfig(modelInfo, [], preferred);
|
||||
|
||||
expect(config?.tier_model_params).toEqual({
|
||||
REASONING: { "gpt-5.6-sol": { reasoning_effort: "xhigh" } },
|
||||
});
|
||||
});
|
||||
|
||||
it("leaves reasoning effort unset when the proxy does not report supported values", () => {
|
||||
const available = ["grok-4.6"];
|
||||
const availability = buildModelAvailability(available, []);
|
||||
const preferred = buildPreferredTierModels([], availability);
|
||||
|
||||
const config = buildAutomaticRouterConfig(models(...available), [], preferred, availability);
|
||||
const config = buildAutomaticRouterConfig(models(...available), [], preferred);
|
||||
|
||||
expect(tierModels(config)).toEqual(expected);
|
||||
expect(config?.tier_model_params).toEqual({
|
||||
REASONING: { [expected[3]]: { reasoning_effort: effort } },
|
||||
});
|
||||
expect(config?.tier_model_params).toBeUndefined();
|
||||
});
|
||||
|
||||
it("reuses the closest available tier when a tier has no match", () => {
|
||||
|
|
|
|||
|
|
@ -7,6 +7,8 @@ const TIER_NAMES = ["SIMPLE", "MEDIUM", "COMPLEX", "REASONING"] as const;
|
|||
type TierName = (typeof TIER_NAMES)[number];
|
||||
export type PreferredTierModels = Record<TierName, string[]>;
|
||||
|
||||
const REASONING_EFFORT_STRENGTH = ["max", "xhigh", "high", "medium", "low", "minimal", "none"] as const;
|
||||
|
||||
const CURRENT_TIER_MODELS: PreferredTierModels = {
|
||||
SIMPLE: ["gpt-5.6-luna", "claude-haiku-4-5", "gemini-3.5-flash-lite", "deepseek-v4-flash"],
|
||||
MEDIUM: ["gpt-5.6-terra", "claude-sonnet-5", "gemini-3.8-flash", "deepseek-v4-flash"],
|
||||
|
|
@ -14,15 +16,6 @@ const CURRENT_TIER_MODELS: PreferredTierModels = {
|
|||
REASONING: ["gpt-6-astra", "gpt-5.6-sol", "claude-opus-5", "gemini-3.1-pro-preview", "deepseek-v4-pro", "grok-4.6"],
|
||||
};
|
||||
|
||||
const MAX_REASONING_EFFORT: Record<string, string> = {
|
||||
"gpt-6-astra": "max",
|
||||
"gpt-5.6-sol": "max",
|
||||
"claude-opus-5": "max",
|
||||
"gemini-3.1-pro-preview": "high",
|
||||
"deepseek-v4-pro": "high",
|
||||
"grok-4.6": "xhigh",
|
||||
};
|
||||
|
||||
export const buildPreferredTierModels = (
|
||||
presets: AutoRouterPreset[],
|
||||
availability: ModelAvailability,
|
||||
|
|
@ -63,7 +56,6 @@ export const buildAutomaticRouterConfig = (
|
|||
models: ModelGroup[],
|
||||
deployments: AutoRouterDeployment[],
|
||||
preferredByTier: PreferredTierModels,
|
||||
availability?: ModelAvailability,
|
||||
): ComplexityRouterConfigValue | null => {
|
||||
const autoRouterNames: ReadonlySet<string> = new Set(
|
||||
deployments
|
||||
|
|
@ -83,11 +75,10 @@ export const buildAutomaticRouterConfig = (
|
|||
const selected = selectPreferredTierModels(preferredByTier, usableNames);
|
||||
if (selected === null) return null;
|
||||
|
||||
const reasoningEffort =
|
||||
availability &&
|
||||
Object.entries(MAX_REASONING_EFFORT).find(
|
||||
([model]) => resolveAvailableModel(model, availability) === selected[3],
|
||||
)?.[1];
|
||||
const supportedReasoningEfforts = models.find(
|
||||
(model) => model.model_group === selected[3],
|
||||
)?.supported_reasoning_efforts;
|
||||
const reasoningEffort = REASONING_EFFORT_STRENGTH.find((effort) => supportedReasoningEfforts?.includes(effort));
|
||||
|
||||
return {
|
||||
tiers: {
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue