feat: add GLM-4.6-turbo model to Chutes provider

- Added zai-org/GLM-4.6-turbo to ChutesModelId type definition
- Added model configuration with 200K context window and turbo pricing
- Maintains consistency with existing GLM-4.5-turbo pricing model

Fixes #8515
This commit is contained in:
Roo Code 2025-10-05 00:43:22 +00:00
parent 97f968673f
commit 27b3314988

View file

@ -34,6 +34,7 @@ export type ChutesModelId =
| "zai-org/GLM-4.5-FP8"
| "zai-org/GLM-4.5-turbo"
| "zai-org/GLM-4.6-FP8"
| "zai-org/GLM-4.6-turbo"
| "moonshotai/Kimi-K2-Instruct-75k"
| "moonshotai/Kimi-K2-Instruct-0905"
| "Qwen/Qwen3-235B-A22B-Thinking-2507"
@ -87,7 +88,8 @@ export const chutesModels = {
supportsPromptCache: false,
inputPrice: 0.23,
outputPrice: 0.9,
description: "DeepSeekV3.1Terminus is an update to V3.1 that improves language consistency by reducing CN/EN mixups and eliminating random characters, while strengthening agent capabilities with notably better Code Agent and Search Agent performance.",
description:
"DeepSeekV3.1Terminus is an update to V3.1 that improves language consistency by reducing CN/EN mixups and eliminating random characters, while strengthening agent capabilities with notably better Code Agent and Search Agent performance.",
},
"deepseek-ai/DeepSeek-V3.1-turbo": {
maxTokens: 32768,
@ -96,7 +98,8 @@ export const chutesModels = {
supportsPromptCache: false,
inputPrice: 1.0,
outputPrice: 3.0,
description: "DeepSeek-V3.1-turbo is an FP8, speculative-decoding turbo variant optimized for ultra-fast single-shot queries (~200 TPS), with outputs close to the originals and solid function calling/reasoning/structured output, priced at $1/M input and $3/M output tokens, using 2× quota per request and not intended for bulk workloads.",
description:
"DeepSeek-V3.1-turbo is an FP8, speculative-decoding turbo variant optimized for ultra-fast single-shot queries (~200 TPS), with outputs close to the originals and solid function calling/reasoning/structured output, priced at $1/M input and $3/M output tokens, using 2× quota per request and not intended for bulk workloads.",
},
"deepseek-ai/DeepSeek-V3.2-Exp": {
maxTokens: 163840,
@ -105,7 +108,8 @@ export const chutesModels = {
supportsPromptCache: false,
inputPrice: 0.25,
outputPrice: 0.35,
description: "DeepSeek-V3.2-Exp is an experimental LLM that introduces DeepSeek Sparse Attention to improve longcontext training and inference efficiency while maintaining performance comparable to V3.1Terminus.",
description:
"DeepSeek-V3.2-Exp is an experimental LLM that introduces DeepSeek Sparse Attention to improve longcontext training and inference efficiency while maintaining performance comparable to V3.1Terminus.",
},
"unsloth/Llama-3.3-70B-Instruct": {
maxTokens: 32768, // From Groq
@ -326,6 +330,16 @@ export const chutesModels = {
description:
"GLM-4.6 introduces major upgrades over GLM-4.5, including a longer 200K-token context window for complex tasks, stronger coding performance in benchmarks and real-world tools (such as Claude Code, Cline, Roo Code, and Kilo Code), improved reasoning with tool use during inference, more capable and efficient agent integration, and refined writing that better matches human style, readability, and natural role-play scenarios.",
},
"zai-org/GLM-4.6-turbo": {
maxTokens: 32768,
contextWindow: 202752,
supportsImages: false,
supportsPromptCache: false,
inputPrice: 1,
outputPrice: 3,
description:
"GLM-4.6-turbo model with 200K token context window, optimized for fast inference with GLM-4.6 capabilities.",
},
"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": {
maxTokens: 32768,
contextWindow: 262144,
@ -387,8 +401,9 @@ export const chutesModels = {
contextWindow: 262144,
supportsImages: true,
supportsPromptCache: false,
inputPrice: 0.1600,
outputPrice: 0.6500,
description: "Qwen3VL235BA22BThinking is an openweight MoE visionlanguage model (235B total, ~22B activated) optimized for deliberate multistep reasoning with strong textimagevideo understanding and longcontext capabilities.",
inputPrice: 0.16,
outputPrice: 0.65,
description:
"Qwen3VL235BA22BThinking is an openweight MoE visionlanguage model (235B total, ~22B activated) optimized for deliberate multistep reasoning with strong textimagevideo understanding and longcontext capabilities.",
},
} as const satisfies Record<string, ModelInfo>