feat: add LongCat-Flash-Thinking-FP8 models to Chutes AI provider (#8426)

Co-authored-by: Roo Code <roomote@roocode.com>
Co-authored-by: daniel-lxs <ricciodaniel98@gmail.com>
This commit is contained in:
roomote[bot] 2025-10-27 13:48:10 -04:00 committed by GitHub
parent 634b1df175
commit e76ac42455
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
2 changed files with 79 additions and 0 deletions

View file

@ -35,6 +35,7 @@ export type ChutesModelId =
| "zai-org/GLM-4.5-turbo"
| "zai-org/GLM-4.6-FP8"
| "zai-org/GLM-4.6-turbo"
| "meituan-longcat/LongCat-Flash-Thinking-FP8"
| "moonshotai/Kimi-K2-Instruct-75k"
| "moonshotai/Kimi-K2-Instruct-0905"
| "Qwen/Qwen3-235B-A22B-Thinking-2507"
@ -339,6 +340,16 @@ export const chutesModels = {
outputPrice: 3.25,
description: "GLM-4.6-turbo model with 200K-token context window, optimized for fast inference.",
},
"meituan-longcat/LongCat-Flash-Thinking-FP8": {
maxTokens: 32768,
contextWindow: 128000,
supportsImages: false,
supportsPromptCache: false,
inputPrice: 0,
outputPrice: 0,
description:
"LongCat Flash Thinking FP8 model with 128K context window, optimized for complex reasoning and coding tasks.",
},
"Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": {
maxTokens: 32768,
contextWindow: 262144,

View file

@ -275,6 +275,74 @@ describe("ChutesHandler", () => {
)
})
it("should return zai-org/GLM-4.6-FP8 model with correct configuration", () => {
const testModelId: ChutesModelId = "zai-org/GLM-4.6-FP8"
const handlerWithModel = new ChutesHandler({
apiModelId: testModelId,
chutesApiKey: "test-chutes-api-key",
})
const model = handlerWithModel.getModel()
expect(model.id).toBe(testModelId)
expect(model.info).toEqual(
expect.objectContaining({
maxTokens: 32768,
contextWindow: 202752,
supportsImages: false,
supportsPromptCache: false,
inputPrice: 0,
outputPrice: 0,
description:
"GLM-4.6 introduces major upgrades over GLM-4.5, including a longer 200K-token context window for complex tasks, stronger coding performance in benchmarks and real-world tools (such as Claude Code, Cline, Roo Code, and Kilo Code), improved reasoning with tool use during inference, more capable and efficient agent integration, and refined writing that better matches human style, readability, and natural role-play scenarios.",
temperature: 0.5, // Default temperature for non-DeepSeek models
}),
)
})
it("should return zai-org/GLM-4.6-turbo model with correct configuration", () => {
const testModelId: ChutesModelId = "zai-org/GLM-4.6-turbo"
const handlerWithModel = new ChutesHandler({
apiModelId: testModelId,
chutesApiKey: "test-chutes-api-key",
})
const model = handlerWithModel.getModel()
expect(model.id).toBe(testModelId)
expect(model.info).toEqual(
expect.objectContaining({
maxTokens: 202752,
contextWindow: 202752,
supportsImages: false,
supportsPromptCache: false,
inputPrice: 1.15,
outputPrice: 3.25,
description: "GLM-4.6-turbo model with 200K-token context window, optimized for fast inference.",
temperature: 0.5, // Default temperature for non-DeepSeek models
}),
)
})
it("should return meituan-longcat/LongCat-Flash-Thinking-FP8 model with correct configuration", () => {
const testModelId: ChutesModelId = "meituan-longcat/LongCat-Flash-Thinking-FP8"
const handlerWithModel = new ChutesHandler({
apiModelId: testModelId,
chutesApiKey: "test-chutes-api-key",
})
const model = handlerWithModel.getModel()
expect(model.id).toBe(testModelId)
expect(model.info).toEqual(
expect.objectContaining({
maxTokens: 32768,
contextWindow: 128000,
supportsImages: false,
supportsPromptCache: false,
inputPrice: 0,
outputPrice: 0,
description:
"LongCat Flash Thinking FP8 model with 128K context window, optimized for complex reasoning and coding tasks.",
temperature: 0.5, // Default temperature for non-DeepSeek models
}),
)
})
it("should return Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8 model with correct configuration", () => {
const testModelId: ChutesModelId = "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8"
const handlerWithModel = new ChutesHandler({