fix: add GLM model detection to OpenAI Compatible provider

This commit adds GLM model detection and related optimizations to the
OpenAI Compatible provider (OpenAiHandler), which handles the "OpenAI
Compatible" option in the UI when users set a custom base URL.

Changes:
- Import GLM detection utilities and Z.ai format converter
- Add glmConfig property to track GLM model configuration
- Detect GLM model on construction when model ID is available
- Re-detect GLM model in createMessage if model ID changes
- For GLM models:
  - Use convertToZAiFormat with mergeToolResultText to prevent
    conversation flow disruption
  - Disable parallel_tool_calls as GLM models may not support it
  - Add thinking parameter for GLM-4.7 models
- Add console logging for detection results and applied optimizations

This addresses the feedback in issue #11071 where the user clarified
they are using the "OpenAI Compatible" provider, not the "OpenAI"
provider. The previous changes in this PR only affected the
base-openai-compatible-provider.ts (used by Z.ai, Groq, etc.) and
lm-studio.ts, but not openai.ts which handles OpenAI Compatible.
This commit is contained in:
Roo Code 2026-01-30 12:12:44 +00:00
parent 96b68450b2
commit 9e96578df4

View file

@ -15,6 +15,7 @@ import type { ApiHandlerOptions } from "../../shared/api"
import { TagMatcher } from "../../utils/tag-matcher"
import { convertToOpenAiMessages } from "../transform/openai-format"
import { convertToZAiFormat } from "../transform/zai-format"
import { convertToR1Format } from "../transform/r1-format"
import { ApiStream, ApiStreamUsageChunk } from "../transform/stream"
import { getModelParams } from "../transform/model-params"
@ -24,6 +25,7 @@ import { BaseProvider } from "./base-provider"
import type { SingleCompletionHandler, ApiHandlerCreateMessageMetadata } from "../index"
import { getApiRequestTimeout } from "./utils/timeout-config"
import { handleOpenAIError } from "./utils/openai-error-handler"
import { detectGlmModel, logGlmDetection, type GlmModelConfig } from "./utils/glm-model-detection"
// TODO: Rename this to OpenAICompatibleHandler. Also, I think the
// `OpenAINativeHandler` can subclass from this, since it's obviously
@ -31,7 +33,8 @@ import { handleOpenAIError } from "./utils/openai-error-handler"
export class OpenAiHandler extends BaseProvider implements SingleCompletionHandler {
protected options: ApiHandlerOptions
protected client: OpenAI
private readonly providerName = "OpenAI"
private readonly providerName = "OpenAI Compatible"
private glmConfig: GlmModelConfig | null = null
constructor(options: ApiHandlerOptions) {
super()
@ -77,6 +80,13 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl
timeout,
})
}
// Detect GLM model on construction if model ID is available
const modelId = this.options.openAiModelId || ""
if (modelId) {
this.glmConfig = detectGlmModel(modelId)
logGlmDetection(this.providerName, modelId, this.glmConfig)
}
}
override async *createMessage(
@ -91,6 +101,12 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl
const isAzureAiInference = this._isAzureAiInference(modelUrl)
const deepseekReasoner = modelId.includes("deepseek-reasoner") || enabledR1Format
// Re-detect GLM model if not already done or if model ID changed
if (!this.glmConfig || this.glmConfig.originalModelId !== modelId) {
this.glmConfig = detectGlmModel(modelId)
logGlmDetection(this.providerName, modelId, this.glmConfig)
}
if (modelId.includes("o1") || modelId.includes("o3") || modelId.includes("o4")) {
yield* this.handleO3FamilyMessage(modelId, systemPrompt, messages, metadata)
return
@ -106,6 +122,12 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl
if (deepseekReasoner) {
convertedMessages = convertToR1Format([{ role: "user", content: systemPrompt }, ...messages])
} else if (this.glmConfig.isGlmModel) {
// GLM models benefit from mergeToolResultText to prevent reasoning_content loss
const glmConvertedMessages = convertToZAiFormat(messages, {
mergeToolResultText: this.glmConfig.mergeToolResultText,
})
convertedMessages = [systemMessage, ...glmConvertedMessages]
} else {
if (modelInfo.supportsPromptCache) {
systemMessage = {
@ -152,6 +174,16 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl
const isGrokXAI = this._isGrokXAI(this.options.openAiBaseUrl)
// Determine parallel_tool_calls setting
// Disable for GLM models as they may not support it properly
let parallelToolCalls: boolean
if (this.glmConfig.isGlmModel && this.glmConfig.disableParallelToolCalls) {
parallelToolCalls = false
console.log(`[${this.providerName}] parallel_tool_calls disabled for GLM model`)
} else {
parallelToolCalls = metadata?.parallelToolCalls ?? true
}
const requestOptions: OpenAI.Chat.Completions.ChatCompletionCreateParamsStreaming = {
model: modelId,
temperature: this.options.modelTemperature ?? (deepseekReasoner ? DEEP_SEEK_DEFAULT_TEMPERATURE : 0),
@ -161,12 +193,19 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl
...(reasoning && reasoning),
tools: this.convertToolsForOpenAI(metadata?.tools),
tool_choice: metadata?.tool_choice,
parallel_tool_calls: metadata?.parallelToolCalls ?? true,
parallel_tool_calls: parallelToolCalls,
}
// Add max_tokens if needed
this.addMaxTokensIfNeeded(requestOptions, modelInfo)
// For GLM-4.7 models with thinking support, add thinking parameter
if (this.glmConfig.isGlmModel && this.glmConfig.supportsThinking) {
const useReasoning = this.options.enableReasoningEffort !== false // Default to enabled for GLM-4.7
;(requestOptions as any).thinking = useReasoning ? { type: "enabled" } : { type: "disabled" }
console.log(`[${this.providerName}] GLM-4.7 thinking mode: ${useReasoning ? "enabled" : "disabled"}`)
}
let stream
try {
stream = await this.client.chat.completions.create(
@ -221,20 +260,46 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl
yield this.processUsageMetrics(lastUsage, modelInfo)
}
} else {
// Determine message conversion based on model type
let nonStreamingMessages
if (deepseekReasoner) {
nonStreamingMessages = convertToR1Format([{ role: "user", content: systemPrompt }, ...messages])
} else if (this.glmConfig.isGlmModel) {
// GLM models benefit from mergeToolResultText to prevent reasoning_content loss
const glmConvertedMessages = convertToZAiFormat(messages, {
mergeToolResultText: this.glmConfig.mergeToolResultText,
})
nonStreamingMessages = [systemMessage, ...glmConvertedMessages]
} else {
nonStreamingMessages = [systemMessage, ...convertToOpenAiMessages(messages)]
}
// Determine parallel_tool_calls setting for non-streaming
let nonStreamingParallelToolCalls: boolean
if (this.glmConfig.isGlmModel && this.glmConfig.disableParallelToolCalls) {
nonStreamingParallelToolCalls = false
} else {
nonStreamingParallelToolCalls = metadata?.parallelToolCalls ?? true
}
const requestOptions: OpenAI.Chat.Completions.ChatCompletionCreateParamsNonStreaming = {
model: modelId,
messages: deepseekReasoner
? convertToR1Format([{ role: "user", content: systemPrompt }, ...messages])
: [systemMessage, ...convertToOpenAiMessages(messages)],
messages: nonStreamingMessages,
// Tools are always present (minimum ALWAYS_AVAILABLE_TOOLS)
tools: this.convertToolsForOpenAI(metadata?.tools),
tool_choice: metadata?.tool_choice,
parallel_tool_calls: metadata?.parallelToolCalls ?? true,
parallel_tool_calls: nonStreamingParallelToolCalls,
}
// Add max_tokens if needed
this.addMaxTokensIfNeeded(requestOptions, modelInfo)
// For GLM-4.7 models with thinking support, add thinking parameter
if (this.glmConfig.isGlmModel && this.glmConfig.supportsThinking) {
const useReasoning = this.options.enableReasoningEffort !== false
;(requestOptions as any).thinking = useReasoning ? { type: "enabled" } : { type: "disabled" }
}
let response
try {
response = await this.client.chat.completions.create(