mirror of
https://github.com/RooVetGit/Roo-Code.git
synced 2026-10-07 02:58:15 +00:00
fix: add GLM model detection to OpenAI Compatible provider
This commit adds GLM model detection and related optimizations to the
OpenAI Compatible provider (OpenAiHandler), which handles the "OpenAI
Compatible" option in the UI when users set a custom base URL.
Changes:
- Import GLM detection utilities and Z.ai format converter
- Add glmConfig property to track GLM model configuration
- Detect GLM model on construction when model ID is available
- Re-detect GLM model in createMessage if model ID changes
- For GLM models:
- Use convertToZAiFormat with mergeToolResultText to prevent
conversation flow disruption
- Disable parallel_tool_calls as GLM models may not support it
- Add thinking parameter for GLM-4.7 models
- Add console logging for detection results and applied optimizations
This addresses the feedback in issue #11071 where the user clarified
they are using the "OpenAI Compatible" provider, not the "OpenAI"
provider. The previous changes in this PR only affected the
base-openai-compatible-provider.ts (used by Z.ai, Groq, etc.) and
lm-studio.ts, but not openai.ts which handles OpenAI Compatible.
This commit is contained in:
parent
96b68450b2
commit
9e96578df4
1 changed files with 71 additions and 6 deletions
|
|
@ -15,6 +15,7 @@ import type { ApiHandlerOptions } from "../../shared/api"
|
|||
import { TagMatcher } from "../../utils/tag-matcher"
|
||||
|
||||
import { convertToOpenAiMessages } from "../transform/openai-format"
|
||||
import { convertToZAiFormat } from "../transform/zai-format"
|
||||
import { convertToR1Format } from "../transform/r1-format"
|
||||
import { ApiStream, ApiStreamUsageChunk } from "../transform/stream"
|
||||
import { getModelParams } from "../transform/model-params"
|
||||
|
|
@ -24,6 +25,7 @@ import { BaseProvider } from "./base-provider"
|
|||
import type { SingleCompletionHandler, ApiHandlerCreateMessageMetadata } from "../index"
|
||||
import { getApiRequestTimeout } from "./utils/timeout-config"
|
||||
import { handleOpenAIError } from "./utils/openai-error-handler"
|
||||
import { detectGlmModel, logGlmDetection, type GlmModelConfig } from "./utils/glm-model-detection"
|
||||
|
||||
// TODO: Rename this to OpenAICompatibleHandler. Also, I think the
|
||||
// `OpenAINativeHandler` can subclass from this, since it's obviously
|
||||
|
|
@ -31,7 +33,8 @@ import { handleOpenAIError } from "./utils/openai-error-handler"
|
|||
export class OpenAiHandler extends BaseProvider implements SingleCompletionHandler {
|
||||
protected options: ApiHandlerOptions
|
||||
protected client: OpenAI
|
||||
private readonly providerName = "OpenAI"
|
||||
private readonly providerName = "OpenAI Compatible"
|
||||
private glmConfig: GlmModelConfig | null = null
|
||||
|
||||
constructor(options: ApiHandlerOptions) {
|
||||
super()
|
||||
|
|
@ -77,6 +80,13 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl
|
|||
timeout,
|
||||
})
|
||||
}
|
||||
|
||||
// Detect GLM model on construction if model ID is available
|
||||
const modelId = this.options.openAiModelId || ""
|
||||
if (modelId) {
|
||||
this.glmConfig = detectGlmModel(modelId)
|
||||
logGlmDetection(this.providerName, modelId, this.glmConfig)
|
||||
}
|
||||
}
|
||||
|
||||
override async *createMessage(
|
||||
|
|
@ -91,6 +101,12 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl
|
|||
const isAzureAiInference = this._isAzureAiInference(modelUrl)
|
||||
const deepseekReasoner = modelId.includes("deepseek-reasoner") || enabledR1Format
|
||||
|
||||
// Re-detect GLM model if not already done or if model ID changed
|
||||
if (!this.glmConfig || this.glmConfig.originalModelId !== modelId) {
|
||||
this.glmConfig = detectGlmModel(modelId)
|
||||
logGlmDetection(this.providerName, modelId, this.glmConfig)
|
||||
}
|
||||
|
||||
if (modelId.includes("o1") || modelId.includes("o3") || modelId.includes("o4")) {
|
||||
yield* this.handleO3FamilyMessage(modelId, systemPrompt, messages, metadata)
|
||||
return
|
||||
|
|
@ -106,6 +122,12 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl
|
|||
|
||||
if (deepseekReasoner) {
|
||||
convertedMessages = convertToR1Format([{ role: "user", content: systemPrompt }, ...messages])
|
||||
} else if (this.glmConfig.isGlmModel) {
|
||||
// GLM models benefit from mergeToolResultText to prevent reasoning_content loss
|
||||
const glmConvertedMessages = convertToZAiFormat(messages, {
|
||||
mergeToolResultText: this.glmConfig.mergeToolResultText,
|
||||
})
|
||||
convertedMessages = [systemMessage, ...glmConvertedMessages]
|
||||
} else {
|
||||
if (modelInfo.supportsPromptCache) {
|
||||
systemMessage = {
|
||||
|
|
@ -152,6 +174,16 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl
|
|||
|
||||
const isGrokXAI = this._isGrokXAI(this.options.openAiBaseUrl)
|
||||
|
||||
// Determine parallel_tool_calls setting
|
||||
// Disable for GLM models as they may not support it properly
|
||||
let parallelToolCalls: boolean
|
||||
if (this.glmConfig.isGlmModel && this.glmConfig.disableParallelToolCalls) {
|
||||
parallelToolCalls = false
|
||||
console.log(`[${this.providerName}] parallel_tool_calls disabled for GLM model`)
|
||||
} else {
|
||||
parallelToolCalls = metadata?.parallelToolCalls ?? true
|
||||
}
|
||||
|
||||
const requestOptions: OpenAI.Chat.Completions.ChatCompletionCreateParamsStreaming = {
|
||||
model: modelId,
|
||||
temperature: this.options.modelTemperature ?? (deepseekReasoner ? DEEP_SEEK_DEFAULT_TEMPERATURE : 0),
|
||||
|
|
@ -161,12 +193,19 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl
|
|||
...(reasoning && reasoning),
|
||||
tools: this.convertToolsForOpenAI(metadata?.tools),
|
||||
tool_choice: metadata?.tool_choice,
|
||||
parallel_tool_calls: metadata?.parallelToolCalls ?? true,
|
||||
parallel_tool_calls: parallelToolCalls,
|
||||
}
|
||||
|
||||
// Add max_tokens if needed
|
||||
this.addMaxTokensIfNeeded(requestOptions, modelInfo)
|
||||
|
||||
// For GLM-4.7 models with thinking support, add thinking parameter
|
||||
if (this.glmConfig.isGlmModel && this.glmConfig.supportsThinking) {
|
||||
const useReasoning = this.options.enableReasoningEffort !== false // Default to enabled for GLM-4.7
|
||||
;(requestOptions as any).thinking = useReasoning ? { type: "enabled" } : { type: "disabled" }
|
||||
console.log(`[${this.providerName}] GLM-4.7 thinking mode: ${useReasoning ? "enabled" : "disabled"}`)
|
||||
}
|
||||
|
||||
let stream
|
||||
try {
|
||||
stream = await this.client.chat.completions.create(
|
||||
|
|
@ -221,20 +260,46 @@ export class OpenAiHandler extends BaseProvider implements SingleCompletionHandl
|
|||
yield this.processUsageMetrics(lastUsage, modelInfo)
|
||||
}
|
||||
} else {
|
||||
// Determine message conversion based on model type
|
||||
let nonStreamingMessages
|
||||
if (deepseekReasoner) {
|
||||
nonStreamingMessages = convertToR1Format([{ role: "user", content: systemPrompt }, ...messages])
|
||||
} else if (this.glmConfig.isGlmModel) {
|
||||
// GLM models benefit from mergeToolResultText to prevent reasoning_content loss
|
||||
const glmConvertedMessages = convertToZAiFormat(messages, {
|
||||
mergeToolResultText: this.glmConfig.mergeToolResultText,
|
||||
})
|
||||
nonStreamingMessages = [systemMessage, ...glmConvertedMessages]
|
||||
} else {
|
||||
nonStreamingMessages = [systemMessage, ...convertToOpenAiMessages(messages)]
|
||||
}
|
||||
|
||||
// Determine parallel_tool_calls setting for non-streaming
|
||||
let nonStreamingParallelToolCalls: boolean
|
||||
if (this.glmConfig.isGlmModel && this.glmConfig.disableParallelToolCalls) {
|
||||
nonStreamingParallelToolCalls = false
|
||||
} else {
|
||||
nonStreamingParallelToolCalls = metadata?.parallelToolCalls ?? true
|
||||
}
|
||||
|
||||
const requestOptions: OpenAI.Chat.Completions.ChatCompletionCreateParamsNonStreaming = {
|
||||
model: modelId,
|
||||
messages: deepseekReasoner
|
||||
? convertToR1Format([{ role: "user", content: systemPrompt }, ...messages])
|
||||
: [systemMessage, ...convertToOpenAiMessages(messages)],
|
||||
messages: nonStreamingMessages,
|
||||
// Tools are always present (minimum ALWAYS_AVAILABLE_TOOLS)
|
||||
tools: this.convertToolsForOpenAI(metadata?.tools),
|
||||
tool_choice: metadata?.tool_choice,
|
||||
parallel_tool_calls: metadata?.parallelToolCalls ?? true,
|
||||
parallel_tool_calls: nonStreamingParallelToolCalls,
|
||||
}
|
||||
|
||||
// Add max_tokens if needed
|
||||
this.addMaxTokensIfNeeded(requestOptions, modelInfo)
|
||||
|
||||
// For GLM-4.7 models with thinking support, add thinking parameter
|
||||
if (this.glmConfig.isGlmModel && this.glmConfig.supportsThinking) {
|
||||
const useReasoning = this.options.enableReasoningEffort !== false
|
||||
;(requestOptions as any).thinking = useReasoning ? { type: "enabled" } : { type: "disabled" }
|
||||
}
|
||||
|
||||
let response
|
||||
try {
|
||||
response = await this.client.chat.completions.create(
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue