From b5109e932918443584644f0e5911284d872a0767 Mon Sep 17 00:00:00 2001 From: Roo Code Date: Thu, 14 Aug 2025 18:20:39 +0000 Subject: [PATCH] fix: address PR review feedback for OpenAI Native streaming toggle - Added comprehensive test coverage for openAiNativeStreamingEnabled option - Added translations for all 17 supported languages - Refactored code to reduce duplication with helper methods - Improved error handling for non-streaming responses - Added documentation comments for all major methods - Enhanced type safety with proper error handling - Resolved merge conflicts with main branch --- .../providers/__tests__/openai-native.spec.ts | 428 ++++++++++++++ src/api/providers/openai-native.ts | 545 +++++++++++------- webview-ui/src/i18n/locales/ca/settings.json | 1 + webview-ui/src/i18n/locales/de/settings.json | 1 + webview-ui/src/i18n/locales/es/settings.json | 1 + webview-ui/src/i18n/locales/fr/settings.json | 1 + webview-ui/src/i18n/locales/hi/settings.json | 1 + webview-ui/src/i18n/locales/id/settings.json | 1 + webview-ui/src/i18n/locales/it/settings.json | 1 + webview-ui/src/i18n/locales/ja/settings.json | 1 + webview-ui/src/i18n/locales/ko/settings.json | 1 + webview-ui/src/i18n/locales/nl/settings.json | 1 + webview-ui/src/i18n/locales/pl/settings.json | 1 + .../src/i18n/locales/pt-BR/settings.json | 1 + webview-ui/src/i18n/locales/ru/settings.json | 1 + webview-ui/src/i18n/locales/tr/settings.json | 1 + webview-ui/src/i18n/locales/vi/settings.json | 1 + .../src/i18n/locales/zh-CN/settings.json | 1 + .../src/i18n/locales/zh-TW/settings.json | 1 + 19 files changed, 767 insertions(+), 223 deletions(-) diff --git a/src/api/providers/__tests__/openai-native.spec.ts b/src/api/providers/__tests__/openai-native.spec.ts index 0acdb6202e..65169fa55d 100644 --- a/src/api/providers/__tests__/openai-native.spec.ts +++ b/src/api/providers/__tests__/openai-native.spec.ts @@ -1815,4 +1815,432 @@ describe("GPT-5 streaming event coverage (additional)", () => { delete (global as any).fetch }) }) + + describe("openAiNativeStreamingEnabled option", () => { + let handler: OpenAiNativeHandler + const systemPrompt = "You are a helpful assistant." + const messages: Anthropic.Messages.MessageParam[] = [ + { + role: "user", + content: "Hello!", + }, + ] + + beforeEach(() => { + mockCreate.mockClear() + }) + + it("should use streaming by default when openAiNativeStreamingEnabled is not specified", async () => { + handler = new OpenAiNativeHandler({ + apiModelId: "gpt-4.1", + openAiNativeApiKey: "test-api-key", + // openAiNativeStreamingEnabled not specified, should default to true + }) + + const stream = handler.createMessage(systemPrompt, messages) + for await (const chunk of stream) { + // consume stream + } + + expect(mockCreate).toHaveBeenCalledWith( + expect.objectContaining({ + stream: true, + stream_options: { include_usage: true }, + }), + ) + }) + + it("should use streaming when openAiNativeStreamingEnabled is true", async () => { + handler = new OpenAiNativeHandler({ + apiModelId: "gpt-4.1", + openAiNativeApiKey: "test-api-key", + openAiNativeStreamingEnabled: true, + }) + + const stream = handler.createMessage(systemPrompt, messages) + for await (const chunk of stream) { + // consume stream + } + + expect(mockCreate).toHaveBeenCalledWith( + expect.objectContaining({ + stream: true, + stream_options: { include_usage: true }, + }), + ) + }) + + it("should disable streaming when openAiNativeStreamingEnabled is false for standard models", async () => { + handler = new OpenAiNativeHandler({ + apiModelId: "gpt-4.1", + openAiNativeApiKey: "test-api-key", + openAiNativeStreamingEnabled: false, + }) + + mockCreate.mockResolvedValueOnce({ + id: "test-completion", + choices: [ + { + message: { role: "assistant", content: "Non-streaming response" }, + finish_reason: "stop", + index: 0, + }, + ], + usage: { + prompt_tokens: 10, + completion_tokens: 5, + total_tokens: 15, + }, + }) + + const stream = handler.createMessage(systemPrompt, messages) + const chunks: any[] = [] + for await (const chunk of stream) { + chunks.push(chunk) + } + + // Should call create without stream parameter + expect(mockCreate).toHaveBeenCalledWith( + expect.objectContaining({ + model: "gpt-4.1", + temperature: 0, + messages: [ + { role: "system", content: systemPrompt }, + { role: "user", content: "Hello!" }, + ], + }), + ) + expect(mockCreate).toHaveBeenCalledWith( + expect.not.objectContaining({ + stream: true, + }), + ) + + // Should yield text and usage from non-streaming response + expect(chunks).toHaveLength(2) + expect(chunks[0]).toMatchObject({ type: "text", text: "Non-streaming response" }) + expect(chunks[1]).toMatchObject({ + type: "usage", + inputTokens: 10, + outputTokens: 5, + }) + }) + + it("should disable streaming for O1 models when openAiNativeStreamingEnabled is false", async () => { + handler = new OpenAiNativeHandler({ + apiModelId: "o1", + openAiNativeApiKey: "test-api-key", + openAiNativeStreamingEnabled: false, + }) + + mockCreate.mockResolvedValueOnce({ + id: "test-completion", + choices: [ + { + message: { role: "assistant", content: "O1 non-streaming response" }, + finish_reason: "stop", + index: 0, + }, + ], + usage: { + prompt_tokens: 20, + completion_tokens: 10, + total_tokens: 30, + }, + }) + + const stream = handler.createMessage(systemPrompt, messages) + const chunks: any[] = [] + for await (const chunk of stream) { + chunks.push(chunk) + } + + // Should not include stream parameter for O1 models + expect(mockCreate).toHaveBeenCalledWith( + expect.objectContaining({ + model: "o1", + messages: [ + { role: "developer", content: "Formatting re-enabled\n" + systemPrompt }, + { role: "user", content: "Hello!" }, + ], + }), + ) + expect(mockCreate).toHaveBeenCalledWith( + expect.not.objectContaining({ + stream: true, + }), + ) + + // Should yield text and usage + expect(chunks).toHaveLength(2) + expect(chunks[0]).toMatchObject({ type: "text", text: "O1 non-streaming response" }) + expect(chunks[1]).toMatchObject({ + type: "usage", + inputTokens: 20, + outputTokens: 10, + }) + }) + + it("should disable streaming for O3-mini models when openAiNativeStreamingEnabled is false", async () => { + handler = new OpenAiNativeHandler({ + apiModelId: "o3-mini", + openAiNativeApiKey: "test-api-key", + openAiNativeStreamingEnabled: false, + }) + + mockCreate.mockResolvedValueOnce({ + id: "test-completion", + choices: [ + { + message: { role: "assistant", content: "O3-mini non-streaming response" }, + finish_reason: "stop", + index: 0, + }, + ], + usage: { + prompt_tokens: 15, + completion_tokens: 8, + total_tokens: 23, + }, + }) + + const stream = handler.createMessage(systemPrompt, messages) + const chunks: any[] = [] + for await (const chunk of stream) { + chunks.push(chunk) + } + + // Should include reasoning_effort but not stream + expect(mockCreate).toHaveBeenCalledWith( + expect.objectContaining({ + model: "o3-mini", + reasoning_effort: "medium", + messages: [ + { role: "developer", content: "Formatting re-enabled\n" + systemPrompt }, + { role: "user", content: "Hello!" }, + ], + }), + ) + expect(mockCreate).toHaveBeenCalledWith( + expect.not.objectContaining({ + stream: true, + }), + ) + + // Should yield text and usage + expect(chunks).toHaveLength(2) + expect(chunks[0]).toMatchObject({ type: "text", text: "O3-mini non-streaming response" }) + expect(chunks[1]).toMatchObject({ + type: "usage", + inputTokens: 15, + outputTokens: 8, + }) + }) + + it("should disable streaming for GPT-5 models when openAiNativeStreamingEnabled is false", async () => { + // Mock fetch for non-streaming Responses API + const mockFetch = vitest.fn().mockResolvedValue({ + ok: true, + json: async () => ({ + response: { + id: "resp_001", + output: [ + { + type: "text", + content: [{ type: "text", text: "GPT-5 non-streaming response" }], + }, + ], + usage: { + prompt_tokens: 30, + completion_tokens: 15, + }, + }, + }), + }) + global.fetch = mockFetch as any + + handler = new OpenAiNativeHandler({ + apiModelId: "gpt-5-2025-08-07", + openAiNativeApiKey: "test-api-key", + openAiNativeStreamingEnabled: false, + }) + + const stream = handler.createMessage(systemPrompt, messages) + const chunks: any[] = [] + for await (const chunk of stream) { + chunks.push(chunk) + } + + // Should call Responses API without stream parameter + expect(mockFetch).toHaveBeenCalledWith( + "https://api.openai.com/v1/responses", + expect.objectContaining({ + method: "POST", + headers: expect.objectContaining({ + "Content-Type": "application/json", + Authorization: "Bearer test-api-key", + // No Accept: text/event-stream header for non-streaming + }), + body: expect.any(String), + }), + ) + + const requestBody = JSON.parse(mockFetch.mock.calls[0][1].body) + expect(requestBody.stream).toBe(false) + expect(requestBody.model).toBe("gpt-5-2025-08-07") + + // Should yield text and usage + expect(chunks).toHaveLength(2) + expect(chunks[0]).toMatchObject({ type: "text", text: "GPT-5 non-streaming response" }) + expect(chunks[1]).toMatchObject({ + type: "usage", + inputTokens: 30, + outputTokens: 15, + }) + + // Clean up + delete (global as any).fetch + }) + + it("should handle cache tokens in non-streaming response", async () => { + handler = new OpenAiNativeHandler({ + apiModelId: "gpt-4.1", + openAiNativeApiKey: "test-api-key", + openAiNativeStreamingEnabled: false, + }) + + mockCreate.mockResolvedValueOnce({ + id: "test-completion", + choices: [ + { + message: { role: "assistant", content: "Cached response" }, + finish_reason: "stop", + index: 0, + }, + ], + usage: { + prompt_tokens: 100, + completion_tokens: 10, + prompt_tokens_details: { + cached_tokens: 80, + }, + cache_creation_input_tokens: 20, + }, + }) + + const stream = handler.createMessage(systemPrompt, messages) + const chunks: any[] = [] + for await (const chunk of stream) { + chunks.push(chunk) + } + + // Should include cache tokens in usage + const usageChunk = chunks.find((c) => c.type === "usage") + expect(usageChunk).toMatchObject({ + type: "usage", + inputTokens: 100, + outputTokens: 10, + cacheReadTokens: 80, + cacheWriteTokens: 20, + }) + }) + + it("should handle error in non-streaming response", async () => { + handler = new OpenAiNativeHandler({ + apiModelId: "gpt-4.1", + openAiNativeApiKey: "test-api-key", + openAiNativeStreamingEnabled: false, + }) + + mockCreate.mockRejectedValueOnce(new Error("API Error")) + + const stream = handler.createMessage(systemPrompt, messages) + await expect(async () => { + for await (const chunk of stream) { + // Should throw before yielding + } + }).rejects.toThrow("API Error") + }) + + it("should handle empty response in non-streaming mode", async () => { + handler = new OpenAiNativeHandler({ + apiModelId: "gpt-4.1", + openAiNativeApiKey: "test-api-key", + openAiNativeStreamingEnabled: false, + }) + + mockCreate.mockResolvedValueOnce({ + id: "test-completion", + choices: [ + { + message: { role: "assistant", content: "" }, + finish_reason: "stop", + index: 0, + }, + ], + usage: { + prompt_tokens: 10, + completion_tokens: 0, + }, + }) + + const stream = handler.createMessage(systemPrompt, messages) + const chunks: any[] = [] + for await (const chunk of stream) { + chunks.push(chunk) + } + + // Should handle empty content gracefully + expect(chunks).toHaveLength(1) + expect(chunks[0]).toMatchObject({ + type: "usage", + inputTokens: 10, + outputTokens: 0, + }) + }) + + it("should handle verbosity parameter correctly in non-streaming mode for GPT-5", async () => { + // Mock fetch for non-streaming Responses API + const mockFetch = vitest.fn().mockResolvedValue({ + ok: true, + json: async () => ({ + response: { + id: "resp_001", + output: [ + { + type: "text", + content: [{ type: "text", text: "High verbosity response" }], + }, + ], + usage: { + prompt_tokens: 25, + completion_tokens: 12, + }, + }, + }), + }) + global.fetch = mockFetch as any + + handler = new OpenAiNativeHandler({ + apiModelId: "gpt-5-2025-08-07", + openAiNativeApiKey: "test-api-key", + openAiNativeStreamingEnabled: false, + verbosity: "high", + }) + + const stream = handler.createMessage(systemPrompt, messages) + const chunks: any[] = [] + for await (const chunk of stream) { + chunks.push(chunk) + } + + // Should include verbosity in request + const requestBody = JSON.parse(mockFetch.mock.calls[0][1].body) + expect(requestBody.verbosity).toBe("high") + expect(requestBody.stream).toBe(false) + + // Clean up + delete (global as any).fetch + }) + }) }) diff --git a/src/api/providers/openai-native.ts b/src/api/providers/openai-native.ts index c05277ac5d..1c41acd391 100644 --- a/src/api/providers/openai-native.ts +++ b/src/api/providers/openai-native.ts @@ -97,6 +97,15 @@ export class OpenAiNativeHandler extends BaseProvider implements SingleCompletio } } + /** + * Creates a message stream for the OpenAI Native API. + * Routes to the appropriate handler based on the model type. + * + * @param systemPrompt - The system prompt to use + * @param messages - The conversation messages + * @param metadata - Optional metadata for conversation continuity + * @yields API stream chunks including text, reasoning, and usage data + */ override async *createMessage( systemPrompt: string, messages: Anthropic.Messages.MessageParam[], @@ -125,188 +134,307 @@ export class OpenAiNativeHandler extends BaseProvider implements SingleCompletio } } - private async *handleO1FamilyMessage( + /** + * Helper method to check if streaming is enabled. + * Defaults to true for backward compatibility. + */ + private isStreamingEnabled(): boolean { + return this.options.openAiNativeStreamingEnabled ?? true + } + + /** + * Helper method to handle non-streaming responses uniformly. + * Extracts text content and usage data from the response. + * + * @param response - The OpenAI completion response + * @param model - The model information + * @yields Text and usage chunks + */ + private async *handleNonStreamingResponse( + response: OpenAI.Chat.Completions.ChatCompletion, model: OpenAiNativeModel, - systemPrompt: string, - messages: Anthropic.Messages.MessageParam[], ): ApiStream { - // o1 supports developer prompt with formatting - // o1-preview and o1-mini only support user messages - const isOriginalO1 = model.id === "o1" - const { reasoning } = this.getModel() - const streamingEnabled = this.options.openAiNativeStreamingEnabled ?? true - - if (streamingEnabled) { - const response = await this.client.chat.completions.create({ - model: model.id, - messages: [ - { - role: isOriginalO1 ? "developer" : "user", - content: isOriginalO1 ? `Formatting re-enabled\n${systemPrompt}` : systemPrompt, - }, - ...convertToOpenAiMessages(messages), - ], - stream: true, - stream_options: { include_usage: true }, - ...(reasoning && reasoning), - }) - - yield* this.handleStreamResponse(response, model) - } else { - // Non-streaming request - const response = await this.client.chat.completions.create({ - model: model.id, - messages: [ - { - role: isOriginalO1 ? "developer" : "user", - content: isOriginalO1 ? `Formatting re-enabled\n${systemPrompt}` : systemPrompt, - }, - ...convertToOpenAiMessages(messages), - ], - ...(reasoning && reasoning), - }) - - yield { - type: "text", - text: response.choices[0]?.message.content || "", + try { + const content = response.choices[0]?.message?.content + if (content) { + yield { + type: "text", + text: content, + } } if (response.usage) { yield* this.yieldUsage(model.info, response.usage) } + } catch (error) { + throw new Error( + `Error processing non-streaming response: ${error instanceof Error ? error.message : "Unknown error"}`, + ) } } + /** + * Handles message creation for O1 family models (o1, o1-preview, o1-mini). + * These models have special requirements for system prompts and don't support temperature. + * + * @param model - The model information + * @param systemPrompt - The system prompt + * @param messages - The conversation messages + * @yields API stream chunks + */ + private async *handleO1FamilyMessage( + model: OpenAiNativeModel, + systemPrompt: string, + messages: Anthropic.Messages.MessageParam[], + ): ApiStream { + try { + // o1 supports developer prompt with formatting + // o1-preview and o1-mini only support user messages + const isOriginalO1 = model.id === "o1" + const { reasoning } = this.getModel() + + const formattedMessages = [ + { + role: isOriginalO1 ? "developer" : "user", + content: isOriginalO1 ? `Formatting re-enabled\n${systemPrompt}` : systemPrompt, + } as any, + ...convertToOpenAiMessages(messages), + ] + + if (this.isStreamingEnabled()) { + const response = await this.client.chat.completions.create({ + model: model.id, + messages: formattedMessages, + stream: true, + stream_options: { include_usage: true }, + ...(reasoning && reasoning), + }) + yield* this.handleStreamResponse(response, model) + } else { + const response = await this.client.chat.completions.create({ + model: model.id, + messages: formattedMessages, + ...(reasoning && reasoning), + }) + yield* this.handleNonStreamingResponse(response, model) + } + } catch (error) { + throw new Error(`O1 family model error: ${error instanceof Error ? error.message : "Unknown error"}`) + } + } + + /** + * Handles message creation for O3/O4 reasoner models. + * These models use the developer role and support reasoning effort parameters. + * + * @param model - The model information + * @param family - The model family identifier + * @param systemPrompt - The system prompt + * @param messages - The conversation messages + * @yields API stream chunks + */ private async *handleReasonerMessage( model: OpenAiNativeModel, family: "o3-mini" | "o3" | "o4-mini", systemPrompt: string, messages: Anthropic.Messages.MessageParam[], ): ApiStream { - const { reasoning } = this.getModel() - const streamingEnabled = this.options.openAiNativeStreamingEnabled ?? true + try { + const { reasoning } = this.getModel() - if (streamingEnabled) { - const stream = await this.client.chat.completions.create({ - model: family, - messages: [ - { - role: "developer", - content: `Formatting re-enabled\n${systemPrompt}`, - }, - ...convertToOpenAiMessages(messages), - ], - stream: true, - stream_options: { include_usage: true }, - ...(reasoning && reasoning), - }) + const formattedMessages = [ + { + role: "developer", + content: `Formatting re-enabled\n${systemPrompt}`, + } as any, + ...convertToOpenAiMessages(messages), + ] - yield* this.handleStreamResponse(stream, model) - } else { - // Non-streaming request - const response = await this.client.chat.completions.create({ - model: family, - messages: [ - { - role: "developer", - content: `Formatting re-enabled\n${systemPrompt}`, - }, - ...convertToOpenAiMessages(messages), - ], - ...(reasoning && reasoning), - }) - - yield { - type: "text", - text: response.choices[0]?.message.content || "", - } - - if (response.usage) { - yield* this.yieldUsage(model.info, response.usage) + if (this.isStreamingEnabled()) { + const stream = await this.client.chat.completions.create({ + model: family, + messages: formattedMessages, + stream: true, + stream_options: { include_usage: true }, + ...(reasoning && reasoning), + }) + yield* this.handleStreamResponse(stream, model) + } else { + const response = await this.client.chat.completions.create({ + model: family, + messages: formattedMessages, + ...(reasoning && reasoning), + }) + yield* this.handleNonStreamingResponse(response, model) } + } catch (error) { + throw new Error(`Reasoner model error: ${error instanceof Error ? error.message : "Unknown error"}`) } } + /** + * Handles message creation for standard OpenAI models (GPT-4, GPT-3.5, etc). + * Supports both streaming and non-streaming modes with full parameter support. + * + * @param model - The model information + * @param systemPrompt - The system prompt + * @param messages - The conversation messages + * @yields API stream chunks + */ private async *handleDefaultModelMessage( model: OpenAiNativeModel, systemPrompt: string, messages: Anthropic.Messages.MessageParam[], ): ApiStream { - const { reasoning, verbosity } = this.getModel() - const streamingEnabled = this.options.openAiNativeStreamingEnabled ?? true + try { + const { reasoning, verbosity } = this.getModel() - if (streamingEnabled) { - // Prepare the request parameters for streaming - const params: any = { + const formattedMessages = [ + { role: "system", content: systemPrompt } as any, + ...convertToOpenAiMessages(messages), + ] + + // Build base parameters + const baseParams: any = { model: model.id, temperature: this.options.modelTemperature ?? OPENAI_NATIVE_DEFAULT_TEMPERATURE, - messages: [{ role: "system", content: systemPrompt }, ...convertToOpenAiMessages(messages)], - stream: true, - stream_options: { include_usage: true }, + messages: formattedMessages, ...(reasoning && reasoning), } // Add verbosity only if the model supports it if (verbosity && model.info.supportsVerbosity) { - params.verbosity = verbosity + baseParams.verbosity = verbosity } - const stream = await this.client.chat.completions.create(params) + if (this.isStreamingEnabled()) { + const params = { + ...baseParams, + stream: true, + stream_options: { include_usage: true }, + } - if (typeof (stream as any)[Symbol.asyncIterator] !== "function") { - throw new Error( - "OpenAI SDK did not return an AsyncIterable for streaming response. Please check SDK version and usage.", + const stream = await this.client.chat.completions.create(params) + + if (typeof (stream as any)[Symbol.asyncIterator] !== "function") { + throw new Error( + "OpenAI SDK did not return an AsyncIterable for streaming response. Please check SDK version and usage.", + ) + } + + yield* this.handleStreamResponse( + stream as unknown as AsyncIterable, + model, ) + } else { + const response = await this.client.chat.completions.create(baseParams) + yield* this.handleNonStreamingResponse(response, model) } - - yield* this.handleStreamResponse( - stream as unknown as AsyncIterable, - model, - ) - } else { - // Non-streaming request - const params: any = { - model: model.id, - temperature: this.options.modelTemperature ?? OPENAI_NATIVE_DEFAULT_TEMPERATURE, - messages: [{ role: "system", content: systemPrompt }, ...convertToOpenAiMessages(messages)], - ...(reasoning && reasoning), - } - - // Add verbosity only if the model supports it - if (verbosity && model.info.supportsVerbosity) { - params.verbosity = verbosity - } - - const response = await this.client.chat.completions.create(params) - - yield { - type: "text", - text: response.choices[0]?.message.content || "", - } - - if (response.usage) { - yield* this.yieldUsage(model.info, response.usage) - } + } catch (error) { + throw new Error(`Default model error: ${error instanceof Error ? error.message : "Unknown error"}`) } } + /** + * Handles message creation for models using the Responses API (GPT-5 and Codex Mini). + * Supports conversation continuity via previous_response_id and both streaming/non-streaming modes. + * Automatically falls back to SSE if SDK fails. + * + * @param model - The model information + * @param systemPrompt - The system prompt + * @param messages - The conversation messages + * @param metadata - Optional metadata for conversation continuity + * @yields API stream chunks + */ private async *handleResponsesApiMessage( model: OpenAiNativeModel, systemPrompt: string, messages: Anthropic.Messages.MessageParam[], metadata?: ApiHandlerCreateMessageMetadata, ): ApiStream { - // Prefer the official SDK Responses API with streaming; fall back to fetch-based SSE if needed. - const { verbosity } = this.getModel() - const streamingEnabled = this.options.openAiNativeStreamingEnabled ?? true + try { + // Prepare the request body for the Responses API + const requestBody = await this.prepareResponsesApiRequest(model, systemPrompt, messages, metadata) - // Both GPT-5 and Codex Mini use the same v1/responses endpoint format + // Check if streaming is enabled + if (!this.isStreamingEnabled()) { + requestBody.stream = false + yield* this.makeGpt5ResponsesAPIRequest(requestBody, model, metadata) + return + } - // Resolve reasoning effort (supports "minimal" for GPT‑5) + // Try SDK first, fall back to SSE on error + yield* this.tryResponsesApiWithFallback(requestBody, model, metadata) + } catch (error) { + throw new Error(`Responses API error: ${error instanceof Error ? error.message : "Unknown error"}`) + } + } + + /** + * Prepares the request body for the Responses API. + * Handles conversation continuity and all required parameters. + * + * @private + */ + private async prepareResponsesApiRequest( + model: OpenAiNativeModel, + systemPrompt: string, + messages: Anthropic.Messages.MessageParam[], + metadata?: ApiHandlerCreateMessageMetadata, + ): Promise { + const { verbosity } = model const reasoningEffort = this.getGpt5ReasoningEffort(model) - // Wait for any pending response ID from a previous request to be available - // This handles the race condition with fast nano model responses + // Handle conversation continuity + const effectivePreviousResponseId = await this.resolvePreviousResponseId(metadata) + + // Format input and capture continuity id + const { formattedInput, previousResponseId } = this.prepareGpt5Input(systemPrompt, messages, metadata) + const requestPreviousResponseId = effectivePreviousResponseId ?? previousResponseId + + // Create a new promise for this request's response ID + this.responseIdPromise = new Promise((resolve) => { + this.responseIdResolver = resolve + }) + + // Build the request body + interface Gpt5RequestBody { + model: string + input: string + stream: boolean + reasoning?: { effort: ReasoningEffortWithMinimal; summary?: "auto" } + text?: { verbosity: VerbosityLevel } + temperature?: number + max_output_tokens?: number + previous_response_id?: string + } + + const requestBody: Gpt5RequestBody = { + model: model.id, + input: formattedInput, + stream: true, + ...(reasoningEffort && { + reasoning: { + effort: reasoningEffort, + ...(this.options.enableGpt5ReasoningSummary ? { summary: "auto" as const } : {}), + }, + }), + text: { verbosity: (verbosity || "medium") as VerbosityLevel }, + temperature: this.options.modelTemperature ?? GPT5_DEFAULT_TEMPERATURE, + ...(model.maxTokens ? { max_output_tokens: model.maxTokens } : {}), + ...(requestPreviousResponseId && { previous_response_id: requestPreviousResponseId }), + } + + return requestBody + } + + /** + * Resolves the previous response ID for conversation continuity. + * Handles race conditions with pending response IDs. + * + * @private + */ + private async resolvePreviousResponseId(metadata?: ApiHandlerCreateMessageMetadata): Promise { let effectivePreviousResponseId = metadata?.previousResponseId // Only allow fallback to pending/last response id when not explicitly suppressed @@ -333,63 +461,20 @@ export class OpenAiNativeHandler extends BaseProvider implements SingleCompletio } } - // Format input and capture continuity id - const { formattedInput, previousResponseId } = this.prepareGpt5Input(systemPrompt, messages, metadata) - const requestPreviousResponseId = effectivePreviousResponseId ?? previousResponseId - - // Create a new promise for this request's response ID - this.responseIdPromise = new Promise((resolve) => { - this.responseIdResolver = resolve - }) - - // Build a request body (also used for fallback) - // Ensure we explicitly pass max_output_tokens for GPT‑5 based on Roo's reserved model response calculation - // so requests do not default to very large limits (e.g., 120k). - interface Gpt5RequestBody { - model: string - input: string - stream: boolean - reasoning?: { effort: ReasoningEffortWithMinimal; summary?: "auto" } - text?: { verbosity: VerbosityLevel } - temperature?: number - max_output_tokens?: number - previous_response_id?: string - } - - const requestBody: Gpt5RequestBody = { - model: model.id, - input: formattedInput, - stream: true, - ...(reasoningEffort && { - reasoning: { - effort: reasoningEffort, - ...(this.options.enableGpt5ReasoningSummary ? { summary: "auto" as const } : {}), - }, - }), - text: { verbosity: (verbosity || "medium") as VerbosityLevel }, - temperature: this.options.modelTemperature ?? GPT5_DEFAULT_TEMPERATURE, - // Explicitly include the calculated max output tokens for GPT‑5. - // Use the per-request reserved output computed by Roo (params.maxTokens from getModelParams). - ...(model.maxTokens ? { max_output_tokens: model.maxTokens } : {}), - ...(requestPreviousResponseId && { previous_response_id: requestPreviousResponseId }), - } - - // Check if streaming is enabled - if (!streamingEnabled) { - // For non-streaming, we need to modify the request body - requestBody.stream = false - - // Make non-streaming request using the makeGpt5ResponsesAPIRequest method - // Note: The method signature expects the requestBody, not params - const responseIterator = this.makeGpt5ResponsesAPIRequest(requestBody, model, metadata) - - // Process the non-streaming response - for await (const chunk of responseIterator) { - yield chunk - } - return - } + return effectivePreviousResponseId + } + /** + * Tries to use the SDK for Responses API, with fallback to SSE on error. + * Handles previous_response_id errors with automatic retry. + * + * @private + */ + private async *tryResponsesApiWithFallback( + requestBody: any, + model: OpenAiNativeModel, + metadata?: ApiHandlerCreateMessageMetadata, + ): ApiStream { try { // Use the official SDK for streaming const stream = (await (this.client as any).responses.create(requestBody)) as AsyncIterable @@ -406,52 +491,66 @@ export class OpenAiNativeHandler extends BaseProvider implements SingleCompletio } } } catch (sdkErr: any) { - // Check if this is a 400 error about previous_response_id not found - const errorMessage = sdkErr?.message || sdkErr?.error?.message || "" - const is400Error = sdkErr?.status === 400 || sdkErr?.response?.status === 400 - const isPreviousResponseError = - errorMessage.includes("Previous response") || errorMessage.includes("not found") + // Handle previous_response_id errors with retry + if (this.isPreviousResponseIdError(sdkErr) && requestBody.previous_response_id) { + yield* this.retryWithoutPreviousResponseId(requestBody, model, metadata) + } else { + // For other errors, fallback to manual SSE via fetch + yield* this.makeGpt5ResponsesAPIRequest(requestBody, model, metadata) + } + } + } - if (is400Error && requestBody.previous_response_id && isPreviousResponseError) { - // Log the error and retry without the previous_response_id - console.warn( - `[GPT-5] Previous response ID not found (${requestBody.previous_response_id}), retrying without it`, - ) + /** + * Checks if an error is related to previous_response_id not found. + * + * @private + */ + private isPreviousResponseIdError(error: any): boolean { + const errorMessage = error?.message || error?.error?.message || "" + const is400Error = error?.status === 400 || error?.response?.status === 400 + return is400Error && (errorMessage.includes("Previous response") || errorMessage.includes("not found")) + } - // Remove the problematic previous_response_id and retry - const retryRequestBody = { ...requestBody } - delete retryRequestBody.previous_response_id + /** + * Retries the request without the previous_response_id after an error. + * + * @private + */ + private async *retryWithoutPreviousResponseId( + requestBody: any, + model: OpenAiNativeModel, + metadata?: ApiHandlerCreateMessageMetadata, + ): ApiStream { + console.warn( + `[GPT-5] Previous response ID not found (${requestBody.previous_response_id}), retrying without it`, + ) - // Clear the stored lastResponseId to prevent using it again - this.lastResponseId = undefined + // Remove the problematic previous_response_id and retry + const retryRequestBody = { ...requestBody } + delete retryRequestBody.previous_response_id - try { - // Retry with the SDK - const retryStream = (await (this.client as any).responses.create( - retryRequestBody, - )) as AsyncIterable + // Clear the stored lastResponseId to prevent using it again + this.lastResponseId = undefined - if (typeof (retryStream as any)[Symbol.asyncIterator] !== "function") { - // If SDK fails, fall back to SSE - yield* this.makeGpt5ResponsesAPIRequest(retryRequestBody, model, metadata) - return - } + try { + // Retry with the SDK + const retryStream = (await (this.client as any).responses.create(retryRequestBody)) as AsyncIterable - for await (const event of retryStream) { - for await (const outChunk of this.processGpt5Event(event, model)) { - yield outChunk - } - } - return - } catch (retryErr) { - // If retry also fails, fall back to SSE - yield* this.makeGpt5ResponsesAPIRequest(retryRequestBody, model, metadata) - return - } + if (typeof (retryStream as any)[Symbol.asyncIterator] !== "function") { + // If SDK fails, fall back to SSE + yield* this.makeGpt5ResponsesAPIRequest(retryRequestBody, model, metadata) + return } - // For other errors, fallback to manual SSE via fetch - yield* this.makeGpt5ResponsesAPIRequest(requestBody, model, metadata) + for await (const event of retryStream) { + for await (const outChunk of this.processGpt5Event(event, model)) { + yield outChunk + } + } + } catch (retryErr) { + // If retry also fails, fall back to SSE + yield* this.makeGpt5ResponsesAPIRequest(retryRequestBody, model, metadata) } } diff --git a/webview-ui/src/i18n/locales/ca/settings.json b/webview-ui/src/i18n/locales/ca/settings.json index 5964798684..bfc2a16135 100644 --- a/webview-ui/src/i18n/locales/ca/settings.json +++ b/webview-ui/src/i18n/locales/ca/settings.json @@ -738,6 +738,7 @@ "cacheReadsPrice": "Preu de lectures de caché", "cacheWritesPrice": "Preu d'escriptures de caché", "enableStreaming": "Habilitar streaming", + "enableStreamingDescription": "Desactiva el streaming si trobes errors de verificació d'organització amb models avançats. Les sol·licituds sense streaming poden funcionar sense verificació.", "enableR1Format": "Activar els paràmetres del model R1", "enableR1FormatTips": "S'ha d'activat quan s'utilitzen models R1 com el QWQ per evitar errors 400", "useAzure": "Utilitzar Azure", diff --git a/webview-ui/src/i18n/locales/de/settings.json b/webview-ui/src/i18n/locales/de/settings.json index 28f3e847fa..15d473a27a 100644 --- a/webview-ui/src/i18n/locales/de/settings.json +++ b/webview-ui/src/i18n/locales/de/settings.json @@ -738,6 +738,7 @@ "cacheReadsPrice": "Cache-Lesepreis", "cacheWritesPrice": "Cache-Schreibpreis", "enableStreaming": "Streaming aktivieren", + "enableStreamingDescription": "Deaktiviere Streaming, wenn du bei erweiterten Modellen auf Organisationsverifizierungsfehler stößt. Nicht-Streaming-Anfragen funktionieren möglicherweise ohne Verifizierung.", "enableR1Format": "R1-Modellparameter aktivieren", "enableR1FormatTips": "Muss bei Verwendung von R1-Modellen wie QWQ aktiviert werden, um 400er-Fehler zu vermeiden", "useAzure": "Azure verwenden", diff --git a/webview-ui/src/i18n/locales/es/settings.json b/webview-ui/src/i18n/locales/es/settings.json index fd6e1e2715..c419b0189b 100644 --- a/webview-ui/src/i18n/locales/es/settings.json +++ b/webview-ui/src/i18n/locales/es/settings.json @@ -738,6 +738,7 @@ "cacheReadsPrice": "Precio de lecturas de caché", "cacheWritesPrice": "Precio de escrituras de caché", "enableStreaming": "Habilitar streaming", + "enableStreamingDescription": "Desactiva el streaming si encuentras errores de verificación de organización con modelos avanzados. Las solicitudes sin streaming pueden funcionar sin verificación.", "enableR1Format": "Habilitar parámetros del modelo R1", "enableR1FormatTips": "Debe habilitarse al utilizar modelos R1 como QWQ, para evitar el error 400", "useAzure": "Usar Azure", diff --git a/webview-ui/src/i18n/locales/fr/settings.json b/webview-ui/src/i18n/locales/fr/settings.json index 451e51d084..e81bc07b6a 100644 --- a/webview-ui/src/i18n/locales/fr/settings.json +++ b/webview-ui/src/i18n/locales/fr/settings.json @@ -738,6 +738,7 @@ "cacheReadsPrice": "Prix des lectures de cache", "cacheWritesPrice": "Prix des écritures de cache", "enableStreaming": "Activer le streaming", + "enableStreamingDescription": "Désactive le streaming si tu rencontres des erreurs de vérification d'organisation avec des modèles avancés. Les requêtes sans streaming peuvent fonctionner sans vérification.", "enableR1Format": "Activer les paramètres du modèle R1", "enableR1FormatTips": "Doit être activé lors de l'utilisation de modèles R1 tels que QWQ, pour éviter l'erreur 400", "useAzure": "Utiliser Azure", diff --git a/webview-ui/src/i18n/locales/hi/settings.json b/webview-ui/src/i18n/locales/hi/settings.json index 83a4b7b81b..5037b15d6e 100644 --- a/webview-ui/src/i18n/locales/hi/settings.json +++ b/webview-ui/src/i18n/locales/hi/settings.json @@ -739,6 +739,7 @@ "cacheReadsPrice": "कैश रीड्स मूल्य", "cacheWritesPrice": "कैश राइट्स मूल्य", "enableStreaming": "स्ट्रीमिंग सक्षम करें", + "enableStreamingDescription": "यदि आपको उन्नत मॉडल के साथ संगठन सत्यापन त्रुटियों का सामना करना पड़ता है तो स्ट्रीमिंग अक्षम करें। गैर-स्ट्रीमिंग अनुरोध सत्यापन के बिना काम कर सकते हैं।", "enableR1Format": "R1 मॉडल पैरामीटर सक्षम करें", "enableR1FormatTips": "QWQ जैसी R1 मॉडलों का उपयोग करते समय इसे सक्षम करना आवश्यक है, ताकि 400 त्रुटि से बचा जा सके", "useAzure": "Azure का उपयोग करें", diff --git a/webview-ui/src/i18n/locales/id/settings.json b/webview-ui/src/i18n/locales/id/settings.json index fd7edf40cf..73b37a20c0 100644 --- a/webview-ui/src/i18n/locales/id/settings.json +++ b/webview-ui/src/i18n/locales/id/settings.json @@ -768,6 +768,7 @@ "cacheReadsPrice": "Harga cache reads", "cacheWritesPrice": "Harga cache writes", "enableStreaming": "Aktifkan streaming", + "enableStreamingDescription": "Nonaktifkan streaming jika kamu mengalami kesalahan verifikasi organisasi dengan model lanjutan. Permintaan non-streaming mungkin berfungsi tanpa verifikasi.", "enableR1Format": "Aktifkan parameter model R1", "enableR1FormatTips": "Harus diaktifkan saat menggunakan model R1 seperti QWQ untuk mencegah error 400", "useAzure": "Gunakan Azure", diff --git a/webview-ui/src/i18n/locales/it/settings.json b/webview-ui/src/i18n/locales/it/settings.json index c8a8800e4f..a32810c736 100644 --- a/webview-ui/src/i18n/locales/it/settings.json +++ b/webview-ui/src/i18n/locales/it/settings.json @@ -739,6 +739,7 @@ "cacheReadsPrice": "Prezzo letture cache", "cacheWritesPrice": "Prezzo scritture cache", "enableStreaming": "Abilita streaming", + "enableStreamingDescription": "Disabilita lo streaming se riscontri errori di verifica dell'organizzazione con modelli avanzati. Le richieste non in streaming potrebbero funzionare senza verifica.", "enableR1Format": "Abilita i parametri del modello R1", "enableR1FormatTips": "Deve essere abilitato quando si utilizzano modelli R1 come QWQ, per evitare l'errore 400", "useAzure": "Usa Azure", diff --git a/webview-ui/src/i18n/locales/ja/settings.json b/webview-ui/src/i18n/locales/ja/settings.json index d3e7b04baf..9fd32aeed1 100644 --- a/webview-ui/src/i18n/locales/ja/settings.json +++ b/webview-ui/src/i18n/locales/ja/settings.json @@ -739,6 +739,7 @@ "cacheReadsPrice": "キャッシュ読み取り価格", "cacheWritesPrice": "キャッシュ書き込み価格", "enableStreaming": "ストリーミングを有効化", + "enableStreamingDescription": "高度なモデルで組織の検証エラーが発生した場合は、ストリーミングを無効にしてください。非ストリーミングリクエストは検証なしで動作する可能性があります。", "enableR1Format": "R1モデルパラメータを有効にする", "enableR1FormatTips": "QWQなどのR1モデルを使用する際には、有効にする必要があります。400エラーを防ぐために", "useAzure": "Azureを使用", diff --git a/webview-ui/src/i18n/locales/ko/settings.json b/webview-ui/src/i18n/locales/ko/settings.json index a5bcd1f385..4d93ce352a 100644 --- a/webview-ui/src/i18n/locales/ko/settings.json +++ b/webview-ui/src/i18n/locales/ko/settings.json @@ -739,6 +739,7 @@ "cacheReadsPrice": "캐시 읽기 가격", "cacheWritesPrice": "캐시 쓰기 가격", "enableStreaming": "스트리밍 활성화", + "enableStreamingDescription": "고급 모델에서 조직 확인 오류가 발생하면 스트리밍을 비활성화하세요. 비스트리밍 요청은 확인 없이 작동할 수 있습니다.", "enableR1Format": "R1 모델 매개변수 활성화", "enableR1FormatTips": "QWQ와 같은 R1 모델을 사용할 때 활성화해야 하며, 400 오류를 방지합니다", "useAzure": "Azure 사용", diff --git a/webview-ui/src/i18n/locales/nl/settings.json b/webview-ui/src/i18n/locales/nl/settings.json index b54e021be0..e88769fb92 100644 --- a/webview-ui/src/i18n/locales/nl/settings.json +++ b/webview-ui/src/i18n/locales/nl/settings.json @@ -739,6 +739,7 @@ "cacheReadsPrice": "Cache-leesprijs", "cacheWritesPrice": "Cache-schrijfprijs", "enableStreaming": "Streaming inschakelen", + "enableStreamingDescription": "Schakel streaming uit als je organisatieverificatiefouten tegenkomt met geavanceerde modellen. Niet-streaming verzoeken kunnen zonder verificatie werken.", "enableR1Format": "R1-modelparameters inschakelen", "enableR1FormatTips": "Moet ingeschakeld zijn bij gebruik van R1-modellen zoals QWQ om 400-fouten te voorkomen", "useAzure": "Azure gebruiken", diff --git a/webview-ui/src/i18n/locales/pl/settings.json b/webview-ui/src/i18n/locales/pl/settings.json index 194bd9029d..322eb98aa6 100644 --- a/webview-ui/src/i18n/locales/pl/settings.json +++ b/webview-ui/src/i18n/locales/pl/settings.json @@ -739,6 +739,7 @@ "cacheReadsPrice": "Cena odczytów bufora", "cacheWritesPrice": "Cena zapisów bufora", "enableStreaming": "Włącz strumieniowanie", + "enableStreamingDescription": "Wyłącz strumieniowanie, jeśli napotkasz błędy weryfikacji organizacji z zaawansowanymi modelami. Żądania bez strumieniowania mogą działać bez weryfikacji.", "enableR1Format": "Włącz parametry modelu R1", "enableR1FormatTips": "Należy włączyć podczas korzystania z modeli R1, takich jak QWQ, aby uniknąć błędu 400", "useAzure": "Użyj Azure", diff --git a/webview-ui/src/i18n/locales/pt-BR/settings.json b/webview-ui/src/i18n/locales/pt-BR/settings.json index 7879dd5154..1096a65310 100644 --- a/webview-ui/src/i18n/locales/pt-BR/settings.json +++ b/webview-ui/src/i18n/locales/pt-BR/settings.json @@ -739,6 +739,7 @@ "cacheReadsPrice": "Preço de leituras de cache", "cacheWritesPrice": "Preço de escritas de cache", "enableStreaming": "Ativar streaming", + "enableStreamingDescription": "Desative o streaming se você encontrar erros de verificação de organização com modelos avançados. Solicitações sem streaming podem funcionar sem verificação.", "enableR1Format": "Ativar parâmetros do modelo R1", "enableR1FormatTips": "Deve ser ativado ao usar modelos R1 como QWQ, para evitar erro 400", "useAzure": "Usar Azure", diff --git a/webview-ui/src/i18n/locales/ru/settings.json b/webview-ui/src/i18n/locales/ru/settings.json index 3744ecedff..3e591135e0 100644 --- a/webview-ui/src/i18n/locales/ru/settings.json +++ b/webview-ui/src/i18n/locales/ru/settings.json @@ -739,6 +739,7 @@ "cacheReadsPrice": "Цена чтения из кэша", "cacheWritesPrice": "Цена записи в кэш", "enableStreaming": "Включить потоковую передачу", + "enableStreamingDescription": "Отключи потоковую передачу, если сталкиваешься с ошибками проверки организации при использовании продвинутых моделей. Запросы без потоковой передачи могут работать без проверки.", "enableR1Format": "Включить параметры модели R1", "enableR1FormatTips": "Необходимо включить при использовании моделей R1 (например, QWQ), чтобы избежать ошибок 400", "useAzure": "Использовать Azure", diff --git a/webview-ui/src/i18n/locales/tr/settings.json b/webview-ui/src/i18n/locales/tr/settings.json index be1a7a0169..940e29bde9 100644 --- a/webview-ui/src/i18n/locales/tr/settings.json +++ b/webview-ui/src/i18n/locales/tr/settings.json @@ -739,6 +739,7 @@ "cacheReadsPrice": "Önbellek okuma fiyatı", "cacheWritesPrice": "Önbellek yazma fiyatı", "enableStreaming": "Akışı etkinleştir", + "enableStreamingDescription": "Gelişmiş modellerle kuruluş doğrulama hatalarıyla karşılaşırsan akışı devre dışı bırak. Akış olmayan istekler doğrulama olmadan çalışabilir.", "enableR1Format": "R1 model parametrelerini etkinleştir", "enableR1FormatTips": "QWQ gibi R1 modelleri kullanıldığında etkinleştirilmelidir, 400 hatası alınmaması için", "useAzure": "Azure kullan", diff --git a/webview-ui/src/i18n/locales/vi/settings.json b/webview-ui/src/i18n/locales/vi/settings.json index 7526cf31a7..754fc93c41 100644 --- a/webview-ui/src/i18n/locales/vi/settings.json +++ b/webview-ui/src/i18n/locales/vi/settings.json @@ -739,6 +739,7 @@ "cacheReadsPrice": "Giá đọc bộ nhớ đệm", "cacheWritesPrice": "Giá ghi bộ nhớ đệm", "enableStreaming": "Bật streaming", + "enableStreamingDescription": "Tắt streaming nếu bạn gặp lỗi xác minh tổ chức với các mô hình nâng cao. Các yêu cầu không streaming có thể hoạt động mà không cần xác minh.", "enableR1Format": "Kích hoạt tham số mô hình R1", "enableR1FormatTips": "Cần kích hoạt khi sử dụng các mô hình R1 như QWQ, để tránh lỗi 400", "useAzure": "Sử dụng Azure", diff --git a/webview-ui/src/i18n/locales/zh-CN/settings.json b/webview-ui/src/i18n/locales/zh-CN/settings.json index c1985e7490..7dcfe4b814 100644 --- a/webview-ui/src/i18n/locales/zh-CN/settings.json +++ b/webview-ui/src/i18n/locales/zh-CN/settings.json @@ -739,6 +739,7 @@ "cacheReadsPrice": "缓存读取价格", "cacheWritesPrice": "缓存写入价格", "enableStreaming": "启用流式传输", + "enableStreamingDescription": "如果在使用高级模型时遇到组织验证错误,请禁用流式传输。非流式请求可能无需验证即可工作。", "enableR1Format": "启用 R1 模型参数", "enableR1FormatTips": "使用 QWQ 等 R1 系列模型时必须启用,避免出现 400 错误", "useAzure": "使用 Azure 服务", diff --git a/webview-ui/src/i18n/locales/zh-TW/settings.json b/webview-ui/src/i18n/locales/zh-TW/settings.json index 6a673de60a..a95cc76291 100644 --- a/webview-ui/src/i18n/locales/zh-TW/settings.json +++ b/webview-ui/src/i18n/locales/zh-TW/settings.json @@ -739,6 +739,7 @@ "cacheReadsPrice": "快取讀取價格", "cacheWritesPrice": "快取寫入價格", "enableStreaming": "啟用串流輸出", + "enableStreamingDescription": "如果在使用進階模型時遇到組織驗證錯誤,請停用串流輸出。非串流請求可能無需驗證即可運作。", "enableR1Format": "啟用 R1 模型參數", "enableR1FormatTips": "使用 QWQ 等 R1 模型時必須啟用,以避免發生 400 錯誤", "useAzure": "使用 Azure",