fix: address PR review feedback for OpenAI Native streaming toggle

- Added comprehensive test coverage for openAiNativeStreamingEnabled option
- Added translations for all 17 supported languages
- Refactored code to reduce duplication with helper methods
- Improved error handling for non-streaming responses
- Added documentation comments for all major methods
- Enhanced type safety with proper error handling
- Resolved merge conflicts with main branch
This commit is contained in:
Roo Code 2025-08-14 18:20:39 +00:00
parent 27ecbbabf3
commit b5109e9329
19 changed files with 767 additions and 223 deletions

View file

@ -1815,4 +1815,432 @@ describe("GPT-5 streaming event coverage (additional)", () => {
delete (global as any).fetch
})
})
describe("openAiNativeStreamingEnabled option", () => {
let handler: OpenAiNativeHandler
const systemPrompt = "You are a helpful assistant."
const messages: Anthropic.Messages.MessageParam[] = [
{
role: "user",
content: "Hello!",
},
]
beforeEach(() => {
mockCreate.mockClear()
})
it("should use streaming by default when openAiNativeStreamingEnabled is not specified", async () => {
handler = new OpenAiNativeHandler({
apiModelId: "gpt-4.1",
openAiNativeApiKey: "test-api-key",
// openAiNativeStreamingEnabled not specified, should default to true
})
const stream = handler.createMessage(systemPrompt, messages)
for await (const chunk of stream) {
// consume stream
}
expect(mockCreate).toHaveBeenCalledWith(
expect.objectContaining({
stream: true,
stream_options: { include_usage: true },
}),
)
})
it("should use streaming when openAiNativeStreamingEnabled is true", async () => {
handler = new OpenAiNativeHandler({
apiModelId: "gpt-4.1",
openAiNativeApiKey: "test-api-key",
openAiNativeStreamingEnabled: true,
})
const stream = handler.createMessage(systemPrompt, messages)
for await (const chunk of stream) {
// consume stream
}
expect(mockCreate).toHaveBeenCalledWith(
expect.objectContaining({
stream: true,
stream_options: { include_usage: true },
}),
)
})
it("should disable streaming when openAiNativeStreamingEnabled is false for standard models", async () => {
handler = new OpenAiNativeHandler({
apiModelId: "gpt-4.1",
openAiNativeApiKey: "test-api-key",
openAiNativeStreamingEnabled: false,
})
mockCreate.mockResolvedValueOnce({
id: "test-completion",
choices: [
{
message: { role: "assistant", content: "Non-streaming response" },
finish_reason: "stop",
index: 0,
},
],
usage: {
prompt_tokens: 10,
completion_tokens: 5,
total_tokens: 15,
},
})
const stream = handler.createMessage(systemPrompt, messages)
const chunks: any[] = []
for await (const chunk of stream) {
chunks.push(chunk)
}
// Should call create without stream parameter
expect(mockCreate).toHaveBeenCalledWith(
expect.objectContaining({
model: "gpt-4.1",
temperature: 0,
messages: [
{ role: "system", content: systemPrompt },
{ role: "user", content: "Hello!" },
],
}),
)
expect(mockCreate).toHaveBeenCalledWith(
expect.not.objectContaining({
stream: true,
}),
)
// Should yield text and usage from non-streaming response
expect(chunks).toHaveLength(2)
expect(chunks[0]).toMatchObject({ type: "text", text: "Non-streaming response" })
expect(chunks[1]).toMatchObject({
type: "usage",
inputTokens: 10,
outputTokens: 5,
})
})
it("should disable streaming for O1 models when openAiNativeStreamingEnabled is false", async () => {
handler = new OpenAiNativeHandler({
apiModelId: "o1",
openAiNativeApiKey: "test-api-key",
openAiNativeStreamingEnabled: false,
})
mockCreate.mockResolvedValueOnce({
id: "test-completion",
choices: [
{
message: { role: "assistant", content: "O1 non-streaming response" },
finish_reason: "stop",
index: 0,
},
],
usage: {
prompt_tokens: 20,
completion_tokens: 10,
total_tokens: 30,
},
})
const stream = handler.createMessage(systemPrompt, messages)
const chunks: any[] = []
for await (const chunk of stream) {
chunks.push(chunk)
}
// Should not include stream parameter for O1 models
expect(mockCreate).toHaveBeenCalledWith(
expect.objectContaining({
model: "o1",
messages: [
{ role: "developer", content: "Formatting re-enabled\n" + systemPrompt },
{ role: "user", content: "Hello!" },
],
}),
)
expect(mockCreate).toHaveBeenCalledWith(
expect.not.objectContaining({
stream: true,
}),
)
// Should yield text and usage
expect(chunks).toHaveLength(2)
expect(chunks[0]).toMatchObject({ type: "text", text: "O1 non-streaming response" })
expect(chunks[1]).toMatchObject({
type: "usage",
inputTokens: 20,
outputTokens: 10,
})
})
it("should disable streaming for O3-mini models when openAiNativeStreamingEnabled is false", async () => {
handler = new OpenAiNativeHandler({
apiModelId: "o3-mini",
openAiNativeApiKey: "test-api-key",
openAiNativeStreamingEnabled: false,
})
mockCreate.mockResolvedValueOnce({
id: "test-completion",
choices: [
{
message: { role: "assistant", content: "O3-mini non-streaming response" },
finish_reason: "stop",
index: 0,
},
],
usage: {
prompt_tokens: 15,
completion_tokens: 8,
total_tokens: 23,
},
})
const stream = handler.createMessage(systemPrompt, messages)
const chunks: any[] = []
for await (const chunk of stream) {
chunks.push(chunk)
}
// Should include reasoning_effort but not stream
expect(mockCreate).toHaveBeenCalledWith(
expect.objectContaining({
model: "o3-mini",
reasoning_effort: "medium",
messages: [
{ role: "developer", content: "Formatting re-enabled\n" + systemPrompt },
{ role: "user", content: "Hello!" },
],
}),
)
expect(mockCreate).toHaveBeenCalledWith(
expect.not.objectContaining({
stream: true,
}),
)
// Should yield text and usage
expect(chunks).toHaveLength(2)
expect(chunks[0]).toMatchObject({ type: "text", text: "O3-mini non-streaming response" })
expect(chunks[1]).toMatchObject({
type: "usage",
inputTokens: 15,
outputTokens: 8,
})
})
it("should disable streaming for GPT-5 models when openAiNativeStreamingEnabled is false", async () => {
// Mock fetch for non-streaming Responses API
const mockFetch = vitest.fn().mockResolvedValue({
ok: true,
json: async () => ({
response: {
id: "resp_001",
output: [
{
type: "text",
content: [{ type: "text", text: "GPT-5 non-streaming response" }],
},
],
usage: {
prompt_tokens: 30,
completion_tokens: 15,
},
},
}),
})
global.fetch = mockFetch as any
handler = new OpenAiNativeHandler({
apiModelId: "gpt-5-2025-08-07",
openAiNativeApiKey: "test-api-key",
openAiNativeStreamingEnabled: false,
})
const stream = handler.createMessage(systemPrompt, messages)
const chunks: any[] = []
for await (const chunk of stream) {
chunks.push(chunk)
}
// Should call Responses API without stream parameter
expect(mockFetch).toHaveBeenCalledWith(
"https://api.openai.com/v1/responses",
expect.objectContaining({
method: "POST",
headers: expect.objectContaining({
"Content-Type": "application/json",
Authorization: "Bearer test-api-key",
// No Accept: text/event-stream header for non-streaming
}),
body: expect.any(String),
}),
)
const requestBody = JSON.parse(mockFetch.mock.calls[0][1].body)
expect(requestBody.stream).toBe(false)
expect(requestBody.model).toBe("gpt-5-2025-08-07")
// Should yield text and usage
expect(chunks).toHaveLength(2)
expect(chunks[0]).toMatchObject({ type: "text", text: "GPT-5 non-streaming response" })
expect(chunks[1]).toMatchObject({
type: "usage",
inputTokens: 30,
outputTokens: 15,
})
// Clean up
delete (global as any).fetch
})
it("should handle cache tokens in non-streaming response", async () => {
handler = new OpenAiNativeHandler({
apiModelId: "gpt-4.1",
openAiNativeApiKey: "test-api-key",
openAiNativeStreamingEnabled: false,
})
mockCreate.mockResolvedValueOnce({
id: "test-completion",
choices: [
{
message: { role: "assistant", content: "Cached response" },
finish_reason: "stop",
index: 0,
},
],
usage: {
prompt_tokens: 100,
completion_tokens: 10,
prompt_tokens_details: {
cached_tokens: 80,
},
cache_creation_input_tokens: 20,
},
})
const stream = handler.createMessage(systemPrompt, messages)
const chunks: any[] = []
for await (const chunk of stream) {
chunks.push(chunk)
}
// Should include cache tokens in usage
const usageChunk = chunks.find((c) => c.type === "usage")
expect(usageChunk).toMatchObject({
type: "usage",
inputTokens: 100,
outputTokens: 10,
cacheReadTokens: 80,
cacheWriteTokens: 20,
})
})
it("should handle error in non-streaming response", async () => {
handler = new OpenAiNativeHandler({
apiModelId: "gpt-4.1",
openAiNativeApiKey: "test-api-key",
openAiNativeStreamingEnabled: false,
})
mockCreate.mockRejectedValueOnce(new Error("API Error"))
const stream = handler.createMessage(systemPrompt, messages)
await expect(async () => {
for await (const chunk of stream) {
// Should throw before yielding
}
}).rejects.toThrow("API Error")
})
it("should handle empty response in non-streaming mode", async () => {
handler = new OpenAiNativeHandler({
apiModelId: "gpt-4.1",
openAiNativeApiKey: "test-api-key",
openAiNativeStreamingEnabled: false,
})
mockCreate.mockResolvedValueOnce({
id: "test-completion",
choices: [
{
message: { role: "assistant", content: "" },
finish_reason: "stop",
index: 0,
},
],
usage: {
prompt_tokens: 10,
completion_tokens: 0,
},
})
const stream = handler.createMessage(systemPrompt, messages)
const chunks: any[] = []
for await (const chunk of stream) {
chunks.push(chunk)
}
// Should handle empty content gracefully
expect(chunks).toHaveLength(1)
expect(chunks[0]).toMatchObject({
type: "usage",
inputTokens: 10,
outputTokens: 0,
})
})
it("should handle verbosity parameter correctly in non-streaming mode for GPT-5", async () => {
// Mock fetch for non-streaming Responses API
const mockFetch = vitest.fn().mockResolvedValue({
ok: true,
json: async () => ({
response: {
id: "resp_001",
output: [
{
type: "text",
content: [{ type: "text", text: "High verbosity response" }],
},
],
usage: {
prompt_tokens: 25,
completion_tokens: 12,
},
},
}),
})
global.fetch = mockFetch as any
handler = new OpenAiNativeHandler({
apiModelId: "gpt-5-2025-08-07",
openAiNativeApiKey: "test-api-key",
openAiNativeStreamingEnabled: false,
verbosity: "high",
})
const stream = handler.createMessage(systemPrompt, messages)
const chunks: any[] = []
for await (const chunk of stream) {
chunks.push(chunk)
}
// Should include verbosity in request
const requestBody = JSON.parse(mockFetch.mock.calls[0][1].body)
expect(requestBody.verbosity).toBe("high")
expect(requestBody.stream).toBe(false)
// Clean up
delete (global as any).fetch
})
})
})

View file

@ -97,6 +97,15 @@ export class OpenAiNativeHandler extends BaseProvider implements SingleCompletio
}
}
/**
* Creates a message stream for the OpenAI Native API.
* Routes to the appropriate handler based on the model type.
*
* @param systemPrompt - The system prompt to use
* @param messages - The conversation messages
* @param metadata - Optional metadata for conversation continuity
* @yields API stream chunks including text, reasoning, and usage data
*/
override async *createMessage(
systemPrompt: string,
messages: Anthropic.Messages.MessageParam[],
@ -125,188 +134,307 @@ export class OpenAiNativeHandler extends BaseProvider implements SingleCompletio
}
}
private async *handleO1FamilyMessage(
/**
* Helper method to check if streaming is enabled.
* Defaults to true for backward compatibility.
*/
private isStreamingEnabled(): boolean {
return this.options.openAiNativeStreamingEnabled ?? true
}
/**
* Helper method to handle non-streaming responses uniformly.
* Extracts text content and usage data from the response.
*
* @param response - The OpenAI completion response
* @param model - The model information
* @yields Text and usage chunks
*/
private async *handleNonStreamingResponse(
response: OpenAI.Chat.Completions.ChatCompletion,
model: OpenAiNativeModel,
systemPrompt: string,
messages: Anthropic.Messages.MessageParam[],
): ApiStream {
// o1 supports developer prompt with formatting
// o1-preview and o1-mini only support user messages
const isOriginalO1 = model.id === "o1"
const { reasoning } = this.getModel()
const streamingEnabled = this.options.openAiNativeStreamingEnabled ?? true
if (streamingEnabled) {
const response = await this.client.chat.completions.create({
model: model.id,
messages: [
{
role: isOriginalO1 ? "developer" : "user",
content: isOriginalO1 ? `Formatting re-enabled\n${systemPrompt}` : systemPrompt,
},
...convertToOpenAiMessages(messages),
],
stream: true,
stream_options: { include_usage: true },
...(reasoning && reasoning),
})
yield* this.handleStreamResponse(response, model)
} else {
// Non-streaming request
const response = await this.client.chat.completions.create({
model: model.id,
messages: [
{
role: isOriginalO1 ? "developer" : "user",
content: isOriginalO1 ? `Formatting re-enabled\n${systemPrompt}` : systemPrompt,
},
...convertToOpenAiMessages(messages),
],
...(reasoning && reasoning),
})
yield {
type: "text",
text: response.choices[0]?.message.content || "",
try {
const content = response.choices[0]?.message?.content
if (content) {
yield {
type: "text",
text: content,
}
}
if (response.usage) {
yield* this.yieldUsage(model.info, response.usage)
}
} catch (error) {
throw new Error(
`Error processing non-streaming response: ${error instanceof Error ? error.message : "Unknown error"}`,
)
}
}
/**
* Handles message creation for O1 family models (o1, o1-preview, o1-mini).
* These models have special requirements for system prompts and don't support temperature.
*
* @param model - The model information
* @param systemPrompt - The system prompt
* @param messages - The conversation messages
* @yields API stream chunks
*/
private async *handleO1FamilyMessage(
model: OpenAiNativeModel,
systemPrompt: string,
messages: Anthropic.Messages.MessageParam[],
): ApiStream {
try {
// o1 supports developer prompt with formatting
// o1-preview and o1-mini only support user messages
const isOriginalO1 = model.id === "o1"
const { reasoning } = this.getModel()
const formattedMessages = [
{
role: isOriginalO1 ? "developer" : "user",
content: isOriginalO1 ? `Formatting re-enabled\n${systemPrompt}` : systemPrompt,
} as any,
...convertToOpenAiMessages(messages),
]
if (this.isStreamingEnabled()) {
const response = await this.client.chat.completions.create({
model: model.id,
messages: formattedMessages,
stream: true,
stream_options: { include_usage: true },
...(reasoning && reasoning),
})
yield* this.handleStreamResponse(response, model)
} else {
const response = await this.client.chat.completions.create({
model: model.id,
messages: formattedMessages,
...(reasoning && reasoning),
})
yield* this.handleNonStreamingResponse(response, model)
}
} catch (error) {
throw new Error(`O1 family model error: ${error instanceof Error ? error.message : "Unknown error"}`)
}
}
/**
* Handles message creation for O3/O4 reasoner models.
* These models use the developer role and support reasoning effort parameters.
*
* @param model - The model information
* @param family - The model family identifier
* @param systemPrompt - The system prompt
* @param messages - The conversation messages
* @yields API stream chunks
*/
private async *handleReasonerMessage(
model: OpenAiNativeModel,
family: "o3-mini" | "o3" | "o4-mini",
systemPrompt: string,
messages: Anthropic.Messages.MessageParam[],
): ApiStream {
const { reasoning } = this.getModel()
const streamingEnabled = this.options.openAiNativeStreamingEnabled ?? true
try {
const { reasoning } = this.getModel()
if (streamingEnabled) {
const stream = await this.client.chat.completions.create({
model: family,
messages: [
{
role: "developer",
content: `Formatting re-enabled\n${systemPrompt}`,
},
...convertToOpenAiMessages(messages),
],
stream: true,
stream_options: { include_usage: true },
...(reasoning && reasoning),
})
const formattedMessages = [
{
role: "developer",
content: `Formatting re-enabled\n${systemPrompt}`,
} as any,
...convertToOpenAiMessages(messages),
]
yield* this.handleStreamResponse(stream, model)
} else {
// Non-streaming request
const response = await this.client.chat.completions.create({
model: family,
messages: [
{
role: "developer",
content: `Formatting re-enabled\n${systemPrompt}`,
},
...convertToOpenAiMessages(messages),
],
...(reasoning && reasoning),
})
yield {
type: "text",
text: response.choices[0]?.message.content || "",
}
if (response.usage) {
yield* this.yieldUsage(model.info, response.usage)
if (this.isStreamingEnabled()) {
const stream = await this.client.chat.completions.create({
model: family,
messages: formattedMessages,
stream: true,
stream_options: { include_usage: true },
...(reasoning && reasoning),
})
yield* this.handleStreamResponse(stream, model)
} else {
const response = await this.client.chat.completions.create({
model: family,
messages: formattedMessages,
...(reasoning && reasoning),
})
yield* this.handleNonStreamingResponse(response, model)
}
} catch (error) {
throw new Error(`Reasoner model error: ${error instanceof Error ? error.message : "Unknown error"}`)
}
}
/**
* Handles message creation for standard OpenAI models (GPT-4, GPT-3.5, etc).
* Supports both streaming and non-streaming modes with full parameter support.
*
* @param model - The model information
* @param systemPrompt - The system prompt
* @param messages - The conversation messages
* @yields API stream chunks
*/
private async *handleDefaultModelMessage(
model: OpenAiNativeModel,
systemPrompt: string,
messages: Anthropic.Messages.MessageParam[],
): ApiStream {
const { reasoning, verbosity } = this.getModel()
const streamingEnabled = this.options.openAiNativeStreamingEnabled ?? true
try {
const { reasoning, verbosity } = this.getModel()
if (streamingEnabled) {
// Prepare the request parameters for streaming
const params: any = {
const formattedMessages = [
{ role: "system", content: systemPrompt } as any,
...convertToOpenAiMessages(messages),
]
// Build base parameters
const baseParams: any = {
model: model.id,
temperature: this.options.modelTemperature ?? OPENAI_NATIVE_DEFAULT_TEMPERATURE,
messages: [{ role: "system", content: systemPrompt }, ...convertToOpenAiMessages(messages)],
stream: true,
stream_options: { include_usage: true },
messages: formattedMessages,
...(reasoning && reasoning),
}
// Add verbosity only if the model supports it
if (verbosity && model.info.supportsVerbosity) {
params.verbosity = verbosity
baseParams.verbosity = verbosity
}
const stream = await this.client.chat.completions.create(params)
if (this.isStreamingEnabled()) {
const params = {
...baseParams,
stream: true,
stream_options: { include_usage: true },
}
if (typeof (stream as any)[Symbol.asyncIterator] !== "function") {
throw new Error(
"OpenAI SDK did not return an AsyncIterable for streaming response. Please check SDK version and usage.",
const stream = await this.client.chat.completions.create(params)
if (typeof (stream as any)[Symbol.asyncIterator] !== "function") {
throw new Error(
"OpenAI SDK did not return an AsyncIterable for streaming response. Please check SDK version and usage.",
)
}
yield* this.handleStreamResponse(
stream as unknown as AsyncIterable<OpenAI.Chat.Completions.ChatCompletionChunk>,
model,
)
} else {
const response = await this.client.chat.completions.create(baseParams)
yield* this.handleNonStreamingResponse(response, model)
}
yield* this.handleStreamResponse(
stream as unknown as AsyncIterable<OpenAI.Chat.Completions.ChatCompletionChunk>,
model,
)
} else {
// Non-streaming request
const params: any = {
model: model.id,
temperature: this.options.modelTemperature ?? OPENAI_NATIVE_DEFAULT_TEMPERATURE,
messages: [{ role: "system", content: systemPrompt }, ...convertToOpenAiMessages(messages)],
...(reasoning && reasoning),
}
// Add verbosity only if the model supports it
if (verbosity && model.info.supportsVerbosity) {
params.verbosity = verbosity
}
const response = await this.client.chat.completions.create(params)
yield {
type: "text",
text: response.choices[0]?.message.content || "",
}
if (response.usage) {
yield* this.yieldUsage(model.info, response.usage)
}
} catch (error) {
throw new Error(`Default model error: ${error instanceof Error ? error.message : "Unknown error"}`)
}
}
/**
* Handles message creation for models using the Responses API (GPT-5 and Codex Mini).
* Supports conversation continuity via previous_response_id and both streaming/non-streaming modes.
* Automatically falls back to SSE if SDK fails.
*
* @param model - The model information
* @param systemPrompt - The system prompt
* @param messages - The conversation messages
* @param metadata - Optional metadata for conversation continuity
* @yields API stream chunks
*/
private async *handleResponsesApiMessage(
model: OpenAiNativeModel,
systemPrompt: string,
messages: Anthropic.Messages.MessageParam[],
metadata?: ApiHandlerCreateMessageMetadata,
): ApiStream {
// Prefer the official SDK Responses API with streaming; fall back to fetch-based SSE if needed.
const { verbosity } = this.getModel()
const streamingEnabled = this.options.openAiNativeStreamingEnabled ?? true
try {
// Prepare the request body for the Responses API
const requestBody = await this.prepareResponsesApiRequest(model, systemPrompt, messages, metadata)
// Both GPT-5 and Codex Mini use the same v1/responses endpoint format
// Check if streaming is enabled
if (!this.isStreamingEnabled()) {
requestBody.stream = false
yield* this.makeGpt5ResponsesAPIRequest(requestBody, model, metadata)
return
}
// Resolve reasoning effort (supports "minimal" for GPT‑5)
// Try SDK first, fall back to SSE on error
yield* this.tryResponsesApiWithFallback(requestBody, model, metadata)
} catch (error) {
throw new Error(`Responses API error: ${error instanceof Error ? error.message : "Unknown error"}`)
}
}
/**
* Prepares the request body for the Responses API.
* Handles conversation continuity and all required parameters.
*
* @private
*/
private async prepareResponsesApiRequest(
model: OpenAiNativeModel,
systemPrompt: string,
messages: Anthropic.Messages.MessageParam[],
metadata?: ApiHandlerCreateMessageMetadata,
): Promise<any> {
const { verbosity } = model
const reasoningEffort = this.getGpt5ReasoningEffort(model)
// Wait for any pending response ID from a previous request to be available
// This handles the race condition with fast nano model responses
// Handle conversation continuity
const effectivePreviousResponseId = await this.resolvePreviousResponseId(metadata)
// Format input and capture continuity id
const { formattedInput, previousResponseId } = this.prepareGpt5Input(systemPrompt, messages, metadata)
const requestPreviousResponseId = effectivePreviousResponseId ?? previousResponseId
// Create a new promise for this request's response ID
this.responseIdPromise = new Promise<string | undefined>((resolve) => {
this.responseIdResolver = resolve
})
// Build the request body
interface Gpt5RequestBody {
model: string
input: string
stream: boolean
reasoning?: { effort: ReasoningEffortWithMinimal; summary?: "auto" }
text?: { verbosity: VerbosityLevel }
temperature?: number
max_output_tokens?: number
previous_response_id?: string
}
const requestBody: Gpt5RequestBody = {
model: model.id,
input: formattedInput,
stream: true,
...(reasoningEffort && {
reasoning: {
effort: reasoningEffort,
...(this.options.enableGpt5ReasoningSummary ? { summary: "auto" as const } : {}),
},
}),
text: { verbosity: (verbosity || "medium") as VerbosityLevel },
temperature: this.options.modelTemperature ?? GPT5_DEFAULT_TEMPERATURE,
...(model.maxTokens ? { max_output_tokens: model.maxTokens } : {}),
...(requestPreviousResponseId && { previous_response_id: requestPreviousResponseId }),
}
return requestBody
}
/**
* Resolves the previous response ID for conversation continuity.
* Handles race conditions with pending response IDs.
*
* @private
*/
private async resolvePreviousResponseId(metadata?: ApiHandlerCreateMessageMetadata): Promise<string | undefined> {
let effectivePreviousResponseId = metadata?.previousResponseId
// Only allow fallback to pending/last response id when not explicitly suppressed
@ -333,63 +461,20 @@ export class OpenAiNativeHandler extends BaseProvider implements SingleCompletio
}
}
// Format input and capture continuity id
const { formattedInput, previousResponseId } = this.prepareGpt5Input(systemPrompt, messages, metadata)
const requestPreviousResponseId = effectivePreviousResponseId ?? previousResponseId
// Create a new promise for this request's response ID
this.responseIdPromise = new Promise<string | undefined>((resolve) => {
this.responseIdResolver = resolve
})
// Build a request body (also used for fallback)
// Ensure we explicitly pass max_output_tokens for GPT‑5 based on Roo's reserved model response calculation
// so requests do not default to very large limits (e.g., 120k).
interface Gpt5RequestBody {
model: string
input: string
stream: boolean
reasoning?: { effort: ReasoningEffortWithMinimal; summary?: "auto" }
text?: { verbosity: VerbosityLevel }
temperature?: number
max_output_tokens?: number
previous_response_id?: string
}
const requestBody: Gpt5RequestBody = {
model: model.id,
input: formattedInput,
stream: true,
...(reasoningEffort && {
reasoning: {
effort: reasoningEffort,
...(this.options.enableGpt5ReasoningSummary ? { summary: "auto" as const } : {}),
},
}),
text: { verbosity: (verbosity || "medium") as VerbosityLevel },
temperature: this.options.modelTemperature ?? GPT5_DEFAULT_TEMPERATURE,
// Explicitly include the calculated max output tokens for GPT‑5.
// Use the per-request reserved output computed by Roo (params.maxTokens from getModelParams).
...(model.maxTokens ? { max_output_tokens: model.maxTokens } : {}),
...(requestPreviousResponseId && { previous_response_id: requestPreviousResponseId }),
}
// Check if streaming is enabled
if (!streamingEnabled) {
// For non-streaming, we need to modify the request body
requestBody.stream = false
// Make non-streaming request using the makeGpt5ResponsesAPIRequest method
// Note: The method signature expects the requestBody, not params
const responseIterator = this.makeGpt5ResponsesAPIRequest(requestBody, model, metadata)
// Process the non-streaming response
for await (const chunk of responseIterator) {
yield chunk
}
return
}
return effectivePreviousResponseId
}
/**
* Tries to use the SDK for Responses API, with fallback to SSE on error.
* Handles previous_response_id errors with automatic retry.
*
* @private
*/
private async *tryResponsesApiWithFallback(
requestBody: any,
model: OpenAiNativeModel,
metadata?: ApiHandlerCreateMessageMetadata,
): ApiStream {
try {
// Use the official SDK for streaming
const stream = (await (this.client as any).responses.create(requestBody)) as AsyncIterable<any>
@ -406,52 +491,66 @@ export class OpenAiNativeHandler extends BaseProvider implements SingleCompletio
}
}
} catch (sdkErr: any) {
// Check if this is a 400 error about previous_response_id not found
const errorMessage = sdkErr?.message || sdkErr?.error?.message || ""
const is400Error = sdkErr?.status === 400 || sdkErr?.response?.status === 400
const isPreviousResponseError =
errorMessage.includes("Previous response") || errorMessage.includes("not found")
// Handle previous_response_id errors with retry
if (this.isPreviousResponseIdError(sdkErr) && requestBody.previous_response_id) {
yield* this.retryWithoutPreviousResponseId(requestBody, model, metadata)
} else {
// For other errors, fallback to manual SSE via fetch
yield* this.makeGpt5ResponsesAPIRequest(requestBody, model, metadata)
}
}
}
if (is400Error && requestBody.previous_response_id && isPreviousResponseError) {
// Log the error and retry without the previous_response_id
console.warn(
`[GPT-5] Previous response ID not found (${requestBody.previous_response_id}), retrying without it`,
)
/**
* Checks if an error is related to previous_response_id not found.
*
* @private
*/
private isPreviousResponseIdError(error: any): boolean {
const errorMessage = error?.message || error?.error?.message || ""
const is400Error = error?.status === 400 || error?.response?.status === 400
return is400Error && (errorMessage.includes("Previous response") || errorMessage.includes("not found"))
}
// Remove the problematic previous_response_id and retry
const retryRequestBody = { ...requestBody }
delete retryRequestBody.previous_response_id
/**
* Retries the request without the previous_response_id after an error.
*
* @private
*/
private async *retryWithoutPreviousResponseId(
requestBody: any,
model: OpenAiNativeModel,
metadata?: ApiHandlerCreateMessageMetadata,
): ApiStream {
console.warn(
`[GPT-5] Previous response ID not found (${requestBody.previous_response_id}), retrying without it`,
)
// Clear the stored lastResponseId to prevent using it again
this.lastResponseId = undefined
// Remove the problematic previous_response_id and retry
const retryRequestBody = { ...requestBody }
delete retryRequestBody.previous_response_id
try {
// Retry with the SDK
const retryStream = (await (this.client as any).responses.create(
retryRequestBody,
)) as AsyncIterable<any>
// Clear the stored lastResponseId to prevent using it again
this.lastResponseId = undefined
if (typeof (retryStream as any)[Symbol.asyncIterator] !== "function") {
// If SDK fails, fall back to SSE
yield* this.makeGpt5ResponsesAPIRequest(retryRequestBody, model, metadata)
return
}
try {
// Retry with the SDK
const retryStream = (await (this.client as any).responses.create(retryRequestBody)) as AsyncIterable<any>
for await (const event of retryStream) {
for await (const outChunk of this.processGpt5Event(event, model)) {
yield outChunk
}
}
return
} catch (retryErr) {
// If retry also fails, fall back to SSE
yield* this.makeGpt5ResponsesAPIRequest(retryRequestBody, model, metadata)
return
}
if (typeof (retryStream as any)[Symbol.asyncIterator] !== "function") {
// If SDK fails, fall back to SSE
yield* this.makeGpt5ResponsesAPIRequest(retryRequestBody, model, metadata)
return
}
// For other errors, fallback to manual SSE via fetch
yield* this.makeGpt5ResponsesAPIRequest(requestBody, model, metadata)
for await (const event of retryStream) {
for await (const outChunk of this.processGpt5Event(event, model)) {
yield outChunk
}
}
} catch (retryErr) {
// If retry also fails, fall back to SSE
yield* this.makeGpt5ResponsesAPIRequest(retryRequestBody, model, metadata)
}
}

View file

@ -738,6 +738,7 @@
"cacheReadsPrice": "Preu de lectures de caché",
"cacheWritesPrice": "Preu d'escriptures de caché",
"enableStreaming": "Habilitar streaming",
"enableStreamingDescription": "Desactiva el streaming si trobes errors de verificació d'organització amb models avançats. Les sol·licituds sense streaming poden funcionar sense verificació.",
"enableR1Format": "Activar els paràmetres del model R1",
"enableR1FormatTips": "S'ha d'activat quan s'utilitzen models R1 com el QWQ per evitar errors 400",
"useAzure": "Utilitzar Azure",

View file

@ -738,6 +738,7 @@
"cacheReadsPrice": "Cache-Lesepreis",
"cacheWritesPrice": "Cache-Schreibpreis",
"enableStreaming": "Streaming aktivieren",
"enableStreamingDescription": "Deaktiviere Streaming, wenn du bei erweiterten Modellen auf Organisationsverifizierungsfehler stößt. Nicht-Streaming-Anfragen funktionieren möglicherweise ohne Verifizierung.",
"enableR1Format": "R1-Modellparameter aktivieren",
"enableR1FormatTips": "Muss bei Verwendung von R1-Modellen wie QWQ aktiviert werden, um 400er-Fehler zu vermeiden",
"useAzure": "Azure verwenden",

View file

@ -738,6 +738,7 @@
"cacheReadsPrice": "Precio de lecturas de caché",
"cacheWritesPrice": "Precio de escrituras de caché",
"enableStreaming": "Habilitar streaming",
"enableStreamingDescription": "Desactiva el streaming si encuentras errores de verificación de organización con modelos avanzados. Las solicitudes sin streaming pueden funcionar sin verificación.",
"enableR1Format": "Habilitar parámetros del modelo R1",
"enableR1FormatTips": "Debe habilitarse al utilizar modelos R1 como QWQ, para evitar el error 400",
"useAzure": "Usar Azure",

View file

@ -738,6 +738,7 @@
"cacheReadsPrice": "Prix des lectures de cache",
"cacheWritesPrice": "Prix des écritures de cache",
"enableStreaming": "Activer le streaming",
"enableStreamingDescription": "Désactive le streaming si tu rencontres des erreurs de vérification d'organisation avec des modèles avancés. Les requêtes sans streaming peuvent fonctionner sans vérification.",
"enableR1Format": "Activer les paramètres du modèle R1",
"enableR1FormatTips": "Doit être activé lors de l'utilisation de modèles R1 tels que QWQ, pour éviter l'erreur 400",
"useAzure": "Utiliser Azure",

View file

@ -739,6 +739,7 @@
"cacheReadsPrice": "कैश रीड्स मूल्य",
"cacheWritesPrice": "कैश राइट्स मूल्य",
"enableStreaming": "स्ट्रीमिंग सक्षम करें",
"enableStreamingDescription": "यदि आपको उन्नत मॉडल के साथ संगठन सत्यापन त्रुटियों का सामना करना पड़ता है तो स्ट्रीमिंग अक्षम करें। गैर-स्ट्रीमिंग अनुरोध सत्यापन के बिना काम कर सकते हैं।",
"enableR1Format": "R1 मॉडल पैरामीटर सक्षम करें",
"enableR1FormatTips": "QWQ जैसी R1 मॉडलों का उपयोग करते समय इसे सक्षम करना आवश्यक है, ताकि 400 त्रुटि से बचा जा सके",
"useAzure": "Azure का उपयोग करें",

View file

@ -768,6 +768,7 @@
"cacheReadsPrice": "Harga cache reads",
"cacheWritesPrice": "Harga cache writes",
"enableStreaming": "Aktifkan streaming",
"enableStreamingDescription": "Nonaktifkan streaming jika kamu mengalami kesalahan verifikasi organisasi dengan model lanjutan. Permintaan non-streaming mungkin berfungsi tanpa verifikasi.",
"enableR1Format": "Aktifkan parameter model R1",
"enableR1FormatTips": "Harus diaktifkan saat menggunakan model R1 seperti QWQ untuk mencegah error 400",
"useAzure": "Gunakan Azure",

View file

@ -739,6 +739,7 @@
"cacheReadsPrice": "Prezzo letture cache",
"cacheWritesPrice": "Prezzo scritture cache",
"enableStreaming": "Abilita streaming",
"enableStreamingDescription": "Disabilita lo streaming se riscontri errori di verifica dell'organizzazione con modelli avanzati. Le richieste non in streaming potrebbero funzionare senza verifica.",
"enableR1Format": "Abilita i parametri del modello R1",
"enableR1FormatTips": "Deve essere abilitato quando si utilizzano modelli R1 come QWQ, per evitare l'errore 400",
"useAzure": "Usa Azure",

View file

@ -739,6 +739,7 @@
"cacheReadsPrice": "キャッシュ読み取り価格",
"cacheWritesPrice": "キャッシュ書き込み価格",
"enableStreaming": "ストリーミングを有効化",
"enableStreamingDescription": "高度なモデルで組織の検証エラーが発生した場合は、ストリーミングを無効にしてください。非ストリーミングリクエストは検証なしで動作する可能性があります。",
"enableR1Format": "R1モデルパラメータを有効にする",
"enableR1FormatTips": "QWQなどのR1モデルを使用する際には、有効にする必要があります。400エラーを防ぐために",
"useAzure": "Azureを使用",

View file

@ -739,6 +739,7 @@
"cacheReadsPrice": "캐시 읽기 가격",
"cacheWritesPrice": "캐시 쓰기 가격",
"enableStreaming": "스트리밍 활성화",
"enableStreamingDescription": "고급 모델에서 조직 확인 오류가 발생하면 스트리밍을 비활성화하세요. 비스트리밍 요청은 확인 없이 작동할 수 있습니다.",
"enableR1Format": "R1 모델 매개변수 활성화",
"enableR1FormatTips": "QWQ와 같은 R1 모델을 사용할 때 활성화해야 하며, 400 오류를 방지합니다",
"useAzure": "Azure 사용",

View file

@ -739,6 +739,7 @@
"cacheReadsPrice": "Cache-leesprijs",
"cacheWritesPrice": "Cache-schrijfprijs",
"enableStreaming": "Streaming inschakelen",
"enableStreamingDescription": "Schakel streaming uit als je organisatieverificatiefouten tegenkomt met geavanceerde modellen. Niet-streaming verzoeken kunnen zonder verificatie werken.",
"enableR1Format": "R1-modelparameters inschakelen",
"enableR1FormatTips": "Moet ingeschakeld zijn bij gebruik van R1-modellen zoals QWQ om 400-fouten te voorkomen",
"useAzure": "Azure gebruiken",

View file

@ -739,6 +739,7 @@
"cacheReadsPrice": "Cena odczytów bufora",
"cacheWritesPrice": "Cena zapisów bufora",
"enableStreaming": "Włącz strumieniowanie",
"enableStreamingDescription": "Wyłącz strumieniowanie, jeśli napotkasz błędy weryfikacji organizacji z zaawansowanymi modelami. Żądania bez strumieniowania mogą działać bez weryfikacji.",
"enableR1Format": "Włącz parametry modelu R1",
"enableR1FormatTips": "Należy włączyć podczas korzystania z modeli R1, takich jak QWQ, aby uniknąć błędu 400",
"useAzure": "Użyj Azure",

View file

@ -739,6 +739,7 @@
"cacheReadsPrice": "Preço de leituras de cache",
"cacheWritesPrice": "Preço de escritas de cache",
"enableStreaming": "Ativar streaming",
"enableStreamingDescription": "Desative o streaming se você encontrar erros de verificação de organização com modelos avançados. Solicitações sem streaming podem funcionar sem verificação.",
"enableR1Format": "Ativar parâmetros do modelo R1",
"enableR1FormatTips": "Deve ser ativado ao usar modelos R1 como QWQ, para evitar erro 400",
"useAzure": "Usar Azure",

View file

@ -739,6 +739,7 @@
"cacheReadsPrice": "Цена чтения из кэша",
"cacheWritesPrice": "Цена записи в кэш",
"enableStreaming": "Включить потоковую передачу",
"enableStreamingDescription": "Отключи потоковую передачу, если сталкиваешься с ошибками проверки организации при использовании продвинутых моделей. Запросы без потоковой передачи могут работать без проверки.",
"enableR1Format": "Включить параметры модели R1",
"enableR1FormatTips": "Необходимо включить при использовании моделей R1 (например, QWQ), чтобы избежать ошибок 400",
"useAzure": "Использовать Azure",

View file

@ -739,6 +739,7 @@
"cacheReadsPrice": "Önbellek okuma fiyatı",
"cacheWritesPrice": "Önbellek yazma fiyatı",
"enableStreaming": "Akışı etkinleştir",
"enableStreamingDescription": "Gelişmiş modellerle kuruluş doğrulama hatalarıyla karşılaşırsan akışı devre dışı bırak. Akış olmayan istekler doğrulama olmadan çalışabilir.",
"enableR1Format": "R1 model parametrelerini etkinleştir",
"enableR1FormatTips": "QWQ gibi R1 modelleri kullanıldığında etkinleştirilmelidir, 400 hatası alınmaması için",
"useAzure": "Azure kullan",

View file

@ -739,6 +739,7 @@
"cacheReadsPrice": "Giá đọc bộ nhớ đệm",
"cacheWritesPrice": "Giá ghi bộ nhớ đệm",
"enableStreaming": "Bật streaming",
"enableStreamingDescription": "Tắt streaming nếu bạn gặp lỗi xác minh tổ chức với các mô hình nâng cao. Các yêu cầu không streaming có thể hoạt động mà không cần xác minh.",
"enableR1Format": "Kích hoạt tham số mô hình R1",
"enableR1FormatTips": "Cần kích hoạt khi sử dụng các mô hình R1 như QWQ, để tránh lỗi 400",
"useAzure": "Sử dụng Azure",

View file

@ -739,6 +739,7 @@
"cacheReadsPrice": "缓存读取价格",
"cacheWritesPrice": "缓存写入价格",
"enableStreaming": "启用流式传输",
"enableStreamingDescription": "如果在使用高级模型时遇到组织验证错误,请禁用流式传输。非流式请求可能无需验证即可工作。",
"enableR1Format": "启用 R1 模型参数",
"enableR1FormatTips": "使用 QWQ 等 R1 系列模型时必须启用,避免出现 400 错误",
"useAzure": "使用 Azure 服务",

View file

@ -739,6 +739,7 @@
"cacheReadsPrice": "快取讀取價格",
"cacheWritesPrice": "快取寫入價格",
"enableStreaming": "啟用串流輸出",
"enableStreamingDescription": "如果在使用進階模型時遇到組織驗證錯誤,請停用串流輸出。非串流請求可能無需驗證即可運作。",
"enableR1Format": "啟用 R1 模型參數",
"enableR1FormatTips": "使用 QWQ 等 R1 模型時必須啟用,以避免發生 400 錯誤",
"useAzure": "使用 Azure",