fix: add separate input token limits for GPT-5 models

- Add inputTokens field to ModelInfo type to support models with separate input/output limits
- Configure GPT-5 models with correct 272K input token limit
- Update sliding window logic to respect input token limits when available
- Add comprehensive tests for the new inputTokens field functionality

Fixes #6856
This commit is contained in:
Roo Code 2025-08-08 18:37:28 +00:00
parent ad0e33e2d9
commit 019391fecf
5 changed files with 194 additions and 3 deletions

View file

@ -40,6 +40,7 @@ export const isModelParameter = (value: string): value is ModelParameter =>
export const modelInfoSchema = z.object({
maxTokens: z.number().nullish(),
maxThinkingTokens: z.number().nullish(),
inputTokens: z.number().nullish(),
contextWindow: z.number(),
supportsImages: z.boolean().optional(),
supportsComputerUse: z.boolean().optional(),

View file

@ -8,6 +8,7 @@ export const openAiNativeDefaultModelId: OpenAiNativeModelId = "gpt-5-2025-08-07
export const openAiNativeModels = {
"gpt-5-2025-08-07": {
maxTokens: 128000,
inputTokens: 272000,
contextWindow: 400000,
supportsImages: true,
supportsPromptCache: true,
@ -19,6 +20,7 @@ export const openAiNativeModels = {
},
"gpt-5-mini-2025-08-07": {
maxTokens: 128000,
inputTokens: 272000,
contextWindow: 400000,
supportsImages: true,
supportsPromptCache: true,
@ -30,6 +32,7 @@ export const openAiNativeModels = {
},
"gpt-5-nano-2025-08-07": {
maxTokens: 128000,
inputTokens: 272000,
contextWindow: 400000,
supportsImages: true,
supportsPromptCache: true,

View file

@ -1244,4 +1244,182 @@ describe("Sliding Window", () => {
expect(result2.messages.length).toBe(3) // Truncated with 0.5 fraction
})
})
/**
* Tests for inputTokens field support
*/
describe("inputTokens support", () => {
const createModelInfo = (contextWindow: number, inputTokens?: number, maxTokens?: number): ModelInfo => ({
contextWindow,
inputTokens,
supportsPromptCache: true,
maxTokens,
})
const messages: ApiMessage[] = [
{ role: "user", content: "First message" },
{ role: "assistant", content: "Second message" },
{ role: "user", content: "Third message" },
{ role: "assistant", content: "Fourth message" },
{ role: "user", content: "Fifth message" },
]
it("should use inputTokens limit when available instead of contextWindow", async () => {
// Model with separate input/output limits (like GPT-5)
const modelInfo = createModelInfo(400000, 272000, 128000)
// Create messages with very small content in the last one to avoid token overflow
const messagesWithSmallContent = [
...messages.slice(0, -1),
{ ...messages[messages.length - 1], content: "" },
]
// Set tokens to be below inputTokens limit but would exceed if using contextWindow
// inputTokens: 272000, with buffer and maxTokens: 272000 * 0.9 - 128000 = 116800
const totalTokens = 116000 // Below the inputTokens-based threshold
const result = await truncateConversationIfNeeded({
messages: messagesWithSmallContent,
totalTokens,
contextWindow: modelInfo.contextWindow,
inputTokens: modelInfo.inputTokens,
maxTokens: modelInfo.maxTokens,
apiHandler: mockApiHandler,
autoCondenseContext: false,
autoCondenseContextPercent: 100,
systemPrompt: "System prompt",
taskId,
profileThresholds: {},
currentProfileId: "default",
})
// Should not truncate as we're below the inputTokens limit
expect(result).toEqual({
messages: messagesWithSmallContent,
summary: "",
cost: 0,
prevContextTokens: totalTokens,
})
// Now test with tokens exceeding the inputTokens limit
const totalTokensExceeding = 117000 // Above the threshold
const result2 = await truncateConversationIfNeeded({
messages: messagesWithSmallContent,
totalTokens: totalTokensExceeding,
contextWindow: modelInfo.contextWindow,
inputTokens: modelInfo.inputTokens,
maxTokens: modelInfo.maxTokens,
apiHandler: mockApiHandler,
autoCondenseContext: false,
autoCondenseContextPercent: 100,
systemPrompt: "System prompt",
taskId,
profileThresholds: {},
currentProfileId: "default",
})
// Should truncate as we're above the inputTokens limit
expect(result2.messages).not.toEqual(messagesWithSmallContent)
expect(result2.messages.length).toBe(3) // Truncated with 0.5 fraction
})
it("should use inputTokens for percentage calculation when autoCondenseContext is enabled", async () => {
// Model with separate input/output limits (like GPT-5)
const modelInfo = createModelInfo(400000, 272000, 128000)
// Create messages with very small content in the last one to avoid token overflow
const messagesWithSmallContent = [
...messages.slice(0, -1),
{ ...messages[messages.length - 1], content: "" },
]
// Set tokens to 60% of inputTokens (not contextWindow)
// 60% of 272000 = 163200
const totalTokens = 163200
// Mock the summarizeConversation function
const mockSummary = "Summary based on inputTokens percentage"
const mockCost = 0.05
const mockSummarizeResponse: condenseModule.SummarizeResponse = {
messages: [
{ role: "user", content: "First message" },
{ role: "assistant", content: mockSummary, isSummary: true },
{ role: "user", content: "Last message" },
],
summary: mockSummary,
cost: mockCost,
newContextTokens: 100,
}
const summarizeSpy = vi
.spyOn(condenseModule, "summarizeConversation")
.mockResolvedValue(mockSummarizeResponse)
const result = await truncateConversationIfNeeded({
messages: messagesWithSmallContent,
totalTokens,
contextWindow: modelInfo.contextWindow,
inputTokens: modelInfo.inputTokens,
maxTokens: modelInfo.maxTokens,
apiHandler: mockApiHandler,
autoCondenseContext: true,
autoCondenseContextPercent: 50, // 50% threshold
systemPrompt: "System prompt",
taskId,
profileThresholds: {},
currentProfileId: "default",
})
// Should trigger condensing because 60% > 50% (of inputTokens, not contextWindow)
expect(summarizeSpy).toHaveBeenCalled()
expect(result).toMatchObject({
messages: mockSummarizeResponse.messages,
summary: mockSummary,
cost: mockCost,
prevContextTokens: totalTokens,
})
// Clean up
summarizeSpy.mockRestore()
})
it("should fall back to contextWindow when inputTokens is not provided", async () => {
// Model without inputTokens field (traditional model)
const modelInfo = createModelInfo(100000, undefined, 30000)
// Create messages with very small content in the last one to avoid token overflow
const messagesWithSmallContent = [
...messages.slice(0, -1),
{ ...messages[messages.length - 1], content: "" },
]
// Set tokens based on contextWindow calculation
// contextWindow: 100000, with buffer and maxTokens: 100000 * 0.9 - 30000 = 60000
const totalTokens = 59999 // Just below the contextWindow-based threshold
const result = await truncateConversationIfNeeded({
messages: messagesWithSmallContent,
totalTokens,
contextWindow: modelInfo.contextWindow,
inputTokens: modelInfo.inputTokens,
maxTokens: modelInfo.maxTokens,
apiHandler: mockApiHandler,
autoCondenseContext: false,
autoCondenseContextPercent: 100,
systemPrompt: "System prompt",
taskId,
profileThresholds: {},
currentProfileId: "default",
})
// Should not truncate as we're below the contextWindow limit
expect(result).toEqual({
messages: messagesWithSmallContent,
summary: "",
cost: 0,
prevContextTokens: totalTokens,
})
})
})
})

View file

@ -67,6 +67,7 @@ type TruncateOptions = {
messages: ApiMessage[]
totalTokens: number
contextWindow: number
inputTokens?: number | null
maxTokens?: number | null
apiHandler: ApiHandler
autoCondenseContext: boolean
@ -92,6 +93,7 @@ export async function truncateConversationIfNeeded({
messages,
totalTokens,
contextWindow,
inputTokens,
maxTokens,
apiHandler,
autoCondenseContext,
@ -118,9 +120,13 @@ export async function truncateConversationIfNeeded({
// Calculate total effective tokens (totalTokens never includes the last message)
const prevContextTokens = totalTokens + lastMessageTokens
// Use inputTokens if available (for models with separate input/output limits like GPT-5)
// Otherwise fall back to contextWindow
const effectiveInputLimit = inputTokens || contextWindow
// Calculate available tokens for conversation history
// Truncate if we're within TOKEN_BUFFER_PERCENTAGE of the context window
const allowedTokens = contextWindow * (1 - TOKEN_BUFFER_PERCENTAGE) - reservedTokens
// Truncate if we're within TOKEN_BUFFER_PERCENTAGE of the input limit
const allowedTokens = effectiveInputLimit * (1 - TOKEN_BUFFER_PERCENTAGE) - reservedTokens
// Determine the effective threshold to use
let effectiveThreshold = autoCondenseContextPercent
@ -143,7 +149,8 @@ export async function truncateConversationIfNeeded({
// If no specific threshold is found for the profile, fall back to global setting
if (autoCondenseContext) {
const contextPercent = (100 * prevContextTokens) / contextWindow
// Use inputTokens if available for percentage calculation
const contextPercent = (100 * prevContextTokens) / effectiveInputLimit
if (contextPercent >= effectiveThreshold || prevContextTokens > allowedTokens) {
// Attempt to intelligently condense the context
const result = await summarizeConversation(

View file

@ -1966,6 +1966,7 @@ export class Task extends EventEmitter<TaskEvents> implements TaskLike {
})
const contextWindow = modelInfo.contextWindow
const inputTokens = modelInfo.inputTokens
const currentProfileId =
state?.listApiConfigMeta.find((profile) => profile.name === state?.currentApiConfigName)?.id ??
@ -1976,6 +1977,7 @@ export class Task extends EventEmitter<TaskEvents> implements TaskLike {
totalTokens: contextTokens,
maxTokens,
contextWindow,
inputTokens,
apiHandler: this.api,
autoCondenseContext,
autoCondenseContextPercent,