From 40ad09478b849a79694265b12a9c9b979791986b Mon Sep 17 00:00:00 2001 From: Roo Code Date: Sun, 27 Jul 2025 15:54:19 +0000 Subject: [PATCH] feat: implement token-based file reading to prevent context exhaustion - Add maxReadFileTokens setting to global settings schema (default: 10000) - Update extractTextFromFile to use token counting with tiktoken - Modify UI components to show token-based settings instead of line-based - Update all callers of extractTextFromFile to pass the new parameter - Add comprehensive tests for token-based truncation - Token-based truncation takes precedence over line-based when both are set This addresses issue #6274 by preventing context window exhaustion when reading files with very long lines that contain many tokens. --- packages/types/src/global-settings.ts | 2 + src/core/mentions/index.ts | 11 +- src/core/tools/readFileTool.ts | 11 +- src/core/webview/ClineProvider.ts | 3 + src/core/webview/webviewMessageHandler.ts | 4 + .../extract-text-token-based.spec.ts | 190 ++++++++++++++++++ src/integrations/misc/extract-text.ts | 90 ++++++++- src/shared/ExtensionMessage.ts | 3 + src/shared/WebviewMessage.ts | 1 + .../settings/ContextManagementSettings.tsx | 23 ++- .../src/components/settings/SettingsView.tsx | 3 + .../src/context/ExtensionStateContext.tsx | 4 + webview-ui/src/i18n/locales/en/settings.json | 6 + 13 files changed, 330 insertions(+), 21 deletions(-) create mode 100644 src/integrations/misc/__tests__/extract-text-token-based.spec.ts diff --git a/packages/types/src/global-settings.ts b/packages/types/src/global-settings.ts index d5e76eccea..ee87fd3a14 100644 --- a/packages/types/src/global-settings.ts +++ b/packages/types/src/global-settings.ts @@ -101,6 +101,7 @@ export const globalSettingsSchema = z.object({ maxWorkspaceFiles: z.number().optional(), showRooIgnoredFiles: z.boolean().optional(), maxReadFileLine: z.number().optional(), + maxReadFileTokens: z.number().optional(), terminalOutputLineLimit: z.number().optional(), terminalOutputCharacterLimit: z.number().optional(), @@ -273,6 +274,7 @@ export const EVALS_SETTINGS: RooCodeSettings = { maxWorkspaceFiles: 200, showRooIgnoredFiles: true, maxReadFileLine: -1, // -1 to enable full file reading. + maxReadFileTokens: -1, // -1 to enable full file reading. includeDiagnosticMessages: true, maxDiagnosticMessages: 50, diff --git a/src/core/mentions/index.ts b/src/core/mentions/index.ts index 494fbae4fc..8cdbb0ebb0 100644 --- a/src/core/mentions/index.ts +++ b/src/core/mentions/index.ts @@ -84,6 +84,7 @@ export async function parseMentions( includeDiagnosticMessages: boolean = true, maxDiagnosticMessages: number = 50, maxReadFileLine?: number, + maxReadFileTokens?: number, ): Promise { const mentions: Set = new Set() const commandMentions: Set = new Set() @@ -166,6 +167,7 @@ export async function parseMentions( rooIgnoreController, showRooIgnoredFiles, maxReadFileLine, + maxReadFileTokens, ) if (mention.endsWith("/")) { parsedText += `\n\n\n${content}\n` @@ -244,6 +246,7 @@ async function getFileOrFolderContent( rooIgnoreController?: any, showRooIgnoredFiles: boolean = true, maxReadFileLine?: number, + maxReadFileTokens?: number, ): Promise { const unescapedPath = unescapeSpaces(mentionPath) const absPath = path.resolve(cwd, unescapedPath) @@ -256,7 +259,7 @@ async function getFileOrFolderContent( return `(File ${mentionPath} is ignored by .rooignore)` } try { - const content = await extractTextFromFile(absPath, maxReadFileLine) + const content = await extractTextFromFile(absPath, maxReadFileLine, maxReadFileTokens) return content } catch (error) { return `(Failed to read contents of ${mentionPath}): ${error.message}` @@ -296,7 +299,11 @@ async function getFileOrFolderContent( if (isBinary) { return undefined } - const content = await extractTextFromFile(absoluteFilePath, maxReadFileLine) + const content = await extractTextFromFile( + absoluteFilePath, + maxReadFileLine, + maxReadFileTokens, + ) return `\n${content}\n` } catch (error) { return undefined diff --git a/src/core/tools/readFileTool.ts b/src/core/tools/readFileTool.ts index 6de8dd5642..6f5f63ccb8 100644 --- a/src/core/tools/readFileTool.ts +++ b/src/core/tools/readFileTool.ts @@ -253,7 +253,8 @@ export async function readFileTool( // Handle batch approval if there are multiple files to approve if (filesToApprove.length > 1) { - const { maxReadFileLine = -1 } = (await cline.providerRef.deref()?.getState()) ?? {} + const { maxReadFileLine = -1, maxReadFileTokens = 10000 } = + (await cline.providerRef.deref()?.getState()) ?? {} // Prepare batch file data const batchFiles = filesToApprove.map((fileResult) => { @@ -368,7 +369,8 @@ export async function readFileTool( const relPath = fileResult.path const fullPath = path.resolve(cline.cwd, relPath) const isOutsideWorkspace = isPathOutsideWorkspace(fullPath) - const { maxReadFileLine = -1 } = (await cline.providerRef.deref()?.getState()) ?? {} + const { maxReadFileLine = -1, maxReadFileTokens = 10000 } = + (await cline.providerRef.deref()?.getState()) ?? {} // Create line snippet for approval message let lineSnippet = "" @@ -429,7 +431,8 @@ export async function readFileTool( const relPath = fileResult.path const fullPath = path.resolve(cline.cwd, relPath) - const { maxReadFileLine = -1 } = (await cline.providerRef.deref()?.getState()) ?? {} + const { maxReadFileLine = -1, maxReadFileTokens = 10000 } = + (await cline.providerRef.deref()?.getState()) ?? {} // Process approved files try { @@ -517,7 +520,7 @@ export async function readFileTool( } // Handle normal file read - const content = await extractTextFromFile(fullPath) + const content = await extractTextFromFile(fullPath, maxReadFileTokens) const lineRangeAttr = ` lines="1-${totalLines}"` let xmlInfo = totalLines > 0 ? `\n${content}\n` : `` diff --git a/src/core/webview/ClineProvider.ts b/src/core/webview/ClineProvider.ts index 6bcb85e337..3eca335eb4 100644 --- a/src/core/webview/ClineProvider.ts +++ b/src/core/webview/ClineProvider.ts @@ -1425,6 +1425,7 @@ export class ClineProvider showRooIgnoredFiles, language, maxReadFileLine, + maxReadFileTokens, terminalCompressProgressBar, historyPreviewCollapsed, cloudUserInfo, @@ -1532,6 +1533,7 @@ export class ClineProvider language: language ?? formatLanguage(vscode.env.language), renderContext: this.renderContext, maxReadFileLine: maxReadFileLine ?? -1, + maxReadFileTokens: maxReadFileTokens ?? 10000, maxConcurrentFileReads: maxConcurrentFileReads ?? 5, settingsImportedAt: this.settingsImportedAt, terminalCompressProgressBar: terminalCompressProgressBar ?? true, @@ -1702,6 +1704,7 @@ export class ClineProvider telemetrySetting: stateValues.telemetrySetting || "unset", showRooIgnoredFiles: stateValues.showRooIgnoredFiles ?? true, maxReadFileLine: stateValues.maxReadFileLine ?? -1, + maxReadFileTokens: stateValues.maxReadFileTokens ?? 10000, maxConcurrentFileReads: stateValues.maxConcurrentFileReads ?? 5, historyPreviewCollapsed: stateValues.historyPreviewCollapsed ?? false, cloudUserInfo, diff --git a/src/core/webview/webviewMessageHandler.ts b/src/core/webview/webviewMessageHandler.ts index da73c56920..60c228f622 100644 --- a/src/core/webview/webviewMessageHandler.ts +++ b/src/core/webview/webviewMessageHandler.ts @@ -1265,6 +1265,10 @@ export const webviewMessageHandler = async ( await updateGlobalState("maxReadFileLine", message.value) await provider.postStateToWebview() break + case "maxReadFileTokens": + await updateGlobalState("maxReadFileTokens", message.value) + await provider.postStateToWebview() + break case "maxConcurrentFileReads": const valueToSave = message.value // Capture the value intended for saving await updateGlobalState("maxConcurrentFileReads", valueToSave) diff --git a/src/integrations/misc/__tests__/extract-text-token-based.spec.ts b/src/integrations/misc/__tests__/extract-text-token-based.spec.ts new file mode 100644 index 0000000000..1f8befbd56 --- /dev/null +++ b/src/integrations/misc/__tests__/extract-text-token-based.spec.ts @@ -0,0 +1,190 @@ +// npx vitest run integrations/misc/__tests__/extract-text-token-based.spec.ts + +import { describe, it, expect, vi, beforeEach, Mock } from "vitest" +import * as fs from "fs/promises" +import { Anthropic } from "@anthropic-ai/sdk" +import { extractTextFromFile } from "../extract-text" +import { countFileLines } from "../line-counter" +import { readLines } from "../read-lines" +import { isBinaryFile } from "isbinaryfile" +import { countTokens } from "../../../utils/countTokens" + +// Mock all dependencies +vi.mock("fs/promises") +vi.mock("../line-counter") +vi.mock("../read-lines") +vi.mock("isbinaryfile") +vi.mock("../../../utils/countTokens") + +describe("extractTextFromFile - Token-based Truncation", () => { + // Type the mocks + const mockedFs = vi.mocked(fs) + const mockedCountFileLines = vi.mocked(countFileLines) + const mockedReadLines = vi.mocked(readLines) + const mockedIsBinaryFile = vi.mocked(isBinaryFile) + const mockedCountTokens = vi.mocked(countTokens) + + beforeEach(() => { + vi.clearAllMocks() + // Set default mock behavior + mockedFs.access.mockResolvedValue(undefined) + mockedIsBinaryFile.mockResolvedValue(false) + + // Mock countTokens to return a predictable token count + mockedCountTokens.mockImplementation(async (content: Anthropic.Messages.ContentBlockParam[]) => { + // Simulate token counting based on text content + const text = content + .filter((block) => block.type === "text") + .map((block) => (block as Anthropic.Messages.TextBlockParam).text) + .join("") + const words = text.split(/\s+/).length + return Math.floor(words * 1.5) + }) + }) + + it("should truncate files based on token count when maxReadFileTokens is provided", async () => { + const fileContent = Array(100) + .fill(null) + .map((_, i) => `Line ${i + 1}: This is a test line with some content that has multiple words`) + .join("\n") + + mockedFs.readFile.mockResolvedValue(fileContent as any) + + // Mock token counting to exceed limit after 50 lines + let tokenCount = 0 + mockedCountTokens.mockImplementation(async (content: Anthropic.Messages.ContentBlockParam[]) => { + const text = content + .filter((block) => block.type === "text") + .map((block) => (block as Anthropic.Messages.TextBlockParam).text) + .join("") + const lines = text.split("\n").length + // Each line has ~15 tokens, so 50 lines = 750 tokens + tokenCount = lines * 15 + return tokenCount + }) + + const result = await extractTextFromFile("/test/large-file.ts", -1, 750) + + // Should truncate based on tokens, not lines + expect(result).toContain("1 | Line 1:") + expect(result).toContain("[File truncated") + expect(result).toMatch(/\d+ of ~?\d+ tokens/) + }) + + it("should not truncate when token count is within limit", async () => { + const fileContent = Array(10) + .fill(null) + .map((_, i) => `Line ${i + 1}: Short content`) + .join("\n") + + mockedFs.readFile.mockResolvedValue(fileContent as any) + + // Mock token counting to stay under limit + mockedCountTokens.mockResolvedValue(100) // Well under 10000 default + + const result = await extractTextFromFile("/test/small-file.ts", -1, 10000) + + // Should include all content + expect(result).toContain(" 1 | Line 1: Short content") + expect(result).toContain("10 | Line 10: Short content") + expect(result).not.toContain("[File truncated") + }) + + it("should prioritize token-based truncation over line-based when both limits are set", async () => { + const fileContent = Array(200) + .fill(null) + .map((_, i) => `Line ${i + 1}: This line has many words to increase token count significantly`) + .join("\n") + + mockedCountFileLines.mockResolvedValue(200) + mockedFs.readFile.mockResolvedValue(fileContent as any) + + // Mock to exceed token limit before line limit + let callCount = 0 + mockedCountTokens.mockImplementation(async (content: Anthropic.Messages.ContentBlockParam[]) => { + callCount++ + const text = content + .filter((block) => block.type === "text") + .map((block) => (block as Anthropic.Messages.TextBlockParam).text) + .join("") + const lines = text.split("\n").length + // Make it exceed token limit at ~30 lines (30 * 20 = 600 tokens) + return lines * 20 + }) + + // maxReadFileLine=100, maxReadFileTokens=500 + const result = await extractTextFromFile("/test/file.ts", 100, 500) + + // Should truncate based on tokens (500), not lines (100) + expect(result).toContain("[File truncated") + expect(result).toMatch(/\d+ of ~?\d+ tokens/) + + // Should have stopped before reaching line limit + const resultLines = result.split("\n").filter((line) => line.match(/^\s*\d+\s*\|/)) + expect(resultLines.length).toBeLessThan(100) + }) + + it("should handle maxReadFileTokens of 0 by throwing an error", async () => { + await expect(extractTextFromFile("/test/file.ts", -1, 0)).rejects.toThrow( + "Invalid maxReadFileTokens: 0. Must be a positive integer or -1 for unlimited.", + ) + }) + + it("should handle negative maxReadFileTokens by throwing an error", async () => { + await expect(extractTextFromFile("/test/file.ts", -1, -100)).rejects.toThrow( + "Invalid maxReadFileTokens: -100. Must be a positive integer or -1 for unlimited.", + ) + }) + + it("should work with both line and token limits disabled", async () => { + const fileContent = "Line 1\nLine 2\nLine 3" + mockedFs.readFile.mockResolvedValue(fileContent as any) + + const result = await extractTextFromFile("/test/file.ts", -1, undefined) + + // Should include all content + expect(result).toContain("1 | Line 1") + expect(result).toContain("2 | Line 2") + expect(result).toContain("3 | Line 3") + expect(result).not.toContain("[File truncated") + }) + + it("should handle empty files with token-based truncation", async () => { + mockedFs.readFile.mockResolvedValue("" as any) + mockedCountTokens.mockResolvedValue(0) + + const result = await extractTextFromFile("/test/empty.ts", -1, 1000) + + expect(result).toBe("") + }) + + it("should efficiently handle very large token counts", async () => { + // Simulate a file that would have millions of tokens + const hugeContent = Array(10000) + .fill(null) + .map((_, i) => `Line ${i + 1}: ${Array(100).fill("word").join(" ")}`) + .join("\n") + + mockedFs.readFile.mockResolvedValue(hugeContent as any) + + // Mock progressive token counting + mockedCountTokens.mockImplementation(async (content: Anthropic.Messages.ContentBlockParam[]) => { + const text = content + .filter((block) => block.type === "text") + .map((block) => (block as Anthropic.Messages.TextBlockParam).text) + .join("") + const lines = text.split("\n").length + return lines * 150 // Each line has ~150 tokens + }) + + const result = await extractTextFromFile("/test/huge.ts", -1, 5000) + + // Should truncate early based on tokens + expect(result).toContain("[File truncated") + expect(result).toMatch(/\d+ of ~?\d+ tokens/) + + // Should have stopped processing early + const resultLines = result.split("\n").filter((line) => line.match(/^\s*\d+\s*\|/)) + expect(resultLines.length).toBeLessThan(50) // Should stop around 33 lines (5000/150) + }) +}) diff --git a/src/integrations/misc/extract-text.ts b/src/integrations/misc/extract-text.ts index 8231c609be..3af6b62515 100644 --- a/src/integrations/misc/extract-text.ts +++ b/src/integrations/misc/extract-text.ts @@ -7,6 +7,8 @@ import { isBinaryFile } from "isbinaryfile" import { extractTextFromXLSX } from "./extract-text-from-xlsx" import { countFileLines } from "./line-counter" import { readLines } from "./read-lines" +import { Tiktoken } from "tiktoken/lite" +import o200kBase from "tiktoken/encoders/o200k_base" async function extractTextFromPDF(filePath: string): Promise { const dataBuffer = await fs.readFile(filePath) @@ -50,18 +52,47 @@ export function getSupportedBinaryFormats(): string[] { return Object.keys(SUPPORTED_BINARY_FORMATS) } +// Cache the encoder instance +let encoder: Tiktoken | null = null + +/** + * Gets or creates a tiktoken encoder instance + */ +function getEncoder(): Tiktoken { + if (!encoder) { + encoder = new Tiktoken(o200kBase.bpe_ranks, o200kBase.special_tokens, o200kBase.pat_str) + } + return encoder +} + +/** + * Counts tokens in a string using tiktoken + */ +function countStringTokens(text: string): number { + if (!text) return 0 + const enc = getEncoder() + return enc.encode(text).length +} + /** * Extracts text content from a file, with support for various formats including PDF, DOCX, XLSX, and plain text. - * For large text files, can limit the number of lines read to prevent context exhaustion. + * For large text files, can limit the number of lines or tokens read to prevent context exhaustion. * * @param filePath - Path to the file to extract text from * @param maxReadFileLine - Maximum number of lines to read from text files. * Use UNLIMITED_LINES (-1) or undefined for no limit. * Must be a positive integer or UNLIMITED_LINES. + * @param maxReadFileTokens - Maximum number of tokens to read from text files. + * Use -1 or undefined for no limit. + * Must be a positive integer or -1. * @returns Promise resolving to the extracted text content with line numbers * @throws {Error} If file not found, unsupported format, or invalid parameters */ -export async function extractTextFromFile(filePath: string, maxReadFileLine?: number): Promise { +export async function extractTextFromFile( + filePath: string, + maxReadFileLine?: number, + maxReadFileTokens?: number, +): Promise { // Validate maxReadFileLine parameter if (maxReadFileLine !== undefined && maxReadFileLine !== -1) { if (!Number.isInteger(maxReadFileLine) || maxReadFileLine < 1) { @@ -71,6 +102,15 @@ export async function extractTextFromFile(filePath: string, maxReadFileLine?: nu } } + // Validate maxReadFileTokens parameter + if (maxReadFileTokens !== undefined && maxReadFileTokens !== -1) { + if (!Number.isInteger(maxReadFileTokens) || maxReadFileTokens < 1) { + throw new Error( + `Invalid maxReadFileTokens: ${maxReadFileTokens}. Must be a positive integer or -1 for unlimited.`, + ) + } + } + try { await fs.access(filePath) } catch (error) { @@ -89,8 +129,48 @@ export async function extractTextFromFile(filePath: string, maxReadFileLine?: nu const isBinary = await isBinaryFile(filePath).catch(() => false) if (!isBinary) { - // Check if we need to apply line limit - if (maxReadFileLine !== undefined && maxReadFileLine !== -1) { + // Check if we need to apply token limit first (takes precedence) + if (maxReadFileTokens !== undefined && maxReadFileTokens !== -1) { + const fullContent = await fs.readFile(filePath, "utf8") + const totalTokens = countStringTokens(fullContent) + + if (totalTokens > maxReadFileTokens) { + // Need to truncate based on tokens + const lines = fullContent.split("\n") + let accumulatedContent = "" + let accumulatedTokens = 0 + let lineCount = 0 + + // Add lines until we exceed the token limit + for (const line of lines) { + const lineWithNewline = line + "\n" + const lineTokens = countStringTokens(lineWithNewline) + + if (accumulatedTokens + lineTokens > maxReadFileTokens && lineCount > 0) { + // Would exceed limit, stop here + break + } + + accumulatedContent += lineWithNewline + accumulatedTokens += lineTokens + lineCount++ + } + + // Remove trailing newline if present + if (accumulatedContent.endsWith("\n")) { + accumulatedContent = accumulatedContent.slice(0, -1) + } + + const numberedContent = addLineNumbers(accumulatedContent) + const totalLines = await countFileLines(filePath) + return ( + numberedContent + + `\n\n[File truncated: showing ${lineCount} of ${totalLines} lines (${accumulatedTokens} of ~${totalTokens} tokens). The file is too large and may exhaust the context window if read in full.]` + ) + } + } + // If no token limit or within token limit, check line limit + else if (maxReadFileLine !== undefined && maxReadFileLine !== -1) { const totalLines = await countFileLines(filePath) if (totalLines > maxReadFileLine) { // Read only up to maxReadFileLine (endLine is 0-based and inclusive) @@ -102,7 +182,7 @@ export async function extractTextFromFile(filePath: string, maxReadFileLine?: nu ) } } - // Read the entire file if no limit or file is within limit + // Read the entire file if no limit or file is within limits return addLineNumbers(await fs.readFile(filePath, "utf8")) } else { throw new Error(`Cannot read text for file type: ${fileExtension}`) diff --git a/src/shared/ExtensionMessage.ts b/src/shared/ExtensionMessage.ts index 816069f91f..bf4383e4b5 100644 --- a/src/shared/ExtensionMessage.ts +++ b/src/shared/ExtensionMessage.ts @@ -94,6 +94,7 @@ export interface ExtensionMessage { | "ttsStart" | "ttsStop" | "maxReadFileLine" + | "maxReadFileTokens" | "fileSearchResults" | "toggleApiConfigPin" | "acceptInput" @@ -230,6 +231,7 @@ export type ExtensionState = Pick< // | "maxWorkspaceFiles" // Optional in GlobalSettings, required here. // | "showRooIgnoredFiles" // Optional in GlobalSettings, required here. // | "maxReadFileLine" // Optional in GlobalSettings, required here. + // | "maxReadFileTokens" // Optional in GlobalSettings, required here. | "maxConcurrentFileReads" // Optional in GlobalSettings, required here. | "terminalOutputLineLimit" | "terminalOutputCharacterLimit" @@ -281,6 +283,7 @@ export type ExtensionState = Pick< maxWorkspaceFiles: number // Maximum number of files to include in current working directory details (0-500) showRooIgnoredFiles: boolean // Whether to show .rooignore'd files in listings maxReadFileLine: number // Maximum number of lines to read from a file before truncating + maxReadFileTokens: number // Maximum number of tokens to read from a file before truncating experiments: Experiments // Map of experiment IDs to their enabled state diff --git a/src/shared/WebviewMessage.ts b/src/shared/WebviewMessage.ts index 1304e4c7d5..9a78a712c7 100644 --- a/src/shared/WebviewMessage.ts +++ b/src/shared/WebviewMessage.ts @@ -162,6 +162,7 @@ export interface WebviewMessage { | "remoteBrowserEnabled" | "language" | "maxReadFileLine" + | "maxReadFileTokens" | "maxConcurrentFileReads" | "includeDiagnosticMessages" | "maxDiagnosticMessages" diff --git a/webview-ui/src/components/settings/ContextManagementSettings.tsx b/webview-ui/src/components/settings/ContextManagementSettings.tsx index 4530fdb1ba..4697905d73 100644 --- a/webview-ui/src/components/settings/ContextManagementSettings.tsx +++ b/webview-ui/src/components/settings/ContextManagementSettings.tsx @@ -20,6 +20,7 @@ type ContextManagementSettingsProps = HTMLAttributes & { maxWorkspaceFiles: number showRooIgnoredFiles?: boolean maxReadFileLine?: number + maxReadFileTokens?: number maxConcurrentFileReads?: number profileThresholds?: Record includeDiagnosticMessages?: boolean @@ -32,6 +33,7 @@ type ContextManagementSettingsProps = HTMLAttributes & { | "maxWorkspaceFiles" | "showRooIgnoredFiles" | "maxReadFileLine" + | "maxReadFileTokens" | "maxConcurrentFileReads" | "profileThresholds" | "includeDiagnosticMessages" @@ -49,6 +51,7 @@ export const ContextManagementSettings = ({ showRooIgnoredFiles, setCachedStateField, maxReadFileLine, + maxReadFileTokens, maxConcurrentFileReads, profileThresholds = {}, includeDiagnosticMessages, @@ -172,37 +175,37 @@ export const ContextManagementSettings = ({
- {t("settings:contextManagement.maxReadFile.label")} + {t("settings:contextManagement.maxReadFileTokens.label")}
{ const newValue = parseInt(e.target.value, 10) if (!isNaN(newValue) && newValue >= -1) { - setCachedStateField("maxReadFileLine", newValue) + setCachedStateField("maxReadFileTokens", newValue) } }} onClick={(e) => e.currentTarget.select()} - data-testid="max-read-file-line-input" - disabled={maxReadFileLine === -1} + data-testid="max-read-file-tokens-input" + disabled={maxReadFileTokens === -1} /> - {t("settings:contextManagement.maxReadFile.lines")} + {t("settings:contextManagement.maxReadFileTokens.tokens")} - setCachedStateField("maxReadFileLine", e.target.checked ? -1 : 500) + setCachedStateField("maxReadFileTokens", e.target.checked ? -1 : 10000) } data-testid="max-read-file-always-full-checkbox"> - {t("settings:contextManagement.maxReadFile.always_full_read")} + {t("settings:contextManagement.maxReadFileTokens.always_full_read")}
- {t("settings:contextManagement.maxReadFile.description")} + {t("settings:contextManagement.maxReadFileTokens.description")}
diff --git a/webview-ui/src/components/settings/SettingsView.tsx b/webview-ui/src/components/settings/SettingsView.tsx index a416afd48d..7be63f79ce 100644 --- a/webview-ui/src/components/settings/SettingsView.tsx +++ b/webview-ui/src/components/settings/SettingsView.tsx @@ -168,6 +168,7 @@ const SettingsView = forwardRef(({ onDone, t showRooIgnoredFiles, remoteBrowserEnabled, maxReadFileLine, + maxReadFileTokens, terminalCompressProgressBar, maxConcurrentFileReads, condensingApiConfigId, @@ -321,6 +322,7 @@ const SettingsView = forwardRef(({ onDone, t vscode.postMessage({ type: "maxWorkspaceFiles", value: maxWorkspaceFiles ?? 200 }) vscode.postMessage({ type: "showRooIgnoredFiles", bool: showRooIgnoredFiles }) vscode.postMessage({ type: "maxReadFileLine", value: maxReadFileLine ?? -1 }) + vscode.postMessage({ type: "maxReadFileTokens", value: maxReadFileTokens ?? -1 }) vscode.postMessage({ type: "maxConcurrentFileReads", value: cachedState.maxConcurrentFileReads ?? 5 }) vscode.postMessage({ type: "includeDiagnosticMessages", bool: includeDiagnosticMessages }) vscode.postMessage({ type: "maxDiagnosticMessages", value: maxDiagnosticMessages ?? 50 }) @@ -667,6 +669,7 @@ const SettingsView = forwardRef(({ onDone, t maxWorkspaceFiles={maxWorkspaceFiles ?? 200} showRooIgnoredFiles={showRooIgnoredFiles} maxReadFileLine={maxReadFileLine} + maxReadFileTokens={maxReadFileTokens} maxConcurrentFileReads={maxConcurrentFileReads} profileThresholds={profileThresholds} includeDiagnosticMessages={includeDiagnosticMessages} diff --git a/webview-ui/src/context/ExtensionStateContext.tsx b/webview-ui/src/context/ExtensionStateContext.tsx index cb51fe9c9a..dfc03f74f0 100644 --- a/webview-ui/src/context/ExtensionStateContext.tsx +++ b/webview-ui/src/context/ExtensionStateContext.tsx @@ -121,6 +121,8 @@ export interface ExtensionStateContextType extends ExtensionState { setAwsUsePromptCache: (value: boolean) => void maxReadFileLine: number setMaxReadFileLine: (value: number) => void + maxReadFileTokens: number + setMaxReadFileTokens: (value: number) => void machineId?: string pinnedApiConfigs?: Record setPinnedApiConfigs: (value: Record) => void @@ -209,6 +211,7 @@ export const ExtensionStateContextProvider: React.FC<{ children: React.ReactNode showRooIgnoredFiles: true, // Default to showing .rooignore'd files with lock symbol (current behavior). renderContext: "sidebar", maxReadFileLine: -1, // Default max read file line limit + maxReadFileTokens: -1, // Default max read file token limit pinnedApiConfigs: {}, // Empty object for pinned API configs terminalZshOhMy: false, // Default Oh My Zsh integration setting maxConcurrentFileReads: 5, // Default concurrent file reads @@ -455,6 +458,7 @@ export const ExtensionStateContextProvider: React.FC<{ children: React.ReactNode setRemoteBrowserEnabled: (value) => setState((prevState) => ({ ...prevState, remoteBrowserEnabled: value })), setAwsUsePromptCache: (value) => setState((prevState) => ({ ...prevState, awsUsePromptCache: value })), setMaxReadFileLine: (value) => setState((prevState) => ({ ...prevState, maxReadFileLine: value })), + setMaxReadFileTokens: (value) => setState((prevState) => ({ ...prevState, maxReadFileTokens: value })), setPinnedApiConfigs: (value) => setState((prevState) => ({ ...prevState, pinnedApiConfigs: value })), setTerminalCompressProgressBar: (value) => setState((prevState) => ({ ...prevState, terminalCompressProgressBar: value })), diff --git a/webview-ui/src/i18n/locales/en/settings.json b/webview-ui/src/i18n/locales/en/settings.json index 2b4d3b8fe0..21de04492a 100644 --- a/webview-ui/src/i18n/locales/en/settings.json +++ b/webview-ui/src/i18n/locales/en/settings.json @@ -502,6 +502,12 @@ "lines": "lines", "always_full_read": "Always read entire file" }, + "maxReadFileTokens": { + "label": "File read token limit", + "description": "Maximum number of tokens to read from a file when the model omits start/end values. This provides more consistent context usage across files with varying line lengths. Special cases: -1 instructs Roo to read the entire file (without truncation). Lower values minimize initial context usage, enabling more precise subsequent reads. Explicit start/end requests are not limited by this setting.", + "tokens": "tokens", + "always_full_read": "Always read entire file" + }, "diagnostics": { "includeMessages": { "label": "Automatically include diagnostics in context",