mirror of
https://github.com/RooVetGit/Roo-Code.git
synced 2026-10-11 03:38:15 +00:00
feat: add automatic checkpoint culling for inactive tasks
- Add purgeOldCheckpoints() function with hardcoded 30-day threshold - Add startBackgroundCheckpointPurge() for fire-and-forget activation - Culls only checkpoints/ subdirectory, preserves task history - Runs silently on extension activation (no user notification) - Coexists with existing taskHistoryRetention setting - Add 6 new tests for checkpoint culling functionality
This commit is contained in:
parent
b5e3a3d898
commit
19efb33e4d
3 changed files with 389 additions and 2 deletions
|
|
@ -10,7 +10,7 @@ vi.mock("../utils/storage", () => ({
|
|||
getStorageBasePath: (p: string) => Promise.resolve(p),
|
||||
}))
|
||||
|
||||
import { purgeOldTasks } from "../utils/task-history-retention"
|
||||
import { purgeOldTasks, purgeOldCheckpoints } from "../utils/task-history-retention"
|
||||
import { GlobalFileNames } from "../shared/globalFileNames"
|
||||
|
||||
// Helpers
|
||||
|
|
@ -263,3 +263,172 @@ describe("utils/task-history-retention.ts purgeOldTasks()", () => {
|
|||
}
|
||||
})
|
||||
})
|
||||
|
||||
// Helper to create task with checkpoints
|
||||
async function createTaskWithCheckpoints(
|
||||
base: string,
|
||||
id: string,
|
||||
ts: number,
|
||||
): Promise<{ taskDir: string; checkpointsDir: string }> {
|
||||
const taskDir = path.join(base, "tasks", id)
|
||||
await fs.mkdir(taskDir, { recursive: true })
|
||||
const metadataPath = path.join(taskDir, GlobalFileNames.taskMetadata)
|
||||
const metadata = JSON.stringify({ ts }, null, 2)
|
||||
await fs.writeFile(metadataPath, metadata, "utf8")
|
||||
const checkpointsDir = path.join(taskDir, "checkpoints")
|
||||
await fs.mkdir(checkpointsDir, { recursive: true })
|
||||
// Add some checkpoint content
|
||||
await fs.writeFile(path.join(checkpointsDir, "checkpoint-1.json"), "{}", "utf8")
|
||||
return { taskDir, checkpointsDir }
|
||||
}
|
||||
|
||||
describe("utils/task-history-retention.ts purgeOldCheckpoints()", () => {
|
||||
it("culls checkpoints from tasks older than 30 days", async () => {
|
||||
const base = await mkTempBase()
|
||||
try {
|
||||
const now = Date.now()
|
||||
const days = (n: number) => n * 24 * 60 * 60 * 1000
|
||||
|
||||
// Old task (31 days) - checkpoints should be culled
|
||||
const old = await createTaskWithCheckpoints(base, "task-old", now - days(31))
|
||||
// Recent task (29 days) - checkpoints should be kept
|
||||
const recent = await createTaskWithCheckpoints(base, "task-recent", now - days(29))
|
||||
|
||||
const { culledCount } = await purgeOldCheckpoints(base, () => {}, false)
|
||||
|
||||
expect(culledCount).toBe(1)
|
||||
// Old task checkpoints should be removed, but task dir should remain
|
||||
expect(await exists(old.taskDir)).toBe(true)
|
||||
expect(await exists(old.checkpointsDir)).toBe(false)
|
||||
// Recent task should be completely intact
|
||||
expect(await exists(recent.taskDir)).toBe(true)
|
||||
expect(await exists(recent.checkpointsDir)).toBe(true)
|
||||
} finally {
|
||||
await fs.rm(base, { recursive: true, force: true })
|
||||
}
|
||||
})
|
||||
|
||||
it("does not delete checkpoints in dry run mode but reports count", async () => {
|
||||
const base = await mkTempBase()
|
||||
try {
|
||||
const now = Date.now()
|
||||
const days = (n: number) => n * 24 * 60 * 60 * 1000
|
||||
|
||||
const old = await createTaskWithCheckpoints(base, "task-old", now - days(31))
|
||||
|
||||
const { culledCount } = await purgeOldCheckpoints(base, () => {}, true)
|
||||
|
||||
expect(culledCount).toBe(1)
|
||||
// In dry run, checkpoints should still exist
|
||||
expect(await exists(old.checkpointsDir)).toBe(true)
|
||||
} finally {
|
||||
await fs.rm(base, { recursive: true, force: true })
|
||||
}
|
||||
})
|
||||
|
||||
it("skips tasks without checkpoints directory", async () => {
|
||||
const base = await mkTempBase()
|
||||
try {
|
||||
const now = Date.now()
|
||||
const days = (n: number) => n * 24 * 60 * 60 * 1000
|
||||
|
||||
// Create an old task WITHOUT checkpoints
|
||||
const taskDir = await createTask(base, "task-no-checkpoints", now - days(31))
|
||||
|
||||
const { culledCount } = await purgeOldCheckpoints(base, () => {}, false)
|
||||
|
||||
expect(culledCount).toBe(0)
|
||||
// Task should be completely intact
|
||||
expect(await exists(taskDir)).toBe(true)
|
||||
} finally {
|
||||
await fs.rm(base, { recursive: true, force: true })
|
||||
}
|
||||
})
|
||||
|
||||
it("uses mtime fallback when no metadata timestamp", async () => {
|
||||
const base = await mkTempBase()
|
||||
try {
|
||||
const now = Date.now()
|
||||
const days = (n: number) => n * 24 * 60 * 60 * 1000
|
||||
|
||||
// Create a task without metadata but with checkpoints
|
||||
const taskDir = path.join(base, "tasks", "task-no-metadata")
|
||||
await fs.mkdir(taskDir, { recursive: true })
|
||||
const checkpointsDir = path.join(taskDir, "checkpoints")
|
||||
await fs.mkdir(checkpointsDir, { recursive: true })
|
||||
await fs.writeFile(path.join(checkpointsDir, "checkpoint.json"), "{}", "utf8")
|
||||
// Set old mtime
|
||||
const oldTime = new Date(now - days(31))
|
||||
await fs.utimes(taskDir, oldTime, oldTime)
|
||||
|
||||
const { culledCount } = await purgeOldCheckpoints(base, () => {}, false)
|
||||
|
||||
expect(culledCount).toBe(1)
|
||||
// Task dir should remain, checkpoints should be gone
|
||||
expect(await exists(taskDir)).toBe(true)
|
||||
expect(await exists(checkpointsDir)).toBe(false)
|
||||
} finally {
|
||||
await fs.rm(base, { recursive: true, force: true })
|
||||
}
|
||||
})
|
||||
|
||||
it("always uses 30-day hardcoded cutoff", async () => {
|
||||
const base = await mkTempBase()
|
||||
try {
|
||||
const now = Date.now()
|
||||
const days = (n: number) => n * 24 * 60 * 60 * 1000
|
||||
|
||||
// Tasks at various ages around the 30-day boundary
|
||||
const day29 = await createTaskWithCheckpoints(base, "task-29d", now - days(29))
|
||||
const day30 = await createTaskWithCheckpoints(base, "task-30d", now - days(30))
|
||||
const day31 = await createTaskWithCheckpoints(base, "task-31d", now - days(31))
|
||||
|
||||
const { culledCount, cutoff } = await purgeOldCheckpoints(base, () => {}, false)
|
||||
|
||||
// Check cutoff is approximately 30 days ago
|
||||
const expectedCutoff = now - days(30)
|
||||
expect(cutoff).toBeGreaterThan(expectedCutoff - 1000) // Allow 1 second margin
|
||||
expect(cutoff).toBeLessThan(expectedCutoff + 1000)
|
||||
|
||||
// 29 day task should keep checkpoints (younger than 30 days)
|
||||
expect(await exists(day29.checkpointsDir)).toBe(true)
|
||||
// 30 and 31 day tasks should lose checkpoints (>= 30 days)
|
||||
expect(await exists(day30.checkpointsDir)).toBe(false)
|
||||
expect(await exists(day31.checkpointsDir)).toBe(false)
|
||||
expect(culledCount).toBe(2)
|
||||
} finally {
|
||||
await fs.rm(base, { recursive: true, force: true })
|
||||
}
|
||||
})
|
||||
|
||||
it("preserves task metadata and other files when culling checkpoints", async () => {
|
||||
const base = await mkTempBase()
|
||||
try {
|
||||
const now = Date.now()
|
||||
const days = (n: number) => n * 24 * 60 * 60 * 1000
|
||||
|
||||
// Create task with checkpoints and other content
|
||||
const taskDir = path.join(base, "tasks", "task-with-content")
|
||||
await fs.mkdir(taskDir, { recursive: true })
|
||||
const metadataPath = path.join(taskDir, GlobalFileNames.taskMetadata)
|
||||
await fs.writeFile(metadataPath, JSON.stringify({ ts: now - days(31) }), "utf8")
|
||||
const checkpointsDir = path.join(taskDir, "checkpoints")
|
||||
await fs.mkdir(checkpointsDir, { recursive: true })
|
||||
await fs.writeFile(path.join(checkpointsDir, "checkpoint.json"), "{}", "utf8")
|
||||
// Add conversation history
|
||||
await fs.writeFile(path.join(taskDir, "conversation.json"), "[]", "utf8")
|
||||
|
||||
const { culledCount } = await purgeOldCheckpoints(base, () => {}, false)
|
||||
|
||||
expect(culledCount).toBe(1)
|
||||
// Task dir and metadata should remain
|
||||
expect(await exists(taskDir)).toBe(true)
|
||||
expect(await exists(metadataPath)).toBe(true)
|
||||
expect(await exists(path.join(taskDir, "conversation.json"))).toBe(true)
|
||||
// Only checkpoints should be removed
|
||||
expect(await exists(checkpointsDir)).toBe(false)
|
||||
} finally {
|
||||
await fs.rm(base, { recursive: true, force: true })
|
||||
}
|
||||
})
|
||||
})
|
||||
|
|
|
|||
|
|
@ -44,7 +44,7 @@ import {
|
|||
} from "./activate"
|
||||
import { initializeI18n } from "./i18n"
|
||||
import { flushModels, initializeModelCacheRefresh, refreshModels } from "./api/providers/fetchers/modelCache"
|
||||
import { startBackgroundRetentionPurge } from "./utils/task-history-retention"
|
||||
import { startBackgroundRetentionPurge, startBackgroundCheckpointPurge } from "./utils/task-history-retention"
|
||||
import { TASK_HISTORY_RETENTION_OPTIONS, type TaskHistoryRetentionSetting } from "@roo-code/types"
|
||||
|
||||
/**
|
||||
|
|
@ -410,6 +410,13 @@ export async function activate(context: vscode.ExtensionContext) {
|
|||
})
|
||||
}
|
||||
|
||||
// Checkpoint culling (runs in background after activation)
|
||||
// Automatically removes checkpoints from tasks not touched in 30 days (non-configurable)
|
||||
startBackgroundCheckpointPurge({
|
||||
globalStoragePath: contextProxy.globalStorageUri.fsPath,
|
||||
log: (m) => outputChannel.appendLine(m),
|
||||
})
|
||||
|
||||
// Implements the `RooCodeAPI` interface.
|
||||
const socketPath = process.env.ROO_CODE_IPC_SOCKET_PATH
|
||||
const enableLogging = typeof socketPath === "string"
|
||||
|
|
|
|||
|
|
@ -17,12 +17,20 @@ export type PurgeResult = {
|
|||
cutoff: number | null
|
||||
}
|
||||
|
||||
export type CheckpointPurgeResult = {
|
||||
culledCount: number
|
||||
cutoff: number
|
||||
}
|
||||
|
||||
/** Concurrency limit for parallel metadata reads */
|
||||
const METADATA_READ_CONCURRENCY = 50
|
||||
|
||||
/** Concurrency limit for parallel task deletions */
|
||||
const DELETION_CONCURRENCY = 10
|
||||
|
||||
/** Hardcoded checkpoint retention: 30 days */
|
||||
const CHECKPOINT_RETENTION_DAYS = 30
|
||||
|
||||
/**
|
||||
* Task metadata read result for batch processing
|
||||
*/
|
||||
|
|
@ -388,3 +396,206 @@ export function startBackgroundRetentionPurge(options: BackgroundPurgeOptions):
|
|||
}
|
||||
})()
|
||||
}
|
||||
|
||||
/**
|
||||
* Metadata result for checkpoint culling - simplified from TaskMetadata
|
||||
*/
|
||||
interface CheckpointTaskMetadata {
|
||||
taskId: string
|
||||
taskDir: string
|
||||
checkpointsDir: string
|
||||
lastActivity: number | null // ts or mtime
|
||||
}
|
||||
|
||||
/**
|
||||
* Read metadata for checkpoint culling - only needs task age, not orphan detection
|
||||
*/
|
||||
async function readCheckpointTaskMetadata(taskId: string, tasksDir: string): Promise<CheckpointTaskMetadata | null> {
|
||||
const taskDir = path.join(tasksDir, taskId)
|
||||
const checkpointsDir = path.join(taskDir, "checkpoints")
|
||||
|
||||
// First check if checkpoints directory exists
|
||||
if (!(await pathExists(checkpointsDir))) {
|
||||
return null // No checkpoints to cull
|
||||
}
|
||||
|
||||
const metadataPath = path.join(taskDir, GlobalFileNames.taskMetadata)
|
||||
let lastActivity: number | null = null
|
||||
|
||||
// Try to read timestamp from metadata file
|
||||
try {
|
||||
const raw = await fs.readFile(metadataPath, "utf8")
|
||||
const meta: unknown = JSON.parse(raw)
|
||||
const maybeTs = Number(
|
||||
typeof meta === "object" && meta !== null && "ts" in meta ? (meta as { ts: unknown }).ts : undefined,
|
||||
)
|
||||
if (Number.isFinite(maybeTs)) {
|
||||
lastActivity = maybeTs
|
||||
}
|
||||
} catch {
|
||||
// Missing or invalid metadata
|
||||
}
|
||||
|
||||
// Fallback to mtime if no valid ts
|
||||
if (lastActivity === null) {
|
||||
try {
|
||||
const stat = await fs.stat(taskDir)
|
||||
lastActivity = stat.mtime.getTime()
|
||||
} catch {
|
||||
// Can't determine age - skip this task
|
||||
return null
|
||||
}
|
||||
}
|
||||
|
||||
return { taskId, taskDir, checkpointsDir, lastActivity }
|
||||
}
|
||||
|
||||
/**
|
||||
* Cull checkpoints from tasks that haven't been touched in CHECKPOINT_RETENTION_DAYS.
|
||||
* This is a non-configurable, always-on feature that removes only the checkpoints/
|
||||
* subdirectory while preserving the task itself (conversation history, metadata).
|
||||
*
|
||||
* @param globalStoragePath VS Code global storage fsPath
|
||||
* @param log Optional logger
|
||||
* @param dryRun When true, logs which checkpoints would be deleted but does not delete
|
||||
* @returns CheckpointPurgeResult with count and cutoff used
|
||||
*/
|
||||
export async function purgeOldCheckpoints(
|
||||
globalStoragePath: string,
|
||||
log?: (message: string) => void,
|
||||
dryRun: boolean = false,
|
||||
): Promise<CheckpointPurgeResult> {
|
||||
const cutoff = Date.now() - CHECKPOINT_RETENTION_DAYS * 24 * 60 * 60 * 1000
|
||||
|
||||
log?.(`[Checkpoints] Starting checkpoint cull (${CHECKPOINT_RETENTION_DAYS} days)${dryRun ? " (dry run)" : ""}`)
|
||||
|
||||
let basePath: string
|
||||
|
||||
try {
|
||||
basePath = await getStorageBasePath(globalStoragePath)
|
||||
} catch (e) {
|
||||
log?.(`[Checkpoints] Failed to resolve storage base path: ${e instanceof Error ? e.message : String(e)}`)
|
||||
return { culledCount: 0, cutoff }
|
||||
}
|
||||
|
||||
const tasksDir = path.join(basePath, "tasks")
|
||||
|
||||
let entries: Dirent[]
|
||||
try {
|
||||
entries = await fs.readdir(tasksDir, { withFileTypes: true })
|
||||
} catch {
|
||||
// No tasks directory yet or unreadable
|
||||
log?.(`[Checkpoints] Tasks directory not found or unreadable`)
|
||||
return { culledCount: 0, cutoff }
|
||||
}
|
||||
|
||||
const taskDirs = entries.filter((d) => d.isDirectory())
|
||||
const totalTasks = taskDirs.length
|
||||
|
||||
if (totalTasks === 0) {
|
||||
return { culledCount: 0, cutoff }
|
||||
}
|
||||
|
||||
// Phase 1: Read metadata for all tasks with checkpoints
|
||||
const metadataLimit = pLimit(METADATA_READ_CONCURRENCY)
|
||||
const metadataResults = await Promise.all(
|
||||
taskDirs.map((d) => metadataLimit(() => readCheckpointTaskMetadata(d.name, tasksDir))),
|
||||
)
|
||||
|
||||
// Phase 2: Filter tasks with checkpoints that need culling
|
||||
const tasksToCull: CheckpointTaskMetadata[] = []
|
||||
|
||||
for (const metadata of metadataResults) {
|
||||
if (!metadata) continue
|
||||
|
||||
// Check if task is older than cutoff
|
||||
if (metadata.lastActivity !== null && metadata.lastActivity < cutoff) {
|
||||
tasksToCull.push(metadata)
|
||||
}
|
||||
}
|
||||
|
||||
if (tasksToCull.length === 0) {
|
||||
log?.(`[Checkpoints] No checkpoints met cull criteria`)
|
||||
return { culledCount: 0, cutoff }
|
||||
}
|
||||
|
||||
// Dry run mode
|
||||
if (dryRun) {
|
||||
for (const metadata of tasksToCull) {
|
||||
log?.(
|
||||
`[Checkpoints][DRY RUN] Would cull checkpoints for task ${metadata.taskId} (last activity: ${new Date(metadata.lastActivity!).toISOString()})`,
|
||||
)
|
||||
}
|
||||
log?.(`[Checkpoints] Would cull checkpoints from ${tasksToCull.length} task(s) (dry run)`)
|
||||
return { culledCount: tasksToCull.length, cutoff }
|
||||
}
|
||||
|
||||
// Phase 3: Delete checkpoints directories in parallel
|
||||
const deleteLimit = pLimit(DELETION_CONCURRENCY)
|
||||
const deleteResults = await Promise.all(
|
||||
tasksToCull.map((metadata) =>
|
||||
deleteLimit(async (): Promise<boolean> => {
|
||||
try {
|
||||
await fs.rm(metadata.checkpointsDir, { recursive: true, force: true })
|
||||
const stillExists = await pathExists(metadata.checkpointsDir)
|
||||
if (!stillExists) {
|
||||
return true
|
||||
}
|
||||
} catch {
|
||||
// Ignore errors
|
||||
}
|
||||
return false
|
||||
}),
|
||||
),
|
||||
)
|
||||
|
||||
const culled = deleteResults.filter(Boolean).length
|
||||
|
||||
if (culled > 0) {
|
||||
log?.(`[Checkpoints] Culled checkpoints from ${culled} task(s); cutoff=${new Date(cutoff).toISOString()}`)
|
||||
}
|
||||
|
||||
return { culledCount: culled, cutoff }
|
||||
}
|
||||
|
||||
/**
|
||||
* Options for starting the background checkpoint purge.
|
||||
*/
|
||||
export interface BackgroundCheckpointPurgeOptions {
|
||||
/** VS Code global storage fsPath */
|
||||
globalStoragePath: string
|
||||
/** Logger function */
|
||||
log: (message: string) => void
|
||||
}
|
||||
|
||||
/**
|
||||
* Starts the checkpoint culling in the background.
|
||||
* This function is designed to be called after extension activation completes,
|
||||
* using a fire-and-forget pattern (void) to avoid blocking activation.
|
||||
*
|
||||
* Checkpoints are culled from tasks that haven't been touched in 30 days.
|
||||
* This is non-configurable and always runs.
|
||||
*
|
||||
* @param options Configuration options for the background checkpoint purge
|
||||
*/
|
||||
export function startBackgroundCheckpointPurge(options: BackgroundCheckpointPurgeOptions): void {
|
||||
const { globalStoragePath, log } = options
|
||||
|
||||
void (async () => {
|
||||
try {
|
||||
log(`[Checkpoints] Starting background checkpoint cull (${CHECKPOINT_RETENTION_DAYS} days)`)
|
||||
|
||||
const result = await purgeOldCheckpoints(globalStoragePath, log, false)
|
||||
|
||||
log(
|
||||
`[Checkpoints] Background checkpoint cull complete: culled=${result.culledCount}, cutoff=${new Date(result.cutoff).toISOString()}`,
|
||||
)
|
||||
|
||||
// No user notification - silent operation as requested
|
||||
} catch (error) {
|
||||
log(
|
||||
`[Checkpoints] Failed during background checkpoint cull: ${error instanceof Error ? error.message : String(error)}`,
|
||||
)
|
||||
}
|
||||
})()
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue