diff --git a/packages/memory-graph/scripts/bench-cluster-assignments-v8.mjs b/packages/memory-graph/scripts/bench-cluster-assignments-v8.mjs new file mode 100644 index 00000000..ff6b4ea1 --- /dev/null +++ b/packages/memory-graph/scripts/bench-cluster-assignments-v8.mjs @@ -0,0 +1,305 @@ +// Companion to bench-cluster-assignments.ts, for the V8 engine specifically +// (Node, and what Chrome/Edge/most browsers actually run -- i.e. where this +// component executes for real users). +// +// scripts/bench-cluster-assignments.ts imports the live computeClusterAssignments +// and is the benchmark to trust for correctness (it runs the real, current +// source). But run under Bun -- this repo's own dev/test runtime -- it shows +// close to NO difference between the old and new implementation. That is +// not a flaw in the fix: Bun's JavaScriptCore engine optimizes Array.shift() +// far better than V8 does, so the O(n^2) behavior this fix removes barely +// shows up there. Do not conclude from the .ts benchmark alone that this +// optimization is a no-op -- run this file too, with plain `node`. +// +// This script is a standalone, verbatim copy of both implementations (not +// an import), specifically so it can run under plain `node` without hitting +// Node's strict ESM extension-resolution rules on the rest of the source +// tree, and without adding a new dev dependency (e.g. tsx) just to make +// that import work. If computeClusterAssignments changes again, this file's +// copies should be refreshed to match. +// +// Usage (plain Node, not bun): +// node scripts/bench-cluster-assignments-v8.mjs + +function hashString(value) { + let hash = 0 + for (let i = 0; i < value.length; i++) { + hash = (Math.imul(31, hash) + value.charCodeAt(i)) | 0 + } + return hash >>> 0 +} + +const CLUSTER_COLORS = [ + "#58C7E8", + "#E7BC52", + "#74D680", + "#D47B75", + "#A789E8", + "#62C5A8", + "#74ABD8", + "#C78AC8", + "#D18A58", + "#8BCB6F", +] + +function getClusterColor(key) { + return CLUSTER_COLORS[hashString(key) % CLUSTER_COLORS.length] +} + +function ensureAdjacency(map, id) { + if (!map.has(id)) map.set(id, new Set()) +} + +function connect(map, a, b) { + ensureAdjacency(map, a) + ensureAdjacency(map, b) + map.get(a)?.add(b) + map.get(b)?.add(a) +} + +function getMemoryRelationTargets(mem) { + if ( + mem.memoryRelations && + typeof mem.memoryRelations === "object" && + Object.keys(mem.memoryRelations).length > 0 + ) { + return mem.memoryRelations + } + if (mem.parentMemoryId) return { [mem.parentMemoryId]: "updates" } + return {} +} + +// Verbatim (pre-fix): BFS frontier drained with Array.shift() -- O(remaining +// length) per call. +function previousImplementation(documents) { + const adjacency = new Map() + const docByMemory = new Map() + const orderByMemory = new Map() + const allMemoryIds = new Set() + let order = 0 + + for (const doc of documents) { + let firstMemoryId = null + for (const mem of doc.memories) { + allMemoryIds.add(mem.id) + docByMemory.set(mem.id, doc.id) + orderByMemory.set(mem.id, order++) + ensureAdjacency(adjacency, mem.id) + if (!firstMemoryId) firstMemoryId = mem.id + else connect(adjacency, firstMemoryId, mem.id) + } + } + + for (const doc of documents) { + for (const mem of doc.memories) { + for (const targetId of Object.keys(getMemoryRelationTargets(mem))) { + if (!allMemoryIds.has(targetId)) continue + connect(adjacency, mem.id, targetId) + } + } + } + + const assignments = new Map() + const visited = new Set() + const memoryIdsByOrder = [...allMemoryIds].sort( + (a, b) => (orderByMemory.get(a) ?? 0) - (orderByMemory.get(b) ?? 0), + ) + + for (const startId of memoryIdsByOrder) { + if (visited.has(startId)) continue + const component = [] + const queue = [startId] + visited.add(startId) + + while (queue.length > 0) { + const id = queue.shift() + component.push(id) + for (const nextId of adjacency.get(id) ?? []) { + if (visited.has(nextId)) continue + visited.add(nextId) + queue.push(nextId) + } + } + + component.sort( + (a, b) => (orderByMemory.get(a) ?? 0) - (orderByMemory.get(b) ?? 0), + ) + const firstId = component[0] ?? startId + const docIds = new Set(component.map((id) => docByMemory.get(id))) + const firstDocId = docByMemory.get(firstId) ?? "unknown" + const key = + docIds.size <= 1 + ? `doc:${firstDocId}` + : `relation:${firstDocId}:${firstId}` + const assignment = { + key, + color: getClusterColor(key), + size: component.length, + } + for (const id of component) assignments.set(id, assignment) + } + + return assignments +} + +// Optimized (current): index-pointer dequeue -- O(1) per call. Identical +// otherwise -- this is the actual diff applied to use-graph-data.ts. +function optimizedImplementation(documents) { + const adjacency = new Map() + const docByMemory = new Map() + const orderByMemory = new Map() + const allMemoryIds = new Set() + let order = 0 + + for (const doc of documents) { + let firstMemoryId = null + for (const mem of doc.memories) { + allMemoryIds.add(mem.id) + docByMemory.set(mem.id, doc.id) + orderByMemory.set(mem.id, order++) + ensureAdjacency(adjacency, mem.id) + if (!firstMemoryId) firstMemoryId = mem.id + else connect(adjacency, firstMemoryId, mem.id) + } + } + + for (const doc of documents) { + for (const mem of doc.memories) { + for (const targetId of Object.keys(getMemoryRelationTargets(mem))) { + if (!allMemoryIds.has(targetId)) continue + connect(adjacency, mem.id, targetId) + } + } + } + + const assignments = new Map() + const visited = new Set() + const memoryIdsByOrder = [...allMemoryIds].sort( + (a, b) => (orderByMemory.get(a) ?? 0) - (orderByMemory.get(b) ?? 0), + ) + + for (const startId of memoryIdsByOrder) { + if (visited.has(startId)) continue + const component = [] + const queue = [startId] + let head = 0 + visited.add(startId) + + while (head < queue.length) { + const id = queue[head++] + component.push(id) + for (const nextId of adjacency.get(id) ?? []) { + if (visited.has(nextId)) continue + visited.add(nextId) + queue.push(nextId) + } + } + + component.sort( + (a, b) => (orderByMemory.get(a) ?? 0) - (orderByMemory.get(b) ?? 0), + ) + const firstId = component[0] ?? startId + const docIds = new Set(component.map((id) => docByMemory.get(id))) + const firstDocId = docByMemory.get(firstId) ?? "unknown" + const key = + docIds.size <= 1 + ? `doc:${firstDocId}` + : `relation:${firstDocId}:${firstId}` + const assignment = { + key, + color: getClusterColor(key), + size: component.length, + } + for (const id of component) assignments.set(id, assignment) + } + + return assignments +} + +function makeMemory(id, relations) { + return { id, memoryRelations: relations ?? null, parentMemoryId: null } +} + +/** + * One hub memory + n-1 memories that `derives` from it, one per document -- + * a single connected component with a wide BFS frontier, matching the + * cross-document relation-merge path in use-graph-data.ts. + */ +function buildStarDataset(n) { + const memories = [makeMemory("mem-hub")] + for (let i = 1; i < n; i++) { + memories.push(makeMemory(`mem-${i}`, { "mem-hub": "derives" })) + } + return memories.map((mem, i) => ({ id: `doc-${i}`, memories: [mem] })) +} + +function timeOnce(fn) { + const start = performance.now() + fn() + return performance.now() - start +} + +console.log(`Engine: ${process.release?.name ?? "unknown"} ${process.version}`) +if (process.versions?.bun) { + console.log( + "WARNING: running under Bun. This is the wrong engine to judge this fix by -- run with plain `node` instead.", + ) +} + +console.log( + "\nCorrectness check: previous vs. optimized produce identical assignments", +) +{ + const dataset = buildStarDataset(2_000) + const prev = previousImplementation(dataset) + const curr = optimizedImplementation(dataset) + let mismatch = prev.size !== curr.size + if (!mismatch) { + for (const [id, assignment] of prev) { + const currAssignment = curr.get(id) + if ( + !currAssignment || + currAssignment.key !== assignment.key || + currAssignment.color !== assignment.color || + currAssignment.size !== assignment.size + ) { + mismatch = true + break + } + } + } + console.log( + mismatch + ? " MISMATCH -- do not trust these numbers" + : " OK, outputs are identical", + ) + if (mismatch) process.exit(1) +} + +const RUNS_PER_SIZE = 5 +const SIZES = [5_000, 10_000, 20_000, 40_000, 80_000, 100_000] + +console.log("\nn\tprevious (ms)\toptimized (ms)\tspeedup") +for (const n of SIZES) { + const dataset = buildStarDataset(n) + let prevBest = Number.POSITIVE_INFINITY + let currBest = Number.POSITIVE_INFINITY + for (let i = 0; i < RUNS_PER_SIZE; i++) { + let pTime + let cTime + // Alternate which implementation runs first each sample, to cancel + // out any heap-warmup/ordering bias between the two. + if (i % 2 === 0) { + pTime = timeOnce(() => previousImplementation(dataset)) + cTime = timeOnce(() => optimizedImplementation(dataset)) + } else { + cTime = timeOnce(() => optimizedImplementation(dataset)) + pTime = timeOnce(() => previousImplementation(dataset)) + } + if (pTime < prevBest) prevBest = pTime + if (cTime < currBest) currBest = cTime + } + console.log( + `${n}\t${prevBest.toFixed(2)}\t\t${currBest.toFixed(2)}\t\t${(prevBest / currBest).toFixed(2)}x`, + ) +} diff --git a/packages/memory-graph/scripts/bench-cluster-assignments.ts b/packages/memory-graph/scripts/bench-cluster-assignments.ts index d4ecc220..0a326310 100644 --- a/packages/memory-graph/scripts/bench-cluster-assignments.ts +++ b/packages/memory-graph/scripts/bench-cluster-assignments.ts @@ -21,8 +21,17 @@ * than itself" purely from GC noise on 40k-node allocations, which is why * this script uses best-of-N sampling at multiple sizes instead. * + * IMPORTANT: this repo's dev tooling runs on Bun, whose JavaScriptCore + * engine optimizes Array.shift() far better than V8 (what Chrome/Edge/most + * browsers -- i.e. this component's actual users -- run). Run under Bun, + * this benchmark will show close to NO difference between old and new. That + * does not mean the fix is a no-op -- see the companion script + * bench-cluster-assignments-v8.mjs, which is engine-independent (plain + * Node, zero dependencies) and shows the real, growing gap. + * * Usage: - * bun run scripts/bench-cluster-assignments.ts + * bun run scripts/bench-cluster-assignments.ts (correctness + Bun numbers) + * node scripts/bench-cluster-assignments-v8.mjs (V8/real-world numbers) */ import { computeClusterAssignments } from "../src/hooks/use-graph-data" import type { GraphApiDocument, GraphApiMemory } from "../src/types" @@ -229,7 +238,16 @@ function bestOf(fn: () => void, runs: number): number { } const RUNS_PER_SIZE = 5 -const SIZES = [5_000, 10_000, 20_000, 40_000, 80_000] +const SIZES = [5_000, 10_000, 20_000, 40_000, 80_000, 100_000] + +if ((process.versions as { bun?: string }).bun) { + console.log( + "WARNING: running under Bun -- its JavaScriptCore engine optimizes\n" + + "Array.shift() well, so the numbers below will look flat regardless\n" + + "of the fix. Run `node scripts/bench-cluster-assignments-v8.mjs` for\n" + + "the engine-independent, real-world (V8) comparison.\n", + ) +} console.log( "Correctness check: previous vs. current produce identical assignments",