fabro/run.json
Fabro ccfab315a2 checkpoint
⚒️ Generated with [Fabro](https://fabro.sh)
2026-05-23 11:20:14 -04:00

701 lines
No EOL
47 KiB
JSON

{
"title": "Agent Compaction API Usage Baseline Plan",
"spec": {
"run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6",
"settings": {
"project": {
"name": null,
"description": null,
"metadata": {}
},
"workflow": {
"name": null,
"description": null,
"graph": "workflow.fabro",
"metadata": {}
},
"run": {
"goal": {
"type": "inline",
"value": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n"
},
"working_dir": null,
"metadata": {},
"inputs": {},
"model": {
"provider": "anthropic",
"name": "claude-sonnet-4-6",
"fallbacks": [],
"controls": {
"reasoning_effort": null,
"speed": null
}
},
"git": {
"author": null
},
"prepare": {
"commands": [],
"timeout_ms": 300000
},
"execution": {
"mode": "normal",
"approval": "prompt"
},
"checkpoint": {
"exclude_globs": [],
"skip_git_hooks": false
},
"clone": {
"enabled": true
},
"run_branch": {
"enabled": true,
"push": true
},
"meta_branch": {
"enabled": true,
"push": true
},
"sandbox": {
"provider": "daytona",
"preserve": false,
"stop_on_terminal": true,
"devcontainer": false,
"env": {},
"docker": {
"image": "buildpack-deps:noble",
"network_mode": null,
"memory_limit": 4000000000,
"cpu_quota": 200000,
"env_vars": {}
},
"daytona": {
"auto_stop_interval": 30,
"labels": {
"repo": "fabro-sh/fabro"
},
"volumes": [],
"snapshot": {
"name": "fabro-v11",
"cpu": 8,
"memory_gb": 16,
"disk_gb": 20,
"dockerfile": {
"type": "inline",
"value": "FROM ubuntu:24.04\n\nRUN apt-get update && apt-get install -y --no-install-recommends \\\n curl git ca-certificates build-essential pkg-config libssl-dev unzip python3 \\\n xvfb xfce4 xfce4-terminal x11vnc novnc dbus-x11 \\\n libx11-6 libxrandr2 libxext6 libxrender1 libxfixes3 libxss1 libxtst6 libxi6 \\\n && rm -rf /var/lib/apt/lists/*\n\n# Install real Chromium (not the snap stub) via xtradeb PPA\nRUN apt-get update && apt-get install -y --no-install-recommends \\\n software-properties-common curl gnupg \\\n && add-apt-repository -y ppa:xtradeb/apps \\\n && apt-get update \\\n && apt-get install -y --no-install-recommends chromium \\\n && rm -rf /var/lib/apt/lists/*\n\n# Wrapper: Chromium needs --no-sandbox when running as root in a container,\n# and --disable-dev-shm-usage avoids crashes from small /dev/shm\nRUN printf '#!/bin/bash\\nexec /usr/bin/chromium --no-sandbox --disable-dev-shm-usage \"$@\"\\n' \\\n > /usr/local/bin/chromium-wrapper \\\n && chmod +x /usr/local/bin/chromium-wrapper\n\n# Make the wrapper the default in the system .desktop file and via alternatives\nRUN sed -i 's|^Exec=.*|Exec=/usr/local/bin/chromium-wrapper %U|' \\\n /usr/share/applications/chromium.desktop \\\n && update-alternatives --install /usr/bin/x-www-browser x-www-browser \\\n /usr/local/bin/chromium-wrapper 100\n\n# Tell XFCE's exo-open that Chromium is the WebBrowser helper (system-wide)\nRUN mkdir -p /etc/xdg/xfce4 /usr/share/xfce4/helpers \\\n && printf 'WebBrowser=custom-WebBrowser\\n' > /etc/xdg/xfce4/helpers.rc \\\n && printf '[Desktop Entry]\\n\\\nVersion=1.0\\n\\\nType=X-XFCE-Helper\\n\\\nName=Chromium\\n\\\nIcon=chromium\\n\\\nX-XFCE-Category=WebBrowser\\n\\\nX-XFCE-CommandsWithParameter=/usr/local/bin/chromium-wrapper \"%%s\"\\n\\\nX-XFCE-Commands=/usr/local/bin/chromium-wrapper\\n' \\\n > /usr/share/xfce4/helpers/custom-WebBrowser.desktop\n\n# GitHub CLI\nRUN curl -fsSL https://cli.github.com/packages/githubcli-archive-keyring.gpg \\\n | dd of=/usr/share/keyrings/githubcli-archive-keyring.gpg \\\n && echo \"deb [arch=$(dpkg --print-architecture) signed-by=/usr/share/keyrings/githubcli-archive-keyring.gpg] https://cli.github.com/packages stable main\" \\\n | tee /etc/apt/sources.list.d/github-cli.list > /dev/null \\\n && apt-get update && apt-get install -y --no-install-recommends gh \\\n && rm -rf /var/lib/apt/lists/*\n\n# Rust\nRUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y\nENV PATH=\"/root/.cargo/bin:${PATH}\"\nRUN rustup toolchain install nightly-2026-04-14 --profile minimal --component clippy,rustfmt\nRUN cargo install cargo-nextest --locked\nENV CARGO_INCREMENTAL=0\n\n# Bun\nRUN curl -fsSL https://bun.sh/install | bash\nENV PATH=\"/root/.bun/bin:${PATH}\"\n\nWORKDIR /root\n"
}
},
"network": null
}
},
"notifications": {},
"interviews": {
"provider": null,
"slack": null
},
"agent": {
"fabro_tools": false,
"permissions": null,
"mcps": {}
},
"hooks": [],
"scm": {
"provider": null,
"owner": null,
"repository": null,
"github": null
},
"pull_request": {
"enabled": true,
"draft": false,
"auto_merge": false,
"merge_strategy": "squash"
},
"artifacts": {
"include": []
},
"integrations": {
"github": {
"permissions": {}
}
}
}
},
"graph": {
"name": "ImplementPlan",
"nodes": {
"implement": {
"id": "implement",
"attrs": {
"provider": {
"String": "openai"
},
"label": {
"String": "Implement"
},
"prompt": {
"String": "Read the plan file referenced in the goal and implement every step. Make all the code changes described in the plan. Use red/green TDD."
},
"model": {
"String": "gpt-5.5"
},
"reasoning_effort": {
"String": "xhigh"
}
}
},
"fix_lints": {
"id": "fix_lints",
"attrs": {
"provider": {
"String": "anthropic"
},
"model": {
"String": "claude-opus-4-7"
},
"prompt": {
"String": "The preflight lint step failed. Read the build output from context and fix all clippy lint warnings."
},
"label": {
"String": "Fix Lints"
},
"max_visits": {
"Integer": 3
}
}
},
"start": {
"id": "start",
"attrs": {
"label": {
"String": "Start"
},
"model": {
"String": "claude-opus-4-7"
},
"provider": {
"String": "anthropic"
},
"shape": {
"String": "Mdiamond"
}
}
},
"preflight_lint": {
"id": "preflight_lint",
"attrs": {
"script": {
"String": "cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1"
},
"shape": {
"String": "parallelogram"
},
"max_retries": {
"Integer": 0
},
"label": {
"String": "Preflight Lint"
},
"model": {
"String": "claude-opus-4-7"
},
"provider": {
"String": "anthropic"
}
}
},
"toolchain": {
"id": "toolchain",
"attrs": {
"label": {
"String": "Toolchain"
},
"max_retries": {
"Integer": 0
},
"script": {
"String": "command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1"
},
"shape": {
"String": "parallelogram"
},
"provider": {
"String": "anthropic"
},
"model": {
"String": "claude-opus-4-7"
}
}
},
"verify": {
"id": "verify",
"attrs": {
"goal_gate": {
"Boolean": true
},
"label": {
"String": "Verify"
},
"model": {
"String": "claude-opus-4-7"
},
"retry_target": {
"String": "fixup"
},
"provider": {
"String": "anthropic"
},
"script": {
"String": "cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1"
},
"shape": {
"String": "parallelogram"
}
}
},
"fmt": {
"id": "fmt",
"attrs": {
"label": {
"String": "Format"
},
"max_retries": {
"Integer": 0
},
"provider": {
"String": "anthropic"
},
"shape": {
"String": "parallelogram"
},
"script": {
"String": "cargo +nightly-2026-04-14 fmt --all 2>&1"
},
"model": {
"String": "claude-opus-4-7"
}
}
},
"simplify_opus": {
"id": "simplify_opus",
"attrs": {
"provider": {
"String": "anthropic"
},
"label": {
"String": "Simplify (Opus)"
},
"prompt": {
"String": "# Simplify: Code Review and Cleanup\n\nReview changes vs. origin for reuse, quality, and efficiency. Fix any issues found.\n\n## Phase 1: Identify Changes\n\nRun git diff (or git diff HEAD if there are staged changes) to see what changed. If there are no git changes, review the most recently modified files that the user mentioned or that you edited earlier in this conversation.\n\n## Phase 2: Launch Three Review Agents in Parallel\n\nUse the Agent tool to launch all three agents concurrently in a single message. Pass each agent the full diff so it has the complete context.\n\n### Agent 1: Code Reuse Review\n\nFor each change:\n\n1. Search for existing utilities and helpers that could replace newly written code. Use Grep to find similar patterns elsewhere in the codebase — common locations are utility directories, shared modules, and files adjacent to the changed ones.\n2. Flag any new function that duplicates existing functionality. Suggest the existing function to use instead.\n3. Flag any inline logic that could use an existing utility — hand-rolled string manipulation, manual path handling, custom environment checks, ad-hoc type guards, and similar patterns are common candidates.\n\nNote: This is a greenfield app, so focus on maximizing simplicity and don't worry about changing things to achieve it.\n\n### Agent 2: Code Quality Review\n\nReview the same changes for hacky patterns:\n\n1. Redundant state: state that duplicates existing state, cached values that could be derived, observers/effects that could be direct calls\n2. Parameter sprawl: adding new parameters to a function instead of generalizing or restructuring existing ones\n3. Copy-paste with slight variation: near-duplicate code blocks that should be unified with a shared abstraction\n4. Leaky abstractions: exposing internal details that should be encapsulated, or breaking existing abstraction boundaries\n5. Stringly-typed code: using raw strings where constants, enums (string unions), or branded types already exist in the codebase\n\nNote: This is a greenfield app, so be aggressive in optimizing quality.\n\n### Agent 3: Efficiency Review\n\nReview the same changes for efficiency:\n\n1. Unnecessary work: redundant computations, repeated file reads, duplicate network/API calls, N+1 patterns\n2. Missed concurrency: independent operations run sequentially when they could run in parallel\n3. Hot-path bloat: new blocking work added to startup or per-request/per-render hot paths\n4. Unnecessary existence checks: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error\n5. Memory: unbounded data structures, missing cleanup, event listener leaks\n6. Overly broad operations: reading entire files when only a portion is needed, loading all items when filtering for one\n\n## Phase 3: Fix Issues\n\nWait for all three agents to complete. Aggregate their findings and fix each issue directly. If a finding is a false positive or not worth addressing, note it and move on — do not argue with the finding, just skip it.\n\nWhen done, briefly summarize what was fixed (or confirm the code was already clean)."
},
"model": {
"String": "claude-opus-4-7"
}
}
},
"fixup": {
"id": "fixup",
"attrs": {
"max_visits": {
"Integer": 3
},
"model": {
"String": "claude-opus-4-7"
},
"label": {
"String": "Fixup"
},
"prompt": {
"String": "The verify step failed. Read the build output from context and fix all clippy lint warnings, test failures, and generated docs errors."
},
"provider": {
"String": "anthropic"
}
}
},
"preflight_compile": {
"id": "preflight_compile",
"attrs": {
"shape": {
"String": "parallelogram"
},
"max_retries": {
"Integer": 0
},
"provider": {
"String": "anthropic"
},
"model": {
"String": "claude-opus-4-7"
},
"label": {
"String": "Preflight Compile"
},
"script": {
"String": "cargo check -q --workspace 2>&1"
}
}
},
"exit": {
"id": "exit",
"attrs": {
"label": {
"String": "Exit"
},
"shape": {
"String": "Msquare"
},
"provider": {
"String": "anthropic"
},
"model": {
"String": "claude-opus-4-7"
}
}
},
"simplify_gpt": {
"id": "simplify_gpt",
"attrs": {
"provider": {
"String": "openai"
},
"prompt": {
"String": "# Simplify: Code Review and Cleanup\n\nReview changes vs. origin for reuse, quality, and efficiency. Fix any issues found.\n\n## Phase 1: Identify Changes\n\nRun git diff (or git diff HEAD if there are staged changes) to see what changed. If there are no git changes, review the most recently modified files that the user mentioned or that you edited earlier in this conversation.\n\n## Phase 2: Launch Three Review Agents in Parallel\n\nUse the Agent tool to launch all three agents concurrently in a single message. Pass each agent the full diff so it has the complete context.\n\n### Agent 1: Code Reuse Review\n\nFor each change:\n\n1. Search for existing utilities and helpers that could replace newly written code. Use Grep to find similar patterns elsewhere in the codebase — common locations are utility directories, shared modules, and files adjacent to the changed ones.\n2. Flag any new function that duplicates existing functionality. Suggest the existing function to use instead.\n3. Flag any inline logic that could use an existing utility — hand-rolled string manipulation, manual path handling, custom environment checks, ad-hoc type guards, and similar patterns are common candidates.\n\nNote: This is a greenfield app, so focus on maximizing simplicity and don't worry about changing things to achieve it.\n\n### Agent 2: Code Quality Review\n\nReview the same changes for hacky patterns:\n\n1. Redundant state: state that duplicates existing state, cached values that could be derived, observers/effects that could be direct calls\n2. Parameter sprawl: adding new parameters to a function instead of generalizing or restructuring existing ones\n3. Copy-paste with slight variation: near-duplicate code blocks that should be unified with a shared abstraction\n4. Leaky abstractions: exposing internal details that should be encapsulated, or breaking existing abstraction boundaries\n5. Stringly-typed code: using raw strings where constants, enums (string unions), or branded types already exist in the codebase\n\nNote: This is a greenfield app, so be aggressive in optimizing quality.\n\n### Agent 3: Efficiency Review\n\nReview the same changes for efficiency:\n\n1. Unnecessary work: redundant computations, repeated file reads, duplicate network/API calls, N+1 patterns\n2. Missed concurrency: independent operations run sequentially when they could run in parallel\n3. Hot-path bloat: new blocking work added to startup or per-request/per-render hot paths\n4. Unnecessary existence checks: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error\n5. Memory: unbounded data structures, missing cleanup, event listener leaks\n6. Overly broad operations: reading entire files when only a portion is needed, loading all items when filtering for one\n\n## Phase 3: Fix Issues\n\nWait for all three agents to complete. Aggregate their findings and fix each issue directly. If a finding is a false positive or not worth addressing, note it and move on — do not argue with the finding, just skip it.\n\nWhen done, briefly summarize what was fixed (or confirm the code was already clean)."
},
"label": {
"String": "Simplify (GPT-55)"
},
"model": {
"String": "gpt-5.5"
}
}
}
},
"edges": [
{
"from": "start",
"to": "toolchain",
"attrs": {}
},
{
"from": "toolchain",
"to": "preflight_compile",
"attrs": {
"condition": {
"String": "outcome=succeeded"
}
}
},
{
"from": "toolchain",
"to": "exit",
"attrs": {}
},
{
"from": "preflight_compile",
"to": "preflight_lint",
"attrs": {
"condition": {
"String": "outcome=succeeded"
}
}
},
{
"from": "preflight_compile",
"to": "exit",
"attrs": {}
},
{
"from": "preflight_lint",
"to": "implement",
"attrs": {
"condition": {
"String": "outcome=succeeded"
}
}
},
{
"from": "preflight_lint",
"to": "fix_lints",
"attrs": {}
},
{
"from": "fix_lints",
"to": "preflight_lint",
"attrs": {}
},
{
"from": "implement",
"to": "simplify_opus",
"attrs": {}
},
{
"from": "simplify_opus",
"to": "simplify_gpt",
"attrs": {}
},
{
"from": "simplify_gpt",
"to": "verify",
"attrs": {}
},
{
"from": "verify",
"to": "fmt",
"attrs": {
"condition": {
"String": "outcome=succeeded"
}
}
},
{
"from": "verify",
"to": "fixup",
"attrs": {}
},
{
"from": "fixup",
"to": "verify",
"attrs": {}
},
{
"from": "fmt",
"to": "exit",
"attrs": {}
}
],
"attrs": {
"model_stylesheet": {
"String": "\n * { model: claude-opus-4-7; }\n "
},
"goal": {
"String": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n"
},
"rankdir": {
"String": "LR"
}
}
},
"graph_source": "digraph ImplementPlan {\n graph [\n goal=\"Implement and simplify\",\n model_stylesheet=\"\n * { model: claude-opus-4-7; }\n \"\n ]\n rankdir=LR\n\n start [shape=Mdiamond, label=\"Start\"]\n exit [shape=Msquare, label=\"Exit\"]\n\n toolchain [label=\"Toolchain\", shape=parallelogram, script=\"command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1\", max_retries=0]\n preflight_compile [label=\"Preflight Compile\", shape=parallelogram, script=\"cargo check -q --workspace 2>&1\", max_retries=0]\n preflight_lint [label=\"Preflight Lint\", shape=parallelogram, script=\"cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1\", max_retries=0]\n fix_lints [label=\"Fix Lints\", prompt=\"The preflight lint step failed. Read the build output from context and fix all clippy lint warnings.\", max_visits=3]\n implement [label=\"Implement\", prompt=\"Read the plan file referenced in the goal and implement every step. Make all the code changes described in the plan. Use red/green TDD.\", model=\"gpt-55\", reasoning_effort=\"xhigh\"]\n simplify_opus [label=\"Simplify (Opus)\", prompt=\"@prompts/simplify.md\"]\n simplify_gpt [label=\"Simplify (GPT-55)\", prompt=\"@prompts/simplify.md\", model=\"gpt-55\"]\n verify [label=\"Verify\", shape=parallelogram, script=\"cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1\", goal_gate=true, retry_target=\"fixup\"]\n fixup [label=\"Fixup\", prompt=\"The verify step failed. Read the build output from context and fix all clippy lint warnings, test failures, and generated docs errors.\", max_visits=3]\n fmt [label=\"Format\", shape=parallelogram, script=\"cargo +nightly-2026-04-14 fmt --all 2>&1\", max_retries=0]\n\n start -> toolchain\n toolchain -> preflight_compile [condition=\"outcome=succeeded\"]\n toolchain -> exit\n preflight_compile -> preflight_lint [condition=\"outcome=succeeded\"]\n preflight_compile -> exit\n preflight_lint -> implement [condition=\"outcome=succeeded\"]\n preflight_lint -> fix_lints\n fix_lints -> preflight_lint\n implement -> simplify_opus -> simplify_gpt -> verify\n verify -> fmt [condition=\"outcome=succeeded\"]\n verify -> fixup\n fixup -> verify\n fmt -> exit\n}\n",
"workflow_slug": "implement-plan",
"source_directory": "/Users/bhelmkamp/p/fabro-sh/fabro",
"provenance": {
"server": {
"version": "0.241.0-nightly.1"
},
"client": {
"user_agent": "fabro-cli/0.242.0-nightly.1",
"name": "fabro-cli",
"version": "0.242.0-nightly.1"
},
"subject": {
"kind": "user",
"identity": {
"issuer": "https://github.com",
"subject": "19"
},
"login": "brynary",
"auth_method": "github"
}
},
"manifest_blob": "374123660c1867421008267a5544ea86bf4e9cef249b8f8e9cf7688ca0e81837",
"definition_blob": "8b6afc58f5efffd5b117a31134949c59ed0e00534e561c4fbdf8dddfbe8b4bfe",
"git": {
"origin_url": "https://github.com/fabro-sh/fabro",
"branch": "main",
"sha": "59f80c91901b628927b8905ea7742d245de16ba3",
"dirty": "dirty",
"push_outcome": {
"type": "failed",
"remote": "origin",
"branch": "main",
"message": "Engine error: git push failed: To github.com:fabro-sh/fabro.git\n ! [rejected] main -> main (fetch first)\nerror: failed to push some refs to 'github.com:fabro-sh/fabro.git'\nhint: Updates were rejected because the remote contains work that you do not\nhint: have locally. This is usually caused by another repository pushing to\nhint: the same ref. If you want to integrate the remote changes, use\nhint: 'git pull' before pushing again.\nhint: See the 'Note about fast-forwards' in 'git push --help' for details.\n"
}
}
},
"web_url": "http://127.0.0.1:32276/runs/01KSAPQQSVK4FY3CWHYVZYD7T6",
"start": {
"start_time": "2026-05-23T15:20:09.658986Z",
"run_branch": "fabro/run/01KSAPQQSVK4FY3CWHYVZYD7T6",
"base_sha": "a64a58d567e3de775aff86503703934c64fe2d38"
},
"status": {
"kind": "running"
},
"status_updated_at": "2026-05-23T15:20:09.659052Z",
"last_event_at": "2026-05-23T15:20:11.504629Z",
"pending_control": null,
"checkpoints": [
{
"seq": 19,
"checkpoint": {
"timestamp": "2026-05-23T15:20:11.504317Z",
"current_node": "start",
"completed_nodes": [
"start"
],
"node_retries": {},
"context_values": {
"failure_signature": "",
"graph.rankdir": "LR",
"current_node": "start",
"graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ",
"internal.node_visit_count": 1,
"graph.goal": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n",
"internal.run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6",
"internal.fidelity": "compact",
"internal.retry_count.start": 0,
"internal.thread_id": null,
"failure_class": "",
"internal.work_dir": "/home/daytona/workspace/fabro",
"outcome": "succeeded"
},
"node_outcomes": {
"start": {
"status": "succeeded",
"usage": null
}
},
"next_node_id": "toolchain",
"node_visits": {
"start": 1
}
},
"diff": {}
},
{
"seq": 0,
"checkpoint": {
"timestamp": "2026-05-23T15:20:14.019032Z",
"current_node": "toolchain",
"completed_nodes": [
"start",
"toolchain"
],
"node_retries": {},
"context_values": {
"internal.work_dir": "/home/daytona/workspace/fabro",
"current_node": "toolchain",
"graph.rankdir": "LR",
"internal.run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6",
"outcome": "succeeded",
"internal.thread_id": "start",
"thread.start.current_node": "toolchain",
"internal.retry_count.toolchain": 0,
"internal.node_visit_count": 1,
"failure_class": "",
"graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ",
"internal.retry_count.start": 0,
"internal.fidelity": "compact",
"command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c",
"graph.goal": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n",
"failure_signature": ""
},
"node_outcomes": {
"start": {
"status": "succeeded",
"usage": null
},
"toolchain": {
"status": "succeeded",
"context_updates": {
"command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c"
},
"notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
"usage": null
}
},
"next_node_id": "preflight_compile",
"node_visits": {
"toolchain": 1,
"start": 1
}
},
"diff": {}
}
],
"conclusion": null,
"sandbox": {
"provider": "daytona",
"image": "buildpack-deps:noble",
"snapshot": "fabro-v11",
"runtime": {
"id": "fabro-01KSAPQQSVK4FY3CWHYVZYD7T6",
"working_directory": "/home/daytona/workspace/fabro",
"repo_cloned": true,
"clone_origin_url": "https://github.com/fabro-sh/fabro",
"clone_branch": "main",
"workspace_root": "/home/daytona/workspace",
"repos_root": "/home/daytona/repos",
"primary_repo_path": "/home/daytona/repos/fabro-sh/fabro",
"primary_repo_link": "/home/daytona/workspace/fabro"
}
},
"pull_request": null,
"superseded_by": null,
"pending_interviews": {},
"stages": {
"start@1": {
"first_event_seq": 16,
"prompt": null,
"response": null,
"completion": {
"outcome": "succeeded",
"notes": null,
"failure_reason": null,
"timestamp": "2026-05-23T15:20:11.504080Z"
},
"provider_used": null,
"diff": null,
"script_invocation": null,
"script_timing": null,
"parallel_results": null,
"output": null,
"started_at": "2026-05-23T15:20:11.503608Z",
"handler": "start",
"timing": {
"wall_time_ms": 0,
"inference_time_ms": 0,
"tool_time_ms": 0,
"active_time_ms": 0
},
"usage": {
"input_tokens": 0,
"output_tokens": 0,
"total_tokens": 0,
"reasoning_tokens": 0,
"cache_read_tokens": 0,
"cache_write_tokens": 0
},
"state": "succeeded"
},
"toolchain@1": {
"first_event_seq": 20,
"prompt": null,
"response": null,
"completion": null,
"provider_used": null,
"diff": null,
"script_invocation": {
"script": "command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
"command": "exec 2>&1\ncommand -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
"language": "shell"
},
"script_timing": null,
"parallel_results": null,
"output": null,
"started_at": "2026-05-23T15:20:11.504411Z",
"handler": "command",
"usage": {
"input_tokens": 0,
"output_tokens": 0,
"total_tokens": 0,
"reasoning_tokens": 0,
"cache_read_tokens": 0,
"cache_write_tokens": 0
},
"state": "running"
}
}
}