mirror of
https://github.com/fabro-sh/fabro.git
synced 2026-10-11 03:40:05 +00:00
1856 lines
No EOL
192 KiB
JSON
1856 lines
No EOL
192 KiB
JSON
{
|
|
"title": "Agent Compaction API Usage Baseline Plan",
|
|
"spec": {
|
|
"run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6",
|
|
"settings": {
|
|
"project": {
|
|
"name": null,
|
|
"description": null,
|
|
"metadata": {}
|
|
},
|
|
"workflow": {
|
|
"name": null,
|
|
"description": null,
|
|
"graph": "workflow.fabro",
|
|
"metadata": {}
|
|
},
|
|
"run": {
|
|
"goal": {
|
|
"type": "inline",
|
|
"value": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n"
|
|
},
|
|
"working_dir": null,
|
|
"metadata": {},
|
|
"inputs": {},
|
|
"model": {
|
|
"provider": "anthropic",
|
|
"name": "claude-sonnet-4-6",
|
|
"fallbacks": [],
|
|
"controls": {
|
|
"reasoning_effort": null,
|
|
"speed": null
|
|
}
|
|
},
|
|
"git": {
|
|
"author": null
|
|
},
|
|
"prepare": {
|
|
"commands": [],
|
|
"timeout_ms": 300000
|
|
},
|
|
"execution": {
|
|
"mode": "normal",
|
|
"approval": "prompt"
|
|
},
|
|
"checkpoint": {
|
|
"exclude_globs": [],
|
|
"skip_git_hooks": false
|
|
},
|
|
"clone": {
|
|
"enabled": true
|
|
},
|
|
"run_branch": {
|
|
"enabled": true,
|
|
"push": true
|
|
},
|
|
"meta_branch": {
|
|
"enabled": true,
|
|
"push": true
|
|
},
|
|
"sandbox": {
|
|
"provider": "daytona",
|
|
"preserve": false,
|
|
"stop_on_terminal": true,
|
|
"devcontainer": false,
|
|
"env": {},
|
|
"docker": {
|
|
"image": "buildpack-deps:noble",
|
|
"network_mode": null,
|
|
"memory_limit": 4000000000,
|
|
"cpu_quota": 200000,
|
|
"env_vars": {}
|
|
},
|
|
"daytona": {
|
|
"auto_stop_interval": 30,
|
|
"labels": {
|
|
"repo": "fabro-sh/fabro"
|
|
},
|
|
"volumes": [],
|
|
"snapshot": {
|
|
"name": "fabro-v11",
|
|
"cpu": 8,
|
|
"memory_gb": 16,
|
|
"disk_gb": 20,
|
|
"dockerfile": {
|
|
"type": "inline",
|
|
"value": "FROM ubuntu:24.04\n\nRUN apt-get update && apt-get install -y --no-install-recommends \\\n curl git ca-certificates build-essential pkg-config libssl-dev unzip python3 \\\n xvfb xfce4 xfce4-terminal x11vnc novnc dbus-x11 \\\n libx11-6 libxrandr2 libxext6 libxrender1 libxfixes3 libxss1 libxtst6 libxi6 \\\n && rm -rf /var/lib/apt/lists/*\n\n# Install real Chromium (not the snap stub) via xtradeb PPA\nRUN apt-get update && apt-get install -y --no-install-recommends \\\n software-properties-common curl gnupg \\\n && add-apt-repository -y ppa:xtradeb/apps \\\n && apt-get update \\\n && apt-get install -y --no-install-recommends chromium \\\n && rm -rf /var/lib/apt/lists/*\n\n# Wrapper: Chromium needs --no-sandbox when running as root in a container,\n# and --disable-dev-shm-usage avoids crashes from small /dev/shm\nRUN printf '#!/bin/bash\\nexec /usr/bin/chromium --no-sandbox --disable-dev-shm-usage \"$@\"\\n' \\\n > /usr/local/bin/chromium-wrapper \\\n && chmod +x /usr/local/bin/chromium-wrapper\n\n# Make the wrapper the default in the system .desktop file and via alternatives\nRUN sed -i 's|^Exec=.*|Exec=/usr/local/bin/chromium-wrapper %U|' \\\n /usr/share/applications/chromium.desktop \\\n && update-alternatives --install /usr/bin/x-www-browser x-www-browser \\\n /usr/local/bin/chromium-wrapper 100\n\n# Tell XFCE's exo-open that Chromium is the WebBrowser helper (system-wide)\nRUN mkdir -p /etc/xdg/xfce4 /usr/share/xfce4/helpers \\\n && printf 'WebBrowser=custom-WebBrowser\\n' > /etc/xdg/xfce4/helpers.rc \\\n && printf '[Desktop Entry]\\n\\\nVersion=1.0\\n\\\nType=X-XFCE-Helper\\n\\\nName=Chromium\\n\\\nIcon=chromium\\n\\\nX-XFCE-Category=WebBrowser\\n\\\nX-XFCE-CommandsWithParameter=/usr/local/bin/chromium-wrapper \"%%s\"\\n\\\nX-XFCE-Commands=/usr/local/bin/chromium-wrapper\\n' \\\n > /usr/share/xfce4/helpers/custom-WebBrowser.desktop\n\n# GitHub CLI\nRUN curl -fsSL https://cli.github.com/packages/githubcli-archive-keyring.gpg \\\n | dd of=/usr/share/keyrings/githubcli-archive-keyring.gpg \\\n && echo \"deb [arch=$(dpkg --print-architecture) signed-by=/usr/share/keyrings/githubcli-archive-keyring.gpg] https://cli.github.com/packages stable main\" \\\n | tee /etc/apt/sources.list.d/github-cli.list > /dev/null \\\n && apt-get update && apt-get install -y --no-install-recommends gh \\\n && rm -rf /var/lib/apt/lists/*\n\n# Rust\nRUN curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y\nENV PATH=\"/root/.cargo/bin:${PATH}\"\nRUN rustup toolchain install nightly-2026-04-14 --profile minimal --component clippy,rustfmt\nRUN cargo install cargo-nextest --locked\nENV CARGO_INCREMENTAL=0\n\n# Bun\nRUN curl -fsSL https://bun.sh/install | bash\nENV PATH=\"/root/.bun/bin:${PATH}\"\n\nWORKDIR /root\n"
|
|
}
|
|
},
|
|
"network": null
|
|
}
|
|
},
|
|
"notifications": {},
|
|
"interviews": {
|
|
"provider": null,
|
|
"slack": null
|
|
},
|
|
"agent": {
|
|
"fabro_tools": false,
|
|
"permissions": null,
|
|
"mcps": {}
|
|
},
|
|
"hooks": [],
|
|
"scm": {
|
|
"provider": null,
|
|
"owner": null,
|
|
"repository": null,
|
|
"github": null
|
|
},
|
|
"pull_request": {
|
|
"enabled": true,
|
|
"draft": false,
|
|
"auto_merge": false,
|
|
"merge_strategy": "squash"
|
|
},
|
|
"artifacts": {
|
|
"include": []
|
|
},
|
|
"integrations": {
|
|
"github": {
|
|
"permissions": {}
|
|
}
|
|
}
|
|
}
|
|
},
|
|
"graph": {
|
|
"name": "ImplementPlan",
|
|
"nodes": {
|
|
"implement": {
|
|
"id": "implement",
|
|
"attrs": {
|
|
"provider": {
|
|
"String": "openai"
|
|
},
|
|
"label": {
|
|
"String": "Implement"
|
|
},
|
|
"prompt": {
|
|
"String": "Read the plan file referenced in the goal and implement every step. Make all the code changes described in the plan. Use red/green TDD."
|
|
},
|
|
"model": {
|
|
"String": "gpt-5.5"
|
|
},
|
|
"reasoning_effort": {
|
|
"String": "xhigh"
|
|
}
|
|
}
|
|
},
|
|
"fix_lints": {
|
|
"id": "fix_lints",
|
|
"attrs": {
|
|
"provider": {
|
|
"String": "anthropic"
|
|
},
|
|
"model": {
|
|
"String": "claude-opus-4-7"
|
|
},
|
|
"prompt": {
|
|
"String": "The preflight lint step failed. Read the build output from context and fix all clippy lint warnings."
|
|
},
|
|
"label": {
|
|
"String": "Fix Lints"
|
|
},
|
|
"max_visits": {
|
|
"Integer": 3
|
|
}
|
|
}
|
|
},
|
|
"start": {
|
|
"id": "start",
|
|
"attrs": {
|
|
"label": {
|
|
"String": "Start"
|
|
},
|
|
"model": {
|
|
"String": "claude-opus-4-7"
|
|
},
|
|
"provider": {
|
|
"String": "anthropic"
|
|
},
|
|
"shape": {
|
|
"String": "Mdiamond"
|
|
}
|
|
}
|
|
},
|
|
"preflight_lint": {
|
|
"id": "preflight_lint",
|
|
"attrs": {
|
|
"script": {
|
|
"String": "cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1"
|
|
},
|
|
"shape": {
|
|
"String": "parallelogram"
|
|
},
|
|
"max_retries": {
|
|
"Integer": 0
|
|
},
|
|
"label": {
|
|
"String": "Preflight Lint"
|
|
},
|
|
"model": {
|
|
"String": "claude-opus-4-7"
|
|
},
|
|
"provider": {
|
|
"String": "anthropic"
|
|
}
|
|
}
|
|
},
|
|
"toolchain": {
|
|
"id": "toolchain",
|
|
"attrs": {
|
|
"label": {
|
|
"String": "Toolchain"
|
|
},
|
|
"max_retries": {
|
|
"Integer": 0
|
|
},
|
|
"script": {
|
|
"String": "command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1"
|
|
},
|
|
"shape": {
|
|
"String": "parallelogram"
|
|
},
|
|
"provider": {
|
|
"String": "anthropic"
|
|
},
|
|
"model": {
|
|
"String": "claude-opus-4-7"
|
|
}
|
|
}
|
|
},
|
|
"verify": {
|
|
"id": "verify",
|
|
"attrs": {
|
|
"goal_gate": {
|
|
"Boolean": true
|
|
},
|
|
"label": {
|
|
"String": "Verify"
|
|
},
|
|
"model": {
|
|
"String": "claude-opus-4-7"
|
|
},
|
|
"retry_target": {
|
|
"String": "fixup"
|
|
},
|
|
"provider": {
|
|
"String": "anthropic"
|
|
},
|
|
"script": {
|
|
"String": "cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1"
|
|
},
|
|
"shape": {
|
|
"String": "parallelogram"
|
|
}
|
|
}
|
|
},
|
|
"fmt": {
|
|
"id": "fmt",
|
|
"attrs": {
|
|
"label": {
|
|
"String": "Format"
|
|
},
|
|
"max_retries": {
|
|
"Integer": 0
|
|
},
|
|
"provider": {
|
|
"String": "anthropic"
|
|
},
|
|
"shape": {
|
|
"String": "parallelogram"
|
|
},
|
|
"script": {
|
|
"String": "cargo +nightly-2026-04-14 fmt --all 2>&1"
|
|
},
|
|
"model": {
|
|
"String": "claude-opus-4-7"
|
|
}
|
|
}
|
|
},
|
|
"simplify_opus": {
|
|
"id": "simplify_opus",
|
|
"attrs": {
|
|
"provider": {
|
|
"String": "anthropic"
|
|
},
|
|
"label": {
|
|
"String": "Simplify (Opus)"
|
|
},
|
|
"prompt": {
|
|
"String": "# Simplify: Code Review and Cleanup\n\nReview changes vs. origin for reuse, quality, and efficiency. Fix any issues found.\n\n## Phase 1: Identify Changes\n\nRun git diff (or git diff HEAD if there are staged changes) to see what changed. If there are no git changes, review the most recently modified files that the user mentioned or that you edited earlier in this conversation.\n\n## Phase 2: Launch Three Review Agents in Parallel\n\nUse the Agent tool to launch all three agents concurrently in a single message. Pass each agent the full diff so it has the complete context.\n\n### Agent 1: Code Reuse Review\n\nFor each change:\n\n1. Search for existing utilities and helpers that could replace newly written code. Use Grep to find similar patterns elsewhere in the codebase — common locations are utility directories, shared modules, and files adjacent to the changed ones.\n2. Flag any new function that duplicates existing functionality. Suggest the existing function to use instead.\n3. Flag any inline logic that could use an existing utility — hand-rolled string manipulation, manual path handling, custom environment checks, ad-hoc type guards, and similar patterns are common candidates.\n\nNote: This is a greenfield app, so focus on maximizing simplicity and don't worry about changing things to achieve it.\n\n### Agent 2: Code Quality Review\n\nReview the same changes for hacky patterns:\n\n1. Redundant state: state that duplicates existing state, cached values that could be derived, observers/effects that could be direct calls\n2. Parameter sprawl: adding new parameters to a function instead of generalizing or restructuring existing ones\n3. Copy-paste with slight variation: near-duplicate code blocks that should be unified with a shared abstraction\n4. Leaky abstractions: exposing internal details that should be encapsulated, or breaking existing abstraction boundaries\n5. Stringly-typed code: using raw strings where constants, enums (string unions), or branded types already exist in the codebase\n\nNote: This is a greenfield app, so be aggressive in optimizing quality.\n\n### Agent 3: Efficiency Review\n\nReview the same changes for efficiency:\n\n1. Unnecessary work: redundant computations, repeated file reads, duplicate network/API calls, N+1 patterns\n2. Missed concurrency: independent operations run sequentially when they could run in parallel\n3. Hot-path bloat: new blocking work added to startup or per-request/per-render hot paths\n4. Unnecessary existence checks: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error\n5. Memory: unbounded data structures, missing cleanup, event listener leaks\n6. Overly broad operations: reading entire files when only a portion is needed, loading all items when filtering for one\n\n## Phase 3: Fix Issues\n\nWait for all three agents to complete. Aggregate their findings and fix each issue directly. If a finding is a false positive or not worth addressing, note it and move on — do not argue with the finding, just skip it.\n\nWhen done, briefly summarize what was fixed (or confirm the code was already clean)."
|
|
},
|
|
"model": {
|
|
"String": "claude-opus-4-7"
|
|
}
|
|
}
|
|
},
|
|
"fixup": {
|
|
"id": "fixup",
|
|
"attrs": {
|
|
"max_visits": {
|
|
"Integer": 3
|
|
},
|
|
"model": {
|
|
"String": "claude-opus-4-7"
|
|
},
|
|
"label": {
|
|
"String": "Fixup"
|
|
},
|
|
"prompt": {
|
|
"String": "The verify step failed. Read the build output from context and fix all clippy lint warnings, test failures, and generated docs errors."
|
|
},
|
|
"provider": {
|
|
"String": "anthropic"
|
|
}
|
|
}
|
|
},
|
|
"preflight_compile": {
|
|
"id": "preflight_compile",
|
|
"attrs": {
|
|
"shape": {
|
|
"String": "parallelogram"
|
|
},
|
|
"max_retries": {
|
|
"Integer": 0
|
|
},
|
|
"provider": {
|
|
"String": "anthropic"
|
|
},
|
|
"model": {
|
|
"String": "claude-opus-4-7"
|
|
},
|
|
"label": {
|
|
"String": "Preflight Compile"
|
|
},
|
|
"script": {
|
|
"String": "cargo check -q --workspace 2>&1"
|
|
}
|
|
}
|
|
},
|
|
"exit": {
|
|
"id": "exit",
|
|
"attrs": {
|
|
"label": {
|
|
"String": "Exit"
|
|
},
|
|
"shape": {
|
|
"String": "Msquare"
|
|
},
|
|
"provider": {
|
|
"String": "anthropic"
|
|
},
|
|
"model": {
|
|
"String": "claude-opus-4-7"
|
|
}
|
|
}
|
|
},
|
|
"simplify_gpt": {
|
|
"id": "simplify_gpt",
|
|
"attrs": {
|
|
"provider": {
|
|
"String": "openai"
|
|
},
|
|
"prompt": {
|
|
"String": "# Simplify: Code Review and Cleanup\n\nReview changes vs. origin for reuse, quality, and efficiency. Fix any issues found.\n\n## Phase 1: Identify Changes\n\nRun git diff (or git diff HEAD if there are staged changes) to see what changed. If there are no git changes, review the most recently modified files that the user mentioned or that you edited earlier in this conversation.\n\n## Phase 2: Launch Three Review Agents in Parallel\n\nUse the Agent tool to launch all three agents concurrently in a single message. Pass each agent the full diff so it has the complete context.\n\n### Agent 1: Code Reuse Review\n\nFor each change:\n\n1. Search for existing utilities and helpers that could replace newly written code. Use Grep to find similar patterns elsewhere in the codebase — common locations are utility directories, shared modules, and files adjacent to the changed ones.\n2. Flag any new function that duplicates existing functionality. Suggest the existing function to use instead.\n3. Flag any inline logic that could use an existing utility — hand-rolled string manipulation, manual path handling, custom environment checks, ad-hoc type guards, and similar patterns are common candidates.\n\nNote: This is a greenfield app, so focus on maximizing simplicity and don't worry about changing things to achieve it.\n\n### Agent 2: Code Quality Review\n\nReview the same changes for hacky patterns:\n\n1. Redundant state: state that duplicates existing state, cached values that could be derived, observers/effects that could be direct calls\n2. Parameter sprawl: adding new parameters to a function instead of generalizing or restructuring existing ones\n3. Copy-paste with slight variation: near-duplicate code blocks that should be unified with a shared abstraction\n4. Leaky abstractions: exposing internal details that should be encapsulated, or breaking existing abstraction boundaries\n5. Stringly-typed code: using raw strings where constants, enums (string unions), or branded types already exist in the codebase\n\nNote: This is a greenfield app, so be aggressive in optimizing quality.\n\n### Agent 3: Efficiency Review\n\nReview the same changes for efficiency:\n\n1. Unnecessary work: redundant computations, repeated file reads, duplicate network/API calls, N+1 patterns\n2. Missed concurrency: independent operations run sequentially when they could run in parallel\n3. Hot-path bloat: new blocking work added to startup or per-request/per-render hot paths\n4. Unnecessary existence checks: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error\n5. Memory: unbounded data structures, missing cleanup, event listener leaks\n6. Overly broad operations: reading entire files when only a portion is needed, loading all items when filtering for one\n\n## Phase 3: Fix Issues\n\nWait for all three agents to complete. Aggregate their findings and fix each issue directly. If a finding is a false positive or not worth addressing, note it and move on — do not argue with the finding, just skip it.\n\nWhen done, briefly summarize what was fixed (or confirm the code was already clean)."
|
|
},
|
|
"label": {
|
|
"String": "Simplify (GPT-55)"
|
|
},
|
|
"model": {
|
|
"String": "gpt-5.5"
|
|
}
|
|
}
|
|
}
|
|
},
|
|
"edges": [
|
|
{
|
|
"from": "start",
|
|
"to": "toolchain",
|
|
"attrs": {}
|
|
},
|
|
{
|
|
"from": "toolchain",
|
|
"to": "preflight_compile",
|
|
"attrs": {
|
|
"condition": {
|
|
"String": "outcome=succeeded"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"from": "toolchain",
|
|
"to": "exit",
|
|
"attrs": {}
|
|
},
|
|
{
|
|
"from": "preflight_compile",
|
|
"to": "preflight_lint",
|
|
"attrs": {
|
|
"condition": {
|
|
"String": "outcome=succeeded"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"from": "preflight_compile",
|
|
"to": "exit",
|
|
"attrs": {}
|
|
},
|
|
{
|
|
"from": "preflight_lint",
|
|
"to": "implement",
|
|
"attrs": {
|
|
"condition": {
|
|
"String": "outcome=succeeded"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"from": "preflight_lint",
|
|
"to": "fix_lints",
|
|
"attrs": {}
|
|
},
|
|
{
|
|
"from": "fix_lints",
|
|
"to": "preflight_lint",
|
|
"attrs": {}
|
|
},
|
|
{
|
|
"from": "implement",
|
|
"to": "simplify_opus",
|
|
"attrs": {}
|
|
},
|
|
{
|
|
"from": "simplify_opus",
|
|
"to": "simplify_gpt",
|
|
"attrs": {}
|
|
},
|
|
{
|
|
"from": "simplify_gpt",
|
|
"to": "verify",
|
|
"attrs": {}
|
|
},
|
|
{
|
|
"from": "verify",
|
|
"to": "fmt",
|
|
"attrs": {
|
|
"condition": {
|
|
"String": "outcome=succeeded"
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"from": "verify",
|
|
"to": "fixup",
|
|
"attrs": {}
|
|
},
|
|
{
|
|
"from": "fixup",
|
|
"to": "verify",
|
|
"attrs": {}
|
|
},
|
|
{
|
|
"from": "fmt",
|
|
"to": "exit",
|
|
"attrs": {}
|
|
}
|
|
],
|
|
"attrs": {
|
|
"model_stylesheet": {
|
|
"String": "\n * { model: claude-opus-4-7; }\n "
|
|
},
|
|
"goal": {
|
|
"String": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n"
|
|
},
|
|
"rankdir": {
|
|
"String": "LR"
|
|
}
|
|
}
|
|
},
|
|
"graph_source": "digraph ImplementPlan {\n graph [\n goal=\"Implement and simplify\",\n model_stylesheet=\"\n * { model: claude-opus-4-7; }\n \"\n ]\n rankdir=LR\n\n start [shape=Mdiamond, label=\"Start\"]\n exit [shape=Msquare, label=\"Exit\"]\n\n toolchain [label=\"Toolchain\", shape=parallelogram, script=\"command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1\", max_retries=0]\n preflight_compile [label=\"Preflight Compile\", shape=parallelogram, script=\"cargo check -q --workspace 2>&1\", max_retries=0]\n preflight_lint [label=\"Preflight Lint\", shape=parallelogram, script=\"cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1\", max_retries=0]\n fix_lints [label=\"Fix Lints\", prompt=\"The preflight lint step failed. Read the build output from context and fix all clippy lint warnings.\", max_visits=3]\n implement [label=\"Implement\", prompt=\"Read the plan file referenced in the goal and implement every step. Make all the code changes described in the plan. Use red/green TDD.\", model=\"gpt-55\", reasoning_effort=\"xhigh\"]\n simplify_opus [label=\"Simplify (Opus)\", prompt=\"@prompts/simplify.md\"]\n simplify_gpt [label=\"Simplify (GPT-55)\", prompt=\"@prompts/simplify.md\", model=\"gpt-55\"]\n verify [label=\"Verify\", shape=parallelogram, script=\"cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1\", goal_gate=true, retry_target=\"fixup\"]\n fixup [label=\"Fixup\", prompt=\"The verify step failed. Read the build output from context and fix all clippy lint warnings, test failures, and generated docs errors.\", max_visits=3]\n fmt [label=\"Format\", shape=parallelogram, script=\"cargo +nightly-2026-04-14 fmt --all 2>&1\", max_retries=0]\n\n start -> toolchain\n toolchain -> preflight_compile [condition=\"outcome=succeeded\"]\n toolchain -> exit\n preflight_compile -> preflight_lint [condition=\"outcome=succeeded\"]\n preflight_compile -> exit\n preflight_lint -> implement [condition=\"outcome=succeeded\"]\n preflight_lint -> fix_lints\n fix_lints -> preflight_lint\n implement -> simplify_opus -> simplify_gpt -> verify\n verify -> fmt [condition=\"outcome=succeeded\"]\n verify -> fixup\n fixup -> verify\n fmt -> exit\n}\n",
|
|
"workflow_slug": "implement-plan",
|
|
"source_directory": "/Users/bhelmkamp/p/fabro-sh/fabro",
|
|
"provenance": {
|
|
"server": {
|
|
"version": "0.241.0-nightly.1"
|
|
},
|
|
"client": {
|
|
"user_agent": "fabro-cli/0.242.0-nightly.1",
|
|
"name": "fabro-cli",
|
|
"version": "0.242.0-nightly.1"
|
|
},
|
|
"subject": {
|
|
"kind": "user",
|
|
"identity": {
|
|
"issuer": "https://github.com",
|
|
"subject": "19"
|
|
},
|
|
"login": "brynary",
|
|
"auth_method": "github"
|
|
}
|
|
},
|
|
"manifest_blob": "374123660c1867421008267a5544ea86bf4e9cef249b8f8e9cf7688ca0e81837",
|
|
"definition_blob": "8b6afc58f5efffd5b117a31134949c59ed0e00534e561c4fbdf8dddfbe8b4bfe",
|
|
"git": {
|
|
"origin_url": "https://github.com/fabro-sh/fabro",
|
|
"branch": "main",
|
|
"sha": "59f80c91901b628927b8905ea7742d245de16ba3",
|
|
"dirty": "dirty",
|
|
"push_outcome": {
|
|
"type": "failed",
|
|
"remote": "origin",
|
|
"branch": "main",
|
|
"message": "Engine error: git push failed: To github.com:fabro-sh/fabro.git\n ! [rejected] main -> main (fetch first)\nerror: failed to push some refs to 'github.com:fabro-sh/fabro.git'\nhint: Updates were rejected because the remote contains work that you do not\nhint: have locally. This is usually caused by another repository pushing to\nhint: the same ref. If you want to integrate the remote changes, use\nhint: 'git pull' before pushing again.\nhint: See the 'Note about fast-forwards' in 'git push --help' for details.\n"
|
|
}
|
|
}
|
|
},
|
|
"web_url": "http://127.0.0.1:32276/runs/01KSAPQQSVK4FY3CWHYVZYD7T6",
|
|
"start": {
|
|
"start_time": "2026-05-23T15:20:09.658986Z",
|
|
"run_branch": "fabro/run/01KSAPQQSVK4FY3CWHYVZYD7T6",
|
|
"base_sha": "a64a58d567e3de775aff86503703934c64fe2d38"
|
|
},
|
|
"status": {
|
|
"kind": "running"
|
|
},
|
|
"status_updated_at": "2026-05-23T15:20:09.659052Z",
|
|
"last_event_at": "2026-05-23T15:46:44.407880Z",
|
|
"pending_control": null,
|
|
"checkpoints": [
|
|
{
|
|
"seq": 19,
|
|
"checkpoint": {
|
|
"timestamp": "2026-05-23T15:20:11.504317Z",
|
|
"current_node": "start",
|
|
"completed_nodes": [
|
|
"start"
|
|
],
|
|
"node_retries": {},
|
|
"context_values": {
|
|
"failure_signature": "",
|
|
"graph.rankdir": "LR",
|
|
"current_node": "start",
|
|
"graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ",
|
|
"internal.node_visit_count": 1,
|
|
"graph.goal": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n",
|
|
"internal.run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6",
|
|
"internal.fidelity": "compact",
|
|
"internal.retry_count.start": 0,
|
|
"internal.thread_id": null,
|
|
"failure_class": "",
|
|
"internal.work_dir": "/home/daytona/workspace/fabro",
|
|
"outcome": "succeeded"
|
|
},
|
|
"node_outcomes": {
|
|
"start": {
|
|
"status": "succeeded",
|
|
"usage": null
|
|
}
|
|
},
|
|
"next_node_id": "toolchain",
|
|
"node_visits": {
|
|
"start": 1
|
|
}
|
|
},
|
|
"diff": {}
|
|
},
|
|
{
|
|
"seq": 27,
|
|
"checkpoint": {
|
|
"timestamp": "2026-05-23T15:20:19.856458Z",
|
|
"current_node": "toolchain",
|
|
"completed_nodes": [
|
|
"start",
|
|
"toolchain"
|
|
],
|
|
"node_retries": {},
|
|
"context_values": {
|
|
"graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ",
|
|
"internal.retry_count.start": 0,
|
|
"internal.retry_count.toolchain": 0,
|
|
"command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c",
|
|
"internal.run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6",
|
|
"internal.thread_id": "start",
|
|
"failure_class": "",
|
|
"outcome": "succeeded",
|
|
"internal.fidelity": "compact",
|
|
"failure_signature": "",
|
|
"internal.work_dir": "/home/daytona/workspace/fabro",
|
|
"graph.rankdir": "LR",
|
|
"graph.goal": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n",
|
|
"thread.start.current_node": "toolchain",
|
|
"internal.node_visit_count": 1,
|
|
"current_node": "toolchain"
|
|
},
|
|
"node_outcomes": {
|
|
"toolchain": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c"
|
|
},
|
|
"notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
|
"usage": null
|
|
},
|
|
"start": {
|
|
"status": "succeeded",
|
|
"usage": null
|
|
}
|
|
},
|
|
"next_node_id": "preflight_compile",
|
|
"git_commit_sha": "86dc83bfeb7eea57de8a7dcd55ee8c4893fdad09",
|
|
"node_visits": {
|
|
"start": 1,
|
|
"toolchain": 1
|
|
}
|
|
},
|
|
"diff": {
|
|
"summary": {
|
|
"files_changed": 0,
|
|
"additions": 0,
|
|
"deletions": 0
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"seq": 37,
|
|
"checkpoint": {
|
|
"timestamp": "2026-05-23T15:22:17.483909Z",
|
|
"current_node": "preflight_compile",
|
|
"completed_nodes": [
|
|
"start",
|
|
"toolchain",
|
|
"preflight_compile"
|
|
],
|
|
"node_retries": {},
|
|
"context_values": {
|
|
"internal.node_visit_count": 1,
|
|
"internal.retry_count.preflight_compile": 0,
|
|
"internal.run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6",
|
|
"internal.work_dir": "/home/daytona/workspace/fabro",
|
|
"current_node": "preflight_compile",
|
|
"graph.rankdir": "LR",
|
|
"internal.retry_count.start": 0,
|
|
"thread.toolchain.current_node": "preflight_compile",
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126",
|
|
"failure_signature": "",
|
|
"graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ",
|
|
"internal.thread_id": "toolchain",
|
|
"outcome": "succeeded",
|
|
"internal.retry_count.toolchain": 0,
|
|
"thread.start.current_node": "toolchain",
|
|
"internal.fidelity": "compact",
|
|
"failure_class": "",
|
|
"graph.goal": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n"
|
|
},
|
|
"node_outcomes": {
|
|
"preflight_compile": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
|
},
|
|
"notes": "Script completed: cargo check -q --workspace 2>&1",
|
|
"usage": null
|
|
},
|
|
"start": {
|
|
"status": "succeeded",
|
|
"usage": null
|
|
},
|
|
"toolchain": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c"
|
|
},
|
|
"notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
|
"usage": null
|
|
}
|
|
},
|
|
"next_node_id": "preflight_lint",
|
|
"git_commit_sha": "9663109f79132d7f609ae1955a30f92aadca7101",
|
|
"node_visits": {
|
|
"start": 1,
|
|
"toolchain": 1,
|
|
"preflight_compile": 1
|
|
}
|
|
},
|
|
"diff": {
|
|
"summary": {
|
|
"files_changed": 0,
|
|
"additions": 0,
|
|
"deletions": 0
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"seq": 47,
|
|
"checkpoint": {
|
|
"timestamp": "2026-05-23T15:24:29.559500Z",
|
|
"current_node": "preflight_lint",
|
|
"completed_nodes": [
|
|
"start",
|
|
"toolchain",
|
|
"preflight_compile",
|
|
"preflight_lint"
|
|
],
|
|
"node_retries": {},
|
|
"context_values": {
|
|
"internal.run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6",
|
|
"thread.preflight_compile.current_node": "preflight_lint",
|
|
"internal.work_dir": "/home/daytona/workspace/fabro",
|
|
"thread.start.current_node": "toolchain",
|
|
"internal.retry_count.preflight_compile": 0,
|
|
"internal.fidelity": "compact",
|
|
"graph.goal": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n",
|
|
"graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ",
|
|
"internal.retry_count.preflight_lint": 0,
|
|
"graph.rankdir": "LR",
|
|
"internal.retry_count.toolchain": 0,
|
|
"current_node": "preflight_lint",
|
|
"internal.node_visit_count": 1,
|
|
"internal.thread_id": "preflight_compile",
|
|
"thread.toolchain.current_node": "preflight_compile",
|
|
"failure_class": "",
|
|
"failure_signature": "",
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126",
|
|
"internal.retry_count.start": 0,
|
|
"outcome": "succeeded"
|
|
},
|
|
"node_outcomes": {
|
|
"preflight_compile": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
|
},
|
|
"notes": "Script completed: cargo check -q --workspace 2>&1",
|
|
"usage": null
|
|
},
|
|
"preflight_lint": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
|
},
|
|
"notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1",
|
|
"usage": null
|
|
},
|
|
"start": {
|
|
"status": "succeeded",
|
|
"usage": null
|
|
},
|
|
"toolchain": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c"
|
|
},
|
|
"notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
|
"usage": null
|
|
}
|
|
},
|
|
"next_node_id": "implement",
|
|
"git_commit_sha": "85cb0d5e89b3c3fc4a860c82b85375d71704a686",
|
|
"node_visits": {
|
|
"toolchain": 1,
|
|
"preflight_lint": 1,
|
|
"preflight_compile": 1,
|
|
"start": 1
|
|
}
|
|
},
|
|
"diff": {
|
|
"summary": {
|
|
"files_changed": 0,
|
|
"additions": 0,
|
|
"deletions": 0
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"seq": 253,
|
|
"checkpoint": {
|
|
"timestamp": "2026-05-23T15:32:46.590640Z",
|
|
"current_node": "implement",
|
|
"completed_nodes": [
|
|
"start",
|
|
"toolchain",
|
|
"preflight_compile",
|
|
"preflight_lint",
|
|
"implement"
|
|
],
|
|
"node_retries": {},
|
|
"context_values": {
|
|
"graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ",
|
|
"internal.work_dir": "/home/daytona/workspace/fabro",
|
|
"failure_class": "",
|
|
"response.implement": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.total_tokens() > 0` as baseline.\n - Adds local estimates only for turns after that assistant response.\n - Falls back to existing whole-history `chars / 4` local estimate when no usable assistant usage exists.\n - Adds `estimate_method` to context-window warning details.\n - Uses the improved estimate for `CompactionStarted`.\n - Prevents no-op compaction from emitting `CompactionStarted`.\n- Updated `lib/crates/fabro-agent/src/history.rs`\n - `History::compact()` now resets preserved assistant `usage` to `TokenCounts::default()`.\n - Assistant content, tool calls, response IDs, and non-stripped provider parts remain preserved.\n- Added/updated tests in:\n - `compaction.rs`\n - `history.rs`\n - `session.rs`\n- Confirmed no `llm_client.count_input_tokens()` / provider token-count API calls were added.\n\nVerification:\n- Red phase confirmed initially with failing new compaction tests.\n- Passed:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - `cargo nextest run -p fabro-agent session`\n - `cargo nextest run -p fabro-agent`\n - `cargo +nightly-2026-04-14 fmt --check --all`\n - `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings`",
|
|
"current_node": "implement",
|
|
"thread.preflight_compile.current_node": "preflight_lint",
|
|
"outcome": "succeeded",
|
|
"internal.thread_id": "preflight_lint",
|
|
"failure_signature": "",
|
|
"internal.retry_count.implement": 0,
|
|
"internal.retry_count.preflight_lint": 0,
|
|
"internal.fidelity": "compact",
|
|
"internal.retry_count.preflight_compile": 0,
|
|
"internal.retry_count.start": 0,
|
|
"internal.run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6",
|
|
"thread.preflight_lint.current_node": "implement",
|
|
"thread.toolchain.current_node": "preflight_compile",
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126",
|
|
"last_stage": "implement",
|
|
"thread.start.current_node": "toolchain",
|
|
"last_response": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.tota",
|
|
"graph.rankdir": "LR",
|
|
"internal.node_visit_count": 1,
|
|
"internal.retry_count.toolchain": 0,
|
|
"graph.goal": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n"
|
|
},
|
|
"node_outcomes": {
|
|
"preflight_lint": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
|
},
|
|
"notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1",
|
|
"usage": null
|
|
},
|
|
"implement": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"last_response": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.tota",
|
|
"last_stage": "implement",
|
|
"response.implement": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.total_tokens() > 0` as baseline.\n - Adds local estimates only for turns after that assistant response.\n - Falls back to existing whole-history `chars / 4` local estimate when no usable assistant usage exists.\n - Adds `estimate_method` to context-window warning details.\n - Uses the improved estimate for `CompactionStarted`.\n - Prevents no-op compaction from emitting `CompactionStarted`.\n- Updated `lib/crates/fabro-agent/src/history.rs`\n - `History::compact()` now resets preserved assistant `usage` to `TokenCounts::default()`.\n - Assistant content, tool calls, response IDs, and non-stripped provider parts remain preserved.\n- Added/updated tests in:\n - `compaction.rs`\n - `history.rs`\n - `session.rs`\n- Confirmed no `llm_client.count_input_tokens()` / provider token-count API calls were added.\n\nVerification:\n- Red phase confirmed initially with failing new compaction tests.\n- Passed:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - `cargo nextest run -p fabro-agent session`\n - `cargo nextest run -p fabro-agent`\n - `cargo +nightly-2026-04-14 fmt --check --all`\n - `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings`"
|
|
},
|
|
"notes": "Stage completed: implement",
|
|
"usage": {
|
|
"input": {
|
|
"usage": {
|
|
"model": {
|
|
"provider": "openai",
|
|
"model_id": "gpt-5.5"
|
|
},
|
|
"tokens": {
|
|
"input_tokens": 168644,
|
|
"output_tokens": 9772,
|
|
"reasoning_tokens": 11720,
|
|
"cache_read_tokens": 3959296,
|
|
"cache_write_tokens": 0
|
|
}
|
|
},
|
|
"facts": {
|
|
"algorithm": "openai"
|
|
}
|
|
},
|
|
"total_usd_micros": 3467628
|
|
}
|
|
},
|
|
"preflight_compile": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
|
},
|
|
"notes": "Script completed: cargo check -q --workspace 2>&1",
|
|
"usage": null
|
|
},
|
|
"start": {
|
|
"status": "succeeded",
|
|
"usage": null
|
|
},
|
|
"toolchain": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c"
|
|
},
|
|
"notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
|
"usage": null
|
|
}
|
|
},
|
|
"next_node_id": "simplify_opus",
|
|
"git_commit_sha": "f606520812886a86025654bf4b9b7ef78c316b88",
|
|
"node_visits": {
|
|
"implement": 1,
|
|
"toolchain": 1,
|
|
"start": 1,
|
|
"preflight_compile": 1,
|
|
"preflight_lint": 1
|
|
}
|
|
},
|
|
"diff": {
|
|
"patch": "diff --git a/lib/crates/fabro-agent/src/compaction.rs b/lib/crates/fabro-agent/src/compaction.rs\nindex 01c756dc5..7c2434fa7 100644\n--- a/lib/crates/fabro-agent/src/compaction.rs\n+++ b/lib/crates/fabro-agent/src/compaction.rs\n@@ -11,6 +11,29 @@ use crate::file_tracker::FileTracker;\n use crate::history::History;\n use crate::types::{AgentEvent, Message};\n \n+const APPROX_CHARS_PER_TOKEN: usize = 4;\n+\n+#[derive(Debug, Clone, Copy, PartialEq, Eq)]\n+enum ContextEstimateMethod {\n+ ApiUsagePlusLocalDelta,\n+ LocalEstimate,\n+}\n+\n+impl ContextEstimateMethod {\n+ const fn as_str(self) -> &'static str {\n+ match self {\n+ Self::ApiUsagePlusLocalDelta => \"api_usage_plus_local_delta\",\n+ Self::LocalEstimate => \"local_estimate\",\n+ }\n+ }\n+}\n+\n+#[derive(Debug, Clone, Copy, PartialEq, Eq)]\n+struct ContextEstimate {\n+ tokens: usize,\n+ method: ContextEstimateMethod,\n+}\n+\n /// Check whether the context window usage exceeds the configured threshold.\n /// Emits a `Warning` event with kind `\"context_window\"` when over the\n /// threshold. Returns `true` if the threshold is exceeded.\n@@ -22,21 +45,21 @@ pub fn check_context_usage(\n emitter: &Emitter,\n session_id: &str,\n ) -> bool {\n- let estimated_tokens = estimate_token_count(system_prompt, history);\n+ let estimate = estimate_active_context_usage(system_prompt, history);\n+ let estimated_tokens = estimate.tokens;\n let context_window = provider_profile.context_window_size();\n let threshold = context_window * threshold_percent / 100;\n \n if estimated_tokens > threshold {\n+ let usage_percent = estimated_tokens.saturating_mul(100) / context_window;\n emitter.emit(session_id.to_owned(), AgentEvent::Warning {\n kind: \"context_window\".into(),\n- message: format!(\n- \"Context window usage: {}%\",\n- estimated_tokens * 100 / context_window\n- ),\n+ message: format!(\"Context window usage: {usage_percent}%\"),\n details: serde_json::json!({\n \"estimated_tokens\": estimated_tokens,\n \"context_window_size\": context_window,\n- \"usage_percent\": estimated_tokens * 100 / context_window,\n+ \"usage_percent\": usage_percent,\n+ \"estimate_method\": estimate.method.as_str(),\n }),\n });\n true\n@@ -61,19 +84,21 @@ pub async fn compact_context(\n emitter: &Emitter,\n session_id: &str,\n ) -> Result<(), Error> {\n- let estimated_tokens = estimate_token_count(system_prompt, history);\n- let context_window = provider_profile.context_window_size();\n let original_turn_count = history.turns().len();\n \n+ // Determine turns to summarize. If there are not enough turns to compact,\n+ // do not emit a started event without a matching completion.\n+ if original_turn_count <= preserve_count {\n+ return Ok(());\n+ }\n+\n+ let estimate = estimate_active_context_usage(system_prompt, history);\n+ let context_window = provider_profile.context_window_size();\n emitter.emit(session_id.to_owned(), AgentEvent::CompactionStarted {\n- estimated_tokens,\n+ estimated_tokens: estimate.tokens,\n context_window_size: context_window,\n });\n \n- // Determine turns to summarize\n- if original_turn_count <= preserve_count {\n- return Ok(());\n- }\n let turns_to_summarize = &history.turns()[..original_turn_count - preserve_count];\n let rendered = render_turns_for_summary(turns_to_summarize);\n \n@@ -156,37 +181,63 @@ Build on their progress — do not repeat completed steps.\\n\\n{summary_text}\"\n /// Estimate the total token count of the system prompt and conversation\n /// history. Uses a rough heuristic of ~4 characters per token.\n pub fn estimate_token_count(system_prompt: &str, history: &History) -> usize {\n- let mut total_chars = system_prompt.len();\n+ estimate_local_token_count(system_prompt, history.turns())\n+}\n \n- for turn in history.turns() {\n- match turn {\n- Message::User { content, .. } => total_chars += content.len(),\n- Message::Assistant {\n- content,\n- tool_calls,\n- ..\n- } => {\n- total_chars += content.len();\n- if let Some(r) = turn.reasoning_text() {\n- total_chars += r.len();\n- }\n- for tc in tool_calls {\n- total_chars += tc.name.len();\n- total_chars += tc.arguments.to_string().len();\n- }\n- }\n- Message::ToolResults { results, .. } => {\n- for r in results {\n- total_chars += r.content.to_string().len();\n- }\n- }\n- Message::System { content, .. } | Message::Steering { content, .. } => {\n- total_chars += content.len();\n+fn estimate_active_context_usage(system_prompt: &str, history: &History) -> ContextEstimate {\n+ let turns = history.turns();\n+ if let Some((baseline_index, baseline_tokens)) = latest_assistant_usage_baseline(turns) {\n+ let local_delta = estimate_local_token_count(\"\", &turns[baseline_index + 1..]);\n+ return ContextEstimate {\n+ tokens: baseline_tokens.saturating_add(local_delta),\n+ method: ContextEstimateMethod::ApiUsagePlusLocalDelta,\n+ };\n+ }\n+\n+ ContextEstimate {\n+ tokens: estimate_local_token_count(system_prompt, turns),\n+ method: ContextEstimateMethod::LocalEstimate,\n+ }\n+}\n+\n+fn latest_assistant_usage_baseline(turns: &[Message]) -> Option<(usize, usize)> {\n+ turns.iter().enumerate().rev().find_map(|(index, turn)| {\n+ if let Message::Assistant { usage, .. } = turn {\n+ let total_tokens = usage.total_tokens();\n+ if total_tokens > 0 {\n+ return Some((index, usize::try_from(total_tokens).unwrap_or(usize::MAX)));\n }\n }\n- }\n+ None\n+ })\n+}\n+\n+fn estimate_local_token_count(system_prompt: &str, turns: &[Message]) -> usize {\n+ let turn_chars: usize = turns.iter().map(estimate_turn_chars).sum();\n+ (system_prompt.len() + turn_chars) / APPROX_CHARS_PER_TOKEN\n+}\n \n- total_chars / 4 // rough estimate: ~4 chars per token\n+fn estimate_turn_chars(turn: &Message) -> usize {\n+ match turn {\n+ Message::User { content, .. }\n+ | Message::System { content, .. }\n+ | Message::Steering { content, .. } => content.len(),\n+ Message::Assistant {\n+ content,\n+ tool_calls,\n+ ..\n+ } => {\n+ let reasoning_chars = turn.reasoning_text().map_or(0, str::len);\n+ let tool_call_chars: usize = tool_calls\n+ .iter()\n+ .map(|tc| tc.name.len() + tc.arguments.to_string().len())\n+ .sum();\n+ content.len() + reasoning_chars + tool_call_chars\n+ }\n+ Message::ToolResults { results, .. } => {\n+ results.iter().map(|r| r.content.to_string().len()).sum()\n+ }\n+ }\n }\n \n /// Render conversation turns into a human-readable summary format for the\n@@ -323,6 +374,149 @@ mod tests {\n assert_eq!(estimate_token_count(\"test\", &history), 3);\n }\n \n+ #[test]\n+ fn active_context_estimate_without_assistant_usage_matches_local_history_estimate() {\n+ let mut history = History::default();\n+ history.push(Message::User {\n+ content: \"Hello world\".into(),\n+ timestamp: SystemTime::now(),\n+ });\n+ history.push(Message::Assistant {\n+ content: \"No usage available\".into(),\n+ tool_calls: vec![ToolCall::new(\n+ \"call_1\",\n+ \"read_file\",\n+ serde_json::json!({\"path\": \"foo.rs\"}),\n+ )],\n+ provider_parts: vec![],\n+ usage: Box::new(TokenCounts::default()),\n+ response_id: \"resp_1\".into(),\n+ timestamp: SystemTime::now(),\n+ });\n+ history.push(Message::ToolResults {\n+ results: vec![ToolResult::success(\"call_1\", serde_json::json!(1234))],\n+ timestamp: SystemTime::now(),\n+ });\n+\n+ let estimate = estimate_active_context_usage(\"test\", &history);\n+\n+ assert_eq!(estimate.tokens, estimate_token_count(\"test\", &history));\n+ assert_eq!(estimate.method, ContextEstimateMethod::LocalEstimate);\n+ }\n+\n+ #[test]\n+ fn active_context_estimate_uses_latest_assistant_usage_plus_later_turns() {\n+ let mut history = History::default();\n+ history.push(Message::User {\n+ content: \"ignored before baseline\".repeat(100),\n+ timestamp: SystemTime::now(),\n+ });\n+ history.push(Message::Assistant {\n+ content: \"baseline response\".into(),\n+ tool_calls: vec![],\n+ provider_parts: vec![],\n+ usage: Box::new(TokenCounts {\n+ input_tokens: 50,\n+ ..TokenCounts::default()\n+ }),\n+ response_id: \"resp_1\".into(),\n+ timestamp: SystemTime::now(),\n+ });\n+ history.push(Message::ToolResults {\n+ // JSON number renders as 4 chars => 1 local token.\n+ results: vec![ToolResult::success(\"call_1\", serde_json::json!(1234))],\n+ timestamp: SystemTime::now(),\n+ });\n+ history.push(Message::User {\n+ // 16 chars => 4 local tokens.\n+ content: \"u\".repeat(16),\n+ timestamp: SystemTime::now(),\n+ });\n+ history.push(Message::Steering {\n+ // 8 chars => 2 local tokens.\n+ content: \"s\".repeat(8),\n+ timestamp: SystemTime::now(),\n+ });\n+\n+ let estimate = estimate_active_context_usage(\"ignored system prompt\", &history);\n+\n+ assert_eq!(estimate.tokens, 57);\n+ assert_eq!(\n+ estimate.method,\n+ ContextEstimateMethod::ApiUsagePlusLocalDelta\n+ );\n+ }\n+\n+ #[test]\n+ fn active_context_estimate_uses_total_tokens_including_cache_and_reasoning() {\n+ let mut history = History::default();\n+ history.push(Message::Assistant {\n+ content: \"short\".into(),\n+ tool_calls: vec![],\n+ provider_parts: vec![],\n+ usage: Box::new(TokenCounts {\n+ input_tokens: 10,\n+ output_tokens: 20,\n+ reasoning_tokens: 30,\n+ cache_read_tokens: 40,\n+ cache_write_tokens: 50,\n+ }),\n+ response_id: \"resp_1\".into(),\n+ timestamp: SystemTime::now(),\n+ });\n+\n+ let estimate = estimate_active_context_usage(\"\", &history);\n+\n+ assert_eq!(estimate.tokens, 150);\n+ assert_eq!(\n+ estimate.method,\n+ ContextEstimateMethod::ApiUsagePlusLocalDelta\n+ );\n+ }\n+\n+ #[test]\n+ fn active_context_estimate_ignores_earlier_usage_when_later_usage_exists() {\n+ let mut history = History::default();\n+ history.push(Message::Assistant {\n+ content: \"older response\".into(),\n+ tool_calls: vec![],\n+ provider_parts: vec![],\n+ usage: Box::new(TokenCounts {\n+ input_tokens: 1_000,\n+ ..TokenCounts::default()\n+ }),\n+ response_id: \"resp_old\".into(),\n+ timestamp: SystemTime::now(),\n+ });\n+ history.push(Message::User {\n+ content: \"ignored before latest baseline\".repeat(100),\n+ timestamp: SystemTime::now(),\n+ });\n+ history.push(Message::Assistant {\n+ content: \"latest response\".into(),\n+ tool_calls: vec![],\n+ provider_parts: vec![],\n+ usage: Box::new(TokenCounts {\n+ input_tokens: 20,\n+ ..TokenCounts::default()\n+ }),\n+ response_id: \"resp_new\".into(),\n+ timestamp: SystemTime::now(),\n+ });\n+ history.push(Message::User {\n+ content: \"u\".repeat(8),\n+ timestamp: SystemTime::now(),\n+ });\n+\n+ let estimate = estimate_active_context_usage(\"\", &history);\n+\n+ assert_eq!(estimate.tokens, 22);\n+ assert_eq!(\n+ estimate.method,\n+ ContextEstimateMethod::ApiUsagePlusLocalDelta\n+ );\n+ }\n+\n #[test]\n fn check_context_usage_below_threshold() {\n let history = History::default();\n@@ -350,6 +544,7 @@ mod tests {\n \n // Should have emitted a Warning\n let event = rx.try_recv().unwrap();\n- assert!(matches!(event.event, AgentEvent::Warning { .. }));\n+ assert!(matches!(event.event, AgentEvent::Warning { details, .. }\n+ if details[\"estimate_method\"] == \"local_estimate\"));\n }\n }\ndiff --git a/lib/crates/fabro-agent/src/history.rs b/lib/crates/fabro-agent/src/history.rs\nindex 497900eae..33182ae73 100644\n--- a/lib/crates/fabro-agent/src/history.rs\n+++ b/lib/crates/fabro-agent/src/history.rs\n@@ -1,4 +1,4 @@\n-use fabro_llm::types::{ContentPart, Message as LlmMessage, Role};\n+use fabro_llm::types::{ContentPart, Message as LlmMessage, Role, TokenCounts};\n use fabro_types::SessionMessage;\n \n use crate::types::Message;\n@@ -36,7 +36,8 @@ impl History {\n if self.turns.len() <= preserve_count {\n return;\n }\n- let preserved = self.turns.split_off(self.turns.len() - preserve_count);\n+ let mut preserved = self.turns.split_off(self.turns.len() - preserve_count);\n+ invalidate_assistant_usage(&mut preserved);\n let discarded = std::mem::take(&mut self.turns);\n let extracted_user_messages =\n extract_recent_user_messages(discarded, COMPACTION_USER_MESSAGE_TOKEN_BUDGET);\n@@ -119,6 +120,14 @@ impl History {\n }\n }\n \n+fn invalidate_assistant_usage(turns: &mut [Message]) {\n+ for turn in turns {\n+ if let Message::Assistant { usage, .. } = turn {\n+ **usage = TokenCounts::default();\n+ }\n+ }\n+}\n+\n /// Maximum token budget for user messages extracted from discarded turns during\n /// compaction.\n const COMPACTION_USER_MESSAGE_TOKEN_BUDGET: usize = 20_000;\n@@ -599,6 +608,60 @@ mod tests {\n }\n }\n \n+ #[test]\n+ fn compact_preserves_assistant_data_but_resets_usage() {\n+ let mut history = History::default();\n+ history.push(Message::User {\n+ content: \"old msg\".into(),\n+ timestamp: SystemTime::now(),\n+ });\n+ let tool_call = ToolCall::new(\"call_1\", \"search\", serde_json::json!({\"query\": \"fabro\"}));\n+ let thinking = ContentPart::Thinking(ThinkingData {\n+ text: \"deep thought\".into(),\n+ signature: Some(\"sig_xyz\".into()),\n+ redacted: false,\n+ });\n+ history.push(Message::Assistant {\n+ content: \"answer\".into(),\n+ tool_calls: vec![tool_call.clone()],\n+ provider_parts: vec![thinking.clone()],\n+ usage: Box::new(TokenCounts {\n+ input_tokens: 10,\n+ output_tokens: 20,\n+ reasoning_tokens: 30,\n+ cache_read_tokens: 40,\n+ cache_write_tokens: 50,\n+ }),\n+ response_id: \"resp_1\".into(),\n+ timestamp: SystemTime::now(),\n+ });\n+\n+ history.compact(1, \"Summary\".into());\n+\n+ let assistant_turn = history\n+ .turns()\n+ .iter()\n+ .find(|turn| matches!(turn, Message::Assistant { .. }))\n+ .expect(\"preserved assistant turn\");\n+ if let Message::Assistant {\n+ content,\n+ tool_calls,\n+ provider_parts,\n+ usage,\n+ response_id,\n+ ..\n+ } = assistant_turn\n+ {\n+ assert_eq!(content, \"answer\");\n+ assert_eq!(tool_calls, &[tool_call]);\n+ assert_eq!(provider_parts, &[thinking]);\n+ assert_eq!(response_id, \"resp_1\");\n+ assert_eq!(**usage, TokenCounts::default());\n+ } else {\n+ panic!(\"expected Assistant turn\");\n+ }\n+ }\n+\n #[test]\n fn compact_strips_reasoning_from_all_preserved_assistant_turns() {\n let mut history = History::default();\ndiff --git a/lib/crates/fabro-agent/src/session.rs b/lib/crates/fabro-agent/src/session.rs\nindex 2d0d0b4e8..de9c58093 100644\n--- a/lib/crates/fabro-agent/src/session.rs\n+++ b/lib/crates/fabro-agent/src/session.rs\n@@ -1853,7 +1853,7 @@ mod tests {\n use fabro_llm::error::{ProviderErrorDetail, ProviderErrorKind};\n use fabro_llm::provider::{ProviderAdapter, StreamEventStream};\n use fabro_llm::types::{\n- ContentPart, ReasoningEffort, Request, Response, Role, StreamEvent, ToolCall,\n+ ContentPart, ReasoningEffort, Request, Response, Role, StreamEvent, TokenCounts, ToolCall,\n ToolDefinition,\n };\n use futures::stream;\n@@ -3653,13 +3653,25 @@ mod tests {\n assert!(found_auth_error_event, \"expected auth error event\");\n }\n \n+ fn response_with_usage(mut response: Response, usage: TokenCounts) -> Response {\n+ response.usage = usage;\n+ response\n+ }\n+\n+ fn response_with_total_usage(response: Response, total_tokens: i64) -> Response {\n+ response_with_usage(response, TokenCounts {\n+ input_tokens: total_tokens,\n+ ..TokenCounts::default()\n+ })\n+ }\n+\n #[tokio::test]\n async fn compaction_triggered_when_over_threshold() {\n // Tiny context window to trigger compaction\n // Responses: [0] conversation response (stream), [1] summarization (complete),\n // [2] unused fallback\n let responses = vec![\n- text_response(\"OK\"),\n+ response_with_usage(text_response(\"OK\"), TokenCounts::default()),\n text_response(\"Here is the summary of the conversation so far.\"),\n text_response(\"fallback\"),\n ];\n@@ -3704,6 +3716,90 @@ mod tests {\n );\n }\n \n+ #[tokio::test]\n+ async fn compaction_uses_assistant_usage_baseline_for_short_response() {\n+ let responses = vec![\n+ response_with_total_usage(text_response(\"OK\"), 90),\n+ text_response(\"Here is the summary of the conversation so far.\"),\n+ text_response(\"fallback\"),\n+ ];\n+\n+ let provider = Arc::new(MockLlmProvider::new(responses));\n+ let client = make_client(provider).await;\n+ let registry = ToolRegistry::new();\n+ let profile = Arc::new(TestProfile::with_context_window(registry, 100));\n+ let env = Arc::new(MockSandbox::default());\n+ let config = SessionOptions {\n+ enable_context_compaction: true,\n+ compaction_preserve_turns: 1,\n+ ..Default::default()\n+ };\n+ let mut session = Session::new(client, profile, env, config, None);\n+ let mut rx = session.subscribe();\n+\n+ session.process_input(\"hi\").await.unwrap();\n+\n+ let mut started = None;\n+ let mut found_completed = false;\n+ while let Ok(event) = rx.try_recv() {\n+ match event.event {\n+ AgentEvent::CompactionStarted {\n+ estimated_tokens,\n+ context_window_size,\n+ } => started = Some((estimated_tokens, context_window_size)),\n+ AgentEvent::CompactionCompleted { .. } => found_completed = true,\n+ _ => {}\n+ }\n+ }\n+\n+ assert_eq!(started, Some((90, 100)));\n+ assert!(\n+ found_completed,\n+ \"CompactionCompleted event should be emitted\"\n+ );\n+ }\n+\n+ #[tokio::test]\n+ async fn compaction_noop_does_not_emit_started() {\n+ let large_input = \"x\".repeat(400);\n+ let responses = vec![text_response(\"OK\")];\n+\n+ let provider = Arc::new(MockLlmProvider::new(responses));\n+ let client = make_client(provider).await;\n+ let registry = ToolRegistry::new();\n+ let profile = Arc::new(TestProfile::with_context_window(registry, 100));\n+ let env = Arc::new(MockSandbox::default());\n+ let config = SessionOptions {\n+ enable_context_compaction: true,\n+ compaction_preserve_turns: 10,\n+ ..Default::default()\n+ };\n+ let mut session = Session::new(client, profile, env, config, None);\n+ let mut rx = session.subscribe();\n+\n+ session.process_input(&large_input).await.unwrap();\n+\n+ let mut found_warning = false;\n+ let mut found_compaction = false;\n+ while let Ok(event) = rx.try_recv() {\n+ match event.event {\n+ AgentEvent::Warning { kind, .. } if kind == \"context_window\" => {\n+ found_warning = true;\n+ }\n+ AgentEvent::CompactionStarted { .. } | AgentEvent::CompactionCompleted { .. } => {\n+ found_compaction = true;\n+ }\n+ _ => {}\n+ }\n+ }\n+\n+ assert!(found_warning, \"threshold should have been exceeded\");\n+ assert!(\n+ !found_compaction,\n+ \"no-op compaction should not emit started or completed events\"\n+ );\n+ }\n+\n #[tokio::test]\n async fn compaction_not_triggered_when_disabled() {\n let large_input = \"x\".repeat(400);\n@@ -3735,6 +3831,49 @@ mod tests {\n assert!(!found_compaction, \"No compaction events when disabled\");\n }\n \n+ #[tokio::test]\n+ async fn compaction_disabled_blocks_api_usage_baseline_compaction() {\n+ let responses = vec![response_with_total_usage(text_response(\"OK\"), 90)];\n+\n+ let provider = Arc::new(MockLlmProvider::new(responses));\n+ let client = make_client(provider).await;\n+ let registry = ToolRegistry::new();\n+ let profile = Arc::new(TestProfile::with_context_window(registry, 100));\n+ let env = Arc::new(MockSandbox::default());\n+ let config = SessionOptions {\n+ enable_context_compaction: false,\n+ compaction_preserve_turns: 1,\n+ ..Default::default()\n+ };\n+ let mut session = Session::new(client, profile, env, config, None);\n+ let mut rx = session.subscribe();\n+\n+ session.process_input(\"hi\").await.unwrap();\n+\n+ let mut found_api_usage_warning = false;\n+ let mut found_compaction = false;\n+ while let Ok(event) = rx.try_recv() {\n+ match event.event {\n+ AgentEvent::Warning { details, .. }\n+ if details[\"estimated_tokens\"] == 90\n+ && details[\"estimate_method\"] == \"api_usage_plus_local_delta\" =>\n+ {\n+ found_api_usage_warning = true;\n+ }\n+ AgentEvent::CompactionStarted { .. } | AgentEvent::CompactionCompleted { .. } => {\n+ found_compaction = true;\n+ }\n+ _ => {}\n+ }\n+ }\n+\n+ assert!(\n+ found_api_usage_warning,\n+ \"API usage baseline should still drive context warning\"\n+ );\n+ assert!(!found_compaction, \"compaction must remain disabled\");\n+ }\n+\n #[tokio::test]\n async fn compaction_failure_is_non_fatal() {\n // Response [0] = conversation response (stream), [1] will be used for\n@@ -3789,7 +3928,10 @@ mod tests {\n }\n \n let large_input = \"x\".repeat(400);\n- let responses = vec![text_response(\"OK\")];\n+ let responses = vec![response_with_usage(\n+ text_response(\"OK\"),\n+ TokenCounts::default(),\n+ )];\n \n let provider = Arc::new(StreamOnlyProvider {\n responses,\n",
|
|
"summary": {
|
|
"files_changed": 3,
|
|
"additions": 446,
|
|
"deletions": 46
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"seq": 703,
|
|
"checkpoint": {
|
|
"timestamp": "2026-05-23T15:43:59.608340Z",
|
|
"current_node": "simplify_opus",
|
|
"completed_nodes": [
|
|
"start",
|
|
"toolchain",
|
|
"preflight_compile",
|
|
"preflight_lint",
|
|
"implement",
|
|
"simplify_opus"
|
|
],
|
|
"node_retries": {},
|
|
"context_values": {
|
|
"thread.preflight_compile.current_node": "preflight_lint",
|
|
"graph.rankdir": "LR",
|
|
"internal.retry_count.implement": 0,
|
|
"internal.run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6",
|
|
"graph.goal": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n",
|
|
"internal.node_visit_count": 1,
|
|
"response.simplify_opus": "All clean. Here's a summary of what was fixed based on the three reviews:\n\n## Summary of cleanup applied\n\n**`lib/crates/fabro-agent/src/compaction.rs`**\n- Removed the no-longer-needed `pub fn estimate_token_count` (no external callers); tests updated to call `estimate_active_context_usage` directly.\n- Replaced the hand-rolled `ContextEstimateMethod::as_str()` with `#[derive(strum::IntoStaticStr)] #[strum(serialize_all = \"snake_case\")]` per repo convention in CLAUDE.md.\n- Split `estimate_local_token_count(system_prompt, turns)` into two single-purpose helpers (`estimate_turns_local_tokens`, `estimate_system_prompt_local_tokens`) to remove the `\"\"` sentinel call from the baseline+delta path.\n- Replaced the literal `4` in `summary_token_estimate = summary_content.len() / 4` with `APPROX_CHARS_PER_TOKEN`.\n- Made `check_context_usage` return `Option<ContextEstimate>` and `compact_context` accept the pre-computed estimate, eliminating a duplicate full estimate scan per actual compaction. Both fns are now `pub(crate)` since they have no external callers. Dropped the redundant `system_prompt` parameter from `compact_context`.\n- Tightened visibility on `ContextEstimate` and `ContextEstimateMethod` to `pub(crate)`.\n\n**`lib/crates/fabro-agent/src/history.rs`**\n- Moved the free `invalidate_assistant_usage` function into `impl History` as a private `invalidate_preserved_usage` method.\n- Added a doc comment on `History::compact` explaining the usage-invalidation invariant.\n\n**`lib/crates/fabro-agent/src/session.rs`**\n- Rewired `compact_if_needed` to consume the `Option<ContextEstimate>` from `check_context_usage` and pass it through.\n- Renamed misleading test helper `response_with_total_usage` → `response_with_input_tokens` (it only set `input_tokens`, never the other usage fields).\n\n**`lib/crates/fabro-agent/Cargo.toml`**\n- Added `strum.workspace = true` (workspace dep) to support the derive.\n\n## Findings skipped\n\n- **Test builder for `Message::Assistant`** (Quality #3): broader scope than just the diff and only marginal cleanup.\n- **Counting-writer for JSON length** (Efficiency #3): the hot-path concern is real but speculative without profiling; the prevailing pattern in fabro-agent uses `Value::to_string()`. Out of scope.\n- **Extract `\"context_window\"` / `\"estimate_method\"` JSON-key constants** (Quality #5): pre-existing pattern beyond this diff.\n- **Reshape `latest_assistant_usage_baseline` to return `(tokens, &[Message])`** (Quality #4): cosmetic; current `+ 1` indexing is local and clear.\n- **`APPROX_CHARS_PER_TOKEN` in `history.rs` `extract_recent_user_messages`** (Reuse #1): out of scope — that code is not part of the change.\n\nVerification: `cargo nextest run -p fabro-agent` → 447/447 pass; `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` → clean; `cargo +nightly-2026-04-14 fmt --check --all` → clean.",
|
|
"thread.preflight_lint.current_node": "implement",
|
|
"internal.retry_count.simplify_opus": 0,
|
|
"internal.fidelity": "compact",
|
|
"thread.implement.current_node": "simplify_opus",
|
|
"failure_class": "",
|
|
"thread.start.current_node": "toolchain",
|
|
"thread.toolchain.current_node": "preflight_compile",
|
|
"internal.retry_count.toolchain": 0,
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126",
|
|
"internal.work_dir": "/home/daytona/workspace/fabro",
|
|
"failure_signature": "",
|
|
"last_stage": "simplify_opus",
|
|
"graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ",
|
|
"internal.retry_count.preflight_lint": 0,
|
|
"internal.retry_count.start": 0,
|
|
"last_response": "All clean. Here's a summary of what was fixed based on the three reviews:\n\n## Summary of cleanup applied\n\n**`lib/crates/fabro-agent/src/compaction.rs`**\n- Removed the no-longer-needed `pub fn estimate",
|
|
"internal.retry_count.preflight_compile": 0,
|
|
"current_node": "simplify_opus",
|
|
"response.implement": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.total_tokens() > 0` as baseline.\n - Adds local estimates only for turns after that assistant response.\n - Falls back to existing whole-history `chars / 4` local estimate when no usable assistant usage exists.\n - Adds `estimate_method` to context-window warning details.\n - Uses the improved estimate for `CompactionStarted`.\n - Prevents no-op compaction from emitting `CompactionStarted`.\n- Updated `lib/crates/fabro-agent/src/history.rs`\n - `History::compact()` now resets preserved assistant `usage` to `TokenCounts::default()`.\n - Assistant content, tool calls, response IDs, and non-stripped provider parts remain preserved.\n- Added/updated tests in:\n - `compaction.rs`\n - `history.rs`\n - `session.rs`\n- Confirmed no `llm_client.count_input_tokens()` / provider token-count API calls were added.\n\nVerification:\n- Red phase confirmed initially with failing new compaction tests.\n- Passed:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - `cargo nextest run -p fabro-agent session`\n - `cargo nextest run -p fabro-agent`\n - `cargo +nightly-2026-04-14 fmt --check --all`\n - `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings`",
|
|
"internal.thread_id": "implement",
|
|
"outcome": "succeeded"
|
|
},
|
|
"node_outcomes": {
|
|
"toolchain": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c"
|
|
},
|
|
"notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
|
"usage": null
|
|
},
|
|
"simplify_opus": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"last_stage": "simplify_opus",
|
|
"last_response": "All clean. Here's a summary of what was fixed based on the three reviews:\n\n## Summary of cleanup applied\n\n**`lib/crates/fabro-agent/src/compaction.rs`**\n- Removed the no-longer-needed `pub fn estimate",
|
|
"response.simplify_opus": "All clean. Here's a summary of what was fixed based on the three reviews:\n\n## Summary of cleanup applied\n\n**`lib/crates/fabro-agent/src/compaction.rs`**\n- Removed the no-longer-needed `pub fn estimate_token_count` (no external callers); tests updated to call `estimate_active_context_usage` directly.\n- Replaced the hand-rolled `ContextEstimateMethod::as_str()` with `#[derive(strum::IntoStaticStr)] #[strum(serialize_all = \"snake_case\")]` per repo convention in CLAUDE.md.\n- Split `estimate_local_token_count(system_prompt, turns)` into two single-purpose helpers (`estimate_turns_local_tokens`, `estimate_system_prompt_local_tokens`) to remove the `\"\"` sentinel call from the baseline+delta path.\n- Replaced the literal `4` in `summary_token_estimate = summary_content.len() / 4` with `APPROX_CHARS_PER_TOKEN`.\n- Made `check_context_usage` return `Option<ContextEstimate>` and `compact_context` accept the pre-computed estimate, eliminating a duplicate full estimate scan per actual compaction. Both fns are now `pub(crate)` since they have no external callers. Dropped the redundant `system_prompt` parameter from `compact_context`.\n- Tightened visibility on `ContextEstimate` and `ContextEstimateMethod` to `pub(crate)`.\n\n**`lib/crates/fabro-agent/src/history.rs`**\n- Moved the free `invalidate_assistant_usage` function into `impl History` as a private `invalidate_preserved_usage` method.\n- Added a doc comment on `History::compact` explaining the usage-invalidation invariant.\n\n**`lib/crates/fabro-agent/src/session.rs`**\n- Rewired `compact_if_needed` to consume the `Option<ContextEstimate>` from `check_context_usage` and pass it through.\n- Renamed misleading test helper `response_with_total_usage` → `response_with_input_tokens` (it only set `input_tokens`, never the other usage fields).\n\n**`lib/crates/fabro-agent/Cargo.toml`**\n- Added `strum.workspace = true` (workspace dep) to support the derive.\n\n## Findings skipped\n\n- **Test builder for `Message::Assistant`** (Quality #3): broader scope than just the diff and only marginal cleanup.\n- **Counting-writer for JSON length** (Efficiency #3): the hot-path concern is real but speculative without profiling; the prevailing pattern in fabro-agent uses `Value::to_string()`. Out of scope.\n- **Extract `\"context_window\"` / `\"estimate_method\"` JSON-key constants** (Quality #5): pre-existing pattern beyond this diff.\n- **Reshape `latest_assistant_usage_baseline` to return `(tokens, &[Message])`** (Quality #4): cosmetic; current `+ 1` indexing is local and clear.\n- **`APPROX_CHARS_PER_TOKEN` in `history.rs` `extract_recent_user_messages`** (Reuse #1): out of scope — that code is not part of the change.\n\nVerification: `cargo nextest run -p fabro-agent` → 447/447 pass; `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` → clean; `cargo +nightly-2026-04-14 fmt --check --all` → clean."
|
|
},
|
|
"notes": "Stage completed: simplify_opus",
|
|
"usage": {
|
|
"input": {
|
|
"usage": {
|
|
"model": {
|
|
"provider": "anthropic",
|
|
"model_id": "claude-opus-4-7"
|
|
},
|
|
"tokens": {
|
|
"input_tokens": 66736,
|
|
"output_tokens": 21289,
|
|
"reasoning_tokens": 0,
|
|
"cache_read_tokens": 1971515,
|
|
"cache_write_tokens": 267947
|
|
}
|
|
},
|
|
"facts": {
|
|
"algorithm": "anthropic",
|
|
"cache_write_5m_tokens": 267947,
|
|
"cache_write_1h_tokens": 0
|
|
}
|
|
},
|
|
"total_usd_micros": 3526330
|
|
},
|
|
"files_touched": [
|
|
"/home/daytona/workspace/fabro/lib/crates/fabro-agent/Cargo.toml",
|
|
"/home/daytona/workspace/fabro/lib/crates/fabro-agent/src/compaction.rs",
|
|
"/home/daytona/workspace/fabro/lib/crates/fabro-agent/src/history.rs",
|
|
"/home/daytona/workspace/fabro/lib/crates/fabro-agent/src/session.rs"
|
|
]
|
|
},
|
|
"implement": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"last_response": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.tota",
|
|
"last_stage": "implement",
|
|
"response.implement": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.total_tokens() > 0` as baseline.\n - Adds local estimates only for turns after that assistant response.\n - Falls back to existing whole-history `chars / 4` local estimate when no usable assistant usage exists.\n - Adds `estimate_method` to context-window warning details.\n - Uses the improved estimate for `CompactionStarted`.\n - Prevents no-op compaction from emitting `CompactionStarted`.\n- Updated `lib/crates/fabro-agent/src/history.rs`\n - `History::compact()` now resets preserved assistant `usage` to `TokenCounts::default()`.\n - Assistant content, tool calls, response IDs, and non-stripped provider parts remain preserved.\n- Added/updated tests in:\n - `compaction.rs`\n - `history.rs`\n - `session.rs`\n- Confirmed no `llm_client.count_input_tokens()` / provider token-count API calls were added.\n\nVerification:\n- Red phase confirmed initially with failing new compaction tests.\n- Passed:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - `cargo nextest run -p fabro-agent session`\n - `cargo nextest run -p fabro-agent`\n - `cargo +nightly-2026-04-14 fmt --check --all`\n - `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings`"
|
|
},
|
|
"notes": "Stage completed: implement",
|
|
"usage": {
|
|
"input": {
|
|
"usage": {
|
|
"model": {
|
|
"provider": "openai",
|
|
"model_id": "gpt-5.5"
|
|
},
|
|
"tokens": {
|
|
"input_tokens": 168644,
|
|
"output_tokens": 9772,
|
|
"reasoning_tokens": 11720,
|
|
"cache_read_tokens": 3959296,
|
|
"cache_write_tokens": 0
|
|
}
|
|
},
|
|
"facts": {
|
|
"algorithm": "openai"
|
|
}
|
|
},
|
|
"total_usd_micros": 3467628
|
|
}
|
|
},
|
|
"preflight_compile": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
|
},
|
|
"notes": "Script completed: cargo check -q --workspace 2>&1",
|
|
"usage": null
|
|
},
|
|
"preflight_lint": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
|
},
|
|
"notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1",
|
|
"usage": null
|
|
},
|
|
"start": {
|
|
"status": "succeeded",
|
|
"usage": null
|
|
}
|
|
},
|
|
"next_node_id": "simplify_gpt",
|
|
"git_commit_sha": "359232c22a87332043e06a4d2446a2e8f30fa08e",
|
|
"node_visits": {
|
|
"preflight_compile": 1,
|
|
"toolchain": 1,
|
|
"implement": 1,
|
|
"preflight_lint": 1,
|
|
"simplify_opus": 1,
|
|
"start": 1
|
|
}
|
|
},
|
|
"diff": {
|
|
"patch": "diff --git a/Cargo.lock b/Cargo.lock\nindex 789a8f4ce..7ec0867dc 100644\n--- a/Cargo.lock\n+++ b/Cargo.lock\n@@ -1629,6 +1629,7 @@ dependencies = [\n \"serde_json\",\n \"sha2\",\n \"shell-escape\",\n+ \"strum\",\n \"tempfile\",\n \"thiserror 2.0.18\",\n \"tokio\",\ndiff --git a/lib/crates/fabro-agent/Cargo.toml b/lib/crates/fabro-agent/Cargo.toml\nindex be015c24d..e1fb7d0ab 100644\n--- a/lib/crates/fabro-agent/Cargo.toml\n+++ b/lib/crates/fabro-agent/Cargo.toml\n@@ -38,6 +38,7 @@ fabro-http.workspace = true\n thiserror.workspace = true\n serde.workspace = true\n serde_json.workspace = true\n+strum.workspace = true\n tokio.workspace = true\n uuid.workspace = true\n futures.workspace = true\ndiff --git a/lib/crates/fabro-agent/src/compaction.rs b/lib/crates/fabro-agent/src/compaction.rs\nindex 7c2434fa7..f610cdd89 100644\n--- a/lib/crates/fabro-agent/src/compaction.rs\n+++ b/lib/crates/fabro-agent/src/compaction.rs\n@@ -13,58 +13,51 @@ use crate::types::{AgentEvent, Message};\n \n const APPROX_CHARS_PER_TOKEN: usize = 4;\n \n-#[derive(Debug, Clone, Copy, PartialEq, Eq)]\n-enum ContextEstimateMethod {\n+#[derive(Debug, Clone, Copy, PartialEq, Eq, strum::IntoStaticStr)]\n+#[strum(serialize_all = \"snake_case\")]\n+pub(crate) enum ContextEstimateMethod {\n ApiUsagePlusLocalDelta,\n LocalEstimate,\n }\n \n-impl ContextEstimateMethod {\n- const fn as_str(self) -> &'static str {\n- match self {\n- Self::ApiUsagePlusLocalDelta => \"api_usage_plus_local_delta\",\n- Self::LocalEstimate => \"local_estimate\",\n- }\n- }\n-}\n-\n #[derive(Debug, Clone, Copy, PartialEq, Eq)]\n-struct ContextEstimate {\n- tokens: usize,\n- method: ContextEstimateMethod,\n+pub(crate) struct ContextEstimate {\n+ pub tokens: usize,\n+ pub method: ContextEstimateMethod,\n }\n \n /// Check whether the context window usage exceeds the configured threshold.\n /// Emits a `Warning` event with kind `\"context_window\"` when over the\n-/// threshold. Returns `true` if the threshold is exceeded.\n-pub fn check_context_usage(\n+/// threshold. Returns `Some(estimate)` if the threshold is exceeded so the\n+/// caller can pass it to `compact_context` without recomputing.\n+pub(crate) fn check_context_usage(\n system_prompt: &str,\n history: &History,\n provider_profile: &dyn AgentProfile,\n threshold_percent: usize,\n emitter: &Emitter,\n session_id: &str,\n-) -> bool {\n+) -> Option<ContextEstimate> {\n let estimate = estimate_active_context_usage(system_prompt, history);\n- let estimated_tokens = estimate.tokens;\n let context_window = provider_profile.context_window_size();\n let threshold = context_window * threshold_percent / 100;\n \n- if estimated_tokens > threshold {\n- let usage_percent = estimated_tokens.saturating_mul(100) / context_window;\n+ if estimate.tokens > threshold {\n+ let usage_percent = estimate.tokens.saturating_mul(100) / context_window;\n+ let method: &'static str = estimate.method.into();\n emitter.emit(session_id.to_owned(), AgentEvent::Warning {\n kind: \"context_window\".into(),\n message: format!(\"Context window usage: {usage_percent}%\"),\n details: serde_json::json!({\n- \"estimated_tokens\": estimated_tokens,\n+ \"estimated_tokens\": estimate.tokens,\n \"context_window_size\": context_window,\n \"usage_percent\": usage_percent,\n- \"estimate_method\": estimate.method.as_str(),\n+ \"estimate_method\": method,\n }),\n });\n- true\n+ Some(estimate)\n } else {\n- false\n+ None\n }\n }\n \n@@ -74,13 +67,13 @@ pub fn check_context_usage(\n clippy::too_many_arguments,\n reason = \"Context compaction needs explicit history, model, tracking, and emission inputs.\"\n )]\n-pub async fn compact_context(\n+pub(crate) async fn compact_context(\n history: &mut History,\n llm_client: &Client,\n provider_profile: &dyn AgentProfile,\n- system_prompt: &str,\n file_tracker: &FileTracker,\n preserve_count: usize,\n+ estimate: ContextEstimate,\n emitter: &Emitter,\n session_id: &str,\n ) -> Result<(), Error> {\n@@ -92,11 +85,9 @@ pub async fn compact_context(\n return Ok(());\n }\n \n- let estimate = estimate_active_context_usage(system_prompt, history);\n- let context_window = provider_profile.context_window_size();\n emitter.emit(session_id.to_owned(), AgentEvent::CompactionStarted {\n estimated_tokens: estimate.tokens,\n- context_window_size: context_window,\n+ context_window_size: provider_profile.context_window_size(),\n });\n \n let turns_to_summarize = &history.turns()[..original_turn_count - preserve_count];\n@@ -164,7 +155,7 @@ function names, error messages, and exact values. Omit pleasantries and conversa\n \"A different assistant began this task and produced the following summary. \\\n Build on their progress — do not repeat completed steps.\\n\\n{summary_text}\"\n );\n- let summary_token_estimate = summary_content.len() / 4;\n+ let summary_token_estimate = summary_content.len() / APPROX_CHARS_PER_TOKEN;\n \n history.compact(preserve_count, summary_content);\n \n@@ -178,16 +169,13 @@ Build on their progress — do not repeat completed steps.\\n\\n{summary_text}\"\n Ok(())\n }\n \n-/// Estimate the total token count of the system prompt and conversation\n-/// history. Uses a rough heuristic of ~4 characters per token.\n-pub fn estimate_token_count(system_prompt: &str, history: &History) -> usize {\n- estimate_local_token_count(system_prompt, history.turns())\n-}\n-\n-fn estimate_active_context_usage(system_prompt: &str, history: &History) -> ContextEstimate {\n+pub(crate) fn estimate_active_context_usage(\n+ system_prompt: &str,\n+ history: &History,\n+) -> ContextEstimate {\n let turns = history.turns();\n if let Some((baseline_index, baseline_tokens)) = latest_assistant_usage_baseline(turns) {\n- let local_delta = estimate_local_token_count(\"\", &turns[baseline_index + 1..]);\n+ let local_delta = estimate_turns_local_tokens(&turns[baseline_index + 1..]);\n return ContextEstimate {\n tokens: baseline_tokens.saturating_add(local_delta),\n method: ContextEstimateMethod::ApiUsagePlusLocalDelta,\n@@ -195,7 +183,8 @@ fn estimate_active_context_usage(system_prompt: &str, history: &History) -> Cont\n }\n \n ContextEstimate {\n- tokens: estimate_local_token_count(system_prompt, turns),\n+ tokens: estimate_system_prompt_local_tokens(system_prompt)\n+ + estimate_turns_local_tokens(turns),\n method: ContextEstimateMethod::LocalEstimate,\n }\n }\n@@ -212,9 +201,12 @@ fn latest_assistant_usage_baseline(turns: &[Message]) -> Option<(usize, usize)>\n })\n }\n \n-fn estimate_local_token_count(system_prompt: &str, turns: &[Message]) -> usize {\n- let turn_chars: usize = turns.iter().map(estimate_turn_chars).sum();\n- (system_prompt.len() + turn_chars) / APPROX_CHARS_PER_TOKEN\n+fn estimate_turns_local_tokens(turns: &[Message]) -> usize {\n+ turns.iter().map(estimate_turn_chars).sum::<usize>() / APPROX_CHARS_PER_TOKEN\n+}\n+\n+fn estimate_system_prompt_local_tokens(system_prompt: &str) -> usize {\n+ system_prompt.len() / APPROX_CHARS_PER_TOKEN\n }\n \n fn estimate_turn_chars(turn: &Message) -> usize {\n@@ -364,24 +356,27 @@ mod tests {\n }\n \n #[test]\n- fn estimate_token_count_basic() {\n+ fn estimate_local_token_count_basic() {\n let mut history = History::default();\n history.push(Message::User {\n content: \"Hello world\".into(), // 11 chars\n timestamp: SystemTime::now(),\n });\n- // system_prompt = \"test\" (4 chars) + 11 chars = 15 chars / 4 = 3 tokens\n- assert_eq!(estimate_token_count(\"test\", &history), 3);\n+ // system_prompt = \"test\" (4/4 = 1 token) + 11 chars / 4 = 2 tokens = 3 tokens\n+ let estimate = estimate_active_context_usage(\"test\", &history);\n+ assert_eq!(estimate.tokens, 3);\n+ assert_eq!(estimate.method, ContextEstimateMethod::LocalEstimate);\n }\n \n #[test]\n- fn active_context_estimate_without_assistant_usage_matches_local_history_estimate() {\n+ fn active_context_estimate_without_assistant_usage_uses_local_estimate() {\n let mut history = History::default();\n history.push(Message::User {\n- content: \"Hello world\".into(),\n+ content: \"Hello world\".into(), // 11 chars => 2 tokens\n timestamp: SystemTime::now(),\n });\n history.push(Message::Assistant {\n+ // 18 chars content + tool call name (9) + args (16) = 43 chars => 10 tokens\n content: \"No usage available\".into(),\n tool_calls: vec![ToolCall::new(\n \"call_1\",\n@@ -394,14 +389,16 @@ mod tests {\n timestamp: SystemTime::now(),\n });\n history.push(Message::ToolResults {\n+ // 4 chars => 1 token\n results: vec![ToolResult::success(\"call_1\", serde_json::json!(1234))],\n timestamp: SystemTime::now(),\n });\n \n let estimate = estimate_active_context_usage(\"test\", &history);\n \n- assert_eq!(estimate.tokens, estimate_token_count(\"test\", &history));\n assert_eq!(estimate.method, ContextEstimateMethod::LocalEstimate);\n+ // sysprompt 4/4 = 1, turns sum = (11 + 18 + 9 + 16 + 4) / 4 = 58/4 = 14\n+ assert_eq!(estimate.tokens, 15);\n }\n \n #[test]\n@@ -523,8 +520,8 @@ mod tests {\n let emitter = Emitter::new();\n let profile = TestProfile::new();\n // Empty history, huge context window => well below threshold\n- let over = check_context_usage(\"short\", &history, &profile, 80, &emitter, \"sess\");\n- assert!(!over);\n+ let result = check_context_usage(\"short\", &history, &profile, 80, &emitter, \"sess\");\n+ assert!(result.is_none());\n }\n \n #[test]\n@@ -539,8 +536,8 @@ mod tests {\n let mut rx = emitter.subscribe();\n // TestProfile has context_window=200_000 by default; use a small one\n let profile = TestProfile::with_context_window(ToolRegistry::new(), 100);\n- let over = check_context_usage(\"prompt\", &history, &profile, 80, &emitter, \"sess\");\n- assert!(over);\n+ let result = check_context_usage(\"prompt\", &history, &profile, 80, &emitter, \"sess\");\n+ assert!(result.is_some());\n \n // Should have emitted a Warning\n let event = rx.try_recv().unwrap();\ndiff --git a/lib/crates/fabro-agent/src/history.rs b/lib/crates/fabro-agent/src/history.rs\nindex 33182ae73..e9dd10476 100644\n--- a/lib/crates/fabro-agent/src/history.rs\n+++ b/lib/crates/fabro-agent/src/history.rs\n@@ -32,12 +32,17 @@ impl History {\n self.turns.iter().map(Message::to_session_message).collect()\n }\n \n+ /// Compact the history by replacing all but the trailing `preserve_count`\n+ /// turns with a summary `System` message. Preserved assistant turns have\n+ /// their `usage` reset to default so a later context-window estimate does\n+ /// not treat pre-compaction provider-reported usage as the new baseline;\n+ /// authoritative billing is recorded via emitted run events.\n pub fn compact(&mut self, preserve_count: usize, summary: String) {\n if self.turns.len() <= preserve_count {\n return;\n }\n let mut preserved = self.turns.split_off(self.turns.len() - preserve_count);\n- invalidate_assistant_usage(&mut preserved);\n+ Self::invalidate_preserved_usage(&mut preserved);\n let discarded = std::mem::take(&mut self.turns);\n let extracted_user_messages =\n extract_recent_user_messages(discarded, COMPACTION_USER_MESSAGE_TOKEN_BUDGET);\n@@ -50,6 +55,14 @@ impl History {\n self.strip_opaque_provider_items();\n }\n \n+ fn invalidate_preserved_usage(preserved: &mut [Message]) {\n+ for turn in preserved {\n+ if let Message::Assistant { usage, .. } = turn {\n+ **usage = TokenCounts::default();\n+ }\n+ }\n+ }\n+\n /// Remove provider-specific opaque items that are no longer valid after\n /// compaction. OpenAI reasoning and message items are opaque round-trip\n /// data tied to specific API responses; after compaction replaces their\n@@ -120,14 +133,6 @@ impl History {\n }\n }\n \n-fn invalidate_assistant_usage(turns: &mut [Message]) {\n- for turn in turns {\n- if let Message::Assistant { usage, .. } = turn {\n- **usage = TokenCounts::default();\n- }\n- }\n-}\n-\n /// Maximum token budget for user messages extracted from discarded turns during\n /// compaction.\n const COMPACTION_USER_MESSAGE_TOKEN_BUDGET: usize = 20_000;\ndiff --git a/lib/crates/fabro-agent/src/session.rs b/lib/crates/fabro-agent/src/session.rs\nindex de9c58093..d999666e4 100644\n--- a/lib/crates/fabro-agent/src/session.rs\n+++ b/lib/crates/fabro-agent/src/session.rs\n@@ -1670,31 +1670,34 @@ impl Session {\n }\n \n async fn compact_if_needed(&mut self) {\n- let over_threshold = check_context_usage(\n+ let Some(estimate) = check_context_usage(\n &self.system_prompt,\n &self.history,\n self.provider_profile.as_ref(),\n self.config.compaction_threshold_percent,\n &self.event_emitter,\n &self.id,\n- );\n- if over_threshold && self.config.enable_context_compaction {\n- if let Err(e) = compact_context(\n- &mut self.history,\n- &self.llm_client,\n- self.provider_profile.as_ref(),\n- &self.system_prompt,\n- &self.file_tracker,\n- self.config.compaction_preserve_turns,\n- &self.event_emitter,\n- &self.id,\n- )\n- .await\n- {\n- self.event_emitter.emit(self.id.clone(), AgentEvent::Error {\n- error: Error::InvalidState(format!(\"Context compaction failed: {e}\")),\n- });\n- }\n+ ) else {\n+ return;\n+ };\n+ if !self.config.enable_context_compaction {\n+ return;\n+ }\n+ if let Err(e) = compact_context(\n+ &mut self.history,\n+ &self.llm_client,\n+ self.provider_profile.as_ref(),\n+ &self.file_tracker,\n+ self.config.compaction_preserve_turns,\n+ estimate,\n+ &self.event_emitter,\n+ &self.id,\n+ )\n+ .await\n+ {\n+ self.event_emitter.emit(self.id.clone(), AgentEvent::Error {\n+ error: Error::InvalidState(format!(\"Context compaction failed: {e}\")),\n+ });\n }\n }\n \n@@ -3658,9 +3661,9 @@ mod tests {\n response\n }\n \n- fn response_with_total_usage(response: Response, total_tokens: i64) -> Response {\n+ fn response_with_input_tokens(response: Response, input_tokens: i64) -> Response {\n response_with_usage(response, TokenCounts {\n- input_tokens: total_tokens,\n+ input_tokens,\n ..TokenCounts::default()\n })\n }\n@@ -3719,7 +3722,7 @@ mod tests {\n #[tokio::test]\n async fn compaction_uses_assistant_usage_baseline_for_short_response() {\n let responses = vec![\n- response_with_total_usage(text_response(\"OK\"), 90),\n+ response_with_input_tokens(text_response(\"OK\"), 90),\n text_response(\"Here is the summary of the conversation so far.\"),\n text_response(\"fallback\"),\n ];\n@@ -3833,7 +3836,7 @@ mod tests {\n \n #[tokio::test]\n async fn compaction_disabled_blocks_api_usage_baseline_compaction() {\n- let responses = vec![response_with_total_usage(text_response(\"OK\"), 90)];\n+ let responses = vec![response_with_input_tokens(text_response(\"OK\"), 90)];\n \n let provider = Arc::new(MockLlmProvider::new(responses));\n let client = make_client(provider).await;\n",
|
|
"summary": {
|
|
"files_changed": 5,
|
|
"additions": 493,
|
|
"deletions": 86
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"seq": 880,
|
|
"checkpoint": {
|
|
"timestamp": "2026-05-23T15:46:44.405805Z",
|
|
"current_node": "simplify_gpt",
|
|
"completed_nodes": [
|
|
"start",
|
|
"toolchain",
|
|
"preflight_compile",
|
|
"preflight_lint",
|
|
"implement",
|
|
"simplify_opus",
|
|
"simplify_gpt"
|
|
],
|
|
"node_retries": {},
|
|
"context_values": {
|
|
"internal.fidelity": "compact",
|
|
"internal.retry_count.simplify_opus": 0,
|
|
"thread.preflight_compile.current_node": "preflight_lint",
|
|
"graph.rankdir": "LR",
|
|
"current_node": "simplify_gpt",
|
|
"thread.start.current_node": "toolchain",
|
|
"internal.retry_count.preflight_compile": 0,
|
|
"internal.run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6",
|
|
"response.simplify_opus": "All clean. Here's a summary of what was fixed based on the three reviews:\n\n## Summary of cleanup applied\n\n**`lib/crates/fabro-agent/src/compaction.rs`**\n- Removed the no-longer-needed `pub fn estimate_token_count` (no external callers); tests updated to call `estimate_active_context_usage` directly.\n- Replaced the hand-rolled `ContextEstimateMethod::as_str()` with `#[derive(strum::IntoStaticStr)] #[strum(serialize_all = \"snake_case\")]` per repo convention in CLAUDE.md.\n- Split `estimate_local_token_count(system_prompt, turns)` into two single-purpose helpers (`estimate_turns_local_tokens`, `estimate_system_prompt_local_tokens`) to remove the `\"\"` sentinel call from the baseline+delta path.\n- Replaced the literal `4` in `summary_token_estimate = summary_content.len() / 4` with `APPROX_CHARS_PER_TOKEN`.\n- Made `check_context_usage` return `Option<ContextEstimate>` and `compact_context` accept the pre-computed estimate, eliminating a duplicate full estimate scan per actual compaction. Both fns are now `pub(crate)` since they have no external callers. Dropped the redundant `system_prompt` parameter from `compact_context`.\n- Tightened visibility on `ContextEstimate` and `ContextEstimateMethod` to `pub(crate)`.\n\n**`lib/crates/fabro-agent/src/history.rs`**\n- Moved the free `invalidate_assistant_usage` function into `impl History` as a private `invalidate_preserved_usage` method.\n- Added a doc comment on `History::compact` explaining the usage-invalidation invariant.\n\n**`lib/crates/fabro-agent/src/session.rs`**\n- Rewired `compact_if_needed` to consume the `Option<ContextEstimate>` from `check_context_usage` and pass it through.\n- Renamed misleading test helper `response_with_total_usage` → `response_with_input_tokens` (it only set `input_tokens`, never the other usage fields).\n\n**`lib/crates/fabro-agent/Cargo.toml`**\n- Added `strum.workspace = true` (workspace dep) to support the derive.\n\n## Findings skipped\n\n- **Test builder for `Message::Assistant`** (Quality #3): broader scope than just the diff and only marginal cleanup.\n- **Counting-writer for JSON length** (Efficiency #3): the hot-path concern is real but speculative without profiling; the prevailing pattern in fabro-agent uses `Value::to_string()`. Out of scope.\n- **Extract `\"context_window\"` / `\"estimate_method\"` JSON-key constants** (Quality #5): pre-existing pattern beyond this diff.\n- **Reshape `latest_assistant_usage_baseline` to return `(tokens, &[Message])`** (Quality #4): cosmetic; current `+ 1` indexing is local and clear.\n- **`APPROX_CHARS_PER_TOKEN` in `history.rs` `extract_recent_user_messages`** (Reuse #1): out of scope — that code is not part of the change.\n\nVerification: `cargo nextest run -p fabro-agent` → 447/447 pass; `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` → clean; `cargo +nightly-2026-04-14 fmt --check --all` → clean.",
|
|
"thread.toolchain.current_node": "preflight_compile",
|
|
"failure_signature": "",
|
|
"response.implement": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.total_tokens() > 0` as baseline.\n - Adds local estimates only for turns after that assistant response.\n - Falls back to existing whole-history `chars / 4` local estimate when no usable assistant usage exists.\n - Adds `estimate_method` to context-window warning details.\n - Uses the improved estimate for `CompactionStarted`.\n - Prevents no-op compaction from emitting `CompactionStarted`.\n- Updated `lib/crates/fabro-agent/src/history.rs`\n - `History::compact()` now resets preserved assistant `usage` to `TokenCounts::default()`.\n - Assistant content, tool calls, response IDs, and non-stripped provider parts remain preserved.\n- Added/updated tests in:\n - `compaction.rs`\n - `history.rs`\n - `session.rs`\n- Confirmed no `llm_client.count_input_tokens()` / provider token-count API calls were added.\n\nVerification:\n- Red phase confirmed initially with failing new compaction tests.\n- Passed:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - `cargo nextest run -p fabro-agent session`\n - `cargo nextest run -p fabro-agent`\n - `cargo +nightly-2026-04-14 fmt --check --all`\n - `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings`",
|
|
"response.simplify_gpt": "Reviewed the compaction changes vs `origin/main` and launched the three requested parallel review agents.\n\nFixed one correctness/quality issue found during review:\n\n- `estimate_active_context_usage()` local fallback was dividing the system prompt and turns separately, which could differ from the previous whole-history `chars / 4` behavior due to integer rounding.\n- Updated it to sum system prompt chars + turn chars first, then divide once.\n- Reused the same local char/token helper for fallback, post-baseline delta, and summary token estimate.\n- Added a regression test proving whole-history rounding is preserved.\n\nVerification run:\n\n- `cargo nextest run -p fabro-agent compaction` ✅\n- `cargo nextest run -p fabro-agent history` ✅\n- `cargo +nightly-2026-04-14 fmt --check --all` ✅\n- `cargo nextest run -p fabro-agent` ✅ 448 passed\n- `cargo +nightly-2026-04-14 clippy -q --package fabro-agent --tests --no-deps -- -D warnings` ✅\n\nNote: a broader `cargo +nightly-2026-04-14 clippy -q -p fabro-agent --all-targets -- -D warnings` hit an unrelated dependency lint in `fabro-sandbox::SandboxSpec` (`large_enum_variant`), so I reran clippy scoped to `fabro-agent` without dependencies, which passed.",
|
|
"internal.node_visit_count": 1,
|
|
"internal.retry_count.toolchain": 0,
|
|
"thread.simplify_opus.current_node": "simplify_gpt",
|
|
"thread.implement.current_node": "simplify_opus",
|
|
"outcome": "succeeded",
|
|
"graph.goal": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n",
|
|
"failure_class": "",
|
|
"internal.retry_count.preflight_lint": 0,
|
|
"last_stage": "simplify_gpt",
|
|
"internal.work_dir": "/home/daytona/workspace/fabro",
|
|
"graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ",
|
|
"internal.thread_id": "simplify_opus",
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126",
|
|
"thread.preflight_lint.current_node": "implement",
|
|
"internal.retry_count.simplify_gpt": 0,
|
|
"internal.retry_count.start": 0,
|
|
"last_response": "Reviewed the compaction changes vs `origin/main` and launched the three requested parallel review agents.\n\nFixed one correctness/quality issue found during review:\n\n- `estimate_active_context_usage()`",
|
|
"internal.retry_count.implement": 0
|
|
},
|
|
"node_outcomes": {
|
|
"simplify_gpt": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"last_response": "Reviewed the compaction changes vs `origin/main` and launched the three requested parallel review agents.\n\nFixed one correctness/quality issue found during review:\n\n- `estimate_active_context_usage()`",
|
|
"response.simplify_gpt": "Reviewed the compaction changes vs `origin/main` and launched the three requested parallel review agents.\n\nFixed one correctness/quality issue found during review:\n\n- `estimate_active_context_usage()` local fallback was dividing the system prompt and turns separately, which could differ from the previous whole-history `chars / 4` behavior due to integer rounding.\n- Updated it to sum system prompt chars + turn chars first, then divide once.\n- Reused the same local char/token helper for fallback, post-baseline delta, and summary token estimate.\n- Added a regression test proving whole-history rounding is preserved.\n\nVerification run:\n\n- `cargo nextest run -p fabro-agent compaction` ✅\n- `cargo nextest run -p fabro-agent history` ✅\n- `cargo +nightly-2026-04-14 fmt --check --all` ✅\n- `cargo nextest run -p fabro-agent` ✅ 448 passed\n- `cargo +nightly-2026-04-14 clippy -q --package fabro-agent --tests --no-deps -- -D warnings` ✅\n\nNote: a broader `cargo +nightly-2026-04-14 clippy -q -p fabro-agent --all-targets -- -D warnings` hit an unrelated dependency lint in `fabro-sandbox::SandboxSpec` (`large_enum_variant`), so I reran clippy scoped to `fabro-agent` without dependencies, which passed.",
|
|
"last_stage": "simplify_gpt"
|
|
},
|
|
"notes": "Stage completed: simplify_gpt",
|
|
"usage": {
|
|
"input": {
|
|
"usage": {
|
|
"model": {
|
|
"provider": "openai",
|
|
"model_id": "gpt-5.5"
|
|
},
|
|
"tokens": {
|
|
"input_tokens": 70433,
|
|
"output_tokens": 3449,
|
|
"reasoning_tokens": 1878,
|
|
"cache_read_tokens": 843776,
|
|
"cache_write_tokens": 0
|
|
}
|
|
},
|
|
"facts": {
|
|
"algorithm": "openai"
|
|
}
|
|
},
|
|
"total_usd_micros": 933863
|
|
}
|
|
},
|
|
"preflight_lint": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
|
},
|
|
"notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1",
|
|
"usage": null
|
|
},
|
|
"implement": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"last_response": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.tota",
|
|
"last_stage": "implement",
|
|
"response.implement": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.total_tokens() > 0` as baseline.\n - Adds local estimates only for turns after that assistant response.\n - Falls back to existing whole-history `chars / 4` local estimate when no usable assistant usage exists.\n - Adds `estimate_method` to context-window warning details.\n - Uses the improved estimate for `CompactionStarted`.\n - Prevents no-op compaction from emitting `CompactionStarted`.\n- Updated `lib/crates/fabro-agent/src/history.rs`\n - `History::compact()` now resets preserved assistant `usage` to `TokenCounts::default()`.\n - Assistant content, tool calls, response IDs, and non-stripped provider parts remain preserved.\n- Added/updated tests in:\n - `compaction.rs`\n - `history.rs`\n - `session.rs`\n- Confirmed no `llm_client.count_input_tokens()` / provider token-count API calls were added.\n\nVerification:\n- Red phase confirmed initially with failing new compaction tests.\n- Passed:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - `cargo nextest run -p fabro-agent session`\n - `cargo nextest run -p fabro-agent`\n - `cargo +nightly-2026-04-14 fmt --check --all`\n - `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings`"
|
|
},
|
|
"notes": "Stage completed: implement",
|
|
"usage": {
|
|
"input": {
|
|
"usage": {
|
|
"model": {
|
|
"provider": "openai",
|
|
"model_id": "gpt-5.5"
|
|
},
|
|
"tokens": {
|
|
"input_tokens": 168644,
|
|
"output_tokens": 9772,
|
|
"reasoning_tokens": 11720,
|
|
"cache_read_tokens": 3959296,
|
|
"cache_write_tokens": 0
|
|
}
|
|
},
|
|
"facts": {
|
|
"algorithm": "openai"
|
|
}
|
|
},
|
|
"total_usd_micros": 3467628
|
|
}
|
|
},
|
|
"simplify_opus": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"last_stage": "simplify_opus",
|
|
"last_response": "All clean. Here's a summary of what was fixed based on the three reviews:\n\n## Summary of cleanup applied\n\n**`lib/crates/fabro-agent/src/compaction.rs`**\n- Removed the no-longer-needed `pub fn estimate",
|
|
"response.simplify_opus": "All clean. Here's a summary of what was fixed based on the three reviews:\n\n## Summary of cleanup applied\n\n**`lib/crates/fabro-agent/src/compaction.rs`**\n- Removed the no-longer-needed `pub fn estimate_token_count` (no external callers); tests updated to call `estimate_active_context_usage` directly.\n- Replaced the hand-rolled `ContextEstimateMethod::as_str()` with `#[derive(strum::IntoStaticStr)] #[strum(serialize_all = \"snake_case\")]` per repo convention in CLAUDE.md.\n- Split `estimate_local_token_count(system_prompt, turns)` into two single-purpose helpers (`estimate_turns_local_tokens`, `estimate_system_prompt_local_tokens`) to remove the `\"\"` sentinel call from the baseline+delta path.\n- Replaced the literal `4` in `summary_token_estimate = summary_content.len() / 4` with `APPROX_CHARS_PER_TOKEN`.\n- Made `check_context_usage` return `Option<ContextEstimate>` and `compact_context` accept the pre-computed estimate, eliminating a duplicate full estimate scan per actual compaction. Both fns are now `pub(crate)` since they have no external callers. Dropped the redundant `system_prompt` parameter from `compact_context`.\n- Tightened visibility on `ContextEstimate` and `ContextEstimateMethod` to `pub(crate)`.\n\n**`lib/crates/fabro-agent/src/history.rs`**\n- Moved the free `invalidate_assistant_usage` function into `impl History` as a private `invalidate_preserved_usage` method.\n- Added a doc comment on `History::compact` explaining the usage-invalidation invariant.\n\n**`lib/crates/fabro-agent/src/session.rs`**\n- Rewired `compact_if_needed` to consume the `Option<ContextEstimate>` from `check_context_usage` and pass it through.\n- Renamed misleading test helper `response_with_total_usage` → `response_with_input_tokens` (it only set `input_tokens`, never the other usage fields).\n\n**`lib/crates/fabro-agent/Cargo.toml`**\n- Added `strum.workspace = true` (workspace dep) to support the derive.\n\n## Findings skipped\n\n- **Test builder for `Message::Assistant`** (Quality #3): broader scope than just the diff and only marginal cleanup.\n- **Counting-writer for JSON length** (Efficiency #3): the hot-path concern is real but speculative without profiling; the prevailing pattern in fabro-agent uses `Value::to_string()`. Out of scope.\n- **Extract `\"context_window\"` / `\"estimate_method\"` JSON-key constants** (Quality #5): pre-existing pattern beyond this diff.\n- **Reshape `latest_assistant_usage_baseline` to return `(tokens, &[Message])`** (Quality #4): cosmetic; current `+ 1` indexing is local and clear.\n- **`APPROX_CHARS_PER_TOKEN` in `history.rs` `extract_recent_user_messages`** (Reuse #1): out of scope — that code is not part of the change.\n\nVerification: `cargo nextest run -p fabro-agent` → 447/447 pass; `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` → clean; `cargo +nightly-2026-04-14 fmt --check --all` → clean."
|
|
},
|
|
"notes": "Stage completed: simplify_opus",
|
|
"usage": {
|
|
"input": {
|
|
"usage": {
|
|
"model": {
|
|
"provider": "anthropic",
|
|
"model_id": "claude-opus-4-7"
|
|
},
|
|
"tokens": {
|
|
"input_tokens": 66736,
|
|
"output_tokens": 21289,
|
|
"reasoning_tokens": 0,
|
|
"cache_read_tokens": 1971515,
|
|
"cache_write_tokens": 267947
|
|
}
|
|
},
|
|
"facts": {
|
|
"algorithm": "anthropic",
|
|
"cache_write_5m_tokens": 267947,
|
|
"cache_write_1h_tokens": 0
|
|
}
|
|
},
|
|
"total_usd_micros": 3526330
|
|
},
|
|
"files_touched": [
|
|
"/home/daytona/workspace/fabro/lib/crates/fabro-agent/Cargo.toml",
|
|
"/home/daytona/workspace/fabro/lib/crates/fabro-agent/src/compaction.rs",
|
|
"/home/daytona/workspace/fabro/lib/crates/fabro-agent/src/history.rs",
|
|
"/home/daytona/workspace/fabro/lib/crates/fabro-agent/src/session.rs"
|
|
]
|
|
},
|
|
"preflight_compile": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
|
},
|
|
"notes": "Script completed: cargo check -q --workspace 2>&1",
|
|
"usage": null
|
|
},
|
|
"start": {
|
|
"status": "succeeded",
|
|
"usage": null
|
|
},
|
|
"toolchain": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c"
|
|
},
|
|
"notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
|
"usage": null
|
|
}
|
|
},
|
|
"next_node_id": "verify",
|
|
"git_commit_sha": "17eb559cfbdf64c81a462136e97ba68e9ad9fbad",
|
|
"node_visits": {
|
|
"simplify_opus": 1,
|
|
"preflight_lint": 1,
|
|
"simplify_gpt": 1,
|
|
"start": 1,
|
|
"toolchain": 1,
|
|
"implement": 1,
|
|
"preflight_compile": 1
|
|
}
|
|
},
|
|
"diff": {
|
|
"patch": "diff --git a/lib/crates/fabro-agent/src/compaction.rs b/lib/crates/fabro-agent/src/compaction.rs\nindex f610cdd89..03d55638b 100644\n--- a/lib/crates/fabro-agent/src/compaction.rs\n+++ b/lib/crates/fabro-agent/src/compaction.rs\n@@ -155,7 +155,7 @@ function names, error messages, and exact values. Omit pleasantries and conversa\n \"A different assistant began this task and produced the following summary. \\\n Build on their progress — do not repeat completed steps.\\n\\n{summary_text}\"\n );\n- let summary_token_estimate = summary_content.len() / APPROX_CHARS_PER_TOKEN;\n+ let summary_token_estimate = estimate_chars_local_tokens(summary_content.len());\n \n history.compact(preserve_count, summary_content);\n \n@@ -183,8 +183,11 @@ pub(crate) fn estimate_active_context_usage(\n }\n \n ContextEstimate {\n- tokens: estimate_system_prompt_local_tokens(system_prompt)\n- + estimate_turns_local_tokens(turns),\n+ tokens: estimate_chars_local_tokens(\n+ system_prompt\n+ .len()\n+ .saturating_add(estimate_turns_local_chars(turns)),\n+ ),\n method: ContextEstimateMethod::LocalEstimate,\n }\n }\n@@ -202,11 +205,17 @@ fn latest_assistant_usage_baseline(turns: &[Message]) -> Option<(usize, usize)>\n }\n \n fn estimate_turns_local_tokens(turns: &[Message]) -> usize {\n- turns.iter().map(estimate_turn_chars).sum::<usize>() / APPROX_CHARS_PER_TOKEN\n+ estimate_chars_local_tokens(estimate_turns_local_chars(turns))\n }\n \n-fn estimate_system_prompt_local_tokens(system_prompt: &str) -> usize {\n- system_prompt.len() / APPROX_CHARS_PER_TOKEN\n+fn estimate_turns_local_chars(turns: &[Message]) -> usize {\n+ turns.iter().fold(0usize, |total, turn| {\n+ total.saturating_add(estimate_turn_chars(turn))\n+ })\n+}\n+\n+fn estimate_chars_local_tokens(chars: usize) -> usize {\n+ chars / APPROX_CHARS_PER_TOKEN\n }\n \n fn estimate_turn_chars(turn: &Message) -> usize {\n@@ -397,10 +406,24 @@ mod tests {\n let estimate = estimate_active_context_usage(\"test\", &history);\n \n assert_eq!(estimate.method, ContextEstimateMethod::LocalEstimate);\n- // sysprompt 4/4 = 1, turns sum = (11 + 18 + 9 + 16 + 4) / 4 = 58/4 = 14\n+ // (system prompt 4 + turn chars 11 + 18 + 9 + 16 + 4) / 4 = 62/4 = 15\n assert_eq!(estimate.tokens, 15);\n }\n \n+ #[test]\n+ fn active_context_local_estimate_matches_whole_history_rounding() {\n+ let mut history = History::default();\n+ history.push(Message::User {\n+ content: \"abc\".into(),\n+ timestamp: SystemTime::now(),\n+ });\n+\n+ let estimate = estimate_active_context_usage(\"x\", &history);\n+\n+ assert_eq!(estimate.method, ContextEstimateMethod::LocalEstimate);\n+ assert_eq!(estimate.tokens, 1);\n+ }\n+\n #[test]\n fn active_context_estimate_uses_latest_assistant_usage_plus_later_turns() {\n let mut history = History::default();\n",
|
|
"summary": {
|
|
"files_changed": 5,
|
|
"additions": 516,
|
|
"deletions": 86
|
|
}
|
|
}
|
|
},
|
|
{
|
|
"seq": 0,
|
|
"checkpoint": {
|
|
"timestamp": "2026-05-23T15:50:48.923842Z",
|
|
"current_node": "verify",
|
|
"completed_nodes": [
|
|
"start",
|
|
"toolchain",
|
|
"preflight_compile",
|
|
"preflight_lint",
|
|
"implement",
|
|
"simplify_opus",
|
|
"simplify_gpt",
|
|
"verify"
|
|
],
|
|
"node_retries": {},
|
|
"context_values": {
|
|
"response.implement": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.total_tokens() > 0` as baseline.\n - Adds local estimates only for turns after that assistant response.\n - Falls back to existing whole-history `chars / 4` local estimate when no usable assistant usage exists.\n - Adds `estimate_method` to context-window warning details.\n - Uses the improved estimate for `CompactionStarted`.\n - Prevents no-op compaction from emitting `CompactionStarted`.\n- Updated `lib/crates/fabro-agent/src/history.rs`\n - `History::compact()` now resets preserved assistant `usage` to `TokenCounts::default()`.\n - Assistant content, tool calls, response IDs, and non-stripped provider parts remain preserved.\n- Added/updated tests in:\n - `compaction.rs`\n - `history.rs`\n - `session.rs`\n- Confirmed no `llm_client.count_input_tokens()` / provider token-count API calls were added.\n\nVerification:\n- Red phase confirmed initially with failing new compaction tests.\n- Passed:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - `cargo nextest run -p fabro-agent session`\n - `cargo nextest run -p fabro-agent`\n - `cargo +nightly-2026-04-14 fmt --check --all`\n - `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings`",
|
|
"internal.retry_count.simplify_gpt": 0,
|
|
"internal.work_dir": "/home/daytona/workspace/fabro",
|
|
"internal.retry_count.implement": 0,
|
|
"current_node": "verify",
|
|
"internal.retry_count.simplify_opus": 0,
|
|
"internal.run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6",
|
|
"thread.implement.current_node": "simplify_opus",
|
|
"internal.retry_count.verify": 0,
|
|
"thread.preflight_lint.current_node": "implement",
|
|
"failure_class": "",
|
|
"internal.retry_count.preflight_lint": 0,
|
|
"response.simplify_opus": "All clean. Here's a summary of what was fixed based on the three reviews:\n\n## Summary of cleanup applied\n\n**`lib/crates/fabro-agent/src/compaction.rs`**\n- Removed the no-longer-needed `pub fn estimate_token_count` (no external callers); tests updated to call `estimate_active_context_usage` directly.\n- Replaced the hand-rolled `ContextEstimateMethod::as_str()` with `#[derive(strum::IntoStaticStr)] #[strum(serialize_all = \"snake_case\")]` per repo convention in CLAUDE.md.\n- Split `estimate_local_token_count(system_prompt, turns)` into two single-purpose helpers (`estimate_turns_local_tokens`, `estimate_system_prompt_local_tokens`) to remove the `\"\"` sentinel call from the baseline+delta path.\n- Replaced the literal `4` in `summary_token_estimate = summary_content.len() / 4` with `APPROX_CHARS_PER_TOKEN`.\n- Made `check_context_usage` return `Option<ContextEstimate>` and `compact_context` accept the pre-computed estimate, eliminating a duplicate full estimate scan per actual compaction. Both fns are now `pub(crate)` since they have no external callers. Dropped the redundant `system_prompt` parameter from `compact_context`.\n- Tightened visibility on `ContextEstimate` and `ContextEstimateMethod` to `pub(crate)`.\n\n**`lib/crates/fabro-agent/src/history.rs`**\n- Moved the free `invalidate_assistant_usage` function into `impl History` as a private `invalidate_preserved_usage` method.\n- Added a doc comment on `History::compact` explaining the usage-invalidation invariant.\n\n**`lib/crates/fabro-agent/src/session.rs`**\n- Rewired `compact_if_needed` to consume the `Option<ContextEstimate>` from `check_context_usage` and pass it through.\n- Renamed misleading test helper `response_with_total_usage` → `response_with_input_tokens` (it only set `input_tokens`, never the other usage fields).\n\n**`lib/crates/fabro-agent/Cargo.toml`**\n- Added `strum.workspace = true` (workspace dep) to support the derive.\n\n## Findings skipped\n\n- **Test builder for `Message::Assistant`** (Quality #3): broader scope than just the diff and only marginal cleanup.\n- **Counting-writer for JSON length** (Efficiency #3): the hot-path concern is real but speculative without profiling; the prevailing pattern in fabro-agent uses `Value::to_string()`. Out of scope.\n- **Extract `\"context_window\"` / `\"estimate_method\"` JSON-key constants** (Quality #5): pre-existing pattern beyond this diff.\n- **Reshape `latest_assistant_usage_baseline` to return `(tokens, &[Message])`** (Quality #4): cosmetic; current `+ 1` indexing is local and clear.\n- **`APPROX_CHARS_PER_TOKEN` in `history.rs` `extract_recent_user_messages`** (Reuse #1): out of scope — that code is not part of the change.\n\nVerification: `cargo nextest run -p fabro-agent` → 447/447 pass; `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` → clean; `cargo +nightly-2026-04-14 fmt --check --all` → clean.",
|
|
"internal.retry_count.start": 0,
|
|
"internal.fidelity": "compact",
|
|
"internal.retry_count.preflight_compile": 0,
|
|
"command.output": "blob://sha256/73ee7d4f71e69fb29ace2db42aeab348b9cc64924624ba42fde460bda17c19f9",
|
|
"failure_signature": "",
|
|
"thread.simplify_opus.current_node": "simplify_gpt",
|
|
"thread.simplify_gpt.current_node": "verify",
|
|
"last_stage": "simplify_gpt",
|
|
"thread.preflight_compile.current_node": "preflight_lint",
|
|
"thread.toolchain.current_node": "preflight_compile",
|
|
"graph.rankdir": "LR",
|
|
"internal.thread_id": "simplify_gpt",
|
|
"outcome": "succeeded",
|
|
"response.simplify_gpt": "Reviewed the compaction changes vs `origin/main` and launched the three requested parallel review agents.\n\nFixed one correctness/quality issue found during review:\n\n- `estimate_active_context_usage()` local fallback was dividing the system prompt and turns separately, which could differ from the previous whole-history `chars / 4` behavior due to integer rounding.\n- Updated it to sum system prompt chars + turn chars first, then divide once.\n- Reused the same local char/token helper for fallback, post-baseline delta, and summary token estimate.\n- Added a regression test proving whole-history rounding is preserved.\n\nVerification run:\n\n- `cargo nextest run -p fabro-agent compaction` ✅\n- `cargo nextest run -p fabro-agent history` ✅\n- `cargo +nightly-2026-04-14 fmt --check --all` ✅\n- `cargo nextest run -p fabro-agent` ✅ 448 passed\n- `cargo +nightly-2026-04-14 clippy -q --package fabro-agent --tests --no-deps -- -D warnings` ✅\n\nNote: a broader `cargo +nightly-2026-04-14 clippy -q -p fabro-agent --all-targets -- -D warnings` hit an unrelated dependency lint in `fabro-sandbox::SandboxSpec` (`large_enum_variant`), so I reran clippy scoped to `fabro-agent` without dependencies, which passed.",
|
|
"thread.start.current_node": "toolchain",
|
|
"internal.retry_count.toolchain": 0,
|
|
"internal.node_visit_count": 1,
|
|
"last_response": "Reviewed the compaction changes vs `origin/main` and launched the three requested parallel review agents.\n\nFixed one correctness/quality issue found during review:\n\n- `estimate_active_context_usage()`",
|
|
"graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ",
|
|
"graph.goal": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n"
|
|
},
|
|
"node_outcomes": {
|
|
"toolchain": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c"
|
|
},
|
|
"notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
|
"usage": null
|
|
},
|
|
"preflight_compile": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
|
},
|
|
"notes": "Script completed: cargo check -q --workspace 2>&1",
|
|
"usage": null
|
|
},
|
|
"simplify_opus": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"last_stage": "simplify_opus",
|
|
"last_response": "All clean. Here's a summary of what was fixed based on the three reviews:\n\n## Summary of cleanup applied\n\n**`lib/crates/fabro-agent/src/compaction.rs`**\n- Removed the no-longer-needed `pub fn estimate",
|
|
"response.simplify_opus": "All clean. Here's a summary of what was fixed based on the three reviews:\n\n## Summary of cleanup applied\n\n**`lib/crates/fabro-agent/src/compaction.rs`**\n- Removed the no-longer-needed `pub fn estimate_token_count` (no external callers); tests updated to call `estimate_active_context_usage` directly.\n- Replaced the hand-rolled `ContextEstimateMethod::as_str()` with `#[derive(strum::IntoStaticStr)] #[strum(serialize_all = \"snake_case\")]` per repo convention in CLAUDE.md.\n- Split `estimate_local_token_count(system_prompt, turns)` into two single-purpose helpers (`estimate_turns_local_tokens`, `estimate_system_prompt_local_tokens`) to remove the `\"\"` sentinel call from the baseline+delta path.\n- Replaced the literal `4` in `summary_token_estimate = summary_content.len() / 4` with `APPROX_CHARS_PER_TOKEN`.\n- Made `check_context_usage` return `Option<ContextEstimate>` and `compact_context` accept the pre-computed estimate, eliminating a duplicate full estimate scan per actual compaction. Both fns are now `pub(crate)` since they have no external callers. Dropped the redundant `system_prompt` parameter from `compact_context`.\n- Tightened visibility on `ContextEstimate` and `ContextEstimateMethod` to `pub(crate)`.\n\n**`lib/crates/fabro-agent/src/history.rs`**\n- Moved the free `invalidate_assistant_usage` function into `impl History` as a private `invalidate_preserved_usage` method.\n- Added a doc comment on `History::compact` explaining the usage-invalidation invariant.\n\n**`lib/crates/fabro-agent/src/session.rs`**\n- Rewired `compact_if_needed` to consume the `Option<ContextEstimate>` from `check_context_usage` and pass it through.\n- Renamed misleading test helper `response_with_total_usage` → `response_with_input_tokens` (it only set `input_tokens`, never the other usage fields).\n\n**`lib/crates/fabro-agent/Cargo.toml`**\n- Added `strum.workspace = true` (workspace dep) to support the derive.\n\n## Findings skipped\n\n- **Test builder for `Message::Assistant`** (Quality #3): broader scope than just the diff and only marginal cleanup.\n- **Counting-writer for JSON length** (Efficiency #3): the hot-path concern is real but speculative without profiling; the prevailing pattern in fabro-agent uses `Value::to_string()`. Out of scope.\n- **Extract `\"context_window\"` / `\"estimate_method\"` JSON-key constants** (Quality #5): pre-existing pattern beyond this diff.\n- **Reshape `latest_assistant_usage_baseline` to return `(tokens, &[Message])`** (Quality #4): cosmetic; current `+ 1` indexing is local and clear.\n- **`APPROX_CHARS_PER_TOKEN` in `history.rs` `extract_recent_user_messages`** (Reuse #1): out of scope — that code is not part of the change.\n\nVerification: `cargo nextest run -p fabro-agent` → 447/447 pass; `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` → clean; `cargo +nightly-2026-04-14 fmt --check --all` → clean."
|
|
},
|
|
"notes": "Stage completed: simplify_opus",
|
|
"usage": {
|
|
"input": {
|
|
"usage": {
|
|
"model": {
|
|
"provider": "anthropic",
|
|
"model_id": "claude-opus-4-7"
|
|
},
|
|
"tokens": {
|
|
"input_tokens": 66736,
|
|
"output_tokens": 21289,
|
|
"reasoning_tokens": 0,
|
|
"cache_read_tokens": 1971515,
|
|
"cache_write_tokens": 267947
|
|
}
|
|
},
|
|
"facts": {
|
|
"algorithm": "anthropic",
|
|
"cache_write_5m_tokens": 267947,
|
|
"cache_write_1h_tokens": 0
|
|
}
|
|
},
|
|
"total_usd_micros": 3526330
|
|
},
|
|
"files_touched": [
|
|
"/home/daytona/workspace/fabro/lib/crates/fabro-agent/Cargo.toml",
|
|
"/home/daytona/workspace/fabro/lib/crates/fabro-agent/src/compaction.rs",
|
|
"/home/daytona/workspace/fabro/lib/crates/fabro-agent/src/history.rs",
|
|
"/home/daytona/workspace/fabro/lib/crates/fabro-agent/src/session.rs"
|
|
]
|
|
},
|
|
"start": {
|
|
"status": "succeeded",
|
|
"usage": null
|
|
},
|
|
"implement": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"last_response": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.tota",
|
|
"last_stage": "implement",
|
|
"response.implement": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.total_tokens() > 0` as baseline.\n - Adds local estimates only for turns after that assistant response.\n - Falls back to existing whole-history `chars / 4` local estimate when no usable assistant usage exists.\n - Adds `estimate_method` to context-window warning details.\n - Uses the improved estimate for `CompactionStarted`.\n - Prevents no-op compaction from emitting `CompactionStarted`.\n- Updated `lib/crates/fabro-agent/src/history.rs`\n - `History::compact()` now resets preserved assistant `usage` to `TokenCounts::default()`.\n - Assistant content, tool calls, response IDs, and non-stripped provider parts remain preserved.\n- Added/updated tests in:\n - `compaction.rs`\n - `history.rs`\n - `session.rs`\n- Confirmed no `llm_client.count_input_tokens()` / provider token-count API calls were added.\n\nVerification:\n- Red phase confirmed initially with failing new compaction tests.\n- Passed:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - `cargo nextest run -p fabro-agent session`\n - `cargo nextest run -p fabro-agent`\n - `cargo +nightly-2026-04-14 fmt --check --all`\n - `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings`"
|
|
},
|
|
"notes": "Stage completed: implement",
|
|
"usage": {
|
|
"input": {
|
|
"usage": {
|
|
"model": {
|
|
"provider": "openai",
|
|
"model_id": "gpt-5.5"
|
|
},
|
|
"tokens": {
|
|
"input_tokens": 168644,
|
|
"output_tokens": 9772,
|
|
"reasoning_tokens": 11720,
|
|
"cache_read_tokens": 3959296,
|
|
"cache_write_tokens": 0
|
|
}
|
|
},
|
|
"facts": {
|
|
"algorithm": "openai"
|
|
}
|
|
},
|
|
"total_usd_micros": 3467628
|
|
}
|
|
},
|
|
"simplify_gpt": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"last_response": "Reviewed the compaction changes vs `origin/main` and launched the three requested parallel review agents.\n\nFixed one correctness/quality issue found during review:\n\n- `estimate_active_context_usage()`",
|
|
"response.simplify_gpt": "Reviewed the compaction changes vs `origin/main` and launched the three requested parallel review agents.\n\nFixed one correctness/quality issue found during review:\n\n- `estimate_active_context_usage()` local fallback was dividing the system prompt and turns separately, which could differ from the previous whole-history `chars / 4` behavior due to integer rounding.\n- Updated it to sum system prompt chars + turn chars first, then divide once.\n- Reused the same local char/token helper for fallback, post-baseline delta, and summary token estimate.\n- Added a regression test proving whole-history rounding is preserved.\n\nVerification run:\n\n- `cargo nextest run -p fabro-agent compaction` ✅\n- `cargo nextest run -p fabro-agent history` ✅\n- `cargo +nightly-2026-04-14 fmt --check --all` ✅\n- `cargo nextest run -p fabro-agent` ✅ 448 passed\n- `cargo +nightly-2026-04-14 clippy -q --package fabro-agent --tests --no-deps -- -D warnings` ✅\n\nNote: a broader `cargo +nightly-2026-04-14 clippy -q -p fabro-agent --all-targets -- -D warnings` hit an unrelated dependency lint in `fabro-sandbox::SandboxSpec` (`large_enum_variant`), so I reran clippy scoped to `fabro-agent` without dependencies, which passed.",
|
|
"last_stage": "simplify_gpt"
|
|
},
|
|
"notes": "Stage completed: simplify_gpt",
|
|
"usage": {
|
|
"input": {
|
|
"usage": {
|
|
"model": {
|
|
"provider": "openai",
|
|
"model_id": "gpt-5.5"
|
|
},
|
|
"tokens": {
|
|
"input_tokens": 70433,
|
|
"output_tokens": 3449,
|
|
"reasoning_tokens": 1878,
|
|
"cache_read_tokens": 843776,
|
|
"cache_write_tokens": 0
|
|
}
|
|
},
|
|
"facts": {
|
|
"algorithm": "openai"
|
|
}
|
|
},
|
|
"total_usd_micros": 933863
|
|
}
|
|
},
|
|
"verify": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/73ee7d4f71e69fb29ace2db42aeab348b9cc64924624ba42fde460bda17c19f9"
|
|
},
|
|
"notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1",
|
|
"usage": null
|
|
},
|
|
"preflight_lint": {
|
|
"status": "succeeded",
|
|
"context_updates": {
|
|
"command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126"
|
|
},
|
|
"notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1",
|
|
"usage": null
|
|
}
|
|
},
|
|
"next_node_id": "fmt",
|
|
"node_visits": {
|
|
"toolchain": 1,
|
|
"verify": 1,
|
|
"start": 1,
|
|
"simplify_opus": 1,
|
|
"implement": 1,
|
|
"preflight_compile": 1,
|
|
"preflight_lint": 1,
|
|
"simplify_gpt": 1
|
|
}
|
|
},
|
|
"diff": {}
|
|
}
|
|
],
|
|
"conclusion": null,
|
|
"sandbox": {
|
|
"provider": "daytona",
|
|
"image": "buildpack-deps:noble",
|
|
"snapshot": "fabro-v11",
|
|
"runtime": {
|
|
"id": "fabro-01KSAPQQSVK4FY3CWHYVZYD7T6",
|
|
"working_directory": "/home/daytona/workspace/fabro",
|
|
"repo_cloned": true,
|
|
"clone_origin_url": "https://github.com/fabro-sh/fabro",
|
|
"clone_branch": "main",
|
|
"workspace_root": "/home/daytona/workspace",
|
|
"repos_root": "/home/daytona/repos",
|
|
"primary_repo_path": "/home/daytona/repos/fabro-sh/fabro",
|
|
"primary_repo_link": "/home/daytona/workspace/fabro"
|
|
}
|
|
},
|
|
"pull_request": null,
|
|
"superseded_by": null,
|
|
"pending_interviews": {},
|
|
"todos_by_list": {
|
|
"anthropic_tasks:70a99891-4c04-4824-8ca5-d9fe373a7111": {
|
|
"kind": "anthropic_tasks",
|
|
"list_id": "anthropic_tasks:70a99891-4c04-4824-8ca5-d9fe373a7111",
|
|
"items": [
|
|
{
|
|
"id": "1",
|
|
"status": "completed",
|
|
"order": 0,
|
|
"subject": "Apply cleanup fixes from code review",
|
|
"description": "1. Use APPROX_CHARS_PER_TOKEN at compaction.rs:167\n2. Delete pub fn estimate_token_count (no external callers)\n3. Adopt strum::IntoStaticStr for ContextEstimateMethod\n4. Rename response_with_total_usage -> response_with_input_tokens\n5. Split estimate_local_token_count to remove sentinel \"\" parameter\n6. Make invalidate_assistant_usage a private method on History\n7. Pass pre-computed ContextEstimate from check_context_usage into compact_context",
|
|
"active_form": "Applying cleanup fixes"
|
|
}
|
|
]
|
|
},
|
|
"openai_plan:bd5288b5-cb6c-4da6-a303-70346a50495c": {
|
|
"kind": "openai_plan",
|
|
"list_id": "openai_plan:bd5288b5-cb6c-4da6-a303-70346a50495c",
|
|
"items": [
|
|
{
|
|
"id": "2771fe0b7d068bfd",
|
|
"status": "completed",
|
|
"order": 0,
|
|
"subject": "Add failing compaction/history/session tests for API-usage baseline behavior"
|
|
},
|
|
{
|
|
"id": "969754ae428a19c5",
|
|
"status": "completed",
|
|
"order": 1,
|
|
"subject": "Implement active-context estimator and warning/event usage changes"
|
|
},
|
|
{
|
|
"id": "10506f244baf24d5",
|
|
"status": "completed",
|
|
"order": 2,
|
|
"subject": "Reset preserved assistant usage during history compaction"
|
|
},
|
|
{
|
|
"id": "4aad9d31be6bc056",
|
|
"status": "completed",
|
|
"order": 3,
|
|
"subject": "Run targeted fabro-agent test suites and fix any regressions"
|
|
}
|
|
]
|
|
}
|
|
},
|
|
"stages": {
|
|
"preflight_lint@1": {
|
|
"first_event_seq": 40,
|
|
"prompt": null,
|
|
"response": null,
|
|
"completion": {
|
|
"outcome": "succeeded",
|
|
"notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1",
|
|
"failure_reason": null,
|
|
"timestamp": "2026-05-23T15:24:23.939755Z"
|
|
},
|
|
"provider_used": null,
|
|
"diff": null,
|
|
"script_invocation": {
|
|
"script": "cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1",
|
|
"command": "exec 2>&1\ncargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1",
|
|
"language": "shell"
|
|
},
|
|
"script_timing": {
|
|
"output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126",
|
|
"exit_code": 0,
|
|
"duration_ms": 126448,
|
|
"termination": "exited",
|
|
"output_bytes": 0,
|
|
"live_streaming": false
|
|
},
|
|
"parallel_results": null,
|
|
"output": null,
|
|
"output_bytes": 0,
|
|
"live_streaming": false,
|
|
"termination": "exited",
|
|
"started_at": "2026-05-23T15:22:17.485222Z",
|
|
"handler": "command",
|
|
"timing": {
|
|
"wall_time_ms": 126453,
|
|
"inference_time_ms": 0,
|
|
"tool_time_ms": 0,
|
|
"active_time_ms": 0
|
|
},
|
|
"usage": {
|
|
"input_tokens": 0,
|
|
"output_tokens": 0,
|
|
"total_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"cache_read_tokens": 0,
|
|
"cache_write_tokens": 0
|
|
},
|
|
"state": "succeeded"
|
|
},
|
|
"start@1": {
|
|
"first_event_seq": 16,
|
|
"prompt": null,
|
|
"response": null,
|
|
"completion": {
|
|
"outcome": "succeeded",
|
|
"notes": null,
|
|
"failure_reason": null,
|
|
"timestamp": "2026-05-23T15:20:11.504080Z"
|
|
},
|
|
"provider_used": null,
|
|
"diff": null,
|
|
"script_invocation": null,
|
|
"script_timing": null,
|
|
"parallel_results": null,
|
|
"output": null,
|
|
"started_at": "2026-05-23T15:20:11.503608Z",
|
|
"handler": "start",
|
|
"timing": {
|
|
"wall_time_ms": 0,
|
|
"inference_time_ms": 0,
|
|
"tool_time_ms": 0,
|
|
"active_time_ms": 0
|
|
},
|
|
"usage": {
|
|
"input_tokens": 0,
|
|
"output_tokens": 0,
|
|
"total_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"cache_read_tokens": 0,
|
|
"cache_write_tokens": 0
|
|
},
|
|
"state": "succeeded"
|
|
},
|
|
"simplify_gpt@1": {
|
|
"first_event_seq": 706,
|
|
"prompt": null,
|
|
"response": null,
|
|
"completion": {
|
|
"outcome": "succeeded",
|
|
"notes": "Stage completed: simplify_gpt",
|
|
"failure_reason": null,
|
|
"timestamp": "2026-05-23T15:46:38.723258Z"
|
|
},
|
|
"provider_used": {
|
|
"mode": "agent",
|
|
"provider": "openai",
|
|
"model": "gpt-5.5"
|
|
},
|
|
"diff": null,
|
|
"script_invocation": null,
|
|
"script_timing": null,
|
|
"parallel_results": null,
|
|
"output": null,
|
|
"started_at": "2026-05-23T15:43:59.610182Z",
|
|
"handler": "agent",
|
|
"timing": {
|
|
"wall_time_ms": 159112,
|
|
"inference_time_ms": 0,
|
|
"tool_time_ms": 0,
|
|
"active_time_ms": 0
|
|
},
|
|
"usage": {
|
|
"input_tokens": 70433,
|
|
"output_tokens": 3449,
|
|
"total_tokens": 919536,
|
|
"reasoning_tokens": 1878,
|
|
"cache_read_tokens": 843776,
|
|
"cache_write_tokens": 0,
|
|
"total_usd_micros": 933863
|
|
},
|
|
"model": {
|
|
"provider": "openai",
|
|
"model_id": "gpt-5.5"
|
|
},
|
|
"state": "succeeded"
|
|
},
|
|
"toolchain@1": {
|
|
"first_event_seq": 20,
|
|
"prompt": null,
|
|
"response": null,
|
|
"completion": {
|
|
"outcome": "succeeded",
|
|
"notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
|
"failure_reason": null,
|
|
"timestamp": "2026-05-23T15:20:14.018526Z"
|
|
},
|
|
"provider_used": null,
|
|
"diff": null,
|
|
"script_invocation": {
|
|
"script": "command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
|
"command": "exec 2>&1\ncommand -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1",
|
|
"language": "shell"
|
|
},
|
|
"script_timing": {
|
|
"output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c",
|
|
"exit_code": 0,
|
|
"duration_ms": 2506,
|
|
"termination": "exited",
|
|
"output_bytes": 36,
|
|
"live_streaming": true
|
|
},
|
|
"parallel_results": null,
|
|
"output": null,
|
|
"output_bytes": 36,
|
|
"live_streaming": true,
|
|
"termination": "exited",
|
|
"started_at": "2026-05-23T15:20:11.504411Z",
|
|
"handler": "command",
|
|
"timing": {
|
|
"wall_time_ms": 2513,
|
|
"inference_time_ms": 0,
|
|
"tool_time_ms": 0,
|
|
"active_time_ms": 0
|
|
},
|
|
"usage": {
|
|
"input_tokens": 0,
|
|
"output_tokens": 0,
|
|
"total_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"cache_read_tokens": 0,
|
|
"cache_write_tokens": 0
|
|
},
|
|
"state": "succeeded"
|
|
},
|
|
"verify@1": {
|
|
"first_event_seq": 883,
|
|
"prompt": null,
|
|
"response": null,
|
|
"completion": null,
|
|
"provider_used": null,
|
|
"diff": null,
|
|
"script_invocation": {
|
|
"script": "cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1",
|
|
"command": "exec 2>&1\ncargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1 && cargo nextest run --cargo-quiet --workspace --status-level fail 2>&1 && cargo dev docs refresh 2>&1 && cargo dev docs check 2>&1",
|
|
"language": "shell"
|
|
},
|
|
"script_timing": null,
|
|
"parallel_results": null,
|
|
"output": null,
|
|
"started_at": "2026-05-23T15:46:44.407613Z",
|
|
"handler": "command",
|
|
"usage": {
|
|
"input_tokens": 0,
|
|
"output_tokens": 0,
|
|
"total_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"cache_read_tokens": 0,
|
|
"cache_write_tokens": 0
|
|
},
|
|
"state": "running"
|
|
},
|
|
"simplify_opus@1": {
|
|
"first_event_seq": 256,
|
|
"prompt": null,
|
|
"response": null,
|
|
"completion": {
|
|
"outcome": "succeeded",
|
|
"notes": "Stage completed: simplify_opus",
|
|
"failure_reason": null,
|
|
"timestamp": "2026-05-23T15:43:53.788041Z"
|
|
},
|
|
"provider_used": {
|
|
"mode": "agent",
|
|
"provider": "anthropic",
|
|
"model": "claude-opus-4-7"
|
|
},
|
|
"diff": null,
|
|
"script_invocation": null,
|
|
"script_timing": null,
|
|
"parallel_results": null,
|
|
"output": null,
|
|
"started_at": "2026-05-23T15:32:46.594117Z",
|
|
"handler": "agent",
|
|
"timing": {
|
|
"wall_time_ms": 667186,
|
|
"inference_time_ms": 0,
|
|
"tool_time_ms": 0,
|
|
"active_time_ms": 0
|
|
},
|
|
"usage": {
|
|
"input_tokens": 66736,
|
|
"output_tokens": 21289,
|
|
"total_tokens": 2327487,
|
|
"reasoning_tokens": 0,
|
|
"cache_read_tokens": 1971515,
|
|
"cache_write_tokens": 267947,
|
|
"total_usd_micros": 3526330
|
|
},
|
|
"model": {
|
|
"provider": "anthropic",
|
|
"model_id": "claude-opus-4-7"
|
|
},
|
|
"state": "succeeded"
|
|
},
|
|
"implement@1": {
|
|
"first_event_seq": 50,
|
|
"prompt": null,
|
|
"response": null,
|
|
"completion": {
|
|
"outcome": "succeeded",
|
|
"notes": "Stage completed: implement",
|
|
"failure_reason": null,
|
|
"timestamp": "2026-05-23T15:32:41.443688Z"
|
|
},
|
|
"provider_used": {
|
|
"mode": "agent",
|
|
"provider": "openai",
|
|
"model": "gpt-5.5"
|
|
},
|
|
"diff": null,
|
|
"script_invocation": null,
|
|
"script_timing": null,
|
|
"parallel_results": null,
|
|
"output": null,
|
|
"started_at": "2026-05-23T15:24:29.561071Z",
|
|
"handler": "agent",
|
|
"timing": {
|
|
"wall_time_ms": 491879,
|
|
"inference_time_ms": 0,
|
|
"tool_time_ms": 0,
|
|
"active_time_ms": 0
|
|
},
|
|
"usage": {
|
|
"input_tokens": 168644,
|
|
"output_tokens": 9772,
|
|
"total_tokens": 4149432,
|
|
"reasoning_tokens": 11720,
|
|
"cache_read_tokens": 3959296,
|
|
"cache_write_tokens": 0,
|
|
"total_usd_micros": 3467628
|
|
},
|
|
"model": {
|
|
"provider": "openai",
|
|
"model_id": "gpt-5.5"
|
|
},
|
|
"state": "succeeded"
|
|
},
|
|
"preflight_compile@1": {
|
|
"first_event_seq": 30,
|
|
"prompt": null,
|
|
"response": null,
|
|
"completion": {
|
|
"outcome": "succeeded",
|
|
"notes": "Script completed: cargo check -q --workspace 2>&1",
|
|
"failure_reason": null,
|
|
"timestamp": "2026-05-23T15:22:12.180967Z"
|
|
},
|
|
"provider_used": null,
|
|
"diff": null,
|
|
"script_invocation": {
|
|
"script": "cargo check -q --workspace 2>&1",
|
|
"command": "exec 2>&1\ncargo check -q --workspace 2>&1",
|
|
"language": "shell"
|
|
},
|
|
"script_timing": {
|
|
"output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126",
|
|
"exit_code": 0,
|
|
"duration_ms": 112316,
|
|
"termination": "exited",
|
|
"output_bytes": 0,
|
|
"live_streaming": false
|
|
},
|
|
"parallel_results": null,
|
|
"output": null,
|
|
"output_bytes": 0,
|
|
"live_streaming": false,
|
|
"termination": "exited",
|
|
"started_at": "2026-05-23T15:20:19.858560Z",
|
|
"handler": "command",
|
|
"timing": {
|
|
"wall_time_ms": 112321,
|
|
"inference_time_ms": 0,
|
|
"tool_time_ms": 0,
|
|
"active_time_ms": 0
|
|
},
|
|
"usage": {
|
|
"input_tokens": 0,
|
|
"output_tokens": 0,
|
|
"total_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"cache_read_tokens": 0,
|
|
"cache_write_tokens": 0
|
|
},
|
|
"state": "succeeded"
|
|
}
|
|
}
|
|
} |