diff --git a/run.json b/run.json index fe6a62b4e..65553c29e 100644 --- a/run.json +++ b/run.json @@ -524,7 +524,7 @@ "kind": "running" }, "status_updated_at": "2026-05-23T15:20:09.659052Z", - "last_event_at": "2026-05-23T15:22:17.485460Z", + "last_event_at": "2026-05-23T15:32:41.358054Z", "pending_control": null, "checkpoints": [ { @@ -691,9 +691,9 @@ } }, { - "seq": 0, + "seq": 47, "checkpoint": { - "timestamp": "2026-05-23T15:24:23.940428Z", + "timestamp": "2026-05-23T15:24:29.559500Z", "current_node": "preflight_lint", "completed_nodes": [ "start", @@ -703,17 +703,104 @@ ], "node_retries": {}, "context_values": { + "internal.run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6", + "thread.preflight_compile.current_node": "preflight_lint", + "internal.work_dir": "/home/daytona/workspace/fabro", + "thread.start.current_node": "toolchain", + "internal.retry_count.preflight_compile": 0, + "internal.fidelity": "compact", + "graph.goal": "# Agent Compaction API Usage Baseline Plan\n\nDate: 2026-05-23\n\n## Summary\n\nChange Fabro's agent compaction trigger from a whole-history `chars / 4`\nestimate to a Claude Code-style hot-path estimate: use the latest real\nassistant response's stored `usage.total_tokens()` as the baseline, then add\nlocal estimates for turns appended after that response. This avoids token-count\nprovider API calls while making compaction sensitive to actual\nprovider-reported context usage, including cache and reasoning tokens.\n\nNo provider token-count API calls should be added in this change.\n\n## Key Changes\n\n- Replace the current compaction estimate in `fabro-agent` with a new\n active-context estimator.\n- Find the newest assistant turn whose `usage.total_tokens() > 0`.\n- Use that `usage.total_tokens()` as the baseline.\n- Add local estimates only for turns after that assistant turn.\n- If no usable assistant usage exists, fall back to the existing local\n whole-history estimate.\n- Reuse a shared per-turn local estimate helper so fallback and post-baseline\n delta counting stay consistent.\n- Keep the estimator local and in-process. Do not call\n `llm_client.count_input_tokens()` from `compact_if_needed()`.\n\n## Implementation Details\n\n- Update `lib/crates/fabro-agent/src/compaction.rs`:\n - Add an estimator that returns both token count and method, for example\n `ApiUsagePlusLocalDelta` or `LocalEstimate`.\n - Make `check_context_usage()` use the new estimator and include the method\n in warning `details`.\n - Make `compact_context()` report the same improved estimate in\n `CompactionStarted`.\n - Move `CompactionStarted` emission after the\n `original_turn_count <= preserve_count` no-op check, so a no-op compact\n cannot emit started without completed.\n- Update `lib/crates/fabro-agent/src/history.rs`:\n - In `History::compact()`, invalidate preserved assistant usage by replacing\n preserved assistant `usage` with `TokenCounts::default()`.\n - Keep provider parts, response IDs, text, and tool calls unchanged.\n - Rationale: preserved assistant usage reflects the pre-compaction context and\n must not become the next baseline. Billing remains available from emitted\n run events, so mutable runtime history should prefer compaction correctness.\n- Leave public run event names and schemas unchanged:\n - `agent.compaction.started`\n - `agent.compaction.completed`\n - Existing warning event remains a warning with richer `details`.\n\n## Test Plan\n\n- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:\n - No assistant usage: estimator matches current local whole-history behavior.\n - Latest assistant usage present: estimator uses `usage.total_tokens()` plus\n only later tool/user/steering turns.\n - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.\n - Earlier assistant usage is ignored when a later assistant usage exists.\n- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:\n - `History::compact()` preserves assistant content, tool calls, and provider\n parts, but resets preserved assistant usage to default.\n - Existing OpenAI opaque stripping and Anthropic thinking preservation tests\n still pass.\n- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:\n - A short assistant response with high `usage.total_tokens()` triggers\n compaction even when text length is small.\n - A compact no-op due to `turns.len() <= preserve_count` does not emit\n `CompactionStarted`.\n - Compaction disabled still prevents compaction even if the API usage\n baseline exceeds threshold.\n- Run targeted verification:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - If those pass, run `cargo nextest run -p fabro-agent`.\n\n## Assumptions\n\n- Runtime/session `Message::Assistant.usage` is safe to invalidate after\n compaction because authoritative billing comes from emitted workflow/run\n events, not preserved mutable agent history.\n- A zero-token `TokenCounts::default()` should be treated as no usable API\n baseline.\n- Provider token-count APIs remain available for future near-threshold\n confirmation, but are intentionally out of scope for this change.\n", + "graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ", + "internal.retry_count.preflight_lint": 0, + "graph.rankdir": "LR", + "internal.retry_count.toolchain": 0, + "current_node": "preflight_lint", + "internal.node_visit_count": 1, + "internal.thread_id": "preflight_compile", + "thread.toolchain.current_node": "preflight_compile", + "failure_class": "", + "failure_signature": "", + "command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126", + "internal.retry_count.start": 0, + "outcome": "succeeded" + }, + "node_outcomes": { + "preflight_compile": { + "status": "succeeded", + "context_updates": { + "command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126" + }, + "notes": "Script completed: cargo check -q --workspace 2>&1", + "usage": null + }, + "preflight_lint": { + "status": "succeeded", + "context_updates": { + "command.output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126" + }, + "notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1", + "usage": null + }, + "start": { + "status": "succeeded", + "usage": null + }, + "toolchain": { + "status": "succeeded", + "context_updates": { + "command.output": "blob://sha256/fc14b2ba2d770e5cd3169df7a29525c962adfc4cfa3097b9098c63ebd61a748c" + }, + "notes": "Script completed: command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1", + "usage": null + } + }, + "next_node_id": "implement", + "git_commit_sha": "85cb0d5e89b3c3fc4a860c82b85375d71704a686", + "node_visits": { + "toolchain": 1, + "preflight_lint": 1, + "preflight_compile": 1, + "start": 1 + } + }, + "diff": { + "summary": { + "files_changed": 0, + "additions": 0, + "deletions": 0 + } + } + }, + { + "seq": 0, + "checkpoint": { + "timestamp": "2026-05-23T15:32:41.444884Z", + "current_node": "implement", + "completed_nodes": [ + "start", + "toolchain", + "preflight_compile", + "preflight_lint", + "implement" + ], + "node_retries": {}, + "context_values": { + "response.implement": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.total_tokens() > 0` as baseline.\n - Adds local estimates only for turns after that assistant response.\n - Falls back to existing whole-history `chars / 4` local estimate when no usable assistant usage exists.\n - Adds `estimate_method` to context-window warning details.\n - Uses the improved estimate for `CompactionStarted`.\n - Prevents no-op compaction from emitting `CompactionStarted`.\n- Updated `lib/crates/fabro-agent/src/history.rs`\n - `History::compact()` now resets preserved assistant `usage` to `TokenCounts::default()`.\n - Assistant content, tool calls, response IDs, and non-stripped provider parts remain preserved.\n- Added/updated tests in:\n - `compaction.rs`\n - `history.rs`\n - `session.rs`\n- Confirmed no `llm_client.count_input_tokens()` / provider token-count API calls were added.\n\nVerification:\n- Red phase confirmed initially with failing new compaction tests.\n- Passed:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - `cargo nextest run -p fabro-agent session`\n - `cargo nextest run -p fabro-agent`\n - `cargo +nightly-2026-04-14 fmt --check --all`\n - `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings`", + "last_stage": "implement", "thread.preflight_compile.current_node": "preflight_lint", "internal.work_dir": "/home/daytona/workspace/fabro", "thread.toolchain.current_node": "preflight_compile", - "current_node": "preflight_lint", + "internal.retry_count.implement": 0, + "current_node": "implement", "graph.rankdir": "LR", "internal.run_id": "01KSAPQQSVK4FY3CWHYVZYD7T6", "outcome": "succeeded", - "internal.thread_id": "preflight_compile", + "internal.thread_id": "preflight_lint", "thread.start.current_node": "toolchain", "internal.retry_count.toolchain": 0, "internal.node_visit_count": 1, + "last_response": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.tota", + "thread.preflight_lint.current_node": "implement", "failure_class": "", "internal.retry_count.preflight_lint": 0, "graph.model_stylesheet": "\n * { model: claude-opus-4-7; }\n ", @@ -745,6 +832,36 @@ "status": "succeeded", "usage": null }, + "implement": { + "status": "succeeded", + "context_updates": { + "last_response": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.tota", + "last_stage": "implement", + "response.implement": "Implemented the Agent Compaction API Usage Baseline plan.\n\nChanges made:\n- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs`\n - Uses newest assistant turn with `usage.total_tokens() > 0` as baseline.\n - Adds local estimates only for turns after that assistant response.\n - Falls back to existing whole-history `chars / 4` local estimate when no usable assistant usage exists.\n - Adds `estimate_method` to context-window warning details.\n - Uses the improved estimate for `CompactionStarted`.\n - Prevents no-op compaction from emitting `CompactionStarted`.\n- Updated `lib/crates/fabro-agent/src/history.rs`\n - `History::compact()` now resets preserved assistant `usage` to `TokenCounts::default()`.\n - Assistant content, tool calls, response IDs, and non-stripped provider parts remain preserved.\n- Added/updated tests in:\n - `compaction.rs`\n - `history.rs`\n - `session.rs`\n- Confirmed no `llm_client.count_input_tokens()` / provider token-count API calls were added.\n\nVerification:\n- Red phase confirmed initially with failing new compaction tests.\n- Passed:\n - `cargo nextest run -p fabro-agent compaction`\n - `cargo nextest run -p fabro-agent history`\n - `cargo nextest run -p fabro-agent session`\n - `cargo nextest run -p fabro-agent`\n - `cargo +nightly-2026-04-14 fmt --check --all`\n - `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings`" + }, + "notes": "Stage completed: implement", + "usage": { + "input": { + "usage": { + "model": { + "provider": "openai", + "model_id": "gpt-5.5" + }, + "tokens": { + "input_tokens": 168644, + "output_tokens": 9772, + "reasoning_tokens": 11720, + "cache_read_tokens": 3959296, + "cache_write_tokens": 0 + } + }, + "facts": { + "algorithm": "openai" + } + }, + "total_usd_micros": 3467628 + } + }, "preflight_lint": { "status": "succeeded", "context_updates": { @@ -754,12 +871,13 @@ "usage": null } }, - "next_node_id": "implement", + "next_node_id": "simplify_opus", "node_visits": { "toolchain": 1, "preflight_compile": 1, "start": 1, - "preflight_lint": 1 + "preflight_lint": 1, + "implement": 1 } }, "diff": {} @@ -785,12 +903,49 @@ "pull_request": null, "superseded_by": null, "pending_interviews": {}, + "todos_by_list": { + "openai_plan:bd5288b5-cb6c-4da6-a303-70346a50495c": { + "kind": "openai_plan", + "list_id": "openai_plan:bd5288b5-cb6c-4da6-a303-70346a50495c", + "items": [ + { + "id": "2771fe0b7d068bfd", + "status": "completed", + "order": 0, + "subject": "Add failing compaction/history/session tests for API-usage baseline behavior" + }, + { + "id": "969754ae428a19c5", + "status": "completed", + "order": 1, + "subject": "Implement active-context estimator and warning/event usage changes" + }, + { + "id": "10506f244baf24d5", + "status": "completed", + "order": 2, + "subject": "Reset preserved assistant usage during history compaction" + }, + { + "id": "4aad9d31be6bc056", + "status": "completed", + "order": 3, + "subject": "Run targeted fabro-agent test suites and fix any regressions" + } + ] + } + }, "stages": { "preflight_lint@1": { "first_event_seq": 40, "prompt": null, "response": null, - "completion": null, + "completion": { + "outcome": "succeeded", + "notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1", + "failure_reason": null, + "timestamp": "2026-05-23T15:24:23.939755Z" + }, "provider_used": null, "diff": null, "script_invocation": { @@ -798,11 +953,27 @@ "command": "exec 2>&1\ncargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1", "language": "shell" }, - "script_timing": null, + "script_timing": { + "output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126", + "exit_code": 0, + "duration_ms": 126448, + "termination": "exited", + "output_bytes": 0, + "live_streaming": false + }, "parallel_results": null, "output": null, + "output_bytes": 0, + "live_streaming": false, + "termination": "exited", "started_at": "2026-05-23T15:22:17.485222Z", "handler": "command", + "timing": { + "wall_time_ms": 126453, + "inference_time_ms": 0, + "tool_time_ms": 0, + "active_time_ms": 0 + }, "usage": { "input_tokens": 0, "output_tokens": 0, @@ -811,6 +982,38 @@ "cache_read_tokens": 0, "cache_write_tokens": 0 }, + "state": "succeeded" + }, + "implement@1": { + "first_event_seq": 50, + "prompt": null, + "response": null, + "completion": null, + "provider_used": { + "mode": "agent", + "provider": "openai", + "model": "gpt-5.5" + }, + "diff": null, + "script_invocation": null, + "script_timing": null, + "parallel_results": null, + "output": null, + "started_at": "2026-05-23T15:24:29.561071Z", + "handler": "agent", + "usage": { + "input_tokens": 168644, + "output_tokens": 9772, + "total_tokens": 4149432, + "reasoning_tokens": 11720, + "cache_read_tokens": 3959296, + "cache_write_tokens": 0, + "total_usd_micros": 3467628 + }, + "model": { + "provider": "openai", + "model_id": "gpt-5.5" + }, "state": "running" }, "start@1": { diff --git a/stages/004-preflight_lint@1/output.log b/stages/004-preflight_lint@1/output.log new file mode 100644 index 000000000..d87ba9545 --- /dev/null +++ b/stages/004-preflight_lint@1/output.log @@ -0,0 +1 @@ +blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126 \ No newline at end of file diff --git a/stages/004-preflight_lint@1/script_timing.json b/stages/004-preflight_lint@1/script_timing.json new file mode 100644 index 000000000..6809bdc4a --- /dev/null +++ b/stages/004-preflight_lint@1/script_timing.json @@ -0,0 +1,8 @@ +{ + "output": "blob://sha256/12ae32cb1ec02d01eda3581b127c1fee3b0dc53572ed6baf239721a03d82e126", + "exit_code": 0, + "duration_ms": 126448, + "termination": "exited", + "output_bytes": 0, + "live_streaming": false +} \ No newline at end of file diff --git a/stages/004-preflight_lint@1/status.json b/stages/004-preflight_lint@1/status.json new file mode 100644 index 000000000..27b573f87 --- /dev/null +++ b/stages/004-preflight_lint@1/status.json @@ -0,0 +1,6 @@ +{ + "outcome": "succeeded", + "notes": "Script completed: cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1", + "failure_reason": null, + "timestamp": "2026-05-23T15:24:23.939755Z" +} \ No newline at end of file diff --git a/stages/005-implement@1/prompt.md b/stages/005-implement@1/prompt.md new file mode 100644 index 000000000..42316d497 --- /dev/null +++ b/stages/005-implement@1/prompt.md @@ -0,0 +1,105 @@ +Goal: # Agent Compaction API Usage Baseline Plan + +Date: 2026-05-23 + +## Summary + +Change Fabro's agent compaction trigger from a whole-history `chars / 4` +estimate to a Claude Code-style hot-path estimate: use the latest real +assistant response's stored `usage.total_tokens()` as the baseline, then add +local estimates for turns appended after that response. This avoids token-count +provider API calls while making compaction sensitive to actual +provider-reported context usage, including cache and reasoning tokens. + +No provider token-count API calls should be added in this change. + +## Key Changes + +- Replace the current compaction estimate in `fabro-agent` with a new + active-context estimator. +- Find the newest assistant turn whose `usage.total_tokens() > 0`. +- Use that `usage.total_tokens()` as the baseline. +- Add local estimates only for turns after that assistant turn. +- If no usable assistant usage exists, fall back to the existing local + whole-history estimate. +- Reuse a shared per-turn local estimate helper so fallback and post-baseline + delta counting stay consistent. +- Keep the estimator local and in-process. Do not call + `llm_client.count_input_tokens()` from `compact_if_needed()`. + +## Implementation Details + +- Update `lib/crates/fabro-agent/src/compaction.rs`: + - Add an estimator that returns both token count and method, for example + `ApiUsagePlusLocalDelta` or `LocalEstimate`. + - Make `check_context_usage()` use the new estimator and include the method + in warning `details`. + - Make `compact_context()` report the same improved estimate in + `CompactionStarted`. + - Move `CompactionStarted` emission after the + `original_turn_count <= preserve_count` no-op check, so a no-op compact + cannot emit started without completed. +- Update `lib/crates/fabro-agent/src/history.rs`: + - In `History::compact()`, invalidate preserved assistant usage by replacing + preserved assistant `usage` with `TokenCounts::default()`. + - Keep provider parts, response IDs, text, and tool calls unchanged. + - Rationale: preserved assistant usage reflects the pre-compaction context and + must not become the next baseline. Billing remains available from emitted + run events, so mutable runtime history should prefer compaction correctness. +- Leave public run event names and schemas unchanged: + - `agent.compaction.started` + - `agent.compaction.completed` + - Existing warning event remains a warning with richer `details`. + +## Test Plan + +- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`: + - No assistant usage: estimator matches current local whole-history behavior. + - Latest assistant usage present: estimator uses `usage.total_tokens()` plus + only later tool/user/steering turns. + - Usage fields include cache and reasoning through `TokenCounts::total_tokens()`. + - Earlier assistant usage is ignored when a later assistant usage exists. +- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`: + - `History::compact()` preserves assistant content, tool calls, and provider + parts, but resets preserved assistant usage to default. + - Existing OpenAI opaque stripping and Anthropic thinking preservation tests + still pass. +- Add session coverage in `lib/crates/fabro-agent/src/session.rs`: + - A short assistant response with high `usage.total_tokens()` triggers + compaction even when text length is small. + - A compact no-op due to `turns.len() <= preserve_count` does not emit + `CompactionStarted`. + - Compaction disabled still prevents compaction even if the API usage + baseline exceeds threshold. +- Run targeted verification: + - `cargo nextest run -p fabro-agent compaction` + - `cargo nextest run -p fabro-agent history` + - If those pass, run `cargo nextest run -p fabro-agent`. + +## Assumptions + +- Runtime/session `Message::Assistant.usage` is safe to invalidate after + compaction because authoritative billing comes from emitted workflow/run + events, not preserved mutable agent history. +- A zero-token `TokenCounts::default()` should be treated as no usable API + baseline. +- Provider token-count APIs remain available for future near-threshold + confirmation, but are intentionally out of scope for this change. + + +## Completed stages +- **toolchain**: succeeded + - Script: `command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1` + - Output: + ``` + cargo 1.95.0 (f2d3ce0bd 2026-03-21) + ``` +- **preflight_compile**: succeeded + - Script: `cargo check -q --workspace 2>&1` + - Output: (empty) +- **preflight_lint**: succeeded + - Script: `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1` + - Output: (empty) + + +Read the plan file referenced in the goal and implement every step. Make all the code changes described in the plan. Use red/green TDD. \ No newline at end of file diff --git a/stages/005-implement@1/provider_used.json b/stages/005-implement@1/provider_used.json new file mode 100644 index 000000000..a04162cbf --- /dev/null +++ b/stages/005-implement@1/provider_used.json @@ -0,0 +1,5 @@ +{ + "mode": "agent", + "provider": "openai", + "model": "gpt-5.5" +} \ No newline at end of file diff --git a/stages/005-implement@1/response.md b/stages/005-implement@1/response.md new file mode 100644 index 000000000..78f77badf --- /dev/null +++ b/stages/005-implement@1/response.md @@ -0,0 +1,28 @@ +Implemented the Agent Compaction API Usage Baseline plan. + +Changes made: +- Added active context estimation in `lib/crates/fabro-agent/src/compaction.rs` + - Uses newest assistant turn with `usage.total_tokens() > 0` as baseline. + - Adds local estimates only for turns after that assistant response. + - Falls back to existing whole-history `chars / 4` local estimate when no usable assistant usage exists. + - Adds `estimate_method` to context-window warning details. + - Uses the improved estimate for `CompactionStarted`. + - Prevents no-op compaction from emitting `CompactionStarted`. +- Updated `lib/crates/fabro-agent/src/history.rs` + - `History::compact()` now resets preserved assistant `usage` to `TokenCounts::default()`. + - Assistant content, tool calls, response IDs, and non-stripped provider parts remain preserved. +- Added/updated tests in: + - `compaction.rs` + - `history.rs` + - `session.rs` +- Confirmed no `llm_client.count_input_tokens()` / provider token-count API calls were added. + +Verification: +- Red phase confirmed initially with failing new compaction tests. +- Passed: + - `cargo nextest run -p fabro-agent compaction` + - `cargo nextest run -p fabro-agent history` + - `cargo nextest run -p fabro-agent session` + - `cargo nextest run -p fabro-agent` + - `cargo +nightly-2026-04-14 fmt --check --all` + - `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings` \ No newline at end of file