checkpoint

⚒️ Generated with [Fabro](https://fabro.sh)
This commit is contained in:
Fabro 2026-05-23 11:43:54 -04:00
parent fbbb0c6303
commit 08a40c3b26
6 changed files with 1065 additions and 17 deletions

253
run.json

File diff suppressed because one or more lines are too long

View file

@ -0,0 +1,630 @@
diff --git a/lib/crates/fabro-agent/src/compaction.rs b/lib/crates/fabro-agent/src/compaction.rs
index 01c756dc5..7c2434fa7 100644
--- a/lib/crates/fabro-agent/src/compaction.rs
+++ b/lib/crates/fabro-agent/src/compaction.rs
@@ -11,6 +11,29 @@ use crate::file_tracker::FileTracker;
use crate::history::History;
use crate::types::{AgentEvent, Message};
+const APPROX_CHARS_PER_TOKEN: usize = 4;
+
+#[derive(Debug, Clone, Copy, PartialEq, Eq)]
+enum ContextEstimateMethod {
+ ApiUsagePlusLocalDelta,
+ LocalEstimate,
+}
+
+impl ContextEstimateMethod {
+ const fn as_str(self) -> &'static str {
+ match self {
+ Self::ApiUsagePlusLocalDelta => "api_usage_plus_local_delta",
+ Self::LocalEstimate => "local_estimate",
+ }
+ }
+}
+
+#[derive(Debug, Clone, Copy, PartialEq, Eq)]
+struct ContextEstimate {
+ tokens: usize,
+ method: ContextEstimateMethod,
+}
+
/// Check whether the context window usage exceeds the configured threshold.
/// Emits a `Warning` event with kind `"context_window"` when over the
/// threshold. Returns `true` if the threshold is exceeded.
@@ -22,21 +45,21 @@ pub fn check_context_usage(
emitter: &Emitter,
session_id: &str,
) -> bool {
- let estimated_tokens = estimate_token_count(system_prompt, history);
+ let estimate = estimate_active_context_usage(system_prompt, history);
+ let estimated_tokens = estimate.tokens;
let context_window = provider_profile.context_window_size();
let threshold = context_window * threshold_percent / 100;
if estimated_tokens > threshold {
+ let usage_percent = estimated_tokens.saturating_mul(100) / context_window;
emitter.emit(session_id.to_owned(), AgentEvent::Warning {
kind: "context_window".into(),
- message: format!(
- "Context window usage: {}%",
- estimated_tokens * 100 / context_window
- ),
+ message: format!("Context window usage: {usage_percent}%"),
details: serde_json::json!({
"estimated_tokens": estimated_tokens,
"context_window_size": context_window,
- "usage_percent": estimated_tokens * 100 / context_window,
+ "usage_percent": usage_percent,
+ "estimate_method": estimate.method.as_str(),
}),
});
true
@@ -61,19 +84,21 @@ pub async fn compact_context(
emitter: &Emitter,
session_id: &str,
) -> Result<(), Error> {
- let estimated_tokens = estimate_token_count(system_prompt, history);
- let context_window = provider_profile.context_window_size();
let original_turn_count = history.turns().len();
+ // Determine turns to summarize. If there are not enough turns to compact,
+ // do not emit a started event without a matching completion.
+ if original_turn_count <= preserve_count {
+ return Ok(());
+ }
+
+ let estimate = estimate_active_context_usage(system_prompt, history);
+ let context_window = provider_profile.context_window_size();
emitter.emit(session_id.to_owned(), AgentEvent::CompactionStarted {
- estimated_tokens,
+ estimated_tokens: estimate.tokens,
context_window_size: context_window,
});
- // Determine turns to summarize
- if original_turn_count <= preserve_count {
- return Ok(());
- }
let turns_to_summarize = &history.turns()[..original_turn_count - preserve_count];
let rendered = render_turns_for_summary(turns_to_summarize);
@@ -156,37 +181,63 @@ Build on their progress — do not repeat completed steps.\n\n{summary_text}"
/// Estimate the total token count of the system prompt and conversation
/// history. Uses a rough heuristic of ~4 characters per token.
pub fn estimate_token_count(system_prompt: &str, history: &History) -> usize {
- let mut total_chars = system_prompt.len();
+ estimate_local_token_count(system_prompt, history.turns())
+}
- for turn in history.turns() {
- match turn {
- Message::User { content, .. } => total_chars += content.len(),
- Message::Assistant {
- content,
- tool_calls,
- ..
- } => {
- total_chars += content.len();
- if let Some(r) = turn.reasoning_text() {
- total_chars += r.len();
- }
- for tc in tool_calls {
- total_chars += tc.name.len();
- total_chars += tc.arguments.to_string().len();
- }
- }
- Message::ToolResults { results, .. } => {
- for r in results {
- total_chars += r.content.to_string().len();
- }
- }
- Message::System { content, .. } | Message::Steering { content, .. } => {
- total_chars += content.len();
+fn estimate_active_context_usage(system_prompt: &str, history: &History) -> ContextEstimate {
+ let turns = history.turns();
+ if let Some((baseline_index, baseline_tokens)) = latest_assistant_usage_baseline(turns) {
+ let local_delta = estimate_local_token_count("", &turns[baseline_index + 1..]);
+ return ContextEstimate {
+ tokens: baseline_tokens.saturating_add(local_delta),
+ method: ContextEstimateMethod::ApiUsagePlusLocalDelta,
+ };
+ }
+
+ ContextEstimate {
+ tokens: estimate_local_token_count(system_prompt, turns),
+ method: ContextEstimateMethod::LocalEstimate,
+ }
+}
+
+fn latest_assistant_usage_baseline(turns: &[Message]) -> Option<(usize, usize)> {
+ turns.iter().enumerate().rev().find_map(|(index, turn)| {
+ if let Message::Assistant { usage, .. } = turn {
+ let total_tokens = usage.total_tokens();
+ if total_tokens > 0 {
+ return Some((index, usize::try_from(total_tokens).unwrap_or(usize::MAX)));
}
}
- }
+ None
+ })
+}
+
+fn estimate_local_token_count(system_prompt: &str, turns: &[Message]) -> usize {
+ let turn_chars: usize = turns.iter().map(estimate_turn_chars).sum();
+ (system_prompt.len() + turn_chars) / APPROX_CHARS_PER_TOKEN
+}
- total_chars / 4 // rough estimate: ~4 chars per token
+fn estimate_turn_chars(turn: &Message) -> usize {
+ match turn {
+ Message::User { content, .. }
+ | Message::System { content, .. }
+ | Message::Steering { content, .. } => content.len(),
+ Message::Assistant {
+ content,
+ tool_calls,
+ ..
+ } => {
+ let reasoning_chars = turn.reasoning_text().map_or(0, str::len);
+ let tool_call_chars: usize = tool_calls
+ .iter()
+ .map(|tc| tc.name.len() + tc.arguments.to_string().len())
+ .sum();
+ content.len() + reasoning_chars + tool_call_chars
+ }
+ Message::ToolResults { results, .. } => {
+ results.iter().map(|r| r.content.to_string().len()).sum()
+ }
+ }
}
/// Render conversation turns into a human-readable summary format for the
@@ -323,6 +374,149 @@ mod tests {
assert_eq!(estimate_token_count("test", &history), 3);
}
+ #[test]
+ fn active_context_estimate_without_assistant_usage_matches_local_history_estimate() {
+ let mut history = History::default();
+ history.push(Message::User {
+ content: "Hello world".into(),
+ timestamp: SystemTime::now(),
+ });
+ history.push(Message::Assistant {
+ content: "No usage available".into(),
+ tool_calls: vec![ToolCall::new(
+ "call_1",
+ "read_file",
+ serde_json::json!({"path": "foo.rs"}),
+ )],
+ provider_parts: vec![],
+ usage: Box::new(TokenCounts::default()),
+ response_id: "resp_1".into(),
+ timestamp: SystemTime::now(),
+ });
+ history.push(Message::ToolResults {
+ results: vec![ToolResult::success("call_1", serde_json::json!(1234))],
+ timestamp: SystemTime::now(),
+ });
+
+ let estimate = estimate_active_context_usage("test", &history);
+
+ assert_eq!(estimate.tokens, estimate_token_count("test", &history));
+ assert_eq!(estimate.method, ContextEstimateMethod::LocalEstimate);
+ }
+
+ #[test]
+ fn active_context_estimate_uses_latest_assistant_usage_plus_later_turns() {
+ let mut history = History::default();
+ history.push(Message::User {
+ content: "ignored before baseline".repeat(100),
+ timestamp: SystemTime::now(),
+ });
+ history.push(Message::Assistant {
+ content: "baseline response".into(),
+ tool_calls: vec![],
+ provider_parts: vec![],
+ usage: Box::new(TokenCounts {
+ input_tokens: 50,
+ ..TokenCounts::default()
+ }),
+ response_id: "resp_1".into(),
+ timestamp: SystemTime::now(),
+ });
+ history.push(Message::ToolResults {
+ // JSON number renders as 4 chars => 1 local token.
+ results: vec![ToolResult::success("call_1", serde_json::json!(1234))],
+ timestamp: SystemTime::now(),
+ });
+ history.push(Message::User {
+ // 16 chars => 4 local tokens.
+ content: "u".repeat(16),
+ timestamp: SystemTime::now(),
+ });
+ history.push(Message::Steering {
+ // 8 chars => 2 local tokens.
+ content: "s".repeat(8),
+ timestamp: SystemTime::now(),
+ });
+
+ let estimate = estimate_active_context_usage("ignored system prompt", &history);
+
+ assert_eq!(estimate.tokens, 57);
+ assert_eq!(
+ estimate.method,
+ ContextEstimateMethod::ApiUsagePlusLocalDelta
+ );
+ }
+
+ #[test]
+ fn active_context_estimate_uses_total_tokens_including_cache_and_reasoning() {
+ let mut history = History::default();
+ history.push(Message::Assistant {
+ content: "short".into(),
+ tool_calls: vec![],
+ provider_parts: vec![],
+ usage: Box::new(TokenCounts {
+ input_tokens: 10,
+ output_tokens: 20,
+ reasoning_tokens: 30,
+ cache_read_tokens: 40,
+ cache_write_tokens: 50,
+ }),
+ response_id: "resp_1".into(),
+ timestamp: SystemTime::now(),
+ });
+
+ let estimate = estimate_active_context_usage("", &history);
+
+ assert_eq!(estimate.tokens, 150);
+ assert_eq!(
+ estimate.method,
+ ContextEstimateMethod::ApiUsagePlusLocalDelta
+ );
+ }
+
+ #[test]
+ fn active_context_estimate_ignores_earlier_usage_when_later_usage_exists() {
+ let mut history = History::default();
+ history.push(Message::Assistant {
+ content: "older response".into(),
+ tool_calls: vec![],
+ provider_parts: vec![],
+ usage: Box::new(TokenCounts {
+ input_tokens: 1_000,
+ ..TokenCounts::default()
+ }),
+ response_id: "resp_old".into(),
+ timestamp: SystemTime::now(),
+ });
+ history.push(Message::User {
+ content: "ignored before latest baseline".repeat(100),
+ timestamp: SystemTime::now(),
+ });
+ history.push(Message::Assistant {
+ content: "latest response".into(),
+ tool_calls: vec![],
+ provider_parts: vec![],
+ usage: Box::new(TokenCounts {
+ input_tokens: 20,
+ ..TokenCounts::default()
+ }),
+ response_id: "resp_new".into(),
+ timestamp: SystemTime::now(),
+ });
+ history.push(Message::User {
+ content: "u".repeat(8),
+ timestamp: SystemTime::now(),
+ });
+
+ let estimate = estimate_active_context_usage("", &history);
+
+ assert_eq!(estimate.tokens, 22);
+ assert_eq!(
+ estimate.method,
+ ContextEstimateMethod::ApiUsagePlusLocalDelta
+ );
+ }
+
#[test]
fn check_context_usage_below_threshold() {
let history = History::default();
@@ -350,6 +544,7 @@ mod tests {
// Should have emitted a Warning
let event = rx.try_recv().unwrap();
- assert!(matches!(event.event, AgentEvent::Warning { .. }));
+ assert!(matches!(event.event, AgentEvent::Warning { details, .. }
+ if details["estimate_method"] == "local_estimate"));
}
}
diff --git a/lib/crates/fabro-agent/src/history.rs b/lib/crates/fabro-agent/src/history.rs
index 497900eae..33182ae73 100644
--- a/lib/crates/fabro-agent/src/history.rs
+++ b/lib/crates/fabro-agent/src/history.rs
@@ -1,4 +1,4 @@
-use fabro_llm::types::{ContentPart, Message as LlmMessage, Role};
+use fabro_llm::types::{ContentPart, Message as LlmMessage, Role, TokenCounts};
use fabro_types::SessionMessage;
use crate::types::Message;
@@ -36,7 +36,8 @@ impl History {
if self.turns.len() <= preserve_count {
return;
}
- let preserved = self.turns.split_off(self.turns.len() - preserve_count);
+ let mut preserved = self.turns.split_off(self.turns.len() - preserve_count);
+ invalidate_assistant_usage(&mut preserved);
let discarded = std::mem::take(&mut self.turns);
let extracted_user_messages =
extract_recent_user_messages(discarded, COMPACTION_USER_MESSAGE_TOKEN_BUDGET);
@@ -119,6 +120,14 @@ impl History {
}
}
+fn invalidate_assistant_usage(turns: &mut [Message]) {
+ for turn in turns {
+ if let Message::Assistant { usage, .. } = turn {
+ **usage = TokenCounts::default();
+ }
+ }
+}
+
/// Maximum token budget for user messages extracted from discarded turns during
/// compaction.
const COMPACTION_USER_MESSAGE_TOKEN_BUDGET: usize = 20_000;
@@ -599,6 +608,60 @@ mod tests {
}
}
+ #[test]
+ fn compact_preserves_assistant_data_but_resets_usage() {
+ let mut history = History::default();
+ history.push(Message::User {
+ content: "old msg".into(),
+ timestamp: SystemTime::now(),
+ });
+ let tool_call = ToolCall::new("call_1", "search", serde_json::json!({"query": "fabro"}));
+ let thinking = ContentPart::Thinking(ThinkingData {
+ text: "deep thought".into(),
+ signature: Some("sig_xyz".into()),
+ redacted: false,
+ });
+ history.push(Message::Assistant {
+ content: "answer".into(),
+ tool_calls: vec![tool_call.clone()],
+ provider_parts: vec![thinking.clone()],
+ usage: Box::new(TokenCounts {
+ input_tokens: 10,
+ output_tokens: 20,
+ reasoning_tokens: 30,
+ cache_read_tokens: 40,
+ cache_write_tokens: 50,
+ }),
+ response_id: "resp_1".into(),
+ timestamp: SystemTime::now(),
+ });
+
+ history.compact(1, "Summary".into());
+
+ let assistant_turn = history
+ .turns()
+ .iter()
+ .find(|turn| matches!(turn, Message::Assistant { .. }))
+ .expect("preserved assistant turn");
+ if let Message::Assistant {
+ content,
+ tool_calls,
+ provider_parts,
+ usage,
+ response_id,
+ ..
+ } = assistant_turn
+ {
+ assert_eq!(content, "answer");
+ assert_eq!(tool_calls, &[tool_call]);
+ assert_eq!(provider_parts, &[thinking]);
+ assert_eq!(response_id, "resp_1");
+ assert_eq!(**usage, TokenCounts::default());
+ } else {
+ panic!("expected Assistant turn");
+ }
+ }
+
#[test]
fn compact_strips_reasoning_from_all_preserved_assistant_turns() {
let mut history = History::default();
diff --git a/lib/crates/fabro-agent/src/session.rs b/lib/crates/fabro-agent/src/session.rs
index 2d0d0b4e8..de9c58093 100644
--- a/lib/crates/fabro-agent/src/session.rs
+++ b/lib/crates/fabro-agent/src/session.rs
@@ -1853,7 +1853,7 @@ mod tests {
use fabro_llm::error::{ProviderErrorDetail, ProviderErrorKind};
use fabro_llm::provider::{ProviderAdapter, StreamEventStream};
use fabro_llm::types::{
- ContentPart, ReasoningEffort, Request, Response, Role, StreamEvent, ToolCall,
+ ContentPart, ReasoningEffort, Request, Response, Role, StreamEvent, TokenCounts, ToolCall,
ToolDefinition,
};
use futures::stream;
@@ -3653,13 +3653,25 @@ mod tests {
assert!(found_auth_error_event, "expected auth error event");
}
+ fn response_with_usage(mut response: Response, usage: TokenCounts) -> Response {
+ response.usage = usage;
+ response
+ }
+
+ fn response_with_total_usage(response: Response, total_tokens: i64) -> Response {
+ response_with_usage(response, TokenCounts {
+ input_tokens: total_tokens,
+ ..TokenCounts::default()
+ })
+ }
+
#[tokio::test]
async fn compaction_triggered_when_over_threshold() {
// Tiny context window to trigger compaction
// Responses: [0] conversation response (stream), [1] summarization (complete),
// [2] unused fallback
let responses = vec![
- text_response("OK"),
+ response_with_usage(text_response("OK"), TokenCounts::default()),
text_response("Here is the summary of the conversation so far."),
text_response("fallback"),
];
@@ -3704,6 +3716,90 @@ mod tests {
);
}
+ #[tokio::test]
+ async fn compaction_uses_assistant_usage_baseline_for_short_response() {
+ let responses = vec![
+ response_with_total_usage(text_response("OK"), 90),
+ text_response("Here is the summary of the conversation so far."),
+ text_response("fallback"),
+ ];
+
+ let provider = Arc::new(MockLlmProvider::new(responses));
+ let client = make_client(provider).await;
+ let registry = ToolRegistry::new();
+ let profile = Arc::new(TestProfile::with_context_window(registry, 100));
+ let env = Arc::new(MockSandbox::default());
+ let config = SessionOptions {
+ enable_context_compaction: true,
+ compaction_preserve_turns: 1,
+ ..Default::default()
+ };
+ let mut session = Session::new(client, profile, env, config, None);
+ let mut rx = session.subscribe();
+
+ session.process_input("hi").await.unwrap();
+
+ let mut started = None;
+ let mut found_completed = false;
+ while let Ok(event) = rx.try_recv() {
+ match event.event {
+ AgentEvent::CompactionStarted {
+ estimated_tokens,
+ context_window_size,
+ } => started = Some((estimated_tokens, context_window_size)),
+ AgentEvent::CompactionCompleted { .. } => found_completed = true,
+ _ => {}
+ }
+ }
+
+ assert_eq!(started, Some((90, 100)));
+ assert!(
+ found_completed,
+ "CompactionCompleted event should be emitted"
+ );
+ }
+
+ #[tokio::test]
+ async fn compaction_noop_does_not_emit_started() {
+ let large_input = "x".repeat(400);
+ let responses = vec![text_response("OK")];
+
+ let provider = Arc::new(MockLlmProvider::new(responses));
+ let client = make_client(provider).await;
+ let registry = ToolRegistry::new();
+ let profile = Arc::new(TestProfile::with_context_window(registry, 100));
+ let env = Arc::new(MockSandbox::default());
+ let config = SessionOptions {
+ enable_context_compaction: true,
+ compaction_preserve_turns: 10,
+ ..Default::default()
+ };
+ let mut session = Session::new(client, profile, env, config, None);
+ let mut rx = session.subscribe();
+
+ session.process_input(&large_input).await.unwrap();
+
+ let mut found_warning = false;
+ let mut found_compaction = false;
+ while let Ok(event) = rx.try_recv() {
+ match event.event {
+ AgentEvent::Warning { kind, .. } if kind == "context_window" => {
+ found_warning = true;
+ }
+ AgentEvent::CompactionStarted { .. } | AgentEvent::CompactionCompleted { .. } => {
+ found_compaction = true;
+ }
+ _ => {}
+ }
+ }
+
+ assert!(found_warning, "threshold should have been exceeded");
+ assert!(
+ !found_compaction,
+ "no-op compaction should not emit started or completed events"
+ );
+ }
+
#[tokio::test]
async fn compaction_not_triggered_when_disabled() {
let large_input = "x".repeat(400);
@@ -3735,6 +3831,49 @@ mod tests {
assert!(!found_compaction, "No compaction events when disabled");
}
+ #[tokio::test]
+ async fn compaction_disabled_blocks_api_usage_baseline_compaction() {
+ let responses = vec![response_with_total_usage(text_response("OK"), 90)];
+
+ let provider = Arc::new(MockLlmProvider::new(responses));
+ let client = make_client(provider).await;
+ let registry = ToolRegistry::new();
+ let profile = Arc::new(TestProfile::with_context_window(registry, 100));
+ let env = Arc::new(MockSandbox::default());
+ let config = SessionOptions {
+ enable_context_compaction: false,
+ compaction_preserve_turns: 1,
+ ..Default::default()
+ };
+ let mut session = Session::new(client, profile, env, config, None);
+ let mut rx = session.subscribe();
+
+ session.process_input("hi").await.unwrap();
+
+ let mut found_api_usage_warning = false;
+ let mut found_compaction = false;
+ while let Ok(event) = rx.try_recv() {
+ match event.event {
+ AgentEvent::Warning { details, .. }
+ if details["estimated_tokens"] == 90
+ && details["estimate_method"] == "api_usage_plus_local_delta" =>
+ {
+ found_api_usage_warning = true;
+ }
+ AgentEvent::CompactionStarted { .. } | AgentEvent::CompactionCompleted { .. } => {
+ found_compaction = true;
+ }
+ _ => {}
+ }
+ }
+
+ assert!(
+ found_api_usage_warning,
+ "API usage baseline should still drive context warning"
+ );
+ assert!(!found_compaction, "compaction must remain disabled");
+ }
+
#[tokio::test]
async fn compaction_failure_is_non_fatal() {
// Response [0] = conversation response (stream), [1] will be used for
@@ -3789,7 +3928,10 @@ mod tests {
}
let large_input = "x".repeat(400);
- let responses = vec![text_response("OK")];
+ let responses = vec![response_with_usage(
+ text_response("OK"),
+ TokenCounts::default(),
+ )];
let provider = Arc::new(StreamOnlyProvider {
responses,

View file

@ -0,0 +1,6 @@
{
"outcome": "succeeded",
"notes": "Stage completed: implement",
"failure_reason": null,
"timestamp": "2026-05-23T15:32:41.443688Z"
}

View file

@ -0,0 +1,156 @@
Goal: # Agent Compaction API Usage Baseline Plan
Date: 2026-05-23
## Summary
Change Fabro's agent compaction trigger from a whole-history `chars / 4`
estimate to a Claude Code-style hot-path estimate: use the latest real
assistant response's stored `usage.total_tokens()` as the baseline, then add
local estimates for turns appended after that response. This avoids token-count
provider API calls while making compaction sensitive to actual
provider-reported context usage, including cache and reasoning tokens.
No provider token-count API calls should be added in this change.
## Key Changes
- Replace the current compaction estimate in `fabro-agent` with a new
active-context estimator.
- Find the newest assistant turn whose `usage.total_tokens() > 0`.
- Use that `usage.total_tokens()` as the baseline.
- Add local estimates only for turns after that assistant turn.
- If no usable assistant usage exists, fall back to the existing local
whole-history estimate.
- Reuse a shared per-turn local estimate helper so fallback and post-baseline
delta counting stay consistent.
- Keep the estimator local and in-process. Do not call
`llm_client.count_input_tokens()` from `compact_if_needed()`.
## Implementation Details
- Update `lib/crates/fabro-agent/src/compaction.rs`:
- Add an estimator that returns both token count and method, for example
`ApiUsagePlusLocalDelta` or `LocalEstimate`.
- Make `check_context_usage()` use the new estimator and include the method
in warning `details`.
- Make `compact_context()` report the same improved estimate in
`CompactionStarted`.
- Move `CompactionStarted` emission after the
`original_turn_count <= preserve_count` no-op check, so a no-op compact
cannot emit started without completed.
- Update `lib/crates/fabro-agent/src/history.rs`:
- In `History::compact()`, invalidate preserved assistant usage by replacing
preserved assistant `usage` with `TokenCounts::default()`.
- Keep provider parts, response IDs, text, and tool calls unchanged.
- Rationale: preserved assistant usage reflects the pre-compaction context and
must not become the next baseline. Billing remains available from emitted
run events, so mutable runtime history should prefer compaction correctness.
- Leave public run event names and schemas unchanged:
- `agent.compaction.started`
- `agent.compaction.completed`
- Existing warning event remains a warning with richer `details`.
## Test Plan
- Add unit coverage in `lib/crates/fabro-agent/src/compaction.rs`:
- No assistant usage: estimator matches current local whole-history behavior.
- Latest assistant usage present: estimator uses `usage.total_tokens()` plus
only later tool/user/steering turns.
- Usage fields include cache and reasoning through `TokenCounts::total_tokens()`.
- Earlier assistant usage is ignored when a later assistant usage exists.
- Add unit coverage in `lib/crates/fabro-agent/src/history.rs`:
- `History::compact()` preserves assistant content, tool calls, and provider
parts, but resets preserved assistant usage to default.
- Existing OpenAI opaque stripping and Anthropic thinking preservation tests
still pass.
- Add session coverage in `lib/crates/fabro-agent/src/session.rs`:
- A short assistant response with high `usage.total_tokens()` triggers
compaction even when text length is small.
- A compact no-op due to `turns.len() <= preserve_count` does not emit
`CompactionStarted`.
- Compaction disabled still prevents compaction even if the API usage
baseline exceeds threshold.
- Run targeted verification:
- `cargo nextest run -p fabro-agent compaction`
- `cargo nextest run -p fabro-agent history`
- If those pass, run `cargo nextest run -p fabro-agent`.
## Assumptions
- Runtime/session `Message::Assistant.usage` is safe to invalidate after
compaction because authoritative billing comes from emitted workflow/run
events, not preserved mutable agent history.
- A zero-token `TokenCounts::default()` should be treated as no usable API
baseline.
- Provider token-count APIs remain available for future near-threshold
confirmation, but are intentionally out of scope for this change.
## Completed stages
- **toolchain**: succeeded
- Script: `command -v cargo >/dev/null || { curl --proto '=https' --tlsv1.2 -sSf https://sh.rustup.rs | sh -s -- -y && sudo ln -sf $HOME/.cargo/bin/* /usr/local/bin/; }; cargo --version 2>&1`
- Output:
```
cargo 1.95.0 (f2d3ce0bd 2026-03-21)
```
- **preflight_compile**: succeeded
- Script: `cargo check -q --workspace 2>&1`
- Output: (empty)
- **preflight_lint**: succeeded
- Script: `cargo +nightly-2026-04-14 clippy -q --workspace --all-targets -- -D warnings 2>&1`
- Output: (empty)
- **implement**: succeeded
- Model: gpt-5.5, 168.6k tokens in / 21.5k out
# Simplify: Code Review and Cleanup
Review changes vs. origin for reuse, quality, and efficiency. Fix any issues found.
## Phase 1: Identify Changes
Run git diff (or git diff HEAD if there are staged changes) to see what changed. If there are no git changes, review the most recently modified files that the user mentioned or that you edited earlier in this conversation.
## Phase 2: Launch Three Review Agents in Parallel
Use the Agent tool to launch all three agents concurrently in a single message. Pass each agent the full diff so it has the complete context.
### Agent 1: Code Reuse Review
For each change:
1. Search for existing utilities and helpers that could replace newly written code. Use Grep to find similar patterns elsewhere in the codebase — common locations are utility directories, shared modules, and files adjacent to the changed ones.
2. Flag any new function that duplicates existing functionality. Suggest the existing function to use instead.
3. Flag any inline logic that could use an existing utility — hand-rolled string manipulation, manual path handling, custom environment checks, ad-hoc type guards, and similar patterns are common candidates.
Note: This is a greenfield app, so focus on maximizing simplicity and don't worry about changing things to achieve it.
### Agent 2: Code Quality Review
Review the same changes for hacky patterns:
1. Redundant state: state that duplicates existing state, cached values that could be derived, observers/effects that could be direct calls
2. Parameter sprawl: adding new parameters to a function instead of generalizing or restructuring existing ones
3. Copy-paste with slight variation: near-duplicate code blocks that should be unified with a shared abstraction
4. Leaky abstractions: exposing internal details that should be encapsulated, or breaking existing abstraction boundaries
5. Stringly-typed code: using raw strings where constants, enums (string unions), or branded types already exist in the codebase
Note: This is a greenfield app, so be aggressive in optimizing quality.
### Agent 3: Efficiency Review
Review the same changes for efficiency:
1. Unnecessary work: redundant computations, repeated file reads, duplicate network/API calls, N+1 patterns
2. Missed concurrency: independent operations run sequentially when they could run in parallel
3. Hot-path bloat: new blocking work added to startup or per-request/per-render hot paths
4. Unnecessary existence checks: pre-checking file/resource existence before operating (TOCTOU anti-pattern) — operate directly and handle the error
5. Memory: unbounded data structures, missing cleanup, event listener leaks
6. Overly broad operations: reading entire files when only a portion is needed, loading all items when filtering for one
## Phase 3: Fix Issues
Wait for all three agents to complete. Aggregate their findings and fix each issue directly. If a finding is a false positive or not worth addressing, note it and move on — do not argue with the finding, just skip it.
When done, briefly summarize what was fixed (or confirm the code was already clean).

View file

@ -0,0 +1,5 @@
{
"mode": "agent",
"provider": "anthropic",
"model": "claude-opus-4-7"
}

View file

@ -0,0 +1,32 @@
All clean. Here's a summary of what was fixed based on the three reviews:
## Summary of cleanup applied
**`lib/crates/fabro-agent/src/compaction.rs`**
- Removed the no-longer-needed `pub fn estimate_token_count` (no external callers); tests updated to call `estimate_active_context_usage` directly.
- Replaced the hand-rolled `ContextEstimateMethod::as_str()` with `#[derive(strum::IntoStaticStr)] #[strum(serialize_all = "snake_case")]` per repo convention in CLAUDE.md.
- Split `estimate_local_token_count(system_prompt, turns)` into two single-purpose helpers (`estimate_turns_local_tokens`, `estimate_system_prompt_local_tokens`) to remove the `""` sentinel call from the baseline+delta path.
- Replaced the literal `4` in `summary_token_estimate = summary_content.len() / 4` with `APPROX_CHARS_PER_TOKEN`.
- Made `check_context_usage` return `Option<ContextEstimate>` and `compact_context` accept the pre-computed estimate, eliminating a duplicate full estimate scan per actual compaction. Both fns are now `pub(crate)` since they have no external callers. Dropped the redundant `system_prompt` parameter from `compact_context`.
- Tightened visibility on `ContextEstimate` and `ContextEstimateMethod` to `pub(crate)`.
**`lib/crates/fabro-agent/src/history.rs`**
- Moved the free `invalidate_assistant_usage` function into `impl History` as a private `invalidate_preserved_usage` method.
- Added a doc comment on `History::compact` explaining the usage-invalidation invariant.
**`lib/crates/fabro-agent/src/session.rs`**
- Rewired `compact_if_needed` to consume the `Option<ContextEstimate>` from `check_context_usage` and pass it through.
- Renamed misleading test helper `response_with_total_usage` → `response_with_input_tokens` (it only set `input_tokens`, never the other usage fields).
**`lib/crates/fabro-agent/Cargo.toml`**
- Added `strum.workspace = true` (workspace dep) to support the derive.
## Findings skipped
- **Test builder for `Message::Assistant`** (Quality #3): broader scope than just the diff and only marginal cleanup.
- **Counting-writer for JSON length** (Efficiency #3): the hot-path concern is real but speculative without profiling; the prevailing pattern in fabro-agent uses `Value::to_string()`. Out of scope.
- **Extract `"context_window"` / `"estimate_method"` JSON-key constants** (Quality #5): pre-existing pattern beyond this diff.
- **Reshape `latest_assistant_usage_baseline` to return `(tokens, &[Message])`** (Quality #4): cosmetic; current `+ 1` indexing is local and clear.
- **`APPROX_CHARS_PER_TOKEN` in `history.rs` `extract_recent_user_messages`** (Reuse #1): out of scope — that code is not part of the change.
Verification: `cargo nextest run -p fabro-agent` → 447/447 pass; `cargo +nightly-2026-04-14 clippy --workspace --all-targets -- -D warnings` → clean; `cargo +nightly-2026-04-14 fmt --check --all` → clean.