From 61956673c3009a9ae4df0999dcc3a35e9bd684ff Mon Sep 17 00:00:00 2001 From: Release Repro Date: Sat, 25 Jul 2026 09:51:14 -0400 Subject: [PATCH] fix(workflow): drop token accounting from agent-facing preamble The stage-summary preamble rendered per-stage token usage for every completed LLM stage: "Model: kimi-k3, 92.6k tokens in / 41.1k out" at compact fidelity and "Tokens: N in / N out" at summary:high. Agents read that as their own remaining budget. In run 01KYCM3EG4KMCVRDYNV93PZWBV an implementation stage stopped after 2 of 9 units, reasoning "We have around 100k tokens, but time constraints are an issue" and recording the rest as halted "within the available execution window". The 92.6k it saw was the preceding plan stage's billing telemetry, the only token quantity anywhere in its context. It had used 11% of a 1,050,000-token window and 0.8% of a 24h stage timeout, and no harness limit was near. These counts have no task value to the agent: they describe a different model's usage on an earlier stage, they are stale by one stage, and nothing in the preamble distinguishes them from a budget. Keep the model id and files touched, which carry provenance the agent can act on. Both tests that asserted the counts now assert their absence, so the regression is caught rather than re-snapshotted. Co-Authored-By: Claude Opus 5 (1M context) --- .../src/handler/llm/preamble.rs | 49 ++++--------------- 1 file changed, 9 insertions(+), 40 deletions(-) diff --git a/lib/components/fabro-workflow/src/handler/llm/preamble.rs b/lib/components/fabro-workflow/src/handler/llm/preamble.rs index fbceff0d7..77cd9a0b4 100644 --- a/lib/components/fabro-workflow/src/handler/llm/preamble.rs +++ b/lib/components/fabro-workflow/src/handler/llm/preamble.rs @@ -115,16 +115,6 @@ fn format_value(val: &serde_json::Value) -> String { } } -fn format_token_count(tokens: i64) -> String { - if tokens >= 1_000_000 { - format!("{:.1}m", tokens as f64 / 1_000_000.0) - } else if tokens >= 1000 { - format!("{:.1}k", tokens as f64 / 1000.0) - } else { - tokens.to_string() - } -} - fn tail_lines(text: &str, max_lines: usize, indent: &str) -> String { use std::fmt::Write; @@ -197,14 +187,7 @@ fn render_compact_stage_details( h if is_llm_handler_type(h) => { let mut lines = Vec::new(); if let Some(usage) = &outcome.usage { - let input = format_token_count(usage.tokens().input_tokens); - let output = format_token_count(usage.tokens().billable_output_tokens()); - lines.push(format!( - " - Model: {}, {} tokens in / {} out", - usage.model_id(), - input, - output - )); + lines.push(format!(" - Model: {}", usage.model_id())); } if !outcome.files_touched.is_empty() { lines.push(format!(" - Files: {}", outcome.files_touched.join(", "))); @@ -265,11 +248,6 @@ fn render_summary_high_stage_section( h if is_llm_handler_type(h) => { if let Some(usage) = &outcome.usage { lines.push(format!("- Model: {}", usage.model_id())); - lines.push(format!( - "- Tokens: {} in / {} out", - format_token_count(usage.tokens().input_tokens), - format_token_count(usage.tokens().billable_output_tokens()) - )); } if !outcome.files_touched.is_empty() { lines.push(format!( @@ -928,8 +906,9 @@ mod tests { "should show model name" ); assert!( - preamble.contains("1.2k tokens in"), - "should show token count" + !preamble.contains("tokens"), + "token accounting must stay out of the agent-facing preamble; agents \ + read it as a budget signal, got:\n{preamble}" ); assert!( preamble.contains("src/lib.rs, src/main.rs"), @@ -1566,7 +1545,11 @@ mod tests { preamble.contains("Model: claude-sonnet-4-20250514"), "should show model" ); - assert!(preamble.contains("1.5k in"), "should show formatted tokens"); + assert!( + !preamble.contains("tokens"), + "token accounting must stay out of the agent-facing preamble; agents \ + read it as a budget signal, got:\n{preamble}" + ); assert!( preamble.contains("Files touched: src/lib.rs"), "should show files" @@ -1641,20 +1624,6 @@ mod tests { ); } - // --- format_token_count --- - - #[test] - fn format_token_count_formatting() { - assert_eq!(format_token_count(500), "500"); - assert_eq!(format_token_count(999), "999"); - assert_eq!(format_token_count(1000), "1.0k"); - assert_eq!(format_token_count(1234), "1.2k"); - assert_eq!(format_token_count(1500), "1.5k"); - assert_eq!(format_token_count(10000), "10.0k"); - assert_eq!(format_token_count(1_000_000), "1.0m"); - assert_eq!(format_token_count(3_456_789), "3.5m"); - } - // --- is_context_key_excluded --- #[test]