Merge pull request #630 from fabro-sh/fix/compaction-reasoning-token-budget

fix(agent): budget compaction summaries for reasoning models
This commit is contained in:
Bryan Helmkamp 2026-07-24 22:26:12 -04:00 • committed by GitHub
commit c914fbbbe0
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
10 changed files with 271 additions and 9 deletions

View file

@ -279,6 +279,7 @@ cache_input_cost_per_mtok = 0.60
| `tools` | boolean | `false` | Whether the model supports tool calls. |
| `vision` | boolean | `false` | Whether the model accepts image inputs. |
| `reasoning` | boolean | `false` | Whether the model has reasoning behavior. |
| `reasoning_by_default` | boolean | effort-capable models: `true`; other models: `false` | Whether requests reason when no `reasoning_effort` is supplied. Set this explicitly for always-reasoning routes that do not expose an effort control, or for effort-capable routes whose provider defaults reasoning off. |
| `reasoning_effort` | `"levels"` \| `"always_adaptive"` \| `"none"` | `"none"` | Whether the model endpoint supports a native reasoning-effort parameter. `levels` accepts discrete effort levels; `always_adaptive` accepts effort levels with natively always-on adaptive thinking; `none` has no native effort parameter. |
| `prompt_cache` | boolean | `false` | Whether prompt cache pricing/usage applies. |
| `sampling_params` | boolean | `true` | Whether the model accepts classic sampling parameters (`temperature`, `top_p`). |

View file

@ -52,6 +52,15 @@ pub trait AgentProfile: Send + Sync {
self.catalog_model().and_then(Model::max_output)
}
fn reasons_by_default(&self) -> bool {
let Some(catalog) = self.catalog() else {
return false;
};
catalog
.model_settings_on_provider(&self.provider_id(), self.model())
.is_some_and(|settings| settings.reasoning_by_default)
}
fn register_subagent_tools(
&mut self,
supervisor: SubAgentSupervisor,

View file

@ -993,6 +993,7 @@ mod tests {
tools: Some(true),
vision: Some(false),
reasoning: Some(false),
reasoning_by_default: None,
reasoning_effort: None,
prompt_cache: None,
cache_control_breakpoints: None,
@ -1079,6 +1080,7 @@ mod tests {
tools: Some(true),
vision: Some(false),
reasoning: Some(false),
reasoning_by_default: None,
reasoning_effort: None,
prompt_cache: None,
cache_control_breakpoints: None,

View file

@ -13,6 +13,15 @@ use crate::types::{AgentEvent, Message};
const APPROX_CHARS_PER_TOKEN: usize = 4;
/// Maximum output budget for the visible summary text itself.
const SUMMARY_MAX_TOKENS: i64 = 4096;
/// Extra output budget for models that reason on every request. `max_tokens`
/// bounds reasoning *plus* visible output, so a reasoning model handed only
/// `SUMMARY_MAX_TOKENS` can spend the whole budget thinking and return a
/// successful response with empty content — a silently empty summary.
const REASONING_HEADROOM_TOKENS: i64 = 16_384;
#[derive(Debug, Clone, Copy, PartialEq, Eq, strum::IntoStaticStr)]
#[strum(serialize_all = "snake_case")]
pub(crate) enum ContextEstimateMethod {
@ -106,6 +115,12 @@ pub(crate) async fn compact_context(
)
};
let max_tokens = summary_max_tokens(
provider_profile.reasons_by_default(),
provider_profile.max_output_tokens(),
);
let visible_max_tokens = SUMMARY_MAX_TOKENS.min(max_tokens);
let summarization_prompt = format!(
"You are creating a handoff document for a different coding assistant that will take over \
this task. That assistant will only see your summary and the most recent messages — nothing else \
@ -117,6 +132,7 @@ Write a summary using EXACTLY these sections:\n\n\
## Failed Approaches\nWhat was tried and didn't work, and why.\n\n\
## Open Issues\nBugs, edge cases, or TODOs that remain.\n\n\
## Next Steps\nWhat should happen next to make progress.\n\n\
Keep the entire response under {visible_max_tokens} tokens.\n\n\
Be thorough and specific — the assistant taking over has no prior context. Include file paths, \
function names, error messages, and exact values. Omit pleasantries and conversational filler.\
{file_ops_section}"
@ -136,7 +152,7 @@ function names, error messages, and exact values. Omit pleasantries and conversa
response_format: None,
temperature: None,
top_p: None,
max_tokens: Some(4096),
max_tokens: Some(max_tokens),
stop_sequences: None,
reasoning_effort: None,
speed: None,
@ -162,9 +178,10 @@ function names, error messages, and exact values. Omit pleasantries and conversa
.into());
}
let (summary_text, summary_truncated) = truncate_summary_text(summary_text);
debug!(
summary_len = summary_text.len(),
"Compaction summary generated"
summary_truncated, max_tokens, "Compaction summary generated"
);
let summary_content = format!(
"A different assistant began this task and produced the following summary. \
@ -184,6 +201,43 @@ Build on their progress — do not repeat completed steps.\n\n{summary_text}"
Ok(())
}
/// Combined reasoning and visible-output budget for the summarization request.
///
/// Compaction runs against the session's own model, so a reasoning session
/// summarizes with reasoning enabled and the budget has to cover the thinking
/// as well as the summary. Provider routes that reason by default get headroom
/// on top of the summary allowance. Every known model budget is capped at its
/// declared `max_output`.
fn summary_max_tokens(reasoning_by_default: bool, max_output: Option<i64>) -> i64 {
let budget = if reasoning_by_default {
SUMMARY_MAX_TOKENS + REASONING_HEADROOM_TOKENS
} else {
SUMMARY_MAX_TOKENS
};
max_output.map_or(budget, |limit| budget.min(limit))
}
/// Bound retained summary text with the same local bytes-per-token heuristic
/// used for context estimates. Provider APIs expose only one combined ceiling
/// for reasoning and visible output, so the larger request budget cannot
/// enforce this limit itself.
fn truncate_summary_text(summary: &str) -> (&str, bool) {
let max_bytes = summary_max_approx_bytes();
if summary.len() <= max_bytes {
return (summary, false);
}
let end = summary.floor_char_boundary(max_bytes);
(&summary[..end], true)
}
fn summary_max_approx_bytes() -> usize {
usize::try_from(SUMMARY_MAX_TOKENS)
.unwrap_or(usize::MAX)
.saturating_mul(APPROX_CHARS_PER_TOKEN)
}
pub(crate) fn estimate_active_context_usage(
system_prompt: &str,
history: &History,
@ -314,6 +368,7 @@ mod tests {
use std::time::SystemTime;
use fabro_llm::types::{TokenCounts, ToolCall, ToolResult};
use fabro_model::{Catalog, Model, ProviderId};
use super::*;
use crate::event::Emitter;
@ -322,6 +377,88 @@ mod tests {
use crate::tool_registry::ToolRegistry;
use crate::types::Message;
fn catalog_model(provider: &ProviderId, id: &str) -> &'static Model {
Catalog::builtin()
.get_on_provider(provider, id)
.unwrap_or_else(|| panic!("{provider}/{id} missing from builtin catalog"))
}
fn builtin_summary_max_tokens(provider: &ProviderId, id: &str) -> i64 {
let catalog = Catalog::builtin();
let model = catalog_model(provider, id);
let settings = catalog
.settings_for(model)
.unwrap_or_else(|| panic!("{provider}/{id} missing catalog settings"));
summary_max_tokens(settings.reasoning_by_default, model.max_output())
}
#[test]
fn summary_budget_without_catalog_model_is_summary_allowance() {
assert_eq!(summary_max_tokens(false, None), SUMMARY_MAX_TOKENS);
// The default agent test profile has no catalog behind it.
let profile = TestProfile::new();
assert_eq!(
summary_max_tokens(profile.reasons_by_default(), profile.max_output_tokens()),
SUMMARY_MAX_TOKENS
);
}
#[test]
fn summary_budget_for_non_reasoning_model_is_summary_allowance() {
// claude-haiku-4-5: reasoning = false.
assert_eq!(
builtin_summary_max_tokens(&ProviderId::anthropic(), "claude-haiku-4-5"),
SUMMARY_MAX_TOKENS
);
}
#[test]
fn summary_budget_for_model_without_effort_feature_is_summary_allowance() {
// claude-sonnet-4-5 reasons only when a request asks for it, and
// compaction never sends a reasoning effort.
let model = catalog_model(&ProviderId::anthropic(), "claude-sonnet-4-5");
assert!(model.supports_reasoning());
assert!(!model.supports_reasoning_effort());
assert_eq!(
builtin_summary_max_tokens(&ProviderId::anthropic(), "claude-sonnet-4-5"),
SUMMARY_MAX_TOKENS
);
}
#[test]
fn summary_budget_for_always_adaptive_model_adds_reasoning_headroom() {
assert_eq!(
builtin_summary_max_tokens(&ProviderId::anthropic(), "claude-fable-5"),
SUMMARY_MAX_TOKENS + REASONING_HEADROOM_TOKENS
);
}
#[test]
fn summary_budget_for_effort_levels_model_adds_reasoning_headroom() {
assert_eq!(
builtin_summary_max_tokens(&ProviderId::anthropic(), "claude-opus-5"),
SUMMARY_MAX_TOKENS + REASONING_HEADROOM_TOKENS
);
}
#[test]
fn summary_budget_for_always_reasoning_route_without_effort_adds_headroom() {
let kimi = ProviderId::new("kimi");
let model = catalog_model(&kimi, "kimi-k2.5");
assert!(model.supports_reasoning());
assert!(!model.supports_reasoning_effort());
assert_eq!(
builtin_summary_max_tokens(&kimi, "kimi-k2.5"),
SUMMARY_MAX_TOKENS + REASONING_HEADROOM_TOKENS
);
}
#[test]
fn summary_budget_never_exceeds_model_max_output() {
assert_eq!(summary_max_tokens(true, Some(8_192)), 8_192);
assert_eq!(summary_max_tokens(false, Some(2_048)), 2_048);
}
#[test]
fn render_turns_produces_labeled_text() {
let turns = vec![
@ -712,4 +849,29 @@ mod tests {
"CompactionCompleted should be emitted on success"
);
}
#[tokio::test]
async fn compaction_bounds_retained_summary_to_visible_budget() {
let max_bytes = summary_max_approx_bytes();
let overlong = format!("{}END", "€".repeat(max_bytes / 3 + 1));
let CompactionTestResult {
result, history, ..
} = compact_with_summary(&overlong).await;
result.expect("an overlong summary should be compacted after truncation");
let summary_turn = history
.turns()
.iter()
.find_map(|turn| match turn {
Message::System { content, .. } => Some(content),
_ => None,
})
.expect("compacted history should contain a summary turn");
let (_, retained_summary) = summary_turn
.split_once("\n\n")
.expect("summary turn should separate its header from the generated text");
assert!(retained_summary.len() <= max_bytes);
assert!(!retained_summary.contains("END"));
}
}

View file

@ -420,6 +420,7 @@ fn model_features_to_catalog(features: &LlmModelFeatures) -> model_catalog::Sett
tools: features.tools,
vision: features.vision,
reasoning: features.reasoning,
reasoning_by_default: features.reasoning_by_default,
reasoning_effort: features.reasoning_effort,
prompt_cache: features.prompt_cache,
cache_control_breakpoints: features.cache_control_breakpoints,

View file

@ -244,6 +244,8 @@ pub struct ModelFeatures {
#[serde(default, skip_serializing_if = "Option::is_none")]
pub reasoning: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub reasoning_by_default: Option<bool>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub reasoning_effort: Option<ReasoningEffortFeature>,
#[serde(default, skip_serializing_if = "Option::is_none")]
pub prompt_cache: Option<bool>,
@ -647,6 +649,7 @@ provider = "bedrock"
tools = true
vision = true
reasoning = true
reasoning_by_default = false
reasoning_effort = "levels"
prompt_cache = false
"#;
@ -663,6 +666,7 @@ prompt_cache = false
features.reasoning_effort,
Some(fabro_model::ReasoningEffortFeature::Levels)
);
assert_eq!(features.reasoning_by_default, Some(false));
assert_eq!(features.prompt_cache, Some(false));
}

View file

@ -326,6 +326,7 @@ cache_input_cost_per_mtok = 0.60
| `tools` | boolean | `false` | Whether the model supports tool calls. |
| `vision` | boolean | `false` | Whether the model accepts image inputs. |
| `reasoning` | boolean | `false` | Whether the model has reasoning behavior. |
| `reasoning_by_default` | boolean | effort-capable models: `true`; other models: `false` | Whether requests reason when no `reasoning_effort` is supplied. Set this explicitly for always-reasoning routes that do not expose an effort control, or for effort-capable routes whose provider defaults reasoning off. |
| `reasoning_effort` | `"levels"` \| `"always_adaptive"` \| `"none"` | `"none"` | Whether the model endpoint supports a native reasoning-effort parameter. `levels` accepts discrete effort levels; `always_adaptive` accepts effort levels with natively always-on adaptive thinking; `none` has no native effort parameter. |
| `prompt_cache` | boolean | `false` | Whether prompt cache pricing/usage applies. |
| `sampling_params` | boolean | `true` | Whether the model accepts classic sampling parameters (`temperature`, `top_p`). |

View file

@ -146,6 +146,11 @@ pub struct SettingsModelFeatures {
pub vision: Option<bool>,
#[serde(default)]
pub reasoning: Option<bool>,
/// Whether requests reason when no effort control is supplied. When
/// omitted, effort-capable models default to `true` and other models to
/// `false`.
#[serde(default)]
pub reasoning_by_default: Option<bool>,
#[serde(default)]
pub reasoning_effort: Option<ReasoningEffortFeature>,
#[serde(default)]
@ -443,17 +448,20 @@ pub struct CatalogModelControls {
#[derive(Debug, Clone, PartialEq)]
pub struct CatalogModelSettings {
pub api_id: String,
pub api_id: String,
/// Wire dialect for this model's route (the provider codec unless the
/// model row overrides it).
pub codec: CodecKind,
pub codec: CodecKind,
/// Billing family for this model (the provider policy unless the model
/// row overrides it).
pub billing_policy: BillingPolicy,
pub agent_profile: AgentProfileKind,
pub controls: CatalogModelControls,
pub speed_costs: HashMap<Speed, ModelCosts>,
probe: bool,
pub billing_policy: BillingPolicy,
pub agent_profile: AgentProfileKind,
/// Whether the provider route reasons when a request omits an effort
/// control.
pub reasoning_by_default: bool,
pub controls: CatalogModelControls,
pub speed_costs: HashMap<Speed, ModelCosts>,
probe: bool,
}
#[derive(Debug, thiserror::Error)]
@ -578,6 +586,8 @@ pub enum CatalogBuildError {
ReasoningEffortControlsWithoutReasoning { model: String },
#[error("model '{model}' declares reasoning_effort feature but features.reasoning is false")]
ReasoningEffortWithoutReasoning { model: String },
#[error("model '{model}' sets reasoning_by_default but features.reasoning is false")]
DefaultReasoningWithoutReasoning { model: String },
#[error(
"model '{model}' declares cache_control_breakpoints but features.prompt_cache is false"
)]
@ -1983,6 +1993,9 @@ fn merge_model_features_settings(
tools: higher.tools.or(fallback.tools),
vision: higher.vision.or(fallback.vision),
reasoning: higher.reasoning.or(fallback.reasoning),
reasoning_by_default: higher
.reasoning_by_default
.or(fallback.reasoning_by_default),
reasoning_effort: higher.reasoning_effort.or(fallback.reasoning_effort),
prompt_cache: higher.prompt_cache.or(fallback.prompt_cache),
cache_control_breakpoints: higher
@ -2214,6 +2227,14 @@ fn build_model(
field: "features",
})?;
let model_features = build_model_features(model_id, features)?;
let reasoning_by_default = features
.reasoning_by_default
.unwrap_or_else(|| model_features.supports_reasoning_effort());
if reasoning_by_default && !model_features.reasoning {
return Err(CatalogBuildError::DefaultReasoningWithoutReasoning {
model: model_id.to_string(),
});
}
let controls = build_model_controls(model_id, &model_features, settings)?;
let costs = build_model_costs(settings.costs.as_ref());
let speed_costs = build_speed_costs(model_id, settings.costs.as_ref(), &controls)?;
@ -2255,6 +2276,7 @@ fn build_model(
codec: resolve_model_codec(model_id, provider, settings.codec)?,
billing_policy: settings.billing_policy.unwrap_or(provider.billing_policy),
agent_profile: settings.agent_profile.unwrap_or(provider.agent_profile),
reasoning_by_default,
controls,
speed_costs,
probe: settings.probe.unwrap_or_default(),
@ -2832,6 +2854,12 @@ enabled = true
.get_on_provider(&bedrock, "claude-fable-5")
.expect("fable row should be present");
assert!(!fable.features.sampling_params);
assert!(
catalog
.settings_for(fable)
.expect("fable settings should be present")
.reasoning_by_default
);
assert_eq!(
catalog
.model_settings_on_provider(&bedrock, "claude-fable-5")
@ -5915,6 +5943,7 @@ context_window = 1000
tools = true
vision = false
reasoning = true
reasoning_by_default = false
reasoning_effort = "levels"
prompt_cache = true
@ -5930,6 +5959,12 @@ reasoning_effort = ["low", "medium"]
crate::ReasoningEffortFeature::Levels
);
assert!(model.features.prompt_cache);
assert!(
!catalog
.model_settings("model")
.unwrap()
.reasoning_by_default
);
assert_eq!(
catalog
.model_settings("model")
@ -5974,6 +6009,12 @@ prompt_cache = true
crate::ReasoningEffortFeature::AlwaysAdaptive
);
assert!(model.supports_reasoning_effort());
assert!(
catalog
.model_settings("model")
.unwrap()
.reasoning_by_default
);
// Always-adaptive models get the full default effort controls, same as
// Levels.
assert_eq!(
@ -6008,6 +6049,7 @@ context_window = 1000
tools = true
vision = false
reasoning = true
reasoning_by_default = true
reasoning_effort = "none"
[models.model.controls]
@ -6021,6 +6063,12 @@ reasoning_effort = ["low"]
model.features.reasoning_effort,
crate::ReasoningEffortFeature::None
);
assert!(
catalog
.model_settings("model")
.unwrap()
.reasoning_by_default
);
assert_eq!(
catalog
.model_settings("model")
@ -6098,6 +6146,38 @@ reasoning_effort = "levels"
));
}
#[test]
fn catalog_from_settings_rejects_default_reasoning_without_reasoning() {
let settings = minimal_settings(
r#"
[providers.test]
display_name = "Test"
adapter = "openai"
agent_profile = "openai"
[models.model]
provider = "test"
display_name = "Model"
family = "test"
[models.model.limits]
context_window = 1000
[models.model.features]
tools = true
vision = false
reasoning = false
reasoning_by_default = true
"#,
);
assert!(matches!(
Catalog::from_settings(&settings).unwrap_err(),
CatalogBuildError::DefaultReasoningWithoutReasoning { model }
if model == "model"
));
}
#[test]
fn catalog_from_settings_rejects_cache_control_breakpoints_without_prompt_cache() {
let settings = minimal_settings(

View file

@ -369,6 +369,7 @@ max_output = 128000
tools = true
vision = true
reasoning = true
reasoning_by_default = true
prompt_cache = true
sampling_params = false

View file

@ -24,6 +24,7 @@ max_output = 32768
tools = true
vision = true
reasoning = true
reasoning_by_default = true
prompt_cache = true
sampling_params = false