From f94955ede5e7852df35aa31ec44a9d3c9f2e688b Mon Sep 17 00:00:00 2001 From: Bryan Helmkamp Date: Fri, 24 Jul 2026 19:11:25 -0400 Subject: [PATCH 1/3] fix(agent): budget compaction summaries for reasoning models MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Compaction summarizes the conversation with the session's own model, but hard-coded `max_tokens: Some(4096)` and sent no `reasoning_effort`. On a reasoning model that ceiling covers thinking *and* visible output, so a long conversation can exhaust it on reasoning alone and return a successful response with empty content — silently replacing the compacted history with an empty summary. The Anthropic codec's existing clamp does not cover this path: it only runs when the request carries a `reasoning_effort` and the model has no native effort parameter. Compaction sends `reasoning_effort: None`, so encoding falls through to the branch that injects `{"type": "adaptive"}` for `levels` models with no clamp at all, and the openai_compatible and openai_responses codecs pass `max_tokens` straight through. Resolve the budget from the catalog instead. Models whose endpoint reasons without being asked (`always_adaptive` natively, `levels` via default adaptive thinking or the provider's default effort) get 16K of reasoning headroom above the 4096-token summary allowance, capped at the model's own `max_output`. Models with no reasoning-effort feature never reason on this path and keep the existing 4096. Co-Authored-By: Claude Opus 5 (1M context) --- lib/components/fabro-agent/src/compaction.rs | 137 ++++++++++++++++++- 1 file changed, 135 insertions(+), 2 deletions(-) diff --git a/lib/components/fabro-agent/src/compaction.rs b/lib/components/fabro-agent/src/compaction.rs index c4395a2e3..9031ec2ca 100644 --- a/lib/components/fabro-agent/src/compaction.rs +++ b/lib/components/fabro-agent/src/compaction.rs @@ -2,6 +2,7 @@ use std::fmt::Write; use fabro_llm::client::Client; use fabro_llm::types::{Message as LlmMessage, Request}; +use fabro_model::Model; use tracing::debug; use crate::agent_profile::AgentProfile; @@ -13,6 +14,15 @@ use crate::types::{AgentEvent, Message}; const APPROX_CHARS_PER_TOKEN: usize = 4; +/// Output budget for the visible summary text itself. +const SUMMARY_MAX_TOKENS: i64 = 4096; + +/// Extra output budget for models that reason on every request. `max_tokens` +/// bounds reasoning *plus* visible output, so a reasoning model handed only +/// `SUMMARY_MAX_TOKENS` can spend the whole budget thinking and return a +/// successful response with empty content — a silently empty summary. +const REASONING_HEADROOM_TOKENS: i64 = 16_384; + #[derive(Debug, Clone, Copy, PartialEq, Eq, strum::IntoStaticStr)] #[strum(serialize_all = "snake_case")] pub(crate) enum ContextEstimateMethod { @@ -122,6 +132,8 @@ function names, error messages, and exact values. Omit pleasantries and conversa {file_ops_section}" ); + let max_tokens = summary_max_tokens(provider_profile.catalog_model()); + let summary_request = Request { model: provider_profile.model().to_string(), messages: vec![ @@ -136,7 +148,7 @@ function names, error messages, and exact values. Omit pleasantries and conversa response_format: None, temperature: None, top_p: None, - max_tokens: Some(4096), + max_tokens: Some(max_tokens), stop_sequences: None, reasoning_effort: None, speed: None, @@ -152,7 +164,7 @@ function names, error messages, and exact values. Omit pleasantries and conversa let summary_text = response.text(); debug!( summary_len = summary_text.len(), - "Compaction summary generated" + max_tokens, "Compaction summary generated" ); let summary_content = format!( "A different assistant began this task and produced the following summary. \ @@ -172,6 +184,28 @@ Build on their progress — do not repeat completed steps.\n\n{summary_text}" Ok(()) } +/// Output budget for the summarization request. +/// +/// Compaction runs against the session's own model, so a reasoning session +/// summarizes with reasoning enabled and the budget has to cover the thinking +/// as well as the summary. Models that reason unconditionally get headroom on +/// top of the summary allowance, capped at the model's own `max_output`. +/// +/// A model only reasons here when its endpoint reasons without being asked: +/// `always_adaptive` models think natively, and `levels` models get thinking +/// enabled by default (adaptive thinking injected by the Anthropic codec, the +/// provider's default effort on OpenAI-style routes). Compaction never sends a +/// `reasoning_effort`, so models without an effort feature stay non-reasoning +/// on this path and keep the plain summary budget. +fn summary_max_tokens(model: Option<&Model>) -> i64 { + let Some(model) = model.filter(|m| m.supports_reasoning_effort()) else { + return SUMMARY_MAX_TOKENS; + }; + + let budget = SUMMARY_MAX_TOKENS.saturating_add(REASONING_HEADROOM_TOKENS); + model.max_output().map_or(budget, |limit| budget.min(limit)) +} + pub(crate) fn estimate_active_context_usage( system_prompt: &str, history: &History, @@ -301,6 +335,10 @@ mod tests { use std::time::SystemTime; use fabro_llm::types::{TokenCounts, ToolCall, ToolResult}; + use fabro_model::{ + Catalog, ModelControls, ModelCosts, ModelFeatures, ModelId, ModelLimits, ProviderId, + ReasoningEffortFeature, + }; use super::*; use crate::event::Emitter; @@ -309,6 +347,101 @@ mod tests { use crate::tool_registry::ToolRegistry; use crate::types::Message; + fn anthropic_model(id: &str) -> &'static Model { + Catalog::builtin() + .get_on_provider(&ProviderId::anthropic(), id) + .unwrap_or_else(|| panic!("{id} missing from builtin catalog")) + } + + /// A model that reasons unconditionally but caps output below the budget + /// compaction would otherwise ask for. + fn small_output_reasoning_model(max_output: i64) -> Model { + Model { + id: ModelId::new("small-output-reasoner"), + provider: ProviderId::anthropic(), + family: "test".into(), + display_name: "Small Output Reasoner".into(), + limits: ModelLimits { + context_window: 200_000, + max_output: Some(max_output), + }, + training: None, + knowledge_cutoff: None, + features: ModelFeatures { + tools: true, + vision: false, + reasoning: true, + reasoning_effort: ReasoningEffortFeature::AlwaysAdaptive, + prompt_cache: false, + cache_control_breakpoints: false, + sampling_params: false, + }, + controls: ModelControls::default(), + costs: ModelCosts { + input_cost_per_mtok: None, + output_cost_per_mtok: None, + cache_input_cost_per_mtok: None, + }, + estimated_output_tps: None, + aliases: vec![], + default: false, + small_default: false, + configured: false, + } + } + + #[test] + fn summary_budget_without_catalog_model_is_summary_allowance() { + assert_eq!(summary_max_tokens(None), 4096); + // The default agent test profile has no catalog behind it. + assert_eq!(summary_max_tokens(TestProfile::new().catalog_model()), 4096); + } + + #[test] + fn summary_budget_for_non_reasoning_model_is_summary_allowance() { + // claude-haiku-4-5: reasoning = false. + assert_eq!( + summary_max_tokens(Some(anthropic_model("claude-haiku-4-5"))), + 4096 + ); + } + + #[test] + fn summary_budget_for_model_without_effort_feature_is_summary_allowance() { + // claude-sonnet-4-5 reasons only when a request asks for it, and + // compaction never sends a reasoning effort. + let model = anthropic_model("claude-sonnet-4-5"); + assert!(model.supports_reasoning()); + assert!(!model.supports_reasoning_effort()); + assert_eq!(summary_max_tokens(Some(model)), 4096); + } + + #[test] + fn summary_budget_for_always_adaptive_model_adds_reasoning_headroom() { + let model = anthropic_model("claude-fable-5"); + assert_eq!( + model.features.reasoning_effort, + ReasoningEffortFeature::AlwaysAdaptive + ); + assert_eq!(summary_max_tokens(Some(model)), 4096 + 16_384); + } + + #[test] + fn summary_budget_for_effort_levels_model_adds_reasoning_headroom() { + let model = anthropic_model("claude-opus-5"); + assert_eq!( + model.features.reasoning_effort, + ReasoningEffortFeature::Levels + ); + assert_eq!(summary_max_tokens(Some(model)), 4096 + 16_384); + } + + #[test] + fn summary_budget_never_exceeds_model_max_output() { + let model = small_output_reasoning_model(8_192); + assert_eq!(summary_max_tokens(Some(&model)), 8_192); + } + #[test] fn render_turns_produces_labeled_text() { let turns = vec![ From 5d0617f5478120a8223a61efe963f5798e759244 Mon Sep 17 00:00:00 2001 From: Release Repro Date: Fri, 24 Jul 2026 21:58:23 -0400 Subject: [PATCH 2/3] fix(agent): harden compaction reasoning budgets Model default reasoning explicitly at the provider-route level so always-reasoning endpoints without effort controls receive summary headroom. Cap all summary requests at model output limits and bound retained visible summaries to the original allowance. Reuse builtin catalog fixtures and named budget constants in tests, and document the new model setting. --- docs/public/reference/user-configuration.mdx | 1 + .../fabro-agent/src/agent_profile.rs | 9 + lib/components/fabro-agent/src/cli.rs | 2 + lib/components/fabro-agent/src/compaction.rs | 180 ++++++++++-------- lib/foundation/fabro-config/src/builders.rs | 1 + lib/foundation/fabro-config/src/layers/llm.rs | 4 + .../src/commands/docs_options_reference.rs | 1 + lib/foundation/fabro-model/src/catalog.rs | 94 ++++++++- .../src/catalog/providers/bedrock.toml | 1 + .../src/catalog/providers/kimi.toml | 1 + 10 files changed, 211 insertions(+), 83 deletions(-) diff --git a/docs/public/reference/user-configuration.mdx b/docs/public/reference/user-configuration.mdx index dd6afd83f..cfc3e09cb 100644 --- a/docs/public/reference/user-configuration.mdx +++ b/docs/public/reference/user-configuration.mdx @@ -279,6 +279,7 @@ cache_input_cost_per_mtok = 0.60 | `tools` | boolean | `false` | Whether the model supports tool calls. | | `vision` | boolean | `false` | Whether the model accepts image inputs. | | `reasoning` | boolean | `false` | Whether the model has reasoning behavior. | +| `reasoning_by_default` | boolean | effort-capable models: `true`; other models: `false` | Whether requests reason when no `reasoning_effort` is supplied. Set this explicitly for always-reasoning routes that do not expose an effort control, or for effort-capable routes whose provider defaults reasoning off. | | `reasoning_effort` | `"levels"` \| `"always_adaptive"` \| `"none"` | `"none"` | Whether the model endpoint supports a native reasoning-effort parameter. `levels` accepts discrete effort levels; `always_adaptive` accepts effort levels with natively always-on adaptive thinking; `none` has no native effort parameter. | | `prompt_cache` | boolean | `false` | Whether prompt cache pricing/usage applies. | | `sampling_params` | boolean | `true` | Whether the model accepts classic sampling parameters (`temperature`, `top_p`). | diff --git a/lib/components/fabro-agent/src/agent_profile.rs b/lib/components/fabro-agent/src/agent_profile.rs index b46095ff1..15140a3e2 100644 --- a/lib/components/fabro-agent/src/agent_profile.rs +++ b/lib/components/fabro-agent/src/agent_profile.rs @@ -52,6 +52,15 @@ pub trait AgentProfile: Send + Sync { self.catalog_model().and_then(Model::max_output) } + fn reasons_by_default(&self) -> bool { + let Some(catalog) = self.catalog() else { + return false; + }; + catalog + .model_settings_on_provider(&self.provider_id(), self.model()) + .is_some_and(|settings| settings.reasoning_by_default) + } + fn register_subagent_tools( &mut self, supervisor: SubAgentSupervisor, diff --git a/lib/components/fabro-agent/src/cli.rs b/lib/components/fabro-agent/src/cli.rs index 12e033e03..61551af4b 100644 --- a/lib/components/fabro-agent/src/cli.rs +++ b/lib/components/fabro-agent/src/cli.rs @@ -993,6 +993,7 @@ mod tests { tools: Some(true), vision: Some(false), reasoning: Some(false), + reasoning_by_default: None, reasoning_effort: None, prompt_cache: None, cache_control_breakpoints: None, @@ -1079,6 +1080,7 @@ mod tests { tools: Some(true), vision: Some(false), reasoning: Some(false), + reasoning_by_default: None, reasoning_effort: None, prompt_cache: None, cache_control_breakpoints: None, diff --git a/lib/components/fabro-agent/src/compaction.rs b/lib/components/fabro-agent/src/compaction.rs index c2256d186..698d68275 100644 --- a/lib/components/fabro-agent/src/compaction.rs +++ b/lib/components/fabro-agent/src/compaction.rs @@ -2,7 +2,6 @@ use std::fmt::Write; use fabro_llm::client::Client; use fabro_llm::types::{Message as LlmMessage, Request}; -use fabro_model::Model; use tracing::debug; use crate::agent_profile::AgentProfile; @@ -127,12 +126,16 @@ Write a summary using EXACTLY these sections:\n\n\ ## Failed Approaches\nWhat was tried and didn't work, and why.\n\n\ ## Open Issues\nBugs, edge cases, or TODOs that remain.\n\n\ ## Next Steps\nWhat should happen next to make progress.\n\n\ +Keep the entire response under {SUMMARY_MAX_TOKENS} tokens.\n\n\ Be thorough and specific — the assistant taking over has no prior context. Include file paths, \ function names, error messages, and exact values. Omit pleasantries and conversational filler.\ {file_ops_section}" ); - let max_tokens = summary_max_tokens(provider_profile.catalog_model()); + let max_tokens = summary_max_tokens( + provider_profile.reasons_by_default(), + provider_profile.max_output_tokens(), + ); let summary_request = Request { model: provider_profile.model().to_string(), @@ -174,9 +177,10 @@ function names, error messages, and exact values. Omit pleasantries and conversa .into()); } + let (summary_text, summary_truncated) = truncate_summary_text(summary_text); debug!( summary_len = summary_text.len(), - max_tokens, "Compaction summary generated" + summary_truncated, max_tokens, "Compaction summary generated" ); let summary_content = format!( "A different assistant began this task and produced the following summary. \ @@ -196,26 +200,41 @@ Build on their progress — do not repeat completed steps.\n\n{summary_text}" Ok(()) } -/// Output budget for the summarization request. +/// Combined reasoning and visible-output budget for the summarization request. /// /// Compaction runs against the session's own model, so a reasoning session /// summarizes with reasoning enabled and the budget has to cover the thinking -/// as well as the summary. Models that reason unconditionally get headroom on -/// top of the summary allowance, capped at the model's own `max_output`. -/// -/// A model only reasons here when its endpoint reasons without being asked: -/// `always_adaptive` models think natively, and `levels` models get thinking -/// enabled by default (adaptive thinking injected by the Anthropic codec, the -/// provider's default effort on OpenAI-style routes). Compaction never sends a -/// `reasoning_effort`, so models without an effort feature stay non-reasoning -/// on this path and keep the plain summary budget. -fn summary_max_tokens(model: Option<&Model>) -> i64 { - let Some(model) = model.filter(|m| m.supports_reasoning_effort()) else { - return SUMMARY_MAX_TOKENS; +/// as well as the summary. Provider routes that reason by default get headroom +/// on top of the summary allowance. Every known model budget is capped at its +/// declared `max_output`. +fn summary_max_tokens(reasoning_by_default: bool, max_output: Option) -> i64 { + let budget = if reasoning_by_default { + SUMMARY_MAX_TOKENS + REASONING_HEADROOM_TOKENS + } else { + SUMMARY_MAX_TOKENS }; - let budget = SUMMARY_MAX_TOKENS.saturating_add(REASONING_HEADROOM_TOKENS); - model.max_output().map_or(budget, |limit| budget.min(limit)) + max_output.map_or(budget, |limit| budget.min(limit)) +} + +/// Bound retained summary text with the same local bytes-per-token heuristic +/// used for context estimates. Provider APIs expose only one combined ceiling +/// for reasoning and visible output, so the larger request budget cannot +/// enforce this limit itself. +fn truncate_summary_text(summary: &str) -> (&str, bool) { + let max_bytes = summary_max_approx_bytes(); + if summary.len() <= max_bytes { + return (summary, false); + } + + let end = summary.floor_char_boundary(max_bytes); + (&summary[..end], true) +} + +fn summary_max_approx_bytes() -> usize { + usize::try_from(SUMMARY_MAX_TOKENS) + .unwrap_or(usize::MAX) + .saturating_mul(APPROX_CHARS_PER_TOKEN) } pub(crate) fn estimate_active_context_usage( @@ -348,10 +367,7 @@ mod tests { use std::time::SystemTime; use fabro_llm::types::{TokenCounts, ToolCall, ToolResult}; - use fabro_model::{ - Catalog, ModelControls, ModelCosts, ModelFeatures, ModelId, ModelLimits, ProviderId, - ReasoningEffortFeature, - }; + use fabro_model::{Catalog, Model, ProviderId}; use super::*; use crate::event::Emitter; @@ -360,62 +376,38 @@ mod tests { use crate::tool_registry::ToolRegistry; use crate::types::Message; - fn anthropic_model(id: &str) -> &'static Model { + fn catalog_model(provider: &ProviderId, id: &str) -> &'static Model { Catalog::builtin() - .get_on_provider(&ProviderId::anthropic(), id) - .unwrap_or_else(|| panic!("{id} missing from builtin catalog")) + .get_on_provider(provider, id) + .unwrap_or_else(|| panic!("{provider}/{id} missing from builtin catalog")) } - /// A model that reasons unconditionally but caps output below the budget - /// compaction would otherwise ask for. - fn small_output_reasoning_model(max_output: i64) -> Model { - Model { - id: ModelId::new("small-output-reasoner"), - provider: ProviderId::anthropic(), - family: "test".into(), - display_name: "Small Output Reasoner".into(), - limits: ModelLimits { - context_window: 200_000, - max_output: Some(max_output), - }, - training: None, - knowledge_cutoff: None, - features: ModelFeatures { - tools: true, - vision: false, - reasoning: true, - reasoning_effort: ReasoningEffortFeature::AlwaysAdaptive, - prompt_cache: false, - cache_control_breakpoints: false, - sampling_params: false, - }, - controls: ModelControls::default(), - costs: ModelCosts { - input_cost_per_mtok: None, - output_cost_per_mtok: None, - cache_input_cost_per_mtok: None, - }, - estimated_output_tps: None, - aliases: vec![], - default: false, - small_default: false, - configured: false, - } + fn builtin_summary_max_tokens(provider: &ProviderId, id: &str) -> i64 { + let catalog = Catalog::builtin(); + let model = catalog_model(provider, id); + let settings = catalog + .settings_for(model) + .unwrap_or_else(|| panic!("{provider}/{id} missing catalog settings")); + summary_max_tokens(settings.reasoning_by_default, model.max_output()) } #[test] fn summary_budget_without_catalog_model_is_summary_allowance() { - assert_eq!(summary_max_tokens(None), 4096); + assert_eq!(summary_max_tokens(false, None), SUMMARY_MAX_TOKENS); // The default agent test profile has no catalog behind it. - assert_eq!(summary_max_tokens(TestProfile::new().catalog_model()), 4096); + let profile = TestProfile::new(); + assert_eq!( + summary_max_tokens(profile.reasons_by_default(), profile.max_output_tokens()), + SUMMARY_MAX_TOKENS + ); } #[test] fn summary_budget_for_non_reasoning_model_is_summary_allowance() { // claude-haiku-4-5: reasoning = false. assert_eq!( - summary_max_tokens(Some(anthropic_model("claude-haiku-4-5"))), - 4096 + builtin_summary_max_tokens(&ProviderId::anthropic(), "claude-haiku-4-5"), + SUMMARY_MAX_TOKENS ); } @@ -423,36 +415,47 @@ mod tests { fn summary_budget_for_model_without_effort_feature_is_summary_allowance() { // claude-sonnet-4-5 reasons only when a request asks for it, and // compaction never sends a reasoning effort. - let model = anthropic_model("claude-sonnet-4-5"); + let model = catalog_model(&ProviderId::anthropic(), "claude-sonnet-4-5"); assert!(model.supports_reasoning()); assert!(!model.supports_reasoning_effort()); - assert_eq!(summary_max_tokens(Some(model)), 4096); + assert_eq!( + builtin_summary_max_tokens(&ProviderId::anthropic(), "claude-sonnet-4-5"), + SUMMARY_MAX_TOKENS + ); } #[test] fn summary_budget_for_always_adaptive_model_adds_reasoning_headroom() { - let model = anthropic_model("claude-fable-5"); assert_eq!( - model.features.reasoning_effort, - ReasoningEffortFeature::AlwaysAdaptive + builtin_summary_max_tokens(&ProviderId::anthropic(), "claude-fable-5"), + SUMMARY_MAX_TOKENS + REASONING_HEADROOM_TOKENS ); - assert_eq!(summary_max_tokens(Some(model)), 4096 + 16_384); } #[test] fn summary_budget_for_effort_levels_model_adds_reasoning_headroom() { - let model = anthropic_model("claude-opus-5"); assert_eq!( - model.features.reasoning_effort, - ReasoningEffortFeature::Levels + builtin_summary_max_tokens(&ProviderId::anthropic(), "claude-opus-5"), + SUMMARY_MAX_TOKENS + REASONING_HEADROOM_TOKENS + ); + } + + #[test] + fn summary_budget_for_always_reasoning_route_without_effort_adds_headroom() { + let kimi = ProviderId::new("kimi"); + let model = catalog_model(&kimi, "kimi-k2.5"); + assert!(model.supports_reasoning()); + assert!(!model.supports_reasoning_effort()); + assert_eq!( + builtin_summary_max_tokens(&kimi, "kimi-k2.5"), + SUMMARY_MAX_TOKENS + REASONING_HEADROOM_TOKENS ); - assert_eq!(summary_max_tokens(Some(model)), 4096 + 16_384); } #[test] fn summary_budget_never_exceeds_model_max_output() { - let model = small_output_reasoning_model(8_192); - assert_eq!(summary_max_tokens(Some(&model)), 8_192); + assert_eq!(summary_max_tokens(true, Some(8_192)), 8_192); + assert_eq!(summary_max_tokens(false, Some(2_048)), 2_048); } #[test] @@ -845,4 +848,29 @@ mod tests { "CompactionCompleted should be emitted on success" ); } + + #[tokio::test] + async fn compaction_bounds_retained_summary_to_visible_budget() { + let max_bytes = summary_max_approx_bytes(); + let overlong = format!("{}END", "€".repeat(max_bytes / 3 + 1)); + let CompactionTestResult { + result, history, .. + } = compact_with_summary(&overlong).await; + + result.expect("an overlong summary should be compacted after truncation"); + + let summary_turn = history + .turns() + .iter() + .find_map(|turn| match turn { + Message::System { content, .. } => Some(content), + _ => None, + }) + .expect("compacted history should contain a summary turn"); + let (_, retained_summary) = summary_turn + .split_once("\n\n") + .expect("summary turn should separate its header from the generated text"); + assert!(retained_summary.len() <= max_bytes); + assert!(!retained_summary.contains("END")); + } } diff --git a/lib/foundation/fabro-config/src/builders.rs b/lib/foundation/fabro-config/src/builders.rs index a1e00907c..980faa3f3 100644 --- a/lib/foundation/fabro-config/src/builders.rs +++ b/lib/foundation/fabro-config/src/builders.rs @@ -420,6 +420,7 @@ fn model_features_to_catalog(features: &LlmModelFeatures) -> model_catalog::Sett tools: features.tools, vision: features.vision, reasoning: features.reasoning, + reasoning_by_default: features.reasoning_by_default, reasoning_effort: features.reasoning_effort, prompt_cache: features.prompt_cache, cache_control_breakpoints: features.cache_control_breakpoints, diff --git a/lib/foundation/fabro-config/src/layers/llm.rs b/lib/foundation/fabro-config/src/layers/llm.rs index 4f89c55fc..d1faa7e8d 100644 --- a/lib/foundation/fabro-config/src/layers/llm.rs +++ b/lib/foundation/fabro-config/src/layers/llm.rs @@ -244,6 +244,8 @@ pub struct ModelFeatures { #[serde(default, skip_serializing_if = "Option::is_none")] pub reasoning: Option, #[serde(default, skip_serializing_if = "Option::is_none")] + pub reasoning_by_default: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] pub reasoning_effort: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub prompt_cache: Option, @@ -647,6 +649,7 @@ provider = "bedrock" tools = true vision = true reasoning = true +reasoning_by_default = false reasoning_effort = "levels" prompt_cache = false "#; @@ -663,6 +666,7 @@ prompt_cache = false features.reasoning_effort, Some(fabro_model::ReasoningEffortFeature::Levels) ); + assert_eq!(features.reasoning_by_default, Some(false)); assert_eq!(features.prompt_cache, Some(false)); } diff --git a/lib/foundation/fabro-dev/src/commands/docs_options_reference.rs b/lib/foundation/fabro-dev/src/commands/docs_options_reference.rs index 9a2414e0a..e547e8e4c 100644 --- a/lib/foundation/fabro-dev/src/commands/docs_options_reference.rs +++ b/lib/foundation/fabro-dev/src/commands/docs_options_reference.rs @@ -326,6 +326,7 @@ cache_input_cost_per_mtok = 0.60 | `tools` | boolean | `false` | Whether the model supports tool calls. | | `vision` | boolean | `false` | Whether the model accepts image inputs. | | `reasoning` | boolean | `false` | Whether the model has reasoning behavior. | +| `reasoning_by_default` | boolean | effort-capable models: `true`; other models: `false` | Whether requests reason when no `reasoning_effort` is supplied. Set this explicitly for always-reasoning routes that do not expose an effort control, or for effort-capable routes whose provider defaults reasoning off. | | `reasoning_effort` | `"levels"` \| `"always_adaptive"` \| `"none"` | `"none"` | Whether the model endpoint supports a native reasoning-effort parameter. `levels` accepts discrete effort levels; `always_adaptive` accepts effort levels with natively always-on adaptive thinking; `none` has no native effort parameter. | | `prompt_cache` | boolean | `false` | Whether prompt cache pricing/usage applies. | | `sampling_params` | boolean | `true` | Whether the model accepts classic sampling parameters (`temperature`, `top_p`). | diff --git a/lib/foundation/fabro-model/src/catalog.rs b/lib/foundation/fabro-model/src/catalog.rs index 7ed634241..7fa9eb161 100644 --- a/lib/foundation/fabro-model/src/catalog.rs +++ b/lib/foundation/fabro-model/src/catalog.rs @@ -146,6 +146,11 @@ pub struct SettingsModelFeatures { pub vision: Option, #[serde(default)] pub reasoning: Option, + /// Whether requests reason when no effort control is supplied. When + /// omitted, effort-capable models default to `true` and other models to + /// `false`. + #[serde(default)] + pub reasoning_by_default: Option, #[serde(default)] pub reasoning_effort: Option, #[serde(default)] @@ -443,17 +448,20 @@ pub struct CatalogModelControls { #[derive(Debug, Clone, PartialEq)] pub struct CatalogModelSettings { - pub api_id: String, + pub api_id: String, /// Wire dialect for this model's route (the provider codec unless the /// model row overrides it). - pub codec: CodecKind, + pub codec: CodecKind, /// Billing family for this model (the provider policy unless the model /// row overrides it). - pub billing_policy: BillingPolicy, - pub agent_profile: AgentProfileKind, - pub controls: CatalogModelControls, - pub speed_costs: HashMap, - probe: bool, + pub billing_policy: BillingPolicy, + pub agent_profile: AgentProfileKind, + /// Whether the provider route reasons when a request omits an effort + /// control. + pub reasoning_by_default: bool, + pub controls: CatalogModelControls, + pub speed_costs: HashMap, + probe: bool, } #[derive(Debug, thiserror::Error)] @@ -578,6 +586,8 @@ pub enum CatalogBuildError { ReasoningEffortControlsWithoutReasoning { model: String }, #[error("model '{model}' declares reasoning_effort feature but features.reasoning is false")] ReasoningEffortWithoutReasoning { model: String }, + #[error("model '{model}' sets reasoning_by_default but features.reasoning is false")] + DefaultReasoningWithoutReasoning { model: String }, #[error( "model '{model}' declares cache_control_breakpoints but features.prompt_cache is false" )] @@ -1983,6 +1993,9 @@ fn merge_model_features_settings( tools: higher.tools.or(fallback.tools), vision: higher.vision.or(fallback.vision), reasoning: higher.reasoning.or(fallback.reasoning), + reasoning_by_default: higher + .reasoning_by_default + .or(fallback.reasoning_by_default), reasoning_effort: higher.reasoning_effort.or(fallback.reasoning_effort), prompt_cache: higher.prompt_cache.or(fallback.prompt_cache), cache_control_breakpoints: higher @@ -2214,6 +2227,14 @@ fn build_model( field: "features", })?; let model_features = build_model_features(model_id, features)?; + let reasoning_by_default = features + .reasoning_by_default + .unwrap_or_else(|| model_features.supports_reasoning_effort()); + if reasoning_by_default && !model_features.reasoning { + return Err(CatalogBuildError::DefaultReasoningWithoutReasoning { + model: model_id.to_string(), + }); + } let controls = build_model_controls(model_id, &model_features, settings)?; let costs = build_model_costs(settings.costs.as_ref()); let speed_costs = build_speed_costs(model_id, settings.costs.as_ref(), &controls)?; @@ -2255,6 +2276,7 @@ fn build_model( codec: resolve_model_codec(model_id, provider, settings.codec)?, billing_policy: settings.billing_policy.unwrap_or(provider.billing_policy), agent_profile: settings.agent_profile.unwrap_or(provider.agent_profile), + reasoning_by_default, controls, speed_costs, probe: settings.probe.unwrap_or_default(), @@ -2832,6 +2854,12 @@ enabled = true .get_on_provider(&bedrock, "claude-fable-5") .expect("fable row should be present"); assert!(!fable.features.sampling_params); + assert!( + catalog + .settings_for(fable) + .expect("fable settings should be present") + .reasoning_by_default + ); assert_eq!( catalog .model_settings_on_provider(&bedrock, "claude-fable-5") @@ -5915,6 +5943,7 @@ context_window = 1000 tools = true vision = false reasoning = true +reasoning_by_default = false reasoning_effort = "levels" prompt_cache = true @@ -5930,6 +5959,12 @@ reasoning_effort = ["low", "medium"] crate::ReasoningEffortFeature::Levels ); assert!(model.features.prompt_cache); + assert!( + !catalog + .model_settings("model") + .unwrap() + .reasoning_by_default + ); assert_eq!( catalog .model_settings("model") @@ -5974,6 +6009,12 @@ prompt_cache = true crate::ReasoningEffortFeature::AlwaysAdaptive ); assert!(model.supports_reasoning_effort()); + assert!( + catalog + .model_settings("model") + .unwrap() + .reasoning_by_default + ); // Always-adaptive models get the full default effort controls, same as // Levels. assert_eq!( @@ -6008,6 +6049,7 @@ context_window = 1000 tools = true vision = false reasoning = true +reasoning_by_default = true reasoning_effort = "none" [models.model.controls] @@ -6021,6 +6063,12 @@ reasoning_effort = ["low"] model.features.reasoning_effort, crate::ReasoningEffortFeature::None ); + assert!( + catalog + .model_settings("model") + .unwrap() + .reasoning_by_default + ); assert_eq!( catalog .model_settings("model") @@ -6098,6 +6146,38 @@ reasoning_effort = "levels" )); } + #[test] + fn catalog_from_settings_rejects_default_reasoning_without_reasoning() { + let settings = minimal_settings( + r#" +[providers.test] +display_name = "Test" +adapter = "openai" +agent_profile = "openai" + +[models.model] +provider = "test" +display_name = "Model" +family = "test" + +[models.model.limits] +context_window = 1000 + +[models.model.features] +tools = true +vision = false +reasoning = false +reasoning_by_default = true +"#, + ); + + assert!(matches!( + Catalog::from_settings(&settings).unwrap_err(), + CatalogBuildError::DefaultReasoningWithoutReasoning { model } + if model == "model" + )); + } + #[test] fn catalog_from_settings_rejects_cache_control_breakpoints_without_prompt_cache() { let settings = minimal_settings( diff --git a/lib/foundation/fabro-model/src/catalog/providers/bedrock.toml b/lib/foundation/fabro-model/src/catalog/providers/bedrock.toml index aeed22e46..71c74d686 100644 --- a/lib/foundation/fabro-model/src/catalog/providers/bedrock.toml +++ b/lib/foundation/fabro-model/src/catalog/providers/bedrock.toml @@ -369,6 +369,7 @@ max_output = 128000 tools = true vision = true reasoning = true +reasoning_by_default = true prompt_cache = true sampling_params = false diff --git a/lib/foundation/fabro-model/src/catalog/providers/kimi.toml b/lib/foundation/fabro-model/src/catalog/providers/kimi.toml index daa4b20c2..9da0d539a 100644 --- a/lib/foundation/fabro-model/src/catalog/providers/kimi.toml +++ b/lib/foundation/fabro-model/src/catalog/providers/kimi.toml @@ -23,6 +23,7 @@ max_output = 32768 tools = true vision = true reasoning = true +reasoning_by_default = true prompt_cache = true sampling_params = false From bf62450a281cf6804758e786fa93cf5dffe00e97 Mon Sep 17 00:00:00 2001 From: Release Repro Date: Fri, 24 Jul 2026 22:21:52 -0400 Subject: [PATCH 3/3] fix(agent): align summary prompt with output cap Interpolate the visible summary allowance after applying the model max_output cap, so low-output models are not asked to produce more text than the request permits. --- lib/components/fabro-agent/src/compaction.rs | 15 ++++++++------- 1 file changed, 8 insertions(+), 7 deletions(-) diff --git a/lib/components/fabro-agent/src/compaction.rs b/lib/components/fabro-agent/src/compaction.rs index 698d68275..2b09192b4 100644 --- a/lib/components/fabro-agent/src/compaction.rs +++ b/lib/components/fabro-agent/src/compaction.rs @@ -13,7 +13,7 @@ use crate::types::{AgentEvent, Message}; const APPROX_CHARS_PER_TOKEN: usize = 4; -/// Output budget for the visible summary text itself. +/// Maximum output budget for the visible summary text itself. const SUMMARY_MAX_TOKENS: i64 = 4096; /// Extra output budget for models that reason on every request. `max_tokens` @@ -115,6 +115,12 @@ pub(crate) async fn compact_context( ) }; + let max_tokens = summary_max_tokens( + provider_profile.reasons_by_default(), + provider_profile.max_output_tokens(), + ); + let visible_max_tokens = SUMMARY_MAX_TOKENS.min(max_tokens); + let summarization_prompt = format!( "You are creating a handoff document for a different coding assistant that will take over \ this task. That assistant will only see your summary and the most recent messages — nothing else \ @@ -126,17 +132,12 @@ Write a summary using EXACTLY these sections:\n\n\ ## Failed Approaches\nWhat was tried and didn't work, and why.\n\n\ ## Open Issues\nBugs, edge cases, or TODOs that remain.\n\n\ ## Next Steps\nWhat should happen next to make progress.\n\n\ -Keep the entire response under {SUMMARY_MAX_TOKENS} tokens.\n\n\ +Keep the entire response under {visible_max_tokens} tokens.\n\n\ Be thorough and specific — the assistant taking over has no prior context. Include file paths, \ function names, error messages, and exact values. Omit pleasantries and conversational filler.\ {file_ops_section}" ); - let max_tokens = summary_max_tokens( - provider_profile.reasons_by_default(), - provider_profile.max_output_tokens(), - ); - let summary_request = Request { model: provider_profile.model().to_string(), messages: vec![