diff --git a/docs/public/reference/user-configuration.mdx b/docs/public/reference/user-configuration.mdx index dd6afd83f..cfc3e09cb 100644 --- a/docs/public/reference/user-configuration.mdx +++ b/docs/public/reference/user-configuration.mdx @@ -279,6 +279,7 @@ cache_input_cost_per_mtok = 0.60 | `tools` | boolean | `false` | Whether the model supports tool calls. | | `vision` | boolean | `false` | Whether the model accepts image inputs. | | `reasoning` | boolean | `false` | Whether the model has reasoning behavior. | +| `reasoning_by_default` | boolean | effort-capable models: `true`; other models: `false` | Whether requests reason when no `reasoning_effort` is supplied. Set this explicitly for always-reasoning routes that do not expose an effort control, or for effort-capable routes whose provider defaults reasoning off. | | `reasoning_effort` | `"levels"` \| `"always_adaptive"` \| `"none"` | `"none"` | Whether the model endpoint supports a native reasoning-effort parameter. `levels` accepts discrete effort levels; `always_adaptive` accepts effort levels with natively always-on adaptive thinking; `none` has no native effort parameter. | | `prompt_cache` | boolean | `false` | Whether prompt cache pricing/usage applies. | | `sampling_params` | boolean | `true` | Whether the model accepts classic sampling parameters (`temperature`, `top_p`). | diff --git a/lib/components/fabro-agent/src/agent_profile.rs b/lib/components/fabro-agent/src/agent_profile.rs index b46095ff1..15140a3e2 100644 --- a/lib/components/fabro-agent/src/agent_profile.rs +++ b/lib/components/fabro-agent/src/agent_profile.rs @@ -52,6 +52,15 @@ pub trait AgentProfile: Send + Sync { self.catalog_model().and_then(Model::max_output) } + fn reasons_by_default(&self) -> bool { + let Some(catalog) = self.catalog() else { + return false; + }; + catalog + .model_settings_on_provider(&self.provider_id(), self.model()) + .is_some_and(|settings| settings.reasoning_by_default) + } + fn register_subagent_tools( &mut self, supervisor: SubAgentSupervisor, diff --git a/lib/components/fabro-agent/src/cli.rs b/lib/components/fabro-agent/src/cli.rs index 12e033e03..61551af4b 100644 --- a/lib/components/fabro-agent/src/cli.rs +++ b/lib/components/fabro-agent/src/cli.rs @@ -993,6 +993,7 @@ mod tests { tools: Some(true), vision: Some(false), reasoning: Some(false), + reasoning_by_default: None, reasoning_effort: None, prompt_cache: None, cache_control_breakpoints: None, @@ -1079,6 +1080,7 @@ mod tests { tools: Some(true), vision: Some(false), reasoning: Some(false), + reasoning_by_default: None, reasoning_effort: None, prompt_cache: None, cache_control_breakpoints: None, diff --git a/lib/components/fabro-agent/src/compaction.rs b/lib/components/fabro-agent/src/compaction.rs index e2331144b..2b09192b4 100644 --- a/lib/components/fabro-agent/src/compaction.rs +++ b/lib/components/fabro-agent/src/compaction.rs @@ -13,6 +13,15 @@ use crate::types::{AgentEvent, Message}; const APPROX_CHARS_PER_TOKEN: usize = 4; +/// Maximum output budget for the visible summary text itself. +const SUMMARY_MAX_TOKENS: i64 = 4096; + +/// Extra output budget for models that reason on every request. `max_tokens` +/// bounds reasoning *plus* visible output, so a reasoning model handed only +/// `SUMMARY_MAX_TOKENS` can spend the whole budget thinking and return a +/// successful response with empty content — a silently empty summary. +const REASONING_HEADROOM_TOKENS: i64 = 16_384; + #[derive(Debug, Clone, Copy, PartialEq, Eq, strum::IntoStaticStr)] #[strum(serialize_all = "snake_case")] pub(crate) enum ContextEstimateMethod { @@ -106,6 +115,12 @@ pub(crate) async fn compact_context( ) }; + let max_tokens = summary_max_tokens( + provider_profile.reasons_by_default(), + provider_profile.max_output_tokens(), + ); + let visible_max_tokens = SUMMARY_MAX_TOKENS.min(max_tokens); + let summarization_prompt = format!( "You are creating a handoff document for a different coding assistant that will take over \ this task. That assistant will only see your summary and the most recent messages — nothing else \ @@ -117,6 +132,7 @@ Write a summary using EXACTLY these sections:\n\n\ ## Failed Approaches\nWhat was tried and didn't work, and why.\n\n\ ## Open Issues\nBugs, edge cases, or TODOs that remain.\n\n\ ## Next Steps\nWhat should happen next to make progress.\n\n\ +Keep the entire response under {visible_max_tokens} tokens.\n\n\ Be thorough and specific — the assistant taking over has no prior context. Include file paths, \ function names, error messages, and exact values. Omit pleasantries and conversational filler.\ {file_ops_section}" @@ -136,7 +152,7 @@ function names, error messages, and exact values. Omit pleasantries and conversa response_format: None, temperature: None, top_p: None, - max_tokens: Some(4096), + max_tokens: Some(max_tokens), stop_sequences: None, reasoning_effort: None, speed: None, @@ -162,9 +178,10 @@ function names, error messages, and exact values. Omit pleasantries and conversa .into()); } + let (summary_text, summary_truncated) = truncate_summary_text(summary_text); debug!( summary_len = summary_text.len(), - "Compaction summary generated" + summary_truncated, max_tokens, "Compaction summary generated" ); let summary_content = format!( "A different assistant began this task and produced the following summary. \ @@ -184,6 +201,43 @@ Build on their progress — do not repeat completed steps.\n\n{summary_text}" Ok(()) } +/// Combined reasoning and visible-output budget for the summarization request. +/// +/// Compaction runs against the session's own model, so a reasoning session +/// summarizes with reasoning enabled and the budget has to cover the thinking +/// as well as the summary. Provider routes that reason by default get headroom +/// on top of the summary allowance. Every known model budget is capped at its +/// declared `max_output`. +fn summary_max_tokens(reasoning_by_default: bool, max_output: Option) -> i64 { + let budget = if reasoning_by_default { + SUMMARY_MAX_TOKENS + REASONING_HEADROOM_TOKENS + } else { + SUMMARY_MAX_TOKENS + }; + + max_output.map_or(budget, |limit| budget.min(limit)) +} + +/// Bound retained summary text with the same local bytes-per-token heuristic +/// used for context estimates. Provider APIs expose only one combined ceiling +/// for reasoning and visible output, so the larger request budget cannot +/// enforce this limit itself. +fn truncate_summary_text(summary: &str) -> (&str, bool) { + let max_bytes = summary_max_approx_bytes(); + if summary.len() <= max_bytes { + return (summary, false); + } + + let end = summary.floor_char_boundary(max_bytes); + (&summary[..end], true) +} + +fn summary_max_approx_bytes() -> usize { + usize::try_from(SUMMARY_MAX_TOKENS) + .unwrap_or(usize::MAX) + .saturating_mul(APPROX_CHARS_PER_TOKEN) +} + pub(crate) fn estimate_active_context_usage( system_prompt: &str, history: &History, @@ -314,6 +368,7 @@ mod tests { use std::time::SystemTime; use fabro_llm::types::{TokenCounts, ToolCall, ToolResult}; + use fabro_model::{Catalog, Model, ProviderId}; use super::*; use crate::event::Emitter; @@ -322,6 +377,88 @@ mod tests { use crate::tool_registry::ToolRegistry; use crate::types::Message; + fn catalog_model(provider: &ProviderId, id: &str) -> &'static Model { + Catalog::builtin() + .get_on_provider(provider, id) + .unwrap_or_else(|| panic!("{provider}/{id} missing from builtin catalog")) + } + + fn builtin_summary_max_tokens(provider: &ProviderId, id: &str) -> i64 { + let catalog = Catalog::builtin(); + let model = catalog_model(provider, id); + let settings = catalog + .settings_for(model) + .unwrap_or_else(|| panic!("{provider}/{id} missing catalog settings")); + summary_max_tokens(settings.reasoning_by_default, model.max_output()) + } + + #[test] + fn summary_budget_without_catalog_model_is_summary_allowance() { + assert_eq!(summary_max_tokens(false, None), SUMMARY_MAX_TOKENS); + // The default agent test profile has no catalog behind it. + let profile = TestProfile::new(); + assert_eq!( + summary_max_tokens(profile.reasons_by_default(), profile.max_output_tokens()), + SUMMARY_MAX_TOKENS + ); + } + + #[test] + fn summary_budget_for_non_reasoning_model_is_summary_allowance() { + // claude-haiku-4-5: reasoning = false. + assert_eq!( + builtin_summary_max_tokens(&ProviderId::anthropic(), "claude-haiku-4-5"), + SUMMARY_MAX_TOKENS + ); + } + + #[test] + fn summary_budget_for_model_without_effort_feature_is_summary_allowance() { + // claude-sonnet-4-5 reasons only when a request asks for it, and + // compaction never sends a reasoning effort. + let model = catalog_model(&ProviderId::anthropic(), "claude-sonnet-4-5"); + assert!(model.supports_reasoning()); + assert!(!model.supports_reasoning_effort()); + assert_eq!( + builtin_summary_max_tokens(&ProviderId::anthropic(), "claude-sonnet-4-5"), + SUMMARY_MAX_TOKENS + ); + } + + #[test] + fn summary_budget_for_always_adaptive_model_adds_reasoning_headroom() { + assert_eq!( + builtin_summary_max_tokens(&ProviderId::anthropic(), "claude-fable-5"), + SUMMARY_MAX_TOKENS + REASONING_HEADROOM_TOKENS + ); + } + + #[test] + fn summary_budget_for_effort_levels_model_adds_reasoning_headroom() { + assert_eq!( + builtin_summary_max_tokens(&ProviderId::anthropic(), "claude-opus-5"), + SUMMARY_MAX_TOKENS + REASONING_HEADROOM_TOKENS + ); + } + + #[test] + fn summary_budget_for_always_reasoning_route_without_effort_adds_headroom() { + let kimi = ProviderId::new("kimi"); + let model = catalog_model(&kimi, "kimi-k2.5"); + assert!(model.supports_reasoning()); + assert!(!model.supports_reasoning_effort()); + assert_eq!( + builtin_summary_max_tokens(&kimi, "kimi-k2.5"), + SUMMARY_MAX_TOKENS + REASONING_HEADROOM_TOKENS + ); + } + + #[test] + fn summary_budget_never_exceeds_model_max_output() { + assert_eq!(summary_max_tokens(true, Some(8_192)), 8_192); + assert_eq!(summary_max_tokens(false, Some(2_048)), 2_048); + } + #[test] fn render_turns_produces_labeled_text() { let turns = vec![ @@ -712,4 +849,29 @@ mod tests { "CompactionCompleted should be emitted on success" ); } + + #[tokio::test] + async fn compaction_bounds_retained_summary_to_visible_budget() { + let max_bytes = summary_max_approx_bytes(); + let overlong = format!("{}END", "€".repeat(max_bytes / 3 + 1)); + let CompactionTestResult { + result, history, .. + } = compact_with_summary(&overlong).await; + + result.expect("an overlong summary should be compacted after truncation"); + + let summary_turn = history + .turns() + .iter() + .find_map(|turn| match turn { + Message::System { content, .. } => Some(content), + _ => None, + }) + .expect("compacted history should contain a summary turn"); + let (_, retained_summary) = summary_turn + .split_once("\n\n") + .expect("summary turn should separate its header from the generated text"); + assert!(retained_summary.len() <= max_bytes); + assert!(!retained_summary.contains("END")); + } } diff --git a/lib/foundation/fabro-config/src/builders.rs b/lib/foundation/fabro-config/src/builders.rs index a1e00907c..980faa3f3 100644 --- a/lib/foundation/fabro-config/src/builders.rs +++ b/lib/foundation/fabro-config/src/builders.rs @@ -420,6 +420,7 @@ fn model_features_to_catalog(features: &LlmModelFeatures) -> model_catalog::Sett tools: features.tools, vision: features.vision, reasoning: features.reasoning, + reasoning_by_default: features.reasoning_by_default, reasoning_effort: features.reasoning_effort, prompt_cache: features.prompt_cache, cache_control_breakpoints: features.cache_control_breakpoints, diff --git a/lib/foundation/fabro-config/src/layers/llm.rs b/lib/foundation/fabro-config/src/layers/llm.rs index 4f89c55fc..d1faa7e8d 100644 --- a/lib/foundation/fabro-config/src/layers/llm.rs +++ b/lib/foundation/fabro-config/src/layers/llm.rs @@ -244,6 +244,8 @@ pub struct ModelFeatures { #[serde(default, skip_serializing_if = "Option::is_none")] pub reasoning: Option, #[serde(default, skip_serializing_if = "Option::is_none")] + pub reasoning_by_default: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] pub reasoning_effort: Option, #[serde(default, skip_serializing_if = "Option::is_none")] pub prompt_cache: Option, @@ -647,6 +649,7 @@ provider = "bedrock" tools = true vision = true reasoning = true +reasoning_by_default = false reasoning_effort = "levels" prompt_cache = false "#; @@ -663,6 +666,7 @@ prompt_cache = false features.reasoning_effort, Some(fabro_model::ReasoningEffortFeature::Levels) ); + assert_eq!(features.reasoning_by_default, Some(false)); assert_eq!(features.prompt_cache, Some(false)); } diff --git a/lib/foundation/fabro-dev/src/commands/docs_options_reference.rs b/lib/foundation/fabro-dev/src/commands/docs_options_reference.rs index 9a2414e0a..e547e8e4c 100644 --- a/lib/foundation/fabro-dev/src/commands/docs_options_reference.rs +++ b/lib/foundation/fabro-dev/src/commands/docs_options_reference.rs @@ -326,6 +326,7 @@ cache_input_cost_per_mtok = 0.60 | `tools` | boolean | `false` | Whether the model supports tool calls. | | `vision` | boolean | `false` | Whether the model accepts image inputs. | | `reasoning` | boolean | `false` | Whether the model has reasoning behavior. | +| `reasoning_by_default` | boolean | effort-capable models: `true`; other models: `false` | Whether requests reason when no `reasoning_effort` is supplied. Set this explicitly for always-reasoning routes that do not expose an effort control, or for effort-capable routes whose provider defaults reasoning off. | | `reasoning_effort` | `"levels"` \| `"always_adaptive"` \| `"none"` | `"none"` | Whether the model endpoint supports a native reasoning-effort parameter. `levels` accepts discrete effort levels; `always_adaptive` accepts effort levels with natively always-on adaptive thinking; `none` has no native effort parameter. | | `prompt_cache` | boolean | `false` | Whether prompt cache pricing/usage applies. | | `sampling_params` | boolean | `true` | Whether the model accepts classic sampling parameters (`temperature`, `top_p`). | diff --git a/lib/foundation/fabro-model/src/catalog.rs b/lib/foundation/fabro-model/src/catalog.rs index 7ed634241..7fa9eb161 100644 --- a/lib/foundation/fabro-model/src/catalog.rs +++ b/lib/foundation/fabro-model/src/catalog.rs @@ -146,6 +146,11 @@ pub struct SettingsModelFeatures { pub vision: Option, #[serde(default)] pub reasoning: Option, + /// Whether requests reason when no effort control is supplied. When + /// omitted, effort-capable models default to `true` and other models to + /// `false`. + #[serde(default)] + pub reasoning_by_default: Option, #[serde(default)] pub reasoning_effort: Option, #[serde(default)] @@ -443,17 +448,20 @@ pub struct CatalogModelControls { #[derive(Debug, Clone, PartialEq)] pub struct CatalogModelSettings { - pub api_id: String, + pub api_id: String, /// Wire dialect for this model's route (the provider codec unless the /// model row overrides it). - pub codec: CodecKind, + pub codec: CodecKind, /// Billing family for this model (the provider policy unless the model /// row overrides it). - pub billing_policy: BillingPolicy, - pub agent_profile: AgentProfileKind, - pub controls: CatalogModelControls, - pub speed_costs: HashMap, - probe: bool, + pub billing_policy: BillingPolicy, + pub agent_profile: AgentProfileKind, + /// Whether the provider route reasons when a request omits an effort + /// control. + pub reasoning_by_default: bool, + pub controls: CatalogModelControls, + pub speed_costs: HashMap, + probe: bool, } #[derive(Debug, thiserror::Error)] @@ -578,6 +586,8 @@ pub enum CatalogBuildError { ReasoningEffortControlsWithoutReasoning { model: String }, #[error("model '{model}' declares reasoning_effort feature but features.reasoning is false")] ReasoningEffortWithoutReasoning { model: String }, + #[error("model '{model}' sets reasoning_by_default but features.reasoning is false")] + DefaultReasoningWithoutReasoning { model: String }, #[error( "model '{model}' declares cache_control_breakpoints but features.prompt_cache is false" )] @@ -1983,6 +1993,9 @@ fn merge_model_features_settings( tools: higher.tools.or(fallback.tools), vision: higher.vision.or(fallback.vision), reasoning: higher.reasoning.or(fallback.reasoning), + reasoning_by_default: higher + .reasoning_by_default + .or(fallback.reasoning_by_default), reasoning_effort: higher.reasoning_effort.or(fallback.reasoning_effort), prompt_cache: higher.prompt_cache.or(fallback.prompt_cache), cache_control_breakpoints: higher @@ -2214,6 +2227,14 @@ fn build_model( field: "features", })?; let model_features = build_model_features(model_id, features)?; + let reasoning_by_default = features + .reasoning_by_default + .unwrap_or_else(|| model_features.supports_reasoning_effort()); + if reasoning_by_default && !model_features.reasoning { + return Err(CatalogBuildError::DefaultReasoningWithoutReasoning { + model: model_id.to_string(), + }); + } let controls = build_model_controls(model_id, &model_features, settings)?; let costs = build_model_costs(settings.costs.as_ref()); let speed_costs = build_speed_costs(model_id, settings.costs.as_ref(), &controls)?; @@ -2255,6 +2276,7 @@ fn build_model( codec: resolve_model_codec(model_id, provider, settings.codec)?, billing_policy: settings.billing_policy.unwrap_or(provider.billing_policy), agent_profile: settings.agent_profile.unwrap_or(provider.agent_profile), + reasoning_by_default, controls, speed_costs, probe: settings.probe.unwrap_or_default(), @@ -2832,6 +2854,12 @@ enabled = true .get_on_provider(&bedrock, "claude-fable-5") .expect("fable row should be present"); assert!(!fable.features.sampling_params); + assert!( + catalog + .settings_for(fable) + .expect("fable settings should be present") + .reasoning_by_default + ); assert_eq!( catalog .model_settings_on_provider(&bedrock, "claude-fable-5") @@ -5915,6 +5943,7 @@ context_window = 1000 tools = true vision = false reasoning = true +reasoning_by_default = false reasoning_effort = "levels" prompt_cache = true @@ -5930,6 +5959,12 @@ reasoning_effort = ["low", "medium"] crate::ReasoningEffortFeature::Levels ); assert!(model.features.prompt_cache); + assert!( + !catalog + .model_settings("model") + .unwrap() + .reasoning_by_default + ); assert_eq!( catalog .model_settings("model") @@ -5974,6 +6009,12 @@ prompt_cache = true crate::ReasoningEffortFeature::AlwaysAdaptive ); assert!(model.supports_reasoning_effort()); + assert!( + catalog + .model_settings("model") + .unwrap() + .reasoning_by_default + ); // Always-adaptive models get the full default effort controls, same as // Levels. assert_eq!( @@ -6008,6 +6049,7 @@ context_window = 1000 tools = true vision = false reasoning = true +reasoning_by_default = true reasoning_effort = "none" [models.model.controls] @@ -6021,6 +6063,12 @@ reasoning_effort = ["low"] model.features.reasoning_effort, crate::ReasoningEffortFeature::None ); + assert!( + catalog + .model_settings("model") + .unwrap() + .reasoning_by_default + ); assert_eq!( catalog .model_settings("model") @@ -6098,6 +6146,38 @@ reasoning_effort = "levels" )); } + #[test] + fn catalog_from_settings_rejects_default_reasoning_without_reasoning() { + let settings = minimal_settings( + r#" +[providers.test] +display_name = "Test" +adapter = "openai" +agent_profile = "openai" + +[models.model] +provider = "test" +display_name = "Model" +family = "test" + +[models.model.limits] +context_window = 1000 + +[models.model.features] +tools = true +vision = false +reasoning = false +reasoning_by_default = true +"#, + ); + + assert!(matches!( + Catalog::from_settings(&settings).unwrap_err(), + CatalogBuildError::DefaultReasoningWithoutReasoning { model } + if model == "model" + )); + } + #[test] fn catalog_from_settings_rejects_cache_control_breakpoints_without_prompt_cache() { let settings = minimal_settings( diff --git a/lib/foundation/fabro-model/src/catalog/providers/bedrock.toml b/lib/foundation/fabro-model/src/catalog/providers/bedrock.toml index aeed22e46..71c74d686 100644 --- a/lib/foundation/fabro-model/src/catalog/providers/bedrock.toml +++ b/lib/foundation/fabro-model/src/catalog/providers/bedrock.toml @@ -369,6 +369,7 @@ max_output = 128000 tools = true vision = true reasoning = true +reasoning_by_default = true prompt_cache = true sampling_params = false diff --git a/lib/foundation/fabro-model/src/catalog/providers/kimi.toml b/lib/foundation/fabro-model/src/catalog/providers/kimi.toml index 66758337e..7826caab9 100644 --- a/lib/foundation/fabro-model/src/catalog/providers/kimi.toml +++ b/lib/foundation/fabro-model/src/catalog/providers/kimi.toml @@ -24,6 +24,7 @@ max_output = 32768 tools = true vision = true reasoning = true +reasoning_by_default = true prompt_cache = true sampling_params = false